agent-shell-py 0.2.5__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/AGENTS.md +53 -1
  2. agent_shell_py-0.2.5/README.md → agent_shell_py-0.3.0/PKG-INFO +49 -1
  3. agent_shell_py-0.2.5/PKG-INFO → agent_shell_py-0.3.0/README.md +40 -10
  4. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/info.md +3 -2
  5. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/invoking-cli-agents/SKILL.md +45 -3
  6. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/invoking-cli-agents/api-reference.md +50 -4
  7. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/_version.py +2 -2
  8. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/claude_code_adapter.py +35 -20
  9. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/codex_adapter.py +33 -16
  10. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/copilot_cli_adapter.py +36 -20
  11. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/cursor_adapter.py +33 -16
  12. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/grok_adapter.py +33 -18
  13. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/opencode_adapter.py +34 -17
  14. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/pi_adapter.py +34 -17
  15. agent_shell_py-0.3.0/src/agent_shell/adapters/process_failure.py +35 -0
  16. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/response.py +6 -0
  17. agent_shell_py-0.3.0/src/agent_shell/execution.py +341 -0
  18. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/models/agent.py +10 -1
  19. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/process_cleanup.py +39 -20
  20. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/shell.py +32 -7
  21. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_cursor_e2e.py +14 -4
  22. agent_shell_py-0.3.0/tests/e2e/test_execution_host_e2e.py +46 -0
  23. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_health_check_e2e.py +2 -3
  24. agent_shell_py-0.3.0/tests/integration/test_execution_host.py +310 -0
  25. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_process_lifecycle.py +95 -9
  26. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cancel.py +4 -6
  27. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_cancel.py +4 -6
  28. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_cancel.py +4 -6
  29. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_cancel.py +4 -6
  30. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_cancel.py +4 -6
  31. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_cancel.py +4 -6
  32. agent_shell_py-0.3.0/tests/unit/test_pi_cancel.py +19 -0
  33. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_process_cleanup.py +32 -36
  34. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_process_group_registration.py +7 -5
  35. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell.py +32 -5
  36. agent_shell_py-0.2.5/tests/unit/test_pi_cancel.py +0 -21
  37. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/build.yml +0 -0
  38. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/ci.yml +0 -0
  39. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.github/workflows/publish.yml +0 -0
  40. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.gitignore +0 -0
  41. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/.python-version +0 -0
  42. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/LICENSE +0 -0
  43. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/assets/skill_banner.png +0 -0
  44. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/agent_parameter_comparison.md +0 -0
  45. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/disabled_tools.md +0 -0
  46. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/docs/development/total_token_count.md +0 -0
  47. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/pyproject.toml +0 -0
  48. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/skills/delegating-code-review/SKILL.md +0 -0
  49. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/__init__.py +0 -0
  50. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/__init__.py +0 -0
  51. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/agent_adapter_protocol.py +0 -0
  52. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/health.py +0 -0
  53. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/model_discovery.py +0 -0
  54. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/outcome.py +0 -0
  55. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/stderr_format.py +0 -0
  56. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/adapters/tool_denial.py +0 -0
  57. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/src/agent_shell/models/__init__.py +0 -0
  58. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/__init__.py +0 -0
  59. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/conftest.py +0 -0
  60. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/__init__.py +0 -0
  61. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_claude_code_e2e.py +0 -0
  62. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_codex_e2e.py +0 -0
  63. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_copilot_cli_e2e.py +0 -0
  64. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_grok_e2e.py +0 -0
  65. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_model_discovery_e2e.py +0 -0
  66. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_opencode_e2e.py +0 -0
  67. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/e2e/test_pi_e2e.py +0 -0
  68. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/__init__.py +0 -0
  69. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_claude_code_integration.py +0 -0
  70. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_claude_code_mcp_integration.py +0 -0
  71. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_codex_integration.py +0 -0
  72. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_codex_mcp_integration.py +0 -0
  73. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_copilot_cli_integration.py +0 -0
  74. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_copilot_cli_mcp_integration.py +0 -0
  75. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_cursor_integration.py +0 -0
  76. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_cursor_mcp_integration.py +0 -0
  77. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_grok_integration.py +0 -0
  78. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_grok_mcp_integration.py +0 -0
  79. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_health_check_integration.py +0 -0
  80. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_model_discovery_integration.py +0 -0
  81. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_opencode_integration.py +0 -0
  82. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_opencode_mcp_integration.py +0 -0
  83. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_pi_integration.py +0 -0
  84. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/integration/test_pi_mcp_integration.py +0 -0
  85. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/__init__.py +0 -0
  86. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/adapter_matrix.py +0 -0
  87. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/codex_fixtures.py +0 -0
  88. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/copilot_fixtures.py +0 -0
  89. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/cursor_fixtures.py +0 -0
  90. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/fixtures.py +0 -0
  91. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/grok_fixtures.py +0 -0
  92. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/opencode_fixtures.py +0 -0
  93. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/pi_fixtures.py +0 -0
  94. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_adapter_transport.py +0 -0
  95. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_execute.py +0 -0
  96. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_parse_event.py +0 -0
  97. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_codex_warnings.py +0 -0
  98. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_execute.py +0 -0
  99. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_parse_event.py +0 -0
  100. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_copilot_cli_stream.py +0 -0
  101. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_execute.py +0 -0
  102. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_parse_event.py +0 -0
  103. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_cursor_warnings.py +0 -0
  104. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_execute.py +0 -0
  105. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_execute_outcome.py +0 -0
  106. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_execute.py +0 -0
  107. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_parse_event.py +0 -0
  108. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_grok_warnings.py +0 -0
  109. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_health_probe.py +0 -0
  110. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_mcp_server_spec.py +0 -0
  111. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_model_discovery.py +0 -0
  112. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_models.py +0 -0
  113. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_execute.py +0 -0
  114. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_parse_event.py +0 -0
  115. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_spawn.py +0 -0
  116. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_opencode_stream.py +0 -0
  117. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_parse_event.py +0 -0
  118. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_execute.py +0 -0
  119. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_parse_event.py +0 -0
  120. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_pi_warnings.py +0 -0
  121. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_response_aggregation.py +0 -0
  122. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell_cancellation.py +0 -0
  123. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_shell_mcp.py +0 -0
  124. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_stderr_format.py +0 -0
  125. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_stream.py +0 -0
  126. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/tests/unit/test_tool_denial.py +0 -0
  127. {agent_shell_py-0.2.5 → agent_shell_py-0.3.0}/uv.lock +0 -0
@@ -8,6 +8,8 @@ A lightweight, async Python package that executes CLI coding agents headlessly a
8
8
  classDiagram
9
9
  class AgentShell {
10
10
  -AgentAdapter _adapter
11
+ +ExecutionHost execution_host
12
+ +IsolationPolicy isolation_policy
11
13
  +execute(cwd, prompt, ...) AgentResponse
12
14
  +stream(cwd, prompt, ...) AsyncIterator~StreamEvent~
13
15
  +health_check(cwd, model, timeout) HealthCheckResult
@@ -49,6 +51,8 @@ classDiagram
49
51
  +str session_id
50
52
  +float duration
51
53
  +int output_tokens
54
+ +int returncode
55
+ +int signal
52
56
  }
53
57
 
54
58
  class MCPServerSpec {
@@ -88,8 +92,39 @@ classDiagram
88
92
  +str session_id
89
93
  +int output_tokens
90
94
  +str error
95
+ +int returncode
96
+ +int signal
91
97
  }
92
98
 
99
+ class ExecutionHost {
100
+ <<Protocol>>
101
+ +launch(command, cwd, env, stdin, isolation_policy) RunHandle
102
+ }
103
+
104
+ class NativeExecutionHost {
105
+ +launch(command, cwd, env, stdin, isolation_policy) NativeRunHandle
106
+ }
107
+
108
+ class IsolationPolicy {
109
+ <<Protocol>>
110
+ +prepare(command, env) PreparedLaunch
111
+ }
112
+
113
+ class NoIsolation
114
+ class LinuxPidNamespaceIsolation
115
+
116
+ class RunHandle {
117
+ <<Protocol>>
118
+ +int pid
119
+ +int returncode
120
+ +wait() int
121
+ +communicate(input) tuple
122
+ +cancel() None
123
+ +release() None
124
+ }
125
+
126
+ class NativeRunHandle
127
+
93
128
  class AgentType {
94
129
  <<StrEnum>>
95
130
  CLAUDE_CODE
@@ -98,9 +133,17 @@ classDiagram
98
133
  CODEX
99
134
  PI
100
135
  CURSOR
136
+ GROK
101
137
  }
102
138
 
103
139
  AgentShell --> AgentAdapter : delegates to
140
+ AgentShell --> ExecutionHost : selects
141
+ AgentShell --> IsolationPolicy : selects
142
+ NativeExecutionHost ..|> ExecutionHost : satisfies
143
+ NativeExecutionHost --> NativeRunHandle : creates
144
+ NativeRunHandle ..|> RunHandle : satisfies
145
+ NoIsolation ..|> IsolationPolicy : satisfies
146
+ LinuxPidNamespaceIsolation ..|> IsolationPolicy : satisfies
104
147
  AgentShell --> AgentType : resolves via
105
148
  ClaudeCodeAdapter ..|> AgentAdapter : satisfies
106
149
  AgentShell ..> AgentResponse : returns on success
@@ -112,7 +155,16 @@ classDiagram
112
155
  ClaudeCodeAdapter ..> StreamEvent : parses NDJSON into
113
156
  ```
114
157
 
115
- The adapter pattern uses Python's `Protocol` (structural typing) rather than ABC, so adapters satisfy the contract implicitly without inheritance. Each adapter manages its own subprocess lifecycle, translating agent-specific CLI flags and NDJSON output into the shared `StreamEvent`/`AgentResponse` models.
158
+ The adapter pattern uses Python's `Protocol` (structural typing) rather than ABC, so adapters satisfy the contract implicitly without inheritance. Each adapter translates agent-specific CLI flags and NDJSON output into the shared `StreamEvent`/`AgentResponse` models, while the selected `ExecutionHost` owns process creation and returns a per-run `RunHandle`.
159
+
160
+ Execution location and protection are separate axes. Existing callers default to
161
+ `NativeExecutionHost()` plus `NoIsolation()`. `LinuxPidNamespaceIsolation` is an opt-in direct
162
+ signal boundary: a tiny init/reaper is PID 1 and the CLI is PID 2 or later, so child-namespace
163
+ processes cannot see or signal AgentShell's ancestors. It requires Linux, `unshare`, and enabled
164
+ unprivileged user/PID namespaces; an unavailable explicit request raises
165
+ `IsolationUnavailableError` and never falls back. This is not a general sandbox and does not
166
+ restrict filesystem, credentials, network, tools, or resources. The host/policy applies to
167
+ `execute()`, `stream()`, and `health_check()`; model discovery and MCP configuration remain local.
116
168
 
117
169
  `output_tokens` is a cost measure — the billed output-token count, which **includes reasoning tokens** (billed at the output rate). Each adapter normalises this so the value is consistent across agents (e.g. OpenCode reports reasoning in a sibling field, so its adapter adds it back).
118
170
 
@@ -1,3 +1,12 @@
1
+ Metadata-Version: 2.5
2
+ Name: agent-shell-py
3
+ Version: 0.3.0
4
+ Summary: A lightweight abstraction for executing CLI coding agents headlessly
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Requires-Python: >=3.12
8
+ Description-Content-Type: text/markdown
9
+
1
10
  # Agent Shell
2
11
  Agent Shell is a light weight abstraction for executing a cli coding agent headlessly
3
12
  and returning the output that can be used programatically as a unified contract
@@ -10,6 +19,8 @@ and returning the output that can be used programatically as a unified contract
10
19
  behind a common adapter protocol.
11
20
  - **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
12
21
  or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
22
+ - **Composable execution policy** — preserve native execution by default, or opt into Linux PID
23
+ namespace isolation without changing an agent adapter.
13
24
  - **Session resumption** — continue any conversation by passing back its `session_id`.
14
25
  - **Normalized cost & tokens** — consistent `cost` and `output_tokens` (reasoning included)
15
26
  regardless of how each CLI reports them.
@@ -70,6 +81,40 @@ separately.
70
81
 
71
82
  ## Examples
72
83
 
84
+ ### Execution host and isolation
85
+
86
+ Existing callers remain unchanged. Omitting both settings means
87
+ `NativeExecutionHost()` plus `NoIsolation()`:
88
+
89
+ ```python
90
+ shell = AgentShell(agent_type=AgentType.CLAUDE_CODE)
91
+ ```
92
+
93
+ To protect the AgentShell owner from broad same-user cleanup commands such as `pkill -f`,
94
+ explicitly request Linux PID namespace isolation:
95
+
96
+ ```python
97
+ from agent_shell.execution import LinuxPidNamespaceIsolation
98
+
99
+ shell = AgentShell(
100
+ agent_type=AgentType.CLAUDE_CODE,
101
+ isolation_policy=LinuxPidNamespaceIsolation(),
102
+ )
103
+ ```
104
+
105
+ The policy runs a tiny namespace init as PID 1 and the real CLI as PID 2 or later. Processes in
106
+ that child namespace cannot see or signal AgentShell's ancestor processes. The feature requires
107
+ Linux, the `unshare` command, and kernel support for unprivileged user/PID namespaces. If any are
108
+ unavailable, launching raises `IsolationUnavailableError`; it never silently falls back.
109
+
110
+ This is **direct-signal protection, not a sandbox**. It does not restrict files, credentials,
111
+ network access, tools, or resource consumption. Any background descendants are also terminated
112
+ when the isolated namespace ends. `execute()`, `stream()`, and `health_check()` use the selected
113
+ host/policy; model discovery and MCP configuration remain local management operations.
114
+
115
+ `NativeExecutionHost` is currently the only host implementation. Host and isolation are separate
116
+ so future tmux/Herdr hosts can compose with policies without creating one class per combination.
117
+
73
118
  ### Execute
74
119
 
75
120
  ```python
@@ -108,7 +153,9 @@ follow_up = await shell.execute(
108
153
  `execute()` raises `AgentExecutionError` instead of returning when a run failed — an `error`
109
154
  event was emitted, the terminal `result` had `content == "error"`, or no terminal `result`
110
155
  arrived at all. `str(e)` is the bare reason; the exception also carries whatever partial
111
- `response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing.
156
+ `response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing. When
157
+ the CLI process itself exits unsuccessfully, `returncode` is also populated; signal termination
158
+ uses Python's negative-returncode convention and supplies the positive signal number separately.
112
159
 
113
160
  ```python
114
161
  from agent_shell.models.agent import AgentExecutionError
@@ -117,6 +164,7 @@ try:
117
164
  response = await shell.execute(cwd="/path/to/project", prompt="Fix the failing test")
118
165
  except AgentExecutionError as e:
119
166
  print(f"run failed: {e}") # e.g. "500 model name=qwen3.6-27b-8Q failed to load"
167
+ print(e.returncode, e.signal) # e.g. -15, 15 for SIGTERM; otherwise None when unavailable
120
168
  ```
121
169
 
122
170
  ### Stream
@@ -1,12 +1,3 @@
1
- Metadata-Version: 2.4
2
- Name: agent-shell-py
3
- Version: 0.2.5
4
- Summary: A lightweight abstraction for executing CLI coding agents headlessly
5
- License-Expression: MIT
6
- License-File: LICENSE
7
- Requires-Python: >=3.12
8
- Description-Content-Type: text/markdown
9
-
10
1
  # Agent Shell
11
2
  Agent Shell is a light weight abstraction for executing a cli coding agent headlessly
12
3
  and returning the output that can be used programatically as a unified contract
@@ -19,6 +10,8 @@ and returning the output that can be used programatically as a unified contract
19
10
  behind a common adapter protocol.
20
11
  - **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
21
12
  or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
13
+ - **Composable execution policy** — preserve native execution by default, or opt into Linux PID
14
+ namespace isolation without changing an agent adapter.
22
15
  - **Session resumption** — continue any conversation by passing back its `session_id`.
23
16
  - **Normalized cost & tokens** — consistent `cost` and `output_tokens` (reasoning included)
24
17
  regardless of how each CLI reports them.
@@ -79,6 +72,40 @@ separately.
79
72
 
80
73
  ## Examples
81
74
 
75
+ ### Execution host and isolation
76
+
77
+ Existing callers remain unchanged. Omitting both settings means
78
+ `NativeExecutionHost()` plus `NoIsolation()`:
79
+
80
+ ```python
81
+ shell = AgentShell(agent_type=AgentType.CLAUDE_CODE)
82
+ ```
83
+
84
+ To protect the AgentShell owner from broad same-user cleanup commands such as `pkill -f`,
85
+ explicitly request Linux PID namespace isolation:
86
+
87
+ ```python
88
+ from agent_shell.execution import LinuxPidNamespaceIsolation
89
+
90
+ shell = AgentShell(
91
+ agent_type=AgentType.CLAUDE_CODE,
92
+ isolation_policy=LinuxPidNamespaceIsolation(),
93
+ )
94
+ ```
95
+
96
+ The policy runs a tiny namespace init as PID 1 and the real CLI as PID 2 or later. Processes in
97
+ that child namespace cannot see or signal AgentShell's ancestor processes. The feature requires
98
+ Linux, the `unshare` command, and kernel support for unprivileged user/PID namespaces. If any are
99
+ unavailable, launching raises `IsolationUnavailableError`; it never silently falls back.
100
+
101
+ This is **direct-signal protection, not a sandbox**. It does not restrict files, credentials,
102
+ network access, tools, or resource consumption. Any background descendants are also terminated
103
+ when the isolated namespace ends. `execute()`, `stream()`, and `health_check()` use the selected
104
+ host/policy; model discovery and MCP configuration remain local management operations.
105
+
106
+ `NativeExecutionHost` is currently the only host implementation. Host and isolation are separate
107
+ so future tmux/Herdr hosts can compose with policies without creating one class per combination.
108
+
82
109
  ### Execute
83
110
 
84
111
  ```python
@@ -117,7 +144,9 @@ follow_up = await shell.execute(
117
144
  `execute()` raises `AgentExecutionError` instead of returning when a run failed — an `error`
118
145
  event was emitted, the terminal `result` had `content == "error"`, or no terminal `result`
119
146
  arrived at all. `str(e)` is the bare reason; the exception also carries whatever partial
120
- `response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing.
147
+ `response`/`cost`/`session_id`/`duration`/`output_tokens` the run produced before failing. When
148
+ the CLI process itself exits unsuccessfully, `returncode` is also populated; signal termination
149
+ uses Python's negative-returncode convention and supplies the positive signal number separately.
121
150
 
122
151
  ```python
123
152
  from agent_shell.models.agent import AgentExecutionError
@@ -126,6 +155,7 @@ try:
126
155
  response = await shell.execute(cwd="/path/to/project", prompt="Fix the failing test")
127
156
  except AgentExecutionError as e:
128
157
  print(f"run failed: {e}") # e.g. "500 model name=qwen3.6-27b-8Q failed to load"
158
+ print(e.returncode, e.signal) # e.g. -15, 15 for SIGTERM; otherwise None when unavailable
129
159
  ```
130
160
 
131
161
  ### Stream
@@ -61,8 +61,9 @@ uv run pytest tests/e2e -v
61
61
 
62
62
  > [!WARNING]
63
63
  > E2E tests may mutate real user configuration files. MCP tests can call an agent's real
64
- > `mcp add` and `mcp remove` commands, affecting files such as `~/.claude.json`,
65
- > `~/.config/opencode/opencode.json`, `~/.copilot/mcp-config.json`, or Codex configuration.
64
+ > MCP commands or edit its config directly, affecting files such as `~/.claude.json`,
65
+ > `~/.config/opencode/opencode.json`, `~/.copilot/mcp-config.json`, `~/.cursor/mcp.json`,
66
+ > or Codex configuration.
66
67
  > Tests use unique names and `finally` cleanup where implemented, but forced termination,
67
68
  > a CLI crash, or a machine failure can prevent cleanup. The CLI may also rewrite config
68
69
  > formatting even when the temporary entry is removed. Review the selected E2E test before
@@ -6,7 +6,7 @@ description: >-
6
6
  restricting tools, or checking agent/model health. Supports Claude Code, OpenCode,
7
7
  Copilot CLI, Codex, Pi, Cursor, and Grok. Keywords: AgentShell, list_models, headless
8
8
  agent, model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
9
- output_tokens.
9
+ output_tokens, execution host, PID namespace, isolation policy.
10
10
  ---
11
11
 
12
12
  # Invoking CLI Agents with AgentShell
@@ -44,7 +44,44 @@ uv add agent-shell-py
44
44
 
45
45
  AgentShell has two invocation methods — `execute()` collects a complete response, `stream()`
46
46
  yields events in real-time — plus helpers for model discovery, health checks, and MCP server
47
- management. All are async.
47
+ management. All are async. Invocation has three independent choices:
48
+
49
+ - `agent_type`: which CLI (Claude Code, Codex, etc.)
50
+ - `execution_host`: where/how the process is owned (`NativeExecutionHost` today)
51
+ - `isolation_policy`: what protection surrounds it (`NoIsolation` or Linux PID isolation)
52
+
53
+ ### Execution Host and Isolation Policy
54
+
55
+ Existing code is backward-compatible: omitting both execution settings selects native execution
56
+ with no isolation.
57
+
58
+ ```python
59
+ shell = AgentShell(agent_type=AgentType.CODEX)
60
+ # Equivalent to NativeExecutionHost() + NoIsolation()
61
+ ```
62
+
63
+ Opt into direct-signal protection when an agent may run broad same-user cleanup commands such as
64
+ `pkill -f`:
65
+
66
+ ```python
67
+ from agent_shell.execution import LinuxPidNamespaceIsolation
68
+
69
+ shell = AgentShell(
70
+ agent_type=AgentType.CODEX,
71
+ isolation_policy=LinuxPidNamespaceIsolation(),
72
+ )
73
+ ```
74
+
75
+ The Linux policy puts a tiny init/reaper at namespace PID 1 and the actual CLI at PID 2 or later.
76
+ The CLI cannot see or signal AgentShell's ancestor processes. It requires Linux, `unshare`, and
77
+ kernel support for unprivileged user/PID namespaces. An unavailable requested policy raises
78
+ `IsolationUnavailableError` before the CLI starts; AgentShell never silently falls back.
79
+
80
+ This policy is **not a sandbox**: filesystem, credentials, network, agent tools, and resource use
81
+ remain available. Background descendants cannot outlive the isolated namespace. It applies to
82
+ `execute()`, `stream()`, and `health_check()`; `list_models()` and MCP configuration are local
83
+ management operations. `NativeExecutionHost` is the only shipped host today. Tmux and Herdr are
84
+ future host possibilities, not current APIs.
48
85
 
49
86
  ### Discover Available Model Strings
50
87
 
@@ -117,6 +154,7 @@ async for event in shell.stream(
117
154
  print(event.content)
118
155
  elif event.type == "error":
119
156
  print(f"[error] {event.content}")
157
+ print(event.returncode, event.signal) # populated for process-level failures
120
158
  elif event.type == "result":
121
159
  print(f"Done ({event.content}). Cost: ${event.cost:.4f}, {event.output_tokens} tok")
122
160
  ```
@@ -187,7 +225,7 @@ Capabilities differ by agent. `output_tokens` is populated on all of them; the r
187
225
  | Copilot CLI | ✅ | ⚠️ `bash`, `edit` only | ✅ | ❌ `0.0` | ✅ real | ✅ |
188
226
  | Codex | ❌ | ⚠️ `web_search` only | ✅ | ❌ `0.0` | ❌ `0.0` | ✅ |
189
227
  | Pi | ✅ | ⚠️ `bash`, `edit`, `read` | ✅ | ⚠️ paid providers only | ❌ `0.0` | ❌ raises |
190
- | Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real | ❌ raises |
228
+ | Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real | ✅ user-scope |
191
229
  | Grok | ✅ | ✅ all canonical | ✅ | ⚠️ may be `0.0` | ✅ real | ✅ user-scope |
192
230
 
193
231
  A `✅` for `allowed_tools` means the flag is passed — but it only *enforces* with
@@ -271,6 +309,7 @@ except AgentExecutionError as e:
271
309
  print(f"failed: {e}") # str(e) == e.reason, the bare cause
272
310
  print(e.response) # text produced before the failure, if any
273
311
  print(e.cost, e.session_id, e.duration, e.output_tokens)
312
+ print(e.returncode, e.signal) # -15 and 15 for SIGTERM; None if not process-derived
274
313
  else:
275
314
  print(response.response)
276
315
  ```
@@ -347,6 +386,7 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
347
386
  | See agent thinking | `include_thinking=True` in `stream()` |
348
387
  | Check an agent/model works | `await shell.health_check(cwd, model=...)` |
349
388
  | Cancel a running agent | `KeyboardInterrupt` (handled automatically) |
389
+ | Protect the owner from broad process kills | `isolation_policy=LinuxPidNamespaceIsolation()` |
350
390
 
351
391
  ## Common Mistakes
352
392
 
@@ -366,6 +406,8 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
366
406
  | Not catching `AgentExecutionError` | `execute()` raises on a failed run — catch it |
367
407
  | Ignoring `UserWarning` on a deny | An unenforceable deny is warned, not applied — the tool is NOT blocked |
368
408
  | Ignoring `session_id` for multi-step work | Without it, each call starts fresh |
409
+ | Treating PID isolation as a sandbox | It only blocks direct signalling of ancestors; separately restrict files, network, credentials, tools, and resources |
410
+ | Assuming isolation silently degrades | Requested isolation raises `IsolationUnavailableError` when unavailable; catch it or fail the operation |
369
411
 
370
412
  ## API Reference
371
413
 
@@ -4,6 +4,7 @@
4
4
  `MCPServerSpec`, `HealthCheckResult`
5
5
  - [StreamEvent types](#event-types)
6
6
  - [AgentShell class](#agentshell-class) — invocation, model discovery, health, MCP management
7
+ - [Execution hosts and isolation](#execution-hosts-and-isolation)
7
8
  - [AgentAdapter protocol](#agentadapter-protocol)
8
9
  - [Agent-specific notes](#agent-specific-notes)
9
10
 
@@ -58,6 +59,8 @@ class AgentExecutionError(Exception):
58
59
  session_id: str | None = None,
59
60
  duration: float = 0.0,
60
61
  output_tokens: int = 0,
62
+ returncode: int | None = None, # raw process status; negative means signal
63
+ signal: int | None = None, # positive signal number for signal termination
61
64
  ): ...
62
65
  ```
63
66
 
@@ -75,6 +78,8 @@ class StreamEvent:
75
78
  session_id: str | None = None # On session-start and "result" events
76
79
  output_tokens: int = 0 # Cumulative generated tokens (on "result" events)
77
80
  error: str | None = None # Why a failing "result" failed, when recoverable (Pi)
81
+ returncode: int | None = None # Set on process-level "error" events
82
+ signal: int | None = None # Positive signal number when returncode is negative
78
83
  ```
79
84
 
80
85
  ### MCPServerSpec
@@ -124,7 +129,8 @@ Canonical event types emitted by `stream()`:
124
129
 
125
130
  > A `result` event carries `cost`, `duration`, `output_tokens` and `session_id` on the agents
126
131
  > that report them. On a failing result, `error` holds the reason when the adapter recovered a
127
- > structured one (Pi); it is `None` otherwise.
132
+ > structured one (Pi); it is `None` otherwise. A process-level `error` also carries
133
+ > `returncode`; if a signal terminated it, `returncode` is negative and `signal` is positive.
128
134
 
129
135
  > Codex emits the session-start event as `type="session"` (not `"system"`). If you branch on
130
136
  > the session event across agents, match both.
@@ -140,7 +146,12 @@ Canonical event types emitted by `stream()`:
140
146
  from agent_shell.shell import AgentShell
141
147
 
142
148
  class AgentShell:
143
- def __init__(self, agent_type: AgentType): ...
149
+ def __init__(
150
+ self,
151
+ agent_type: AgentType,
152
+ execution_host: ExecutionHost | None = None, # default NativeExecutionHost()
153
+ isolation_policy: IsolationPolicy | None = None, # default NoIsolation()
154
+ ): ...
144
155
  # raises ValueError for an AgentType with no registered adapter
145
156
 
146
157
  async def execute(
@@ -174,6 +185,38 @@ class AgentShell:
174
185
  async def list_mcp_servers(self) -> list[MCPServerSpec]: ...
175
186
  ```
176
187
 
188
+ ## Execution Hosts and Isolation
189
+
190
+ ```python
191
+ from agent_shell.execution import (
192
+ ExecutionHost,
193
+ IsolationPolicy,
194
+ IsolationUnavailableError,
195
+ LinuxPidNamespaceIsolation,
196
+ NativeExecutionHost,
197
+ NativeRunHandle,
198
+ NoIsolation,
199
+ PreparedLaunch,
200
+ RunHandle,
201
+ )
202
+ ```
203
+
204
+ `ExecutionHost` creates a `RunHandle` for one command. The handle exposes `pid`, `stdin`,
205
+ `stdout`, `stderr`, `returncode`, `wait()`, `communicate()`, `cancel()`, and `release()`.
206
+ `NativeExecutionHost` is currently the only concrete host. Agent adapters use handles internally;
207
+ existing `execute()` and `stream()` callers do not need to manage them.
208
+
209
+ `NoIsolation` preserves historical native execution. `LinuxPidNamespaceIsolation` uses rootless
210
+ user + PID namespaces, with a tiny PID 1 reaper and the CLI at PID 2 or later. It protects
211
+ AgentShell's ancestors from direct same-user signals sent inside the child namespace. It is not a
212
+ filesystem, credential, network, tool, resource, or general security sandbox. Background
213
+ descendants terminate when the namespace ends.
214
+
215
+ The Linux policy requires `unshare` and supporting kernel configuration. An explicit request that
216
+ cannot be satisfied raises `IsolationUnavailableError` before launching the CLI and never falls
217
+ back to `NoIsolation`. The host/policy selection applies to `execute()`, `stream()`, and
218
+ `health_check()`; model discovery and MCP configuration remain local management operations.
219
+
177
220
  ### Model discovery semantics
178
221
 
179
222
  `list_models()` reads the selected CLI's account/workspace-aware catalog without sending an
@@ -273,8 +316,11 @@ class AgentAdapter(Protocol):
273
316
  it rather than failing (like Pi; unlike Claude Code, OpenCode, Copilot and Codex, which all
274
317
  reject one). A matching id is therefore not proof a prior transcript was continued.
275
318
  - `duration` and `output_tokens` are real (`usage.outputTokens`); `cost` is always `0.0` — Cursor
276
- reports no cost. MCP-management methods raise `NotImplementedError` (`cursor-agent mcp` has no
277
- add/remove, and its `list` returns only name+status, which cannot rebuild an `MCPServerSpec`).
319
+ reports no cost.
320
+ - MCP add/remove/list are supported by directly managing user-scope `~/.cursor/mcp.json`, because
321
+ `cursor-agent mcp` has no add/remove subcommands and its list output lacks full configuration.
322
+ Writes are atomic and user-only. Same-transport updates preserve Cursor-native fields that
323
+ `MCPServerSpec` cannot represent.
278
324
 
279
325
  ### Grok
280
326
  - Headless: `grok -p --output-format streaming-messages-json` (full assistant blocks; not
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.5'
22
- __version_tuple__ = version_tuple = (0, 2, 5)
21
+ __version__ = version = '0.3.0'
22
+ __version_tuple__ = version_tuple = (0, 3, 0)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -1,12 +1,24 @@
1
1
  import asyncio
2
2
  import codecs
3
3
  import json
4
- import os
5
4
  import logging
5
+ import os
6
6
  import warnings
7
7
  from pathlib import Path
8
8
  from typing import AsyncIterator
9
9
 
10
+ from agent_shell.adapters.health import run_health_probe
11
+ from agent_shell.adapters.model_discovery import decode_model_output, run_model_command
12
+ from agent_shell.adapters.process_failure import process_failure_event
13
+ from agent_shell.adapters.response import collect_response
14
+ from agent_shell.adapters.stderr_format import format_stderr
15
+ from agent_shell.adapters.tool_denial import resolve_disallowed_tools
16
+ from agent_shell.execution import (
17
+ ExecutionHost,
18
+ IsolationPolicy,
19
+ NativeExecutionHost,
20
+ NoIsolation,
21
+ )
10
22
  from agent_shell.models.agent import (
11
23
  AgentResponse,
12
24
  HealthCheckResult,
@@ -15,15 +27,8 @@ from agent_shell.models.agent import (
15
27
  StreamEvent,
16
28
  )
17
29
  from agent_shell.process_cleanup import (
18
- create_grouped_process,
19
- kill_process_group,
20
30
  release_process,
21
31
  )
22
- from agent_shell.adapters.health import run_health_probe
23
- from agent_shell.adapters.model_discovery import decode_model_output, run_model_command
24
- from agent_shell.adapters.response import collect_response
25
- from agent_shell.adapters.stderr_format import format_stderr
26
- from agent_shell.adapters.tool_denial import resolve_disallowed_tools
27
32
 
28
33
  logger = logging.getLogger("agent_shell.claude_code_adapter")
29
34
 
@@ -38,8 +43,18 @@ _DISALLOWED_TOOL_MAP = {
38
43
  }
39
44
 
40
45
  class ClaudeCodeAdapter():
41
- def __init__(self):
46
+ def __init__(
47
+ self,
48
+ execution_host: ExecutionHost | None = None,
49
+ isolation_policy: IsolationPolicy | None = None,
50
+ ):
42
51
  self._active_processes = []
52
+ self._execution_host = (
53
+ execution_host if execution_host is not None else NativeExecutionHost()
54
+ )
55
+ self._isolation_policy = (
56
+ isolation_policy if isolation_policy is not None else NoIsolation()
57
+ )
43
58
 
44
59
  async def execute(
45
60
  self,
@@ -114,9 +129,10 @@ class ClaudeCodeAdapter():
114
129
  logger.debug("Command: %s", cmd)
115
130
  logger.info("Process started (cwd=%s)", os.path.abspath(cwd))
116
131
 
117
- process = await create_grouped_process(
132
+ process = await self._execution_host.launch(
118
133
  cmd,
119
134
  cwd=os.path.abspath(cwd),
135
+ isolation_policy=self._isolation_policy,
120
136
  )
121
137
 
122
138
  self._active_processes.append(process)
@@ -174,10 +190,10 @@ class ClaudeCodeAdapter():
174
190
  child_exited = True
175
191
 
176
192
  stderr = await stderr_task
177
- if stderr and process.returncode != 0:
178
- error_msg = format_stderr(stderr)
179
- logger.warning("Process exited with code %d: %s", process.returncode, error_msg)
180
- yield StreamEvent(type="error", content=error_msg)
193
+ failure = process_failure_event(process.returncode, stderr)
194
+ if failure is not None:
195
+ logger.warning("Process failed (%d): %s", process.returncode, failure.content)
196
+ yield failure
181
197
  finally:
182
198
  # Teardown must live here, not after the read loop: on an exception, or when the
183
199
  # consumer abandons the stream, the normal path never runs and the still-running
@@ -188,8 +204,8 @@ class ClaudeCodeAdapter():
188
204
  # the child is still alive and still registered until a later turn of the loop, and
189
205
  # if the loop is torn down first (asyncio.run cancelling pending tasks) it never
190
206
  # runs at all. That last case is what the atexit net in process_cleanup covers.
191
- release_process(process, self._active_processes, stderr_task,
192
- child_exited=child_exited)
207
+ await release_process(process, self._active_processes, stderr_task,
208
+ child_exited=child_exited)
193
209
 
194
210
  def _parse_event(self, event: dict, include_thinking: bool) -> list[StreamEvent]:
195
211
  t = event.get("type", "")
@@ -229,9 +245,10 @@ class ClaudeCodeAdapter():
229
245
  return events
230
246
 
231
247
  async def cancel(self) -> None:
232
- for process in self._active_processes:
233
- kill_process_group(process)
248
+ processes = list(self._active_processes)
234
249
  self._active_processes.clear()
250
+ for process in processes:
251
+ await process.cancel()
235
252
 
236
253
  async def health_check(
237
254
  self,
@@ -437,5 +454,3 @@ class ClaudeCodeAdapter():
437
454
  return {"returncode": process.returncode, "stdout": stdout_text, "stderr": stderr_text}
438
455
 
439
456
 
440
-
441
-