agent-shell-py 0.2.3__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/AGENTS.md +15 -3
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/PKG-INFO +49 -11
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/README.md +48 -10
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/agent_parameter_comparison.md +64 -13
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/invoking-cli-agents/SKILL.md +16 -13
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/invoking-cli-agents/api-reference.md +18 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/_version.py +2 -2
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/copilot_cli_adapter.py +1 -1
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/cursor_adapter.py +168 -7
- agent_shell_py-0.2.5/src/agent_shell/adapters/grok_adapter.py +477 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/response.py +1 -1
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/models/agent.py +1 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/shell.py +2 -0
- agent_shell_py-0.2.5/tests/e2e/test_grok_e2e.py +142 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_health_check_e2e.py +1 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_model_discovery_e2e.py +1 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_copilot_cli_integration.py +13 -0
- agent_shell_py-0.2.5/tests/integration/test_cursor_mcp_integration.py +367 -0
- agent_shell_py-0.2.5/tests/integration/test_grok_integration.py +288 -0
- agent_shell_py-0.2.5/tests/integration/test_grok_mcp_integration.py +222 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_health_check_integration.py +3 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_model_discovery_integration.py +35 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_process_lifecycle.py +12 -5
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/adapter_matrix.py +7 -1
- agent_shell_py-0.2.5/tests/unit/grok_fixtures.py +154 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_adapter_transport.py +9 -2
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_execute_outcome.py +2 -2
- agent_shell_py-0.2.5/tests/unit/test_grok_cancel.py +21 -0
- agent_shell_py-0.2.5/tests/unit/test_grok_execute.py +119 -0
- agent_shell_py-0.2.5/tests/unit/test_grok_parse_event.py +199 -0
- agent_shell_py-0.2.5/tests/unit/test_grok_warnings.py +110 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell.py +8 -0
- agent_shell_py-0.2.3/tests/integration/test_cursor_mcp_integration.py +0 -36
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/build.yml +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/ci.yml +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/publish.yml +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.gitignore +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.python-version +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/LICENSE +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/assets/skill_banner.png +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/disabled_tools.md +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/info.md +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/total_token_count.md +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/pyproject.toml +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/delegating-code-review/SKILL.md +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/agent_adapter_protocol.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/claude_code_adapter.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/codex_adapter.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/health.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/model_discovery.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/opencode_adapter.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/outcome.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/pi_adapter.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/stderr_format.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/tool_denial.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/models/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/process_cleanup.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/conftest.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_claude_code_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_codex_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_copilot_cli_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_cursor_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_opencode_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_pi_e2e.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_claude_code_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_claude_code_mcp_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_codex_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_codex_mcp_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_copilot_cli_mcp_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_cursor_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_opencode_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_opencode_mcp_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_pi_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_pi_mcp_integration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/__init__.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/codex_fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/copilot_fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/cursor_fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/opencode_fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/pi_fixtures.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_warnings.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_stream.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_warnings.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_health_probe.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_mcp_server_spec.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_model_discovery.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_models.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_spawn.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_stream.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_cancel.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_execute.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_parse_event.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_warnings.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_process_cleanup.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_process_group_registration.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_response_aggregation.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell_cancellation.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell_mcp.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_stderr_format.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_stream.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_tool_denial.py +0 -0
- {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/uv.lock +0 -0
|
@@ -142,6 +142,7 @@ command, and never substitutes a static catalog. "Available" means advertised as
|
|
|
142
142
|
- [x] Codex
|
|
143
143
|
- [x] Pi
|
|
144
144
|
- [x] Cursor
|
|
145
|
+
- [x] Grok
|
|
145
146
|
|
|
146
147
|
## MCP Server Configuration
|
|
147
148
|
|
|
@@ -170,8 +171,19 @@ All adapters write to user-scope configuration:
|
|
|
170
171
|
| OpenCode | direct JSON file write | `~/.config/opencode/opencode.json` |
|
|
171
172
|
| Copilot CLI | direct JSON file write | `~/.copilot/mcp-config.json` |
|
|
172
173
|
| Codex | `codex mcp add` subprocess | Codex config |
|
|
173
|
-
|
|
174
|
-
|
|
174
|
+
| Cursor | direct JSON file write | `~/.cursor/mcp.json` |
|
|
175
|
+
| Grok | `grok mcp add --scope user` subprocess | `~/.grok/config.toml` (managed by CLI) |
|
|
176
|
+
|
|
177
|
+
Adds are idempotent (update existing entries with the same name). Cursor preserves native fields
|
|
178
|
+
that `MCPServerSpec` cannot represent when an update keeps the same transport, and writes its file
|
|
179
|
+
atomically with user-only permissions. Removes warn rather than raise when the named server is not
|
|
180
|
+
found. Claude Code listing reads the user-scope `mcpServers`
|
|
181
|
+
entries from `~/.claude.json` directly, avoiding the health checks and human-readable output of
|
|
182
|
+
`claude mcp list`. Cursor manages user-scope MCP entries directly in `~/.cursor/mcp.json` because
|
|
183
|
+
its `mcp` subcommands have no add/remove commands. Grok listing reads user-scope `mcp_servers`
|
|
184
|
+
entries from `~/.grok/config.toml` directly for the same reason. Pi's MCP add/remove/list methods
|
|
185
|
+
raise `NotImplementedError`. Pi manages capability via `pi install` extensions, which needs
|
|
186
|
+
investigation before wiring up.
|
|
175
187
|
|
|
176
188
|
## Test Philosophy
|
|
177
189
|
|
|
@@ -183,7 +195,7 @@ Tests validate real functionality, not code coverage metrics. Three tiers, each
|
|
|
183
195
|
| **Integration** | Full flow through `AgentShell` -> `Adapter` -> parser with mocked subprocess | Yes | No |
|
|
184
196
|
| **E2E** | Real CLI calls; usually real API costs | No (local only) | Yes |
|
|
185
197
|
|
|
186
|
-
The model-discovery E2E test is the exception: it calls all
|
|
198
|
+
The model-discovery E2E test is the exception: it calls all seven real CLIs but only reads
|
|
187
199
|
metadata, so it sends no inference request and incurs no model-token cost.
|
|
188
200
|
|
|
189
201
|
Integration tests mirror the E2E tests but substitute mocked subprocesses emitting captured CLI
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-shell-py
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: A lightweight abstraction for executing CLI coding agents headlessly
|
|
5
5
|
License-Expression: MIT
|
|
6
6
|
License-File: LICENSE
|
|
@@ -15,8 +15,8 @@ and returning the output that can be used programatically as a unified contract
|
|
|
15
15
|
|
|
16
16
|
- **One unified contract** — the same `execute`, `stream`, `health_check`, and
|
|
17
17
|
`list_models` API across every agent; swap the backend without changing consuming code.
|
|
18
|
-
- **
|
|
19
|
-
common adapter protocol.
|
|
18
|
+
- **Seven CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, and Grok
|
|
19
|
+
behind a common adapter protocol.
|
|
20
20
|
- **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
|
|
21
21
|
or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
|
|
22
22
|
- **Session resumption** — continue any conversation by passing back its `session_id`.
|
|
@@ -64,7 +64,7 @@ Or install both skills globally for every coding agent supported by AgentShell:
|
|
|
64
64
|
```bash
|
|
65
65
|
npx skills add ScottRBK/agent-shell --global \
|
|
66
66
|
--skill '*' \
|
|
67
|
-
--agent claude-code opencode github-copilot codex pi cursor \
|
|
67
|
+
--agent claude-code opencode github-copilot codex pi cursor grok \
|
|
68
68
|
--yes
|
|
69
69
|
```
|
|
70
70
|
|
|
@@ -213,10 +213,10 @@ response = await shell.execute(
|
|
|
213
213
|
`mcp__server__tool`, or a harness-specific name like `Write`, or Copilot's `view`).
|
|
214
214
|
- Deny takes precedence over auto-approve on every backend that supports it.
|
|
215
215
|
- Where a backend cannot enforce a deny, the adapter emits a `UserWarning` listing the
|
|
216
|
-
ignored tools rather than failing silently. Coverage varies: Claude and
|
|
217
|
-
all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
|
|
218
|
-
name for its other tools); Codex can only deny `web_search`; Cursor cannot
|
|
219
|
-
per-call deny (its tool policy lives in `.cursor/cli.json`).
|
|
216
|
+
ignored tools rather than failing silently. Coverage varies: Claude, OpenCode, and Grok
|
|
217
|
+
enforce all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
|
|
218
|
+
verbatim name for its other tools); Codex can only deny `web_search`; Cursor cannot
|
|
219
|
+
enforce any per-call deny (its tool policy lives in `.cursor/cli.json`).
|
|
220
220
|
- Denying `edit` or `read` is **best-effort**: a model can still modify or read files through
|
|
221
221
|
the shell, so also deny `bash` when you need a hard file boundary.
|
|
222
222
|
|
|
@@ -276,8 +276,39 @@ print(f"Session: {response.session_id}")
|
|
|
276
276
|
> tools are auto-*rejected* but the run still completes. `allowed_tools`, `effort`, and
|
|
277
277
|
> `disallowed_tools` are **ignored** — Cursor exposes no per-call tool policy or effort flag
|
|
278
278
|
> (tool policy lives in `.cursor/cli.json`), so each emits a `UserWarning`. On a Free plan
|
|
279
|
-
> only `model=None`/`"auto"` works. MCP
|
|
280
|
-
> `
|
|
279
|
+
> only `model=None`/`"auto"` works. MCP add/remove/list are supported by directly managing
|
|
280
|
+
> the user-scope `~/.cursor/mcp.json` file because `cursor-agent mcp` has no add/remove
|
|
281
|
+
> subcommands.
|
|
282
|
+
|
|
283
|
+
### Grok
|
|
284
|
+
|
|
285
|
+
```python
|
|
286
|
+
from agent_shell.shell import AgentShell
|
|
287
|
+
from agent_shell.models.agent import AgentType
|
|
288
|
+
|
|
289
|
+
shell = AgentShell(agent_type=AgentType.GROK)
|
|
290
|
+
|
|
291
|
+
response = await shell.execute(
|
|
292
|
+
cwd="/path/to/project",
|
|
293
|
+
prompt="Can you tell me about this project?",
|
|
294
|
+
model="grok-4.5",
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
print(response.response)
|
|
298
|
+
print(f"Session: {response.session_id}")
|
|
299
|
+
print(f"Cost: ${response.cost:.4f}")
|
|
300
|
+
```
|
|
301
|
+
|
|
302
|
+
> **Note:** Grok runs headlessly via `grok -p --output-format streaming-messages-json`
|
|
303
|
+
> (full assistant blocks — not token-delta `streaming-json`, which would break
|
|
304
|
+
> newline-joined text collection). With `auto_approve=True` (the default) the adapter
|
|
305
|
+
> passes `--always-approve`. `effort` maps to `--reasoning-effort`, `allowed_tools` to
|
|
306
|
+
> `--tools`, and `disallowed_tools` to `--disallowed-tools` (canonical `bash` maps to
|
|
307
|
+
> `run_terminal_cmd` — the working deny id, not init.tools' `run_terminal_command`).
|
|
308
|
+
> The terminal `result` event carries cost (may be `0` on some auth paths), duration, and
|
|
309
|
+
> raw `usage.output_tokens` (reasoning is already inside that figure when reported).
|
|
310
|
+
> MCP add/remove/list are supported via `grok mcp` with **user scope only**
|
|
311
|
+
> (`~/.grok/config.toml`).
|
|
281
312
|
|
|
282
313
|
## MCP Servers
|
|
283
314
|
|
|
@@ -318,7 +349,14 @@ await shell.add_mcp_server(MCPServerSpec(
|
|
|
318
349
|
))
|
|
319
350
|
```
|
|
320
351
|
|
|
321
|
-
`add_mcp_server`
|
|
352
|
+
`add_mcp_server` adds or updates a server with the same name. Cursor preserves native fields
|
|
353
|
+
that `MCPServerSpec` cannot represent when an update keeps the same transport. The configuration
|
|
354
|
+
is written atomically with user-only permissions. `remove_mcp_server` warns rather than raises
|
|
355
|
+
when the named server is not found. `list_mcp_servers()` works for
|
|
356
|
+
Claude Code, OpenCode, Copilot CLI, Codex, Cursor, and Grok. Claude Code reads user-scope
|
|
357
|
+
entries directly from `~/.claude.json`, Cursor from `~/.cursor/mcp.json`, and Grok from
|
|
358
|
+
`~/.grok/config.toml`, so listing does not launch configured servers for health checks. MCP
|
|
359
|
+
is not supported for Pi; all three MCP methods raise `NotImplementedError`.
|
|
322
360
|
|
|
323
361
|
## Logging
|
|
324
362
|
|
|
@@ -6,8 +6,8 @@ and returning the output that can be used programatically as a unified contract
|
|
|
6
6
|
|
|
7
7
|
- **One unified contract** — the same `execute`, `stream`, `health_check`, and
|
|
8
8
|
`list_models` API across every agent; swap the backend without changing consuming code.
|
|
9
|
-
- **
|
|
10
|
-
common adapter protocol.
|
|
9
|
+
- **Seven CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, and Grok
|
|
10
|
+
behind a common adapter protocol.
|
|
11
11
|
- **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
|
|
12
12
|
or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
|
|
13
13
|
- **Session resumption** — continue any conversation by passing back its `session_id`.
|
|
@@ -55,7 +55,7 @@ Or install both skills globally for every coding agent supported by AgentShell:
|
|
|
55
55
|
```bash
|
|
56
56
|
npx skills add ScottRBK/agent-shell --global \
|
|
57
57
|
--skill '*' \
|
|
58
|
-
--agent claude-code opencode github-copilot codex pi cursor \
|
|
58
|
+
--agent claude-code opencode github-copilot codex pi cursor grok \
|
|
59
59
|
--yes
|
|
60
60
|
```
|
|
61
61
|
|
|
@@ -204,10 +204,10 @@ response = await shell.execute(
|
|
|
204
204
|
`mcp__server__tool`, or a harness-specific name like `Write`, or Copilot's `view`).
|
|
205
205
|
- Deny takes precedence over auto-approve on every backend that supports it.
|
|
206
206
|
- Where a backend cannot enforce a deny, the adapter emits a `UserWarning` listing the
|
|
207
|
-
ignored tools rather than failing silently. Coverage varies: Claude and
|
|
208
|
-
all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
|
|
209
|
-
name for its other tools); Codex can only deny `web_search`; Cursor cannot
|
|
210
|
-
per-call deny (its tool policy lives in `.cursor/cli.json`).
|
|
207
|
+
ignored tools rather than failing silently. Coverage varies: Claude, OpenCode, and Grok
|
|
208
|
+
enforce all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
|
|
209
|
+
verbatim name for its other tools); Codex can only deny `web_search`; Cursor cannot
|
|
210
|
+
enforce any per-call deny (its tool policy lives in `.cursor/cli.json`).
|
|
211
211
|
- Denying `edit` or `read` is **best-effort**: a model can still modify or read files through
|
|
212
212
|
the shell, so also deny `bash` when you need a hard file boundary.
|
|
213
213
|
|
|
@@ -267,8 +267,39 @@ print(f"Session: {response.session_id}")
|
|
|
267
267
|
> tools are auto-*rejected* but the run still completes. `allowed_tools`, `effort`, and
|
|
268
268
|
> `disallowed_tools` are **ignored** — Cursor exposes no per-call tool policy or effort flag
|
|
269
269
|
> (tool policy lives in `.cursor/cli.json`), so each emits a `UserWarning`. On a Free plan
|
|
270
|
-
> only `model=None`/`"auto"` works. MCP
|
|
271
|
-
> `
|
|
270
|
+
> only `model=None`/`"auto"` works. MCP add/remove/list are supported by directly managing
|
|
271
|
+
> the user-scope `~/.cursor/mcp.json` file because `cursor-agent mcp` has no add/remove
|
|
272
|
+
> subcommands.
|
|
273
|
+
|
|
274
|
+
### Grok
|
|
275
|
+
|
|
276
|
+
```python
|
|
277
|
+
from agent_shell.shell import AgentShell
|
|
278
|
+
from agent_shell.models.agent import AgentType
|
|
279
|
+
|
|
280
|
+
shell = AgentShell(agent_type=AgentType.GROK)
|
|
281
|
+
|
|
282
|
+
response = await shell.execute(
|
|
283
|
+
cwd="/path/to/project",
|
|
284
|
+
prompt="Can you tell me about this project?",
|
|
285
|
+
model="grok-4.5",
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
print(response.response)
|
|
289
|
+
print(f"Session: {response.session_id}")
|
|
290
|
+
print(f"Cost: ${response.cost:.4f}")
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
> **Note:** Grok runs headlessly via `grok -p --output-format streaming-messages-json`
|
|
294
|
+
> (full assistant blocks — not token-delta `streaming-json`, which would break
|
|
295
|
+
> newline-joined text collection). With `auto_approve=True` (the default) the adapter
|
|
296
|
+
> passes `--always-approve`. `effort` maps to `--reasoning-effort`, `allowed_tools` to
|
|
297
|
+
> `--tools`, and `disallowed_tools` to `--disallowed-tools` (canonical `bash` maps to
|
|
298
|
+
> `run_terminal_cmd` — the working deny id, not init.tools' `run_terminal_command`).
|
|
299
|
+
> The terminal `result` event carries cost (may be `0` on some auth paths), duration, and
|
|
300
|
+
> raw `usage.output_tokens` (reasoning is already inside that figure when reported).
|
|
301
|
+
> MCP add/remove/list are supported via `grok mcp` with **user scope only**
|
|
302
|
+
> (`~/.grok/config.toml`).
|
|
272
303
|
|
|
273
304
|
## MCP Servers
|
|
274
305
|
|
|
@@ -309,7 +340,14 @@ await shell.add_mcp_server(MCPServerSpec(
|
|
|
309
340
|
))
|
|
310
341
|
```
|
|
311
342
|
|
|
312
|
-
`add_mcp_server`
|
|
343
|
+
`add_mcp_server` adds or updates a server with the same name. Cursor preserves native fields
|
|
344
|
+
that `MCPServerSpec` cannot represent when an update keeps the same transport. The configuration
|
|
345
|
+
is written atomically with user-only permissions. `remove_mcp_server` warns rather than raises
|
|
346
|
+
when the named server is not found. `list_mcp_servers()` works for
|
|
347
|
+
Claude Code, OpenCode, Copilot CLI, Codex, Cursor, and Grok. Claude Code reads user-scope
|
|
348
|
+
entries directly from `~/.claude.json`, Cursor from `~/.cursor/mcp.json`, and Grok from
|
|
349
|
+
`~/.grok/config.toml`, so listing does not launch configured servers for health checks. MCP
|
|
350
|
+
is not supported for Pi; all three MCP methods raise `NotImplementedError`.
|
|
313
351
|
|
|
314
352
|
## Logging
|
|
315
353
|
|
{agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/agent_parameter_comparison.md
RENAMED
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
# Agent CLI Parameter Comparison
|
|
2
2
|
|
|
3
3
|
Comparison of headless/non-interactive configuration across supported CLI coding agents.
|
|
4
|
-
Last updated: 2026-08-
|
|
4
|
+
Last updated: 2026-08-08
|
|
5
5
|
|
|
6
|
-
> The summary matrix below has no Pi or
|
|
7
|
-
> per-agent detail sections, the model-discovery section, and the
|
|
6
|
+
> The summary matrix below has no Pi, Cursor, or Grok column (it predates those adapters);
|
|
7
|
+
> see the per-agent detail sections, the model-discovery section, and the
|
|
8
|
+
> `disallowed_tools` table.
|
|
8
9
|
|
|
9
10
|
> Every "measured 2026-07-26" claim below comes from a real three-call run per agent (first
|
|
10
11
|
> turn, resume on the returned id, then a session-less turn), recorded by the resume e2e tests
|
|
@@ -55,6 +56,7 @@ Discovery reads CLI metadata and makes no inference call.
|
|
|
55
56
|
| Codex | `codex debug models` | visible model `slug` |
|
|
56
57
|
| Pi | `pi --no-approve --list-models` | `provider/model` |
|
|
57
58
|
| Cursor | `cursor-agent models` | ID before the first ` - ` |
|
|
59
|
+
| Grok | `grok models` | name after optional `* `, before `(default)` |
|
|
58
60
|
|
|
59
61
|
Important semantics:
|
|
60
62
|
|
|
@@ -74,9 +76,9 @@ Important semantics:
|
|
|
74
76
|
malformed protocol/output failures are surfaced with actionable errors.
|
|
75
77
|
- AgentShell does not cache or replace results with a static fallback catalog.
|
|
76
78
|
|
|
77
|
-
The behavior was verified against
|
|
78
|
-
spend. CI covers captured real outputs through `AgentShell`; a local E2E
|
|
79
|
-
real discovery commands.
|
|
79
|
+
The behavior was verified against the installed CLIs on 2026-08-08 without inference
|
|
80
|
+
spend (Grok included). CI covers captured real outputs through `AgentShell`; a local E2E
|
|
81
|
+
test checks all seven real discovery commands.
|
|
80
82
|
|
|
81
83
|
## Claude Code
|
|
82
84
|
|
|
@@ -126,9 +128,9 @@ real discovery commands.
|
|
|
126
128
|
- **Headless mode**: `-p` / `--prompt` for one-shot; `--acp` for Agent Client Protocol
|
|
127
129
|
- **Model**: `--model <model>`; use `auto` to let Copilot choose
|
|
128
130
|
- **Effort**: `--effort` / `--reasoning-effort` with choices
|
|
129
|
-
`none|minimal|low|medium|high|xhigh|max`. AgentShell
|
|
130
|
-
|
|
131
|
-
|
|
131
|
+
`none|minimal|low|medium|high|xhigh|max`. AgentShell treats `None` and `""` as omitted, so
|
|
132
|
+
Copilot uses its default. Explicit values, including `"none"`, are accepted case-insensitively,
|
|
133
|
+
validated against these choices, and passed to Copilot via `--effort` in lowercase.
|
|
132
134
|
- **Allowed tools**: `--allow-tool`, `--deny-tool`, `--available-tools`,
|
|
133
135
|
`--excluded-tools`, `--allow-all-tools`
|
|
134
136
|
- **Auto-approve**: `--allow-all-tools`; `--allow-all` / `--yolo` also grant path and URL
|
|
@@ -191,13 +193,61 @@ real discovery commands.
|
|
|
191
193
|
cursor-agent ACCEPTS an unknown id — `--resume=<never-seen-uuid>` starts a session under that
|
|
192
194
|
id and echoes it back rather than failing, so id identity proves the flag was passed through
|
|
193
195
|
and honoured, not that a prior transcript was replayed (measured 2026-07-26)
|
|
194
|
-
- **MCP**: `cursor-agent mcp` = login/list/list-tools/enable/disable only (no add/remove)
|
|
195
|
-
servers
|
|
196
|
-
|
|
197
|
-
|
|
196
|
+
- **MCP**: `cursor-agent mcp` = login/list/list-tools/enable/disable only (no add/remove).
|
|
197
|
+
The adapter manages user-scope servers directly in `~/.cursor/mcp.json`, which provides the
|
|
198
|
+
transport details that `mcp list` omits. Writes are atomic and user-only; same-transport
|
|
199
|
+
updates preserve Cursor-native fields that `MCPServerSpec` cannot represent.
|
|
198
200
|
- **Usage**: the terminal `result` event carries `usage.outputTokens` (undocumented but real)
|
|
199
201
|
and `duration_ms`; there is no cost field, so `cost` is `0.0`
|
|
200
202
|
|
|
203
|
+
## Grok (xAI Grok Build)
|
|
204
|
+
|
|
205
|
+
Flags below are the ones the adapter actually emits (`_build_command` in
|
|
206
|
+
`src/agent_shell/adapters/grok_adapter.py`); this is not a full survey of the Grok CLI.
|
|
207
|
+
Measured against grok 1.0.0 (3cd0d0cbce) on 2026-08-08.
|
|
208
|
+
|
|
209
|
+
- **Headless mode**: `-p` / `--single <PROMPT>` with
|
|
210
|
+
`--output-format streaming-messages-json` (Anthropic Messages NDJSON on stdout).
|
|
211
|
+
The adapter deliberately does **not** use `streaming-json`: that format emits token
|
|
212
|
+
`text.data` fragments, and execute()'s `"\n".join` over text events would explode
|
|
213
|
+
replies (same class of bug as issue #6; pi waits for `text_end`, copilot uses full
|
|
214
|
+
`assistant.message`)
|
|
215
|
+
- **Model**: `-m` / `--model` (e.g. `grok-4.5`). `grok models` prints plain text
|
|
216
|
+
(`Default model:` + `Available models:` list); the adapter returns the bare selectors
|
|
217
|
+
- **Effort**: `--reasoning-effort` / `--effort`
|
|
218
|
+
(`none|minimal|low|medium|high|xhigh|max`)
|
|
219
|
+
- **Allowed tools**: `--tools` (comma-joined native tool ids such as `read_file,list_dir`)
|
|
220
|
+
- **Disallowed tools**: `--disallowed-tools` (comma-joined). Canonical map (measured):
|
|
221
|
+
`bash`→`run_terminal_cmd` (NOT `run_terminal_command` — that id appears in
|
|
222
|
+
`system/init.tools` but is a no-op as a deny), `edit`→`search_replace,write`,
|
|
223
|
+
`read`→`read_file`, `web_search`/`web_fetch` one-to-one
|
|
224
|
+
- **Auto-approve**: `--always-approve`
|
|
225
|
+
- **Output format**: adapter uses `streaming-messages-json` only
|
|
226
|
+
- **Working directory**: process cwd (also has `--cwd`, unused by the adapter)
|
|
227
|
+
- **Session**: `--resume` / `-r` `[SESSION_ID_OR_TITLE]`, `--continue`, `--session-id`
|
|
228
|
+
(new id only). The adapter resumes with `--resume <id>`. `system/init` and `result`
|
|
229
|
+
carry `session_id`; a resumed run should report the same id (see e2e resume test)
|
|
230
|
+
- **Usage / cost**: terminal `result` carries `total_cost_usd` (may be `0` on some
|
|
231
|
+
OAuth/pool paths), `duration_ms`, and `usage.output_tokens`. When
|
|
232
|
+
`usage.reasoning_tokens` is present it is a **subset** of `output_tokens` (Grok's
|
|
233
|
+
`total_tokens` math uses `output_tokens` alone) — AgentShell reports raw
|
|
234
|
+
`output_tokens` and must not add them. `duration = duration_ms/1000`
|
|
235
|
+
- **Stream events (adapter mapping)** — Cursor/Claude-shaped:
|
|
236
|
+
- `system/init` → `StreamEvent(type="system", session_id=...)`
|
|
237
|
+
- `assistant.message.content[]` text blocks → `StreamEvent(type="text")`
|
|
238
|
+
- thinking blocks → `StreamEvent(type="thinking")` when `include_thinking`
|
|
239
|
+
- `tool_use` / `server_tool_use` blocks → `StreamEvent(type="tool_use", content=name)`
|
|
240
|
+
- `result` → `StreamEvent(type="result", content="ok"|"error", ...)` via `is_error`;
|
|
241
|
+
optional `errors[]` populates `StreamEvent.error`
|
|
242
|
+
- **MCP**: full CLI — `grok mcp add|remove|list|enable|disable|doctor`. Adapter uses
|
|
243
|
+
`grok mcp add --scope user` (add-or-update; no pre-remove) and
|
|
244
|
+
`grok mcp remove --scope user` (unscoped remove can hit project config). Listing reads
|
|
245
|
+
`~/.grok/config.toml` `[mcp_servers.*]` (stdio: `command`/`args`/`env`; http:
|
|
246
|
+
`url`/`headers`)
|
|
247
|
+
- **Auth**: `grok login` or `XAI_API_KEY`. Model listing works while logged in via
|
|
248
|
+
grok.com account; execution requires auth
|
|
249
|
+
- **Project rules**: reads `AGENTS.md` and `CLAUDE.md` (Claude Code compatible)
|
|
250
|
+
|
|
201
251
|
## Pi
|
|
202
252
|
|
|
203
253
|
Flags below are the ones the adapter actually emits (`_build_command` in
|
|
@@ -253,6 +303,7 @@ through **verbatim** (e.g. `mcp__server__tool`, or a harness-specific name like
|
|
|
253
303
|
| Codex | `-c web_search="disabled"` only | no name-based deny; web_search key verified on codex-cli 0.133.0 but version-fragile (upstream `web_search_mode`), guarded by an e2e test; every other canonical/verbatim name warns and is ignored |
|
|
254
304
|
| Cursor | none (no per-call flag) | tool policy is config-file only (`.cursor/cli.json`); the adapter has no `canonical → native` map, so **every** deny (canonical or verbatim) warns and is ignored |
|
|
255
305
|
| Pi | `--exclude-tools` (comma-joined) | `edit` → `edit,write`; no web tool, so those warn |
|
|
306
|
+
| Grok | `--disallowed-tools` (comma-joined) | `bash`→`run_terminal_cmd` (not init.tools' `run_terminal_command`); `edit`→`search_replace,write`; `read`→`read_file`; web tools one-to-one |
|
|
256
307
|
|
|
257
308
|
When an adapter cannot honor a requested canonical deny it emits a `UserWarning` listing
|
|
258
309
|
the ignored names rather than silently dropping the deny (fail-loud). A caller who knows a
|
|
@@ -4,8 +4,8 @@ description: >-
|
|
|
4
4
|
Use when invoking a CLI coding agent from Python, delegating to a sub-agent, discovering
|
|
5
5
|
available model strings, orchestrating agents, streaming output, resuming sessions,
|
|
6
6
|
restricting tools, or checking agent/model health. Supports Claude Code, OpenCode,
|
|
7
|
-
Copilot CLI, Codex, Pi, and
|
|
8
|
-
model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
|
|
7
|
+
Copilot CLI, Codex, Pi, Cursor, and Grok. Keywords: AgentShell, list_models, headless
|
|
8
|
+
agent, model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
|
|
9
9
|
output_tokens.
|
|
10
10
|
---
|
|
11
11
|
|
|
@@ -21,7 +21,7 @@ passes `model` strings through verbatim and does not manage credentials.
|
|
|
21
21
|
|
|
22
22
|
## When to Use
|
|
23
23
|
|
|
24
|
-
- You need to invoke Claude Code, OpenCode, Copilot CLI, Codex, Pi, or
|
|
24
|
+
- You need to invoke Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, or Grok from Python
|
|
25
25
|
- You want to delegate a coding task to a sub-agent and collect the result
|
|
26
26
|
- You need to orchestrate multi-step workflows across agents
|
|
27
27
|
- You want to stream agent output in real-time
|
|
@@ -151,11 +151,11 @@ Both `execute()` and `stream()` take the same parameters.
|
|
|
151
151
|
|-----------|------|---------|---------|
|
|
152
152
|
| `cwd` | `str` | required | Working directory (must exist, else `ValueError`) |
|
|
153
153
|
| `prompt` | `str` | required | Task or question for the agent |
|
|
154
|
-
| `allowed_tools` | `list[str] \| None` | `None` | Whitelist of tools (agent-native names). `None` = all tools. Honoured by Claude Code, Copilot CLI, Pi; ignored by OpenCode, Codex and Cursor. **Only actually enforced when `auto_approve=False`** (see Tool Restriction). |
|
|
154
|
+
| `allowed_tools` | `list[str] \| None` | `None` | Whitelist of tools (agent-native names). `None` = all tools. Honoured by Claude Code, Copilot CLI, Pi, Grok; ignored by OpenCode, Codex and Cursor. **Only actually enforced when `auto_approve=False`** on some agents (see Tool Restriction). |
|
|
155
155
|
| `disallowed_tools` | `list[str] \| None` | `None` | Denylist using a canonical vocabulary (see Tool Restriction). Deny takes precedence over allow **and** over `auto_approve`, but covers only built-in tools. Enforcement varies per agent; unenforceable denies emit a `UserWarning`. |
|
|
156
156
|
| `model` | `str \| None` | `None` | Model alias or name, passed to the CLI verbatim (e.g. `"sonnet"`, `"opencode/big-pickle"`) |
|
|
157
|
-
| `effort` | `str \| None` | `None` | Reasoning effort: `"low"`, `"medium"`, `"high"`, etc. Claude Code, Copilot, Codex, Pi. **Ignored by OpenCode** (silently) **and Cursor** (warns). |
|
|
158
|
-
| `include_thinking` | `bool` | `False` | Yield `thinking` events in `stream()`. Claude Code, Copilot, Pi, Cursor. **Dropped by `execute()`** (which keeps only text). |
|
|
157
|
+
| `effort` | `str \| None` | `None` | Reasoning effort: `"low"`, `"medium"`, `"high"`, etc. Claude Code, Copilot, Codex, Pi, Grok. **Ignored by OpenCode** (silently) **and Cursor** (warns). |
|
|
158
|
+
| `include_thinking` | `bool` | `False` | Yield `thinking` events in `stream()`. Claude Code, Copilot, Pi, Cursor, Grok. **Dropped by `execute()`** (which keeps only text). |
|
|
159
159
|
| `auto_approve` | `bool` | `True` | Skip tool permission prompts. Mapped on every adapter (Pi *requires* a trust decision — the default avoids a hang). **On Claude Code the default `True` sends `--dangerously-skip-permissions`, which bypasses `allowed_tools`.** |
|
|
160
160
|
| `session_id` | `str \| None` | `None` | Resume a previous session |
|
|
161
161
|
|
|
@@ -175,6 +175,7 @@ AgentType.COPILOT_CLI # GitHub Copilot CLI
|
|
|
175
175
|
AgentType.CODEX # OpenAI Codex CLI
|
|
176
176
|
AgentType.PI # Pi coding agent
|
|
177
177
|
AgentType.CURSOR # Cursor CLI (cursor-agent)
|
|
178
|
+
AgentType.GROK # xAI Grok Build CLI (grok)
|
|
178
179
|
```
|
|
179
180
|
|
|
180
181
|
Capabilities differ by agent. `output_tokens` is populated on all of them; the rest varies:
|
|
@@ -187,6 +188,7 @@ Capabilities differ by agent. `output_tokens` is populated on all of them; the r
|
|
|
187
188
|
| Codex | ❌ | ⚠️ `web_search` only | ✅ | ❌ `0.0` | ❌ `0.0` | ✅ |
|
|
188
189
|
| Pi | ✅ | ⚠️ `bash`, `edit`, `read` | ✅ | ⚠️ paid providers only | ❌ `0.0` | ❌ raises |
|
|
189
190
|
| Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real | ❌ raises |
|
|
191
|
+
| Grok | ✅ | ✅ all canonical | ✅ | ⚠️ may be `0.0` | ✅ real | ✅ user-scope |
|
|
190
192
|
|
|
191
193
|
A `✅` for `allowed_tools` means the flag is passed — but it only *enforces* with
|
|
192
194
|
`auto_approve=False`; `disallowed_tools` covers only built-in tools. See Tool Restriction.
|
|
@@ -300,10 +302,10 @@ succeeded = saw_ok and error is None # absent result => succeeded stays
|
|
|
300
302
|
- `output_tokens` — the portable "how much did it generate" signal; populated on the `result`
|
|
301
303
|
event of every adapter. It reads `0` when the `result` event never arrives (a truncated turn),
|
|
302
304
|
so it is not a standalone liveness check — pair it with the success check above.
|
|
303
|
-
- `cost` — real for Claude Code and paid Pi providers; frequently `0.0` for
|
|
304
|
-
Copilot, Codex, and always `0.0` for Cursor (they don't report it).
|
|
305
|
-
`cost == 0` as "the call failed".
|
|
306
|
-
- `duration` — real for Claude Code, Copilot CLI, and
|
|
305
|
+
- `cost` — real for Claude Code, Grok (usually), and paid Pi providers; frequently `0.0` for
|
|
306
|
+
OpenCode, Copilot, Codex, and always `0.0` for Cursor (they don't report it). Grok may also
|
|
307
|
+
report `0.0` on some OAuth/pool paths. Don't treat `cost == 0` as "the call failed".
|
|
308
|
+
- `duration` — real for Claude Code, Copilot CLI, Cursor, and Grok; `0.0` elsewhere.
|
|
307
309
|
|
|
308
310
|
## Other Capabilities
|
|
309
311
|
|
|
@@ -314,7 +316,8 @@ succeeded = saw_ok and error is None # absent result => succeeded stays
|
|
|
314
316
|
**not** run your prompt, so use the stream-based check above when you care about a specific
|
|
315
317
|
call's outcome.
|
|
316
318
|
- **MCP server management** — `add_mcp_server`, `remove_mcp_server`, `list_mcp_servers` manage
|
|
317
|
-
|
|
319
|
+
user-scope MCP configuration. Pi raises `NotImplementedError`; Cursor is managed directly
|
|
320
|
+
through `~/.cursor/mcp.json` because `cursor-agent mcp` has no add/remove subcommands.
|
|
318
321
|
|
|
319
322
|
See [api-reference.md](api-reference.md) for their full signatures and the `MCPServerSpec` model.
|
|
320
323
|
|
|
@@ -335,8 +338,8 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
|
|
|
335
338
|
| Get a complete answer | `await shell.execute(cwd, prompt)` |
|
|
336
339
|
| Stream events live | `async for event in shell.stream(cwd, prompt)` |
|
|
337
340
|
| Continue a conversation | Pass `session_id=response.session_id` |
|
|
338
|
-
| Whitelist tools (Claude/Copilot/Pi) | `allowed_tools=["Read", "Glob"]` |
|
|
339
|
-
| Deny tools (Claude/OpenCode/Copilot/Pi) | `disallowed_tools=["edit", "bash"]` |
|
|
341
|
+
| Whitelist tools (Claude/Copilot/Pi/Grok) | `allowed_tools=["Read", "Glob"]` |
|
|
342
|
+
| Deny tools (Claude/OpenCode/Copilot/Pi/Grok) | `disallowed_tools=["edit", "bash"]` |
|
|
340
343
|
| Track usage | Read `response.output_tokens` (portable) or `response.cost` |
|
|
341
344
|
| Discover selectable models | `await shell.list_models(cwd)` |
|
|
342
345
|
| Use a specific model | Pass one returned string as `model=...` |
|
|
@@ -21,6 +21,7 @@ class AgentType(StrEnum):
|
|
|
21
21
|
CODEX = "codex"
|
|
22
22
|
PI = "pi"
|
|
23
23
|
CURSOR = "cursor"
|
|
24
|
+
GROK = "grok"
|
|
24
25
|
```
|
|
25
26
|
|
|
26
27
|
### AgentResponse
|
|
@@ -274,3 +275,20 @@ class AgentAdapter(Protocol):
|
|
|
274
275
|
- `duration` and `output_tokens` are real (`usage.outputTokens`); `cost` is always `0.0` — Cursor
|
|
275
276
|
reports no cost. MCP-management methods raise `NotImplementedError` (`cursor-agent mcp` has no
|
|
276
277
|
add/remove, and its `list` returns only name+status, which cannot rebuild an `MCPServerSpec`).
|
|
278
|
+
|
|
279
|
+
### Grok
|
|
280
|
+
- Headless: `grok -p --output-format streaming-messages-json` (full assistant blocks; not
|
|
281
|
+
token-delta `streaming-json`, which would break newline-joined text collection).
|
|
282
|
+
- `model` → `-m` / `--model` (e.g. `"grok-4.5"`). `list_models()` parses `grok models` text.
|
|
283
|
+
- `allowed_tools` → `--tools`; `disallowed_tools` → `--disallowed-tools` with all five canonical
|
|
284
|
+
names. **Important:** `bash` maps to `run_terminal_cmd` (the working deny id). Init lists the
|
|
285
|
+
shell tool as `run_terminal_command`, but denying that longer name is a no-op on grok 1.0.0.
|
|
286
|
+
`edit` → `search_replace,write`.
|
|
287
|
+
- `effort` → `--reasoning-effort`; `auto_approve` → `--always-approve`; `include_thinking` is
|
|
288
|
+
honoured from assistant thinking blocks.
|
|
289
|
+
- Session resume → `--resume <id>`. `system/init` and `result` carry `session_id`.
|
|
290
|
+
- `cost` comes from `result.total_cost_usd` (may be `0.0` on some OAuth/pool paths);
|
|
291
|
+
`duration` from `duration_ms`; `output_tokens` is raw `usage.output_tokens` (reasoning is a
|
|
292
|
+
subset when present — do not add `reasoning_tokens`).
|
|
293
|
+
- MCP via `grok mcp add|remove --scope user` only (unscoped remove can hit project config).
|
|
294
|
+
`list_mcp_servers()` reads `~/.grok/config.toml`.
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '0.2.
|
|
22
|
-
__version_tuple__ = version_tuple = (0, 2,
|
|
21
|
+
__version__ = version = '0.2.5'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 2, 5)
|
|
23
23
|
|
|
24
24
|
__commit_id__ = commit_id = None
|
{agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/copilot_cli_adapter.py
RENAMED
|
@@ -48,7 +48,7 @@ _COPILOT_EFFORTS = ("none", "minimal", "low", "medium", "high", "xhigh", "max")
|
|
|
48
48
|
|
|
49
49
|
|
|
50
50
|
def _normalize_effort(effort: str | None) -> str | None:
|
|
51
|
-
if effort is None:
|
|
51
|
+
if effort is None or effort == "":
|
|
52
52
|
return None
|
|
53
53
|
|
|
54
54
|
normalized = effort.lower() if isinstance(effort, str) else None
|