agent-shell-py 0.2.3__tar.gz → 0.2.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (123) hide show
  1. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/AGENTS.md +15 -3
  2. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/PKG-INFO +49 -11
  3. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/README.md +48 -10
  4. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/agent_parameter_comparison.md +64 -13
  5. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/invoking-cli-agents/SKILL.md +16 -13
  6. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/invoking-cli-agents/api-reference.md +18 -0
  7. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/_version.py +2 -2
  8. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/copilot_cli_adapter.py +1 -1
  9. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/cursor_adapter.py +168 -7
  10. agent_shell_py-0.2.5/src/agent_shell/adapters/grok_adapter.py +477 -0
  11. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/response.py +1 -1
  12. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/models/agent.py +1 -0
  13. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/shell.py +2 -0
  14. agent_shell_py-0.2.5/tests/e2e/test_grok_e2e.py +142 -0
  15. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_health_check_e2e.py +1 -0
  16. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_model_discovery_e2e.py +1 -0
  17. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_copilot_cli_integration.py +13 -0
  18. agent_shell_py-0.2.5/tests/integration/test_cursor_mcp_integration.py +367 -0
  19. agent_shell_py-0.2.5/tests/integration/test_grok_integration.py +288 -0
  20. agent_shell_py-0.2.5/tests/integration/test_grok_mcp_integration.py +222 -0
  21. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_health_check_integration.py +3 -0
  22. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_model_discovery_integration.py +35 -0
  23. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_process_lifecycle.py +12 -5
  24. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/adapter_matrix.py +7 -1
  25. agent_shell_py-0.2.5/tests/unit/grok_fixtures.py +154 -0
  26. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_adapter_transport.py +9 -2
  27. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_execute_outcome.py +2 -2
  28. agent_shell_py-0.2.5/tests/unit/test_grok_cancel.py +21 -0
  29. agent_shell_py-0.2.5/tests/unit/test_grok_execute.py +119 -0
  30. agent_shell_py-0.2.5/tests/unit/test_grok_parse_event.py +199 -0
  31. agent_shell_py-0.2.5/tests/unit/test_grok_warnings.py +110 -0
  32. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell.py +8 -0
  33. agent_shell_py-0.2.3/tests/integration/test_cursor_mcp_integration.py +0 -36
  34. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/build.yml +0 -0
  35. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/ci.yml +0 -0
  36. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.github/workflows/publish.yml +0 -0
  37. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.gitignore +0 -0
  38. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/.python-version +0 -0
  39. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/LICENSE +0 -0
  40. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/assets/skill_banner.png +0 -0
  41. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/disabled_tools.md +0 -0
  42. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/info.md +0 -0
  43. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/docs/development/total_token_count.md +0 -0
  44. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/pyproject.toml +0 -0
  45. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/skills/delegating-code-review/SKILL.md +0 -0
  46. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/__init__.py +0 -0
  47. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/__init__.py +0 -0
  48. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/agent_adapter_protocol.py +0 -0
  49. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/claude_code_adapter.py +0 -0
  50. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/codex_adapter.py +0 -0
  51. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/health.py +0 -0
  52. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/model_discovery.py +0 -0
  53. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/opencode_adapter.py +0 -0
  54. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/outcome.py +0 -0
  55. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/pi_adapter.py +0 -0
  56. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/stderr_format.py +0 -0
  57. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/adapters/tool_denial.py +0 -0
  58. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/models/__init__.py +0 -0
  59. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/src/agent_shell/process_cleanup.py +0 -0
  60. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/__init__.py +0 -0
  61. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/conftest.py +0 -0
  62. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/__init__.py +0 -0
  63. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_claude_code_e2e.py +0 -0
  64. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_codex_e2e.py +0 -0
  65. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_copilot_cli_e2e.py +0 -0
  66. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_cursor_e2e.py +0 -0
  67. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_opencode_e2e.py +0 -0
  68. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/e2e/test_pi_e2e.py +0 -0
  69. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/__init__.py +0 -0
  70. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_claude_code_integration.py +0 -0
  71. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_claude_code_mcp_integration.py +0 -0
  72. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_codex_integration.py +0 -0
  73. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_codex_mcp_integration.py +0 -0
  74. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_copilot_cli_mcp_integration.py +0 -0
  75. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_cursor_integration.py +0 -0
  76. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_opencode_integration.py +0 -0
  77. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_opencode_mcp_integration.py +0 -0
  78. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_pi_integration.py +0 -0
  79. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/integration/test_pi_mcp_integration.py +0 -0
  80. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/__init__.py +0 -0
  81. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/codex_fixtures.py +0 -0
  82. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/copilot_fixtures.py +0 -0
  83. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/cursor_fixtures.py +0 -0
  84. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/fixtures.py +0 -0
  85. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/opencode_fixtures.py +0 -0
  86. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/pi_fixtures.py +0 -0
  87. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cancel.py +0 -0
  88. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_cancel.py +0 -0
  89. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_execute.py +0 -0
  90. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_parse_event.py +0 -0
  91. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_codex_warnings.py +0 -0
  92. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_cancel.py +0 -0
  93. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_execute.py +0 -0
  94. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_parse_event.py +0 -0
  95. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_copilot_cli_stream.py +0 -0
  96. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_cancel.py +0 -0
  97. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_execute.py +0 -0
  98. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_parse_event.py +0 -0
  99. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_cursor_warnings.py +0 -0
  100. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_execute.py +0 -0
  101. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_health_probe.py +0 -0
  102. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_mcp_server_spec.py +0 -0
  103. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_model_discovery.py +0 -0
  104. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_models.py +0 -0
  105. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_cancel.py +0 -0
  106. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_execute.py +0 -0
  107. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_parse_event.py +0 -0
  108. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_spawn.py +0 -0
  109. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_opencode_stream.py +0 -0
  110. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_parse_event.py +0 -0
  111. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_cancel.py +0 -0
  112. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_execute.py +0 -0
  113. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_parse_event.py +0 -0
  114. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_pi_warnings.py +0 -0
  115. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_process_cleanup.py +0 -0
  116. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_process_group_registration.py +0 -0
  117. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_response_aggregation.py +0 -0
  118. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell_cancellation.py +0 -0
  119. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_shell_mcp.py +0 -0
  120. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_stderr_format.py +0 -0
  121. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_stream.py +0 -0
  122. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/tests/unit/test_tool_denial.py +0 -0
  123. {agent_shell_py-0.2.3 → agent_shell_py-0.2.5}/uv.lock +0 -0
@@ -142,6 +142,7 @@ command, and never substitutes a static catalog. "Available" means advertised as
142
142
  - [x] Codex
143
143
  - [x] Pi
144
144
  - [x] Cursor
145
+ - [x] Grok
145
146
 
146
147
  ## MCP Server Configuration
147
148
 
@@ -170,8 +171,19 @@ All adapters write to user-scope configuration:
170
171
  | OpenCode | direct JSON file write | `~/.config/opencode/opencode.json` |
171
172
  | Copilot CLI | direct JSON file write | `~/.copilot/mcp-config.json` |
172
173
  | Codex | `codex mcp add` subprocess | Codex config |
173
-
174
- Adds are idempotent (overwrite existing entries with the same name). Removes warn rather than raise when the named server is not found. Claude Code listing reads the user-scope `mcpServers` entries from `~/.claude.json` directly, avoiding the health checks and human-readable output of `claude mcp list`. MCP is not implemented for Pi or Cursor — all three methods raise `NotImplementedError`. Pi manages capability via `pi install` extensions (which needs investigation before wiring up); Cursor's `mcp` subcommands are login/list/list-tools/enable/disable only (no add/remove — servers are declared in `.cursor/mcp.json`), and `mcp list` reports only `name: status`, not the transport an `MCPServerSpec` needs.
174
+ | Cursor | direct JSON file write | `~/.cursor/mcp.json` |
175
+ | Grok | `grok mcp add --scope user` subprocess | `~/.grok/config.toml` (managed by CLI) |
176
+
177
+ Adds are idempotent (update existing entries with the same name). Cursor preserves native fields
178
+ that `MCPServerSpec` cannot represent when an update keeps the same transport, and writes its file
179
+ atomically with user-only permissions. Removes warn rather than raise when the named server is not
180
+ found. Claude Code listing reads the user-scope `mcpServers`
181
+ entries from `~/.claude.json` directly, avoiding the health checks and human-readable output of
182
+ `claude mcp list`. Cursor manages user-scope MCP entries directly in `~/.cursor/mcp.json` because
183
+ its `mcp` subcommands have no add/remove commands. Grok listing reads user-scope `mcp_servers`
184
+ entries from `~/.grok/config.toml` directly for the same reason. Pi's MCP add/remove/list methods
185
+ raise `NotImplementedError`. Pi manages capability via `pi install` extensions, which needs
186
+ investigation before wiring up.
175
187
 
176
188
  ## Test Philosophy
177
189
 
@@ -183,7 +195,7 @@ Tests validate real functionality, not code coverage metrics. Three tiers, each
183
195
  | **Integration** | Full flow through `AgentShell` -> `Adapter` -> parser with mocked subprocess | Yes | No |
184
196
  | **E2E** | Real CLI calls; usually real API costs | No (local only) | Yes |
185
197
 
186
- The model-discovery E2E test is the exception: it calls all six real CLIs but only reads
198
+ The model-discovery E2E test is the exception: it calls all seven real CLIs but only reads
187
199
  metadata, so it sends no inference request and incurs no model-token cost.
188
200
 
189
201
  Integration tests mirror the E2E tests but substitute mocked subprocesses emitting captured CLI
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-shell-py
3
- Version: 0.2.3
3
+ Version: 0.2.5
4
4
  Summary: A lightweight abstraction for executing CLI coding agents headlessly
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -15,8 +15,8 @@ and returning the output that can be used programatically as a unified contract
15
15
 
16
16
  - **One unified contract** — the same `execute`, `stream`, `health_check`, and
17
17
  `list_models` API across every agent; swap the backend without changing consuming code.
18
- - **Six CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, and Cursor behind a
19
- common adapter protocol.
18
+ - **Seven CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, and Grok
19
+ behind a common adapter protocol.
20
20
  - **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
21
21
  or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
22
22
  - **Session resumption** — continue any conversation by passing back its `session_id`.
@@ -64,7 +64,7 @@ Or install both skills globally for every coding agent supported by AgentShell:
64
64
  ```bash
65
65
  npx skills add ScottRBK/agent-shell --global \
66
66
  --skill '*' \
67
- --agent claude-code opencode github-copilot codex pi cursor \
67
+ --agent claude-code opencode github-copilot codex pi cursor grok \
68
68
  --yes
69
69
  ```
70
70
 
@@ -213,10 +213,10 @@ response = await shell.execute(
213
213
  `mcp__server__tool`, or a harness-specific name like `Write`, or Copilot's `view`).
214
214
  - Deny takes precedence over auto-approve on every backend that supports it.
215
215
  - Where a backend cannot enforce a deny, the adapter emits a `UserWarning` listing the
216
- ignored tools rather than failing silently. Coverage varies: Claude and OpenCode enforce
217
- all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a verbatim
218
- name for its other tools); Codex can only deny `web_search`; Cursor cannot enforce any
219
- per-call deny (its tool policy lives in `.cursor/cli.json`).
216
+ ignored tools rather than failing silently. Coverage varies: Claude, OpenCode, and Grok
217
+ enforce all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
218
+ verbatim name for its other tools); Codex can only deny `web_search`; Cursor cannot
219
+ enforce any per-call deny (its tool policy lives in `.cursor/cli.json`).
220
220
  - Denying `edit` or `read` is **best-effort**: a model can still modify or read files through
221
221
  the shell, so also deny `bash` when you need a hard file boundary.
222
222
 
@@ -276,8 +276,39 @@ print(f"Session: {response.session_id}")
276
276
  > tools are auto-*rejected* but the run still completes. `allowed_tools`, `effort`, and
277
277
  > `disallowed_tools` are **ignored** — Cursor exposes no per-call tool policy or effort flag
278
278
  > (tool policy lives in `.cursor/cli.json`), so each emits a `UserWarning`. On a Free plan
279
- > only `model=None`/`"auto"` works. MCP servers are declared in `.cursor/mcp.json`; the
280
- > `add`/`remove`/`list` MCP methods raise `NotImplementedError`.
279
+ > only `model=None`/`"auto"` works. MCP add/remove/list are supported by directly managing
280
+ > the user-scope `~/.cursor/mcp.json` file because `cursor-agent mcp` has no add/remove
281
+ > subcommands.
282
+
283
+ ### Grok
284
+
285
+ ```python
286
+ from agent_shell.shell import AgentShell
287
+ from agent_shell.models.agent import AgentType
288
+
289
+ shell = AgentShell(agent_type=AgentType.GROK)
290
+
291
+ response = await shell.execute(
292
+ cwd="/path/to/project",
293
+ prompt="Can you tell me about this project?",
294
+ model="grok-4.5",
295
+ )
296
+
297
+ print(response.response)
298
+ print(f"Session: {response.session_id}")
299
+ print(f"Cost: ${response.cost:.4f}")
300
+ ```
301
+
302
+ > **Note:** Grok runs headlessly via `grok -p --output-format streaming-messages-json`
303
+ > (full assistant blocks — not token-delta `streaming-json`, which would break
304
+ > newline-joined text collection). With `auto_approve=True` (the default) the adapter
305
+ > passes `--always-approve`. `effort` maps to `--reasoning-effort`, `allowed_tools` to
306
+ > `--tools`, and `disallowed_tools` to `--disallowed-tools` (canonical `bash` maps to
307
+ > `run_terminal_cmd` — the working deny id, not init.tools' `run_terminal_command`).
308
+ > The terminal `result` event carries cost (may be `0` on some auth paths), duration, and
309
+ > raw `usage.output_tokens` (reasoning is already inside that figure when reported).
310
+ > MCP add/remove/list are supported via `grok mcp` with **user scope only**
311
+ > (`~/.grok/config.toml`).
281
312
 
282
313
  ## MCP Servers
283
314
 
@@ -318,7 +349,14 @@ await shell.add_mcp_server(MCPServerSpec(
318
349
  ))
319
350
  ```
320
351
 
321
- `add_mcp_server` overwrites an existing server with the same name. `remove_mcp_server` warns rather than raises when the named server is not found. `list_mcp_servers()` works for Claude Code, OpenCode, Copilot CLI, and Codex. Claude Code reads user-scope entries directly from `~/.claude.json`, so listing does not launch configured servers for health checks. MCP is not supported for Pi or Cursor — neither CLI exposes an add/remove subcommand, so all three MCP methods raise `NotImplementedError`.
352
+ `add_mcp_server` adds or updates a server with the same name. Cursor preserves native fields
353
+ that `MCPServerSpec` cannot represent when an update keeps the same transport. The configuration
354
+ is written atomically with user-only permissions. `remove_mcp_server` warns rather than raises
355
+ when the named server is not found. `list_mcp_servers()` works for
356
+ Claude Code, OpenCode, Copilot CLI, Codex, Cursor, and Grok. Claude Code reads user-scope
357
+ entries directly from `~/.claude.json`, Cursor from `~/.cursor/mcp.json`, and Grok from
358
+ `~/.grok/config.toml`, so listing does not launch configured servers for health checks. MCP
359
+ is not supported for Pi; all three MCP methods raise `NotImplementedError`.
322
360
 
323
361
  ## Logging
324
362
 
@@ -6,8 +6,8 @@ and returning the output that can be used programatically as a unified contract
6
6
 
7
7
  - **One unified contract** — the same `execute`, `stream`, `health_check`, and
8
8
  `list_models` API across every agent; swap the backend without changing consuming code.
9
- - **Six CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, and Cursor behind a
10
- common adapter protocol.
9
+ - **Seven CLI agents** — Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, and Grok
10
+ behind a common adapter protocol.
11
11
  - **Execute or stream** — get one `AgentResponse` (raises `AgentExecutionError` on a failed run),
12
12
  or async-iterate normalized `StreamEvent`s with optional thinking/reasoning.
13
13
  - **Session resumption** — continue any conversation by passing back its `session_id`.
@@ -55,7 +55,7 @@ Or install both skills globally for every coding agent supported by AgentShell:
55
55
  ```bash
56
56
  npx skills add ScottRBK/agent-shell --global \
57
57
  --skill '*' \
58
- --agent claude-code opencode github-copilot codex pi cursor \
58
+ --agent claude-code opencode github-copilot codex pi cursor grok \
59
59
  --yes
60
60
  ```
61
61
 
@@ -204,10 +204,10 @@ response = await shell.execute(
204
204
  `mcp__server__tool`, or a harness-specific name like `Write`, or Copilot's `view`).
205
205
  - Deny takes precedence over auto-approve on every backend that supports it.
206
206
  - Where a backend cannot enforce a deny, the adapter emits a `UserWarning` listing the
207
- ignored tools rather than failing silently. Coverage varies: Claude and OpenCode enforce
208
- all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a verbatim
209
- name for its other tools); Codex can only deny `web_search`; Cursor cannot enforce any
210
- per-call deny (its tool policy lives in `.cursor/cli.json`).
207
+ ignored tools rather than failing silently. Coverage varies: Claude, OpenCode, and Grok
208
+ enforce all five canonical names; Copilot enforces only `bash`/`edit` canonically (use a
209
+ verbatim name for its other tools); Codex can only deny `web_search`; Cursor cannot
210
+ enforce any per-call deny (its tool policy lives in `.cursor/cli.json`).
211
211
  - Denying `edit` or `read` is **best-effort**: a model can still modify or read files through
212
212
  the shell, so also deny `bash` when you need a hard file boundary.
213
213
 
@@ -267,8 +267,39 @@ print(f"Session: {response.session_id}")
267
267
  > tools are auto-*rejected* but the run still completes. `allowed_tools`, `effort`, and
268
268
  > `disallowed_tools` are **ignored** — Cursor exposes no per-call tool policy or effort flag
269
269
  > (tool policy lives in `.cursor/cli.json`), so each emits a `UserWarning`. On a Free plan
270
- > only `model=None`/`"auto"` works. MCP servers are declared in `.cursor/mcp.json`; the
271
- > `add`/`remove`/`list` MCP methods raise `NotImplementedError`.
270
+ > only `model=None`/`"auto"` works. MCP add/remove/list are supported by directly managing
271
+ > the user-scope `~/.cursor/mcp.json` file because `cursor-agent mcp` has no add/remove
272
+ > subcommands.
273
+
274
+ ### Grok
275
+
276
+ ```python
277
+ from agent_shell.shell import AgentShell
278
+ from agent_shell.models.agent import AgentType
279
+
280
+ shell = AgentShell(agent_type=AgentType.GROK)
281
+
282
+ response = await shell.execute(
283
+ cwd="/path/to/project",
284
+ prompt="Can you tell me about this project?",
285
+ model="grok-4.5",
286
+ )
287
+
288
+ print(response.response)
289
+ print(f"Session: {response.session_id}")
290
+ print(f"Cost: ${response.cost:.4f}")
291
+ ```
292
+
293
+ > **Note:** Grok runs headlessly via `grok -p --output-format streaming-messages-json`
294
+ > (full assistant blocks — not token-delta `streaming-json`, which would break
295
+ > newline-joined text collection). With `auto_approve=True` (the default) the adapter
296
+ > passes `--always-approve`. `effort` maps to `--reasoning-effort`, `allowed_tools` to
297
+ > `--tools`, and `disallowed_tools` to `--disallowed-tools` (canonical `bash` maps to
298
+ > `run_terminal_cmd` — the working deny id, not init.tools' `run_terminal_command`).
299
+ > The terminal `result` event carries cost (may be `0` on some auth paths), duration, and
300
+ > raw `usage.output_tokens` (reasoning is already inside that figure when reported).
301
+ > MCP add/remove/list are supported via `grok mcp` with **user scope only**
302
+ > (`~/.grok/config.toml`).
272
303
 
273
304
  ## MCP Servers
274
305
 
@@ -309,7 +340,14 @@ await shell.add_mcp_server(MCPServerSpec(
309
340
  ))
310
341
  ```
311
342
 
312
- `add_mcp_server` overwrites an existing server with the same name. `remove_mcp_server` warns rather than raises when the named server is not found. `list_mcp_servers()` works for Claude Code, OpenCode, Copilot CLI, and Codex. Claude Code reads user-scope entries directly from `~/.claude.json`, so listing does not launch configured servers for health checks. MCP is not supported for Pi or Cursor — neither CLI exposes an add/remove subcommand, so all three MCP methods raise `NotImplementedError`.
343
+ `add_mcp_server` adds or updates a server with the same name. Cursor preserves native fields
344
+ that `MCPServerSpec` cannot represent when an update keeps the same transport. The configuration
345
+ is written atomically with user-only permissions. `remove_mcp_server` warns rather than raises
346
+ when the named server is not found. `list_mcp_servers()` works for
347
+ Claude Code, OpenCode, Copilot CLI, Codex, Cursor, and Grok. Claude Code reads user-scope
348
+ entries directly from `~/.claude.json`, Cursor from `~/.cursor/mcp.json`, and Grok from
349
+ `~/.grok/config.toml`, so listing does not launch configured servers for health checks. MCP
350
+ is not supported for Pi; all three MCP methods raise `NotImplementedError`.
313
351
 
314
352
  ## Logging
315
353
 
@@ -1,10 +1,11 @@
1
1
  # Agent CLI Parameter Comparison
2
2
 
3
3
  Comparison of headless/non-interactive configuration across supported CLI coding agents.
4
- Last updated: 2026-08-05
4
+ Last updated: 2026-08-08
5
5
 
6
- > The summary matrix below has no Pi or Cursor column (it predates both adapters); see the
7
- > per-agent detail sections, the model-discovery section, and the `disallowed_tools` table.
6
+ > The summary matrix below has no Pi, Cursor, or Grok column (it predates those adapters);
7
+ > see the per-agent detail sections, the model-discovery section, and the
8
+ > `disallowed_tools` table.
8
9
 
9
10
  > Every "measured 2026-07-26" claim below comes from a real three-call run per agent (first
10
11
  > turn, resume on the returned id, then a session-less turn), recorded by the resume e2e tests
@@ -55,6 +56,7 @@ Discovery reads CLI metadata and makes no inference call.
55
56
  | Codex | `codex debug models` | visible model `slug` |
56
57
  | Pi | `pi --no-approve --list-models` | `provider/model` |
57
58
  | Cursor | `cursor-agent models` | ID before the first ` - ` |
59
+ | Grok | `grok models` | name after optional `* `, before `(default)` |
58
60
 
59
61
  Important semantics:
60
62
 
@@ -74,9 +76,9 @@ Important semantics:
74
76
  malformed protocol/output failures are surfaced with actionable errors.
75
77
  - AgentShell does not cache or replace results with a static fallback catalog.
76
78
 
77
- The behavior was verified against all six installed CLIs on 2026-08-02 without inference
78
- spend. CI covers captured real outputs through `AgentShell`; a local E2E test checks all six
79
- real discovery commands.
79
+ The behavior was verified against the installed CLIs on 2026-08-08 without inference
80
+ spend (Grok included). CI covers captured real outputs through `AgentShell`; a local E2E
81
+ test checks all seven real discovery commands.
80
82
 
81
83
  ## Claude Code
82
84
 
@@ -126,9 +128,9 @@ real discovery commands.
126
128
  - **Headless mode**: `-p` / `--prompt` for one-shot; `--acp` for Agent Client Protocol
127
129
  - **Model**: `--model <model>`; use `auto` to let Copilot choose
128
130
  - **Effort**: `--effort` / `--reasoning-effort` with choices
129
- `none|minimal|low|medium|high|xhigh|max`. AgentShell accepts effort values
130
- case-insensitively, validates them against these choices, and passes the normalized lowercase
131
- value to Copilot via `--effort`.
131
+ `none|minimal|low|medium|high|xhigh|max`. AgentShell treats `None` and `""` as omitted, so
132
+ Copilot uses its default. Explicit values, including `"none"`, are accepted case-insensitively,
133
+ validated against these choices, and passed to Copilot via `--effort` in lowercase.
132
134
  - **Allowed tools**: `--allow-tool`, `--deny-tool`, `--available-tools`,
133
135
  `--excluded-tools`, `--allow-all-tools`
134
136
  - **Auto-approve**: `--allow-all-tools`; `--allow-all` / `--yolo` also grant path and URL
@@ -191,13 +193,61 @@ real discovery commands.
191
193
  cursor-agent ACCEPTS an unknown id — `--resume=<never-seen-uuid>` starts a session under that
192
194
  id and echoes it back rather than failing, so id identity proves the flag was passed through
193
195
  and honoured, not that a prior transcript was replayed (measured 2026-07-26)
194
- - **MCP**: `cursor-agent mcp` = login/list/list-tools/enable/disable only (no add/remove);
195
- servers declared in `.cursor/mcp.json`. All three adapter MCP methods raise
196
- `NotImplementedError` — including `list_mcp_servers`, because `mcp list` emits only
197
- `name: status` and cannot rebuild an `MCPServerSpec`
196
+ - **MCP**: `cursor-agent mcp` = login/list/list-tools/enable/disable only (no add/remove).
197
+ The adapter manages user-scope servers directly in `~/.cursor/mcp.json`, which provides the
198
+ transport details that `mcp list` omits. Writes are atomic and user-only; same-transport
199
+ updates preserve Cursor-native fields that `MCPServerSpec` cannot represent.
198
200
  - **Usage**: the terminal `result` event carries `usage.outputTokens` (undocumented but real)
199
201
  and `duration_ms`; there is no cost field, so `cost` is `0.0`
200
202
 
203
+ ## Grok (xAI Grok Build)
204
+
205
+ Flags below are the ones the adapter actually emits (`_build_command` in
206
+ `src/agent_shell/adapters/grok_adapter.py`); this is not a full survey of the Grok CLI.
207
+ Measured against grok 1.0.0 (3cd0d0cbce) on 2026-08-08.
208
+
209
+ - **Headless mode**: `-p` / `--single <PROMPT>` with
210
+ `--output-format streaming-messages-json` (Anthropic Messages NDJSON on stdout).
211
+ The adapter deliberately does **not** use `streaming-json`: that format emits token
212
+ `text.data` fragments, and execute()'s `"\n".join` over text events would explode
213
+ replies (same class of bug as issue #6; pi waits for `text_end`, copilot uses full
214
+ `assistant.message`)
215
+ - **Model**: `-m` / `--model` (e.g. `grok-4.5`). `grok models` prints plain text
216
+ (`Default model:` + `Available models:` list); the adapter returns the bare selectors
217
+ - **Effort**: `--reasoning-effort` / `--effort`
218
+ (`none|minimal|low|medium|high|xhigh|max`)
219
+ - **Allowed tools**: `--tools` (comma-joined native tool ids such as `read_file,list_dir`)
220
+ - **Disallowed tools**: `--disallowed-tools` (comma-joined). Canonical map (measured):
221
+ `bash`→`run_terminal_cmd` (NOT `run_terminal_command` — that id appears in
222
+ `system/init.tools` but is a no-op as a deny), `edit`→`search_replace,write`,
223
+ `read`→`read_file`, `web_search`/`web_fetch` one-to-one
224
+ - **Auto-approve**: `--always-approve`
225
+ - **Output format**: adapter uses `streaming-messages-json` only
226
+ - **Working directory**: process cwd (also has `--cwd`, unused by the adapter)
227
+ - **Session**: `--resume` / `-r` `[SESSION_ID_OR_TITLE]`, `--continue`, `--session-id`
228
+ (new id only). The adapter resumes with `--resume <id>`. `system/init` and `result`
229
+ carry `session_id`; a resumed run should report the same id (see e2e resume test)
230
+ - **Usage / cost**: terminal `result` carries `total_cost_usd` (may be `0` on some
231
+ OAuth/pool paths), `duration_ms`, and `usage.output_tokens`. When
232
+ `usage.reasoning_tokens` is present it is a **subset** of `output_tokens` (Grok's
233
+ `total_tokens` math uses `output_tokens` alone) — AgentShell reports raw
234
+ `output_tokens` and must not add them. `duration = duration_ms/1000`
235
+ - **Stream events (adapter mapping)** — Cursor/Claude-shaped:
236
+ - `system/init` → `StreamEvent(type="system", session_id=...)`
237
+ - `assistant.message.content[]` text blocks → `StreamEvent(type="text")`
238
+ - thinking blocks → `StreamEvent(type="thinking")` when `include_thinking`
239
+ - `tool_use` / `server_tool_use` blocks → `StreamEvent(type="tool_use", content=name)`
240
+ - `result` → `StreamEvent(type="result", content="ok"|"error", ...)` via `is_error`;
241
+ optional `errors[]` populates `StreamEvent.error`
242
+ - **MCP**: full CLI — `grok mcp add|remove|list|enable|disable|doctor`. Adapter uses
243
+ `grok mcp add --scope user` (add-or-update; no pre-remove) and
244
+ `grok mcp remove --scope user` (unscoped remove can hit project config). Listing reads
245
+ `~/.grok/config.toml` `[mcp_servers.*]` (stdio: `command`/`args`/`env`; http:
246
+ `url`/`headers`)
247
+ - **Auth**: `grok login` or `XAI_API_KEY`. Model listing works while logged in via
248
+ grok.com account; execution requires auth
249
+ - **Project rules**: reads `AGENTS.md` and `CLAUDE.md` (Claude Code compatible)
250
+
201
251
  ## Pi
202
252
 
203
253
  Flags below are the ones the adapter actually emits (`_build_command` in
@@ -253,6 +303,7 @@ through **verbatim** (e.g. `mcp__server__tool`, or a harness-specific name like
253
303
  | Codex | `-c web_search="disabled"` only | no name-based deny; web_search key verified on codex-cli 0.133.0 but version-fragile (upstream `web_search_mode`), guarded by an e2e test; every other canonical/verbatim name warns and is ignored |
254
304
  | Cursor | none (no per-call flag) | tool policy is config-file only (`.cursor/cli.json`); the adapter has no `canonical → native` map, so **every** deny (canonical or verbatim) warns and is ignored |
255
305
  | Pi | `--exclude-tools` (comma-joined) | `edit` → `edit,write`; no web tool, so those warn |
306
+ | Grok | `--disallowed-tools` (comma-joined) | `bash`→`run_terminal_cmd` (not init.tools' `run_terminal_command`); `edit`→`search_replace,write`; `read`→`read_file`; web tools one-to-one |
256
307
 
257
308
  When an adapter cannot honor a requested canonical deny it emits a `UserWarning` listing
258
309
  the ignored names rather than silently dropping the deny (fail-loud). A caller who knows a
@@ -4,8 +4,8 @@ description: >-
4
4
  Use when invoking a CLI coding agent from Python, delegating to a sub-agent, discovering
5
5
  available model strings, orchestrating agents, streaming output, resuming sessions,
6
6
  restricting tools, or checking agent/model health. Supports Claude Code, OpenCode,
7
- Copilot CLI, Codex, Pi, and Cursor. Keywords: AgentShell, list_models, headless agent,
8
- model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
7
+ Copilot CLI, Codex, Pi, Cursor, and Grok. Keywords: AgentShell, list_models, headless
8
+ agent, model discovery, subprocess, allowed_tools, disallowed_tools, session_id, cost,
9
9
  output_tokens.
10
10
  ---
11
11
 
@@ -21,7 +21,7 @@ passes `model` strings through verbatim and does not manage credentials.
21
21
 
22
22
  ## When to Use
23
23
 
24
- - You need to invoke Claude Code, OpenCode, Copilot CLI, Codex, Pi, or Cursor from Python
24
+ - You need to invoke Claude Code, OpenCode, Copilot CLI, Codex, Pi, Cursor, or Grok from Python
25
25
  - You want to delegate a coding task to a sub-agent and collect the result
26
26
  - You need to orchestrate multi-step workflows across agents
27
27
  - You want to stream agent output in real-time
@@ -151,11 +151,11 @@ Both `execute()` and `stream()` take the same parameters.
151
151
  |-----------|------|---------|---------|
152
152
  | `cwd` | `str` | required | Working directory (must exist, else `ValueError`) |
153
153
  | `prompt` | `str` | required | Task or question for the agent |
154
- | `allowed_tools` | `list[str] \| None` | `None` | Whitelist of tools (agent-native names). `None` = all tools. Honoured by Claude Code, Copilot CLI, Pi; ignored by OpenCode, Codex and Cursor. **Only actually enforced when `auto_approve=False`** (see Tool Restriction). |
154
+ | `allowed_tools` | `list[str] \| None` | `None` | Whitelist of tools (agent-native names). `None` = all tools. Honoured by Claude Code, Copilot CLI, Pi, Grok; ignored by OpenCode, Codex and Cursor. **Only actually enforced when `auto_approve=False`** on some agents (see Tool Restriction). |
155
155
  | `disallowed_tools` | `list[str] \| None` | `None` | Denylist using a canonical vocabulary (see Tool Restriction). Deny takes precedence over allow **and** over `auto_approve`, but covers only built-in tools. Enforcement varies per agent; unenforceable denies emit a `UserWarning`. |
156
156
  | `model` | `str \| None` | `None` | Model alias or name, passed to the CLI verbatim (e.g. `"sonnet"`, `"opencode/big-pickle"`) |
157
- | `effort` | `str \| None` | `None` | Reasoning effort: `"low"`, `"medium"`, `"high"`, etc. Claude Code, Copilot, Codex, Pi. **Ignored by OpenCode** (silently) **and Cursor** (warns). |
158
- | `include_thinking` | `bool` | `False` | Yield `thinking` events in `stream()`. Claude Code, Copilot, Pi, Cursor. **Dropped by `execute()`** (which keeps only text). |
157
+ | `effort` | `str \| None` | `None` | Reasoning effort: `"low"`, `"medium"`, `"high"`, etc. Claude Code, Copilot, Codex, Pi, Grok. **Ignored by OpenCode** (silently) **and Cursor** (warns). |
158
+ | `include_thinking` | `bool` | `False` | Yield `thinking` events in `stream()`. Claude Code, Copilot, Pi, Cursor, Grok. **Dropped by `execute()`** (which keeps only text). |
159
159
  | `auto_approve` | `bool` | `True` | Skip tool permission prompts. Mapped on every adapter (Pi *requires* a trust decision — the default avoids a hang). **On Claude Code the default `True` sends `--dangerously-skip-permissions`, which bypasses `allowed_tools`.** |
160
160
  | `session_id` | `str \| None` | `None` | Resume a previous session |
161
161
 
@@ -175,6 +175,7 @@ AgentType.COPILOT_CLI # GitHub Copilot CLI
175
175
  AgentType.CODEX # OpenAI Codex CLI
176
176
  AgentType.PI # Pi coding agent
177
177
  AgentType.CURSOR # Cursor CLI (cursor-agent)
178
+ AgentType.GROK # xAI Grok Build CLI (grok)
178
179
  ```
179
180
 
180
181
  Capabilities differ by agent. `output_tokens` is populated on all of them; the rest varies:
@@ -187,6 +188,7 @@ Capabilities differ by agent. `output_tokens` is populated on all of them; the r
187
188
  | Codex | ❌ | ⚠️ `web_search` only | ✅ | ❌ `0.0` | ❌ `0.0` | ✅ |
188
189
  | Pi | ✅ | ⚠️ `bash`, `edit`, `read` | ✅ | ⚠️ paid providers only | ❌ `0.0` | ❌ raises |
189
190
  | Cursor | ❌ warns | ❌ none — warns | ❌ warns | ❌ `0.0` | ✅ real | ❌ raises |
191
+ | Grok | ✅ | ✅ all canonical | ✅ | ⚠️ may be `0.0` | ✅ real | ✅ user-scope |
190
192
 
191
193
  A `✅` for `allowed_tools` means the flag is passed — but it only *enforces* with
192
194
  `auto_approve=False`; `disallowed_tools` covers only built-in tools. See Tool Restriction.
@@ -300,10 +302,10 @@ succeeded = saw_ok and error is None # absent result => succeeded stays
300
302
  - `output_tokens` — the portable "how much did it generate" signal; populated on the `result`
301
303
  event of every adapter. It reads `0` when the `result` event never arrives (a truncated turn),
302
304
  so it is not a standalone liveness check — pair it with the success check above.
303
- - `cost` — real for Claude Code and paid Pi providers; frequently `0.0` for OpenCode,
304
- Copilot, Codex, and always `0.0` for Cursor (they don't report it). Don't treat
305
- `cost == 0` as "the call failed".
306
- - `duration` — real for Claude Code, Copilot CLI, and Cursor; `0.0` elsewhere.
305
+ - `cost` — real for Claude Code, Grok (usually), and paid Pi providers; frequently `0.0` for
306
+ OpenCode, Copilot, Codex, and always `0.0` for Cursor (they don't report it). Grok may also
307
+ report `0.0` on some OAuth/pool paths. Don't treat `cost == 0` as "the call failed".
308
+ - `duration` — real for Claude Code, Copilot CLI, Cursor, and Grok; `0.0` elsewhere.
307
309
 
308
310
  ## Other Capabilities
309
311
 
@@ -314,7 +316,8 @@ succeeded = saw_ok and error is None # absent result => succeeded stays
314
316
  **not** run your prompt, so use the stream-based check above when you care about a specific
315
317
  call's outcome.
316
318
  - **MCP server management** — `add_mcp_server`, `remove_mcp_server`, `list_mcp_servers` manage
317
- the underlying CLI's MCP config (Pi and Cursor raise `NotImplementedError`).
319
+ user-scope MCP configuration. Pi raises `NotImplementedError`; Cursor is managed directly
320
+ through `~/.cursor/mcp.json` because `cursor-agent mcp` has no add/remove subcommands.
318
321
 
319
322
  See [api-reference.md](api-reference.md) for their full signatures and the `MCPServerSpec` model.
320
323
 
@@ -335,8 +338,8 @@ logging.getLogger("agent_shell").addHandler(logging.StreamHandler())
335
338
  | Get a complete answer | `await shell.execute(cwd, prompt)` |
336
339
  | Stream events live | `async for event in shell.stream(cwd, prompt)` |
337
340
  | Continue a conversation | Pass `session_id=response.session_id` |
338
- | Whitelist tools (Claude/Copilot/Pi) | `allowed_tools=["Read", "Glob"]` |
339
- | Deny tools (Claude/OpenCode/Copilot/Pi) | `disallowed_tools=["edit", "bash"]` |
341
+ | Whitelist tools (Claude/Copilot/Pi/Grok) | `allowed_tools=["Read", "Glob"]` |
342
+ | Deny tools (Claude/OpenCode/Copilot/Pi/Grok) | `disallowed_tools=["edit", "bash"]` |
340
343
  | Track usage | Read `response.output_tokens` (portable) or `response.cost` |
341
344
  | Discover selectable models | `await shell.list_models(cwd)` |
342
345
  | Use a specific model | Pass one returned string as `model=...` |
@@ -21,6 +21,7 @@ class AgentType(StrEnum):
21
21
  CODEX = "codex"
22
22
  PI = "pi"
23
23
  CURSOR = "cursor"
24
+ GROK = "grok"
24
25
  ```
25
26
 
26
27
  ### AgentResponse
@@ -274,3 +275,20 @@ class AgentAdapter(Protocol):
274
275
  - `duration` and `output_tokens` are real (`usage.outputTokens`); `cost` is always `0.0` — Cursor
275
276
  reports no cost. MCP-management methods raise `NotImplementedError` (`cursor-agent mcp` has no
276
277
  add/remove, and its `list` returns only name+status, which cannot rebuild an `MCPServerSpec`).
278
+
279
+ ### Grok
280
+ - Headless: `grok -p --output-format streaming-messages-json` (full assistant blocks; not
281
+ token-delta `streaming-json`, which would break newline-joined text collection).
282
+ - `model` → `-m` / `--model` (e.g. `"grok-4.5"`). `list_models()` parses `grok models` text.
283
+ - `allowed_tools` → `--tools`; `disallowed_tools` → `--disallowed-tools` with all five canonical
284
+ names. **Important:** `bash` maps to `run_terminal_cmd` (the working deny id). Init lists the
285
+ shell tool as `run_terminal_command`, but denying that longer name is a no-op on grok 1.0.0.
286
+ `edit` → `search_replace,write`.
287
+ - `effort` → `--reasoning-effort`; `auto_approve` → `--always-approve`; `include_thinking` is
288
+ honoured from assistant thinking blocks.
289
+ - Session resume → `--resume <id>`. `system/init` and `result` carry `session_id`.
290
+ - `cost` comes from `result.total_cost_usd` (may be `0.0` on some OAuth/pool paths);
291
+ `duration` from `duration_ms`; `output_tokens` is raw `usage.output_tokens` (reasoning is a
292
+ subset when present — do not add `reasoning_tokens`).
293
+ - MCP via `grok mcp add|remove --scope user` only (unscoped remove can hit project config).
294
+ `list_mcp_servers()` reads `~/.grok/config.toml`.
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.2.3'
22
- __version_tuple__ = version_tuple = (0, 2, 3)
21
+ __version__ = version = '0.2.5'
22
+ __version_tuple__ = version_tuple = (0, 2, 5)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -48,7 +48,7 @@ _COPILOT_EFFORTS = ("none", "minimal", "low", "medium", "high", "xhigh", "max")
48
48
 
49
49
 
50
50
  def _normalize_effort(effort: str | None) -> str | None:
51
- if effort is None:
51
+ if effort is None or effort == "":
52
52
  return None
53
53
 
54
54
  normalized = effort.lower() if isinstance(effort, str) else None