@enderfga/claw-orchestrator 4.14.1 → 5.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +8 -8
  2. package/dist/src/base-oneshot-session.d.ts +10 -1
  3. package/dist/src/base-oneshot-session.js +13 -1
  4. package/dist/src/base-oneshot-session.js.map +1 -1
  5. package/dist/src/embedded-server.js +1 -0
  6. package/dist/src/embedded-server.js.map +1 -1
  7. package/dist/src/index.d.ts +1 -0
  8. package/dist/src/index.js +1 -0
  9. package/dist/src/index.js.map +1 -1
  10. package/dist/src/models.d.ts +1 -1
  11. package/dist/src/models.js +82 -11
  12. package/dist/src/models.js.map +1 -1
  13. package/dist/src/persistent-agy-session.js +12 -5
  14. package/dist/src/persistent-agy-session.js.map +1 -1
  15. package/dist/src/persistent-codex-app-session.js +11 -1
  16. package/dist/src/persistent-codex-app-session.js.map +1 -1
  17. package/dist/src/persistent-codex-session.js +7 -2
  18. package/dist/src/persistent-codex-session.js.map +1 -1
  19. package/dist/src/persistent-cursor-session.js +11 -3
  20. package/dist/src/persistent-cursor-session.js.map +1 -1
  21. package/dist/src/persistent-custom-session.js +15 -1
  22. package/dist/src/persistent-custom-session.js.map +1 -1
  23. package/dist/src/persistent-gemini-session.js +5 -1
  24. package/dist/src/persistent-gemini-session.js.map +1 -1
  25. package/dist/src/persistent-grok-session.d.ts +40 -0
  26. package/dist/src/persistent-grok-session.js +197 -0
  27. package/dist/src/persistent-grok-session.js.map +1 -0
  28. package/dist/src/persistent-opencode-session.js +8 -3
  29. package/dist/src/persistent-opencode-session.js.map +1 -1
  30. package/dist/src/persistent-session.d.ts +1 -0
  31. package/dist/src/persistent-session.js +9 -0
  32. package/dist/src/persistent-session.js.map +1 -1
  33. package/dist/src/run-ledger.js +4 -1
  34. package/dist/src/run-ledger.js.map +1 -1
  35. package/dist/src/session-manager.d.ts +1 -0
  36. package/dist/src/session-manager.js +31 -2
  37. package/dist/src/session-manager.js.map +1 -1
  38. package/dist/src/types.d.ts +38 -5
  39. package/dist/src/types.js +15 -1
  40. package/dist/src/types.js.map +1 -1
  41. package/package.json +1 -1
  42. package/skills/SKILL.md +5 -5
  43. package/skills/references/acp.md +1 -1
  44. package/skills/references/autoloop.md +9 -7
  45. package/skills/references/cli.md +3 -2
  46. package/skills/references/getting-started.md +1 -1
  47. package/skills/references/multi-engine.md +62 -4
  48. package/skills/references/observability.md +5 -4
  49. package/skills/references/openai-compat.md +9 -3
  50. package/skills/references/sessions.md +25 -2
  51. package/skills/references/tools.md +2 -2
@@ -1 +1 @@
1
- {"version":3,"file":"types.js","sourceRoot":"","sources":["../../src/types.ts"],"names":[],"mappings":"AAAA;;GAEG;AAIH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAEzC,OAAO,EACL,eAAe,EACf,oBAAoB,EACpB,sBAAsB,EACtB,YAAY,EACZ,YAAY,EACZ,qBAAqB,EACrB,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,UAAU,GACX,MAAM,aAAa,CAAC;AAErB,oDAAoD;AACpD,MAAM,CAAC,MAAM,aAAa,GAA2B,UAAU,EAAE,CAAC;AAYlE,+EAA+E;AAE/E,MAAM,CAAC,MAAM,YAAY,GAAG,CAAC,QAAQ,EAAE,OAAO,EAAE,WAAW,EAAE,QAAQ,EAAE,KAAK,EAAE,QAAQ,EAAE,UAAU,EAAE,QAAQ,CAAU,CAAC;AAGvH;;;;;;;;;;;;;;;;;;;;;;;GAuBG;AACH,MAAM,UAAU,2BAA2B,CACzC,MAA8B,EAC9B,YAAiC;IAEjC,QAAQ,MAAM,EAAE,CAAC;QACf,KAAK,QAAQ,CAAC;QACd,KAAK,OAAO,CAAC;QACb,KAAK,WAAW,CAAC;QACjB,KAAK,KAAK,CAAC;QACX,KAAK,UAAU,CAAC;QAChB,KAAK,QAAQ;YACX,OAAO,IAAI,CAAC;QACd,KAAK,QAAQ;YACX,2EAA2E;YAC3E,gEAAgE;YAChE,OAAO,YAAY,EAAE,UAAU,KAAK,IAAI,CAAC;QAC3C;YACE,OAAO,KAAK,CAAC;IACjB,CAAC;AACH,CAAC"}
1
+ {"version":3,"file":"types.js","sourceRoot":"","sources":["../../src/types.ts"],"names":[],"mappings":"AAAA;;GAEG;AAIH,OAAO,EAAE,UAAU,EAAE,MAAM,aAAa,CAAC;AAEzC,OAAO,EACL,eAAe,EACf,oBAAoB,EACpB,sBAAsB,EACtB,YAAY,EACZ,YAAY,EACZ,qBAAqB,EACrB,eAAe,EACf,gBAAgB,EAChB,aAAa,EACb,aAAa,EACb,cAAc,EACd,iBAAiB,EACjB,UAAU,GACX,MAAM,aAAa,CAAC;AAErB,oDAAoD;AACpD,MAAM,CAAC,MAAM,aAAa,GAA2B,UAAU,EAAE,CAAC;AAYlE,+EAA+E;AAE/E,MAAM,CAAC,MAAM,YAAY,GAAG;IAC1B,QAAQ;IACR,OAAO;IACP,WAAW;IACX,QAAQ;IACR,KAAK;IACL,QAAQ;IACR,MAAM;IACN,UAAU;IACV,QAAQ;CACA,CAAC;AAGX;;;;;;;;;;;;;;;;;;;;;;;;;;GA0BG;AACH,MAAM,UAAU,2BAA2B,CACzC,MAA8B,EAC9B,YAAiC;IAEjC,QAAQ,MAAM,EAAE,CAAC;QACf,KAAK,QAAQ,CAAC;QACd,KAAK,OAAO,CAAC;QACb,KAAK,WAAW,CAAC;QACjB,KAAK,KAAK,CAAC;QACX,KAAK,UAAU,CAAC;QAChB,KAAK,QAAQ,CAAC;QACd,KAAK,MAAM;YACT,OAAO,IAAI,CAAC;QACd,KAAK,QAAQ;YACX,2EAA2E;YAC3E,gEAAgE;YAChE,OAAO,YAAY,EAAE,UAAU,KAAK,IAAI,CAAC;QAC3C;YACE,OAAO,KAAK,CAAC;IACjB,CAAC;AACH,CAAC"}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@enderfga/claw-orchestrator",
3
- "version": "4.14.1",
3
+ "version": "5.1.0",
4
4
  "description": "Claw Orchestrator — run Claude Code, Codex, Gemini, Cursor Agent, OpenCode and custom coding CLIs as one unified runtime. Drop into Hermes Agent, Claude Desktop, Cursor, Cline, Continue, Zed, Windsurf, Goose or any Model Context Protocol (MCP) host, install as an OpenClaw plugin, or run standalone. Persistent sessions, multi-agent council, ultraplan, ultrareview, autoloop, tool orchestration.",
5
5
  "type": "module",
6
6
  "main": "./dist/src/index.js",
package/skills/SKILL.md CHANGED
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: claw-orchestrator
3
- description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Cursor, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's 69 tools as an MCP server to Hermes Agent / Claude Desktop / Cursor / Cline / Continue / Zed / Windsurf / Goose, or running as an Agent Client Protocol (ACP) agent that Zed / JetBrains / Neovim / Emacs / VS Code / dsh can drive directly. Triggers on "start a session", "send to session", "run council", "ultraplan", "ultrareview", "autoloop", "ultraapp", "Forge tab", "build a web app", "one-click app", "AppSpec", "autonomous iteration", "iterate until goal", "deep paper review", "auto research", "switch model", "multi-agent", "coding session", "session inbox", "cursor agent", "opencode", "mcp server", "clawo-mcp", "hermes mcp", "model context protocol", "ultracode", "dynamic workflow", "fanout", "fan-out", "best-of-N", "steer turn", "interrupt turn", "fork thread", "rollback turns", "acp", "agent client protocol", "clawo acp", "zed agent", "jetbrains agent", "external agent", "dsh subagent", "deepseek harness", "clawo runs", "run ledger", "how much did it cost", "token usage", "spend cap", "budget limit", "maxBudgetUsd".
3
+ description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Grok Build, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's 69 tools as an MCP server to Hermes Agent / Claude Desktop / Cursor / Cline / Continue / Zed / Windsurf / Goose, or running as an Agent Client Protocol (ACP) agent that Zed / JetBrains / Neovim / Emacs / VS Code / dsh can drive directly. Triggers on "start a session", "send to session", "run council", "ultraplan", "ultrareview", "autoloop", "ultraapp", "Forge tab", "build a web app", "one-click app", "AppSpec", "autonomous iteration", "iterate until goal", "deep paper review", "auto research", "switch model", "multi-agent", "coding session", "session inbox", "grok", "grok build", "opencode", "mcp server", "clawo-mcp", "hermes mcp", "model context protocol", "ultracode", "dynamic workflow", "fanout", "fan-out", "best-of-N", "steer turn", "interrupt turn", "fork thread", "rollback turns", "acp", "agent client protocol", "clawo acp", "zed agent", "jetbrains agent", "external agent", "dsh subagent", "deepseek harness", "clawo runs", "run ledger", "how much did it cost", "token usage", "spend cap", "budget limit", "maxBudgetUsd".
4
4
  metadata:
5
5
  {
6
6
  "openclaw":
@@ -36,7 +36,7 @@ metadata:
36
36
 
37
37
  # Claw Orchestrator Skill
38
38
 
39
- Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, and custom CLIs into headless agentic engines with 69 tools.
39
+ Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Grok Build, OpenCode, and custom CLIs into headless agentic engines with 69 tools.
40
40
 
41
41
  ## Engine Quick Reference
42
42
 
@@ -45,7 +45,7 @@ Claw Orchestrator — persistent multi-engine coding session manager for claw-st
45
45
  | `claude` | `claude` | Persistent subprocess | Multi-turn, complex tasks |
46
46
  | `codex` | `codex exec` | Per-message spawn | One-shot execution |
47
47
  | `agy` | `agy -p` | Per-message spawn | Google Antigravity; plain-text, auto conversation resume |
48
- | `cursor` | `agent -p` | Per-message spawn | One-shot execution |
48
+ | `grok` | `grok -p` | Per-message spawn | xAI Grok Build; engine-reported cost, resumable session |
49
49
  | `opencode` | `opencode run` | Per-message spawn | Provider-agnostic (`provider/model`) |
50
50
 
51
51
  ## Core Workflow
@@ -55,7 +55,7 @@ Claw Orchestrator — persistent multi-engine coding session manager for claw-st
55
55
  session_start({ name: "myproject", cwd: "/path/to/project", engine: "claude" })
56
56
  session_start({ name: "codex-task", cwd: "/path/to/project", engine: "codex" })
57
57
  session_start({ name: "agy-task", cwd: "/path/to/project", engine: "agy" })
58
- session_start({ name: "cursor-task", cwd: "/path/to/project", engine: "cursor" })
58
+ session_start({ name: "grok-task", cwd: "/path/to/project", engine: "grok" })
59
59
  session_start({ name: "opencode-task", cwd: "/path/to/project", engine: "opencode", model: "anthropic/claude-sonnet-4" })
60
60
 
61
61
  // 2. Send messages
@@ -231,5 +231,5 @@ Each engine requires its own auth before use:
231
231
  - **Claude**: `claude /login` or `ANTHROPIC_API_KEY`
232
232
  - **Codex**: `codex login` or `OPENAI_API_KEY`
233
233
  - **Antigravity**: run `agy` once and complete the Google OAuth login
234
- - **Cursor**: `agent login` or `CURSOR_API_KEY`
234
+ - **Grok**: run `grok` once and sign in (grok.com account or `XAI_API_KEY`)
235
235
  - **OpenCode**: `opencode auth login` (provider-agnostic; use `provider/model` form for `model`)
@@ -94,7 +94,7 @@ cancelling one abandons the poll rather than stopping the work.
94
94
 
95
95
  `session/new` also returns a `category: "model"` config option whose values are
96
96
  **grouped by engine**, built from the shared registry in `src/models.ts`. One
97
- dropdown holds Claude, Codex and Cursor models at once. Changing it restarts the
97
+ dropdown holds Claude, Codex and Grok models at once. Changing it restarts the
98
98
  underlying session on the new engine; the ACP session id is unaffected.
99
99
 
100
100
  Two engines are absent for different reasons:
@@ -34,15 +34,17 @@ uses its own default model rather than receiving the Claude `opus` / `sonnet`
34
34
  defaults. Role instructions are included in-band for engines that do not expose a
35
35
  native system-prompt flag.
36
36
 
37
- Engines without native multi-turn conversation (Cursor, OpenCode, one-shot custom
38
- engines) spawn a fresh process per send, so the dispatcher replays that role's
39
- transcript in-band as a `<conversation_history>` block, oldest turns dropped past a
40
- character budget. Claude, Codex and Antigravity keep context themselves and get no
41
- replay.
37
+ Engines without native multi-turn conversation (Gemini and one-shot custom engines)
38
+ spawn a fresh process per send with nothing to resume, so the dispatcher replays that
39
+ role's transcript in-band as a `<conversation_history>` block, oldest turns dropped
40
+ past a character budget. Claude, Codex, Antigravity, Grok, OpenCode and Cursor each
41
+ resume their own conversation by id and get no replay — see
42
+ `engineHasNativeConversation` in `types.ts`, which is the single source of truth for
43
+ this and is checked with a two-turn recall test per engine.
42
44
 
43
45
  The Planner runs read-only so strategy cannot turn into source edits, and that is
44
46
  enforced by the engine rather than requested politely: Claude uses plan mode,
45
- Antigravity and Cursor use their plan modes, and OpenCode gets a generated
47
+ Antigravity uses its plan mode, and OpenCode gets a generated
46
48
  `clawo-readonly` agent that denies `edit`/`bash`/`external_directory` (its built-in
47
49
  `plan` agent is a user-overridable preset that denies neither, so a "read-only"
48
50
  session could otherwise still author files through a shell heredoc). A custom
@@ -176,7 +178,7 @@ Reviewer 70 %. Override per run via `compactThresholds`. A 30 s debounce
176
178
  prevents re-fire while post-compact stats settle. Events: `compact` is
177
179
  emitted on the dispatcher EventEmitter AND appended to `decisions.jsonl`.
178
180
 
179
- One-shot engines (`codex`, `agy`, `cursor`, `opencode`) cannot compact — their
181
+ One-shot engines (`codex`, `agy`, `grok`, `opencode`) cannot compact — their
180
182
  CLIs expose no such command. The threshold is still meaningful there because
181
183
  `contextPercent` now tracks real occupancy, but crossing it cannot free space:
182
184
  the session emits a single warning on its log channel the first time compaction
@@ -42,7 +42,8 @@ The server exposes an OpenAI-compatible chat completions endpoint, enabling any
42
42
  **Model routing:** The `model` field auto-routes to the correct engine:
43
43
  - `claude-*`, `opus`, `sonnet`, `haiku` → Claude engine
44
44
  - `gpt-*` → Codex engine
45
- - `composer-*` → Cursor engine
45
+ - `grok-*` → Grok engine
46
+ - `composer-*` → Cursor engine (legacy)
46
47
  - `gemini-3.5-flash`, `gemini-3.1-pro`, `agy-*`, `agy/*` → Antigravity (`agy`) engine
47
48
  - other `gemini-*` → the legacy `gemini` engine (Gemini CLI is sunset; prefer `agy`)
48
49
 
@@ -61,7 +62,7 @@ clawo session-start [name] [options]
61
62
  | Flag | Description |
62
63
  |------|-------------|
63
64
  | `-d, --cwd <dir>` | Working directory |
64
- | `-e, --engine <engine>` | Engine: `claude` (default), `codex`, `codex-app`, `agy`, `cursor`, `opencode`, or `custom` |
65
+ | `-e, --engine <engine>` | Engine: `claude` (default), `codex`, `codex-app`, `agy`, `grok`, `opencode`, or `custom` |
65
66
  | `-m, --model <model>` | Model name or alias |
66
67
  | `--permission-mode <mode>` | `acceptEdits`, `plan`, `auto`, `bypassPermissions`, `manual`, `dontAsk` |
67
68
  | `--effort <level>` | `low`, `medium`, `high`, `max`, `auto` |
@@ -23,7 +23,7 @@ openclaw plugins install @enderfga/claw-orchestrator --dangerously-force-unsafe-
23
23
  openclaw gateway restart
24
24
  ```
25
25
 
26
- > **Why `--dangerously-force-unsafe-install`?** Claw Orchestrator spawns Claude Code / Codex / Antigravity / Cursor Agent / OpenCode CLI subprocesses via `child_process`, which OpenClaw's security scanner flags by design. The flag is required — there is no way to drive coding CLIs without process spawning.
26
+ > **Why `--dangerously-force-unsafe-install`?** Claw Orchestrator spawns Claude Code / Codex / Antigravity / Grok Build / OpenCode CLI subprocesses via `child_process`, which OpenClaw's security scanner flags by design. The flag is required — there is no way to drive coding CLIs without process spawning.
27
27
 
28
28
  Agents automatically get access to all session, council, and management tools.
29
29
 
@@ -14,7 +14,9 @@ SessionManager
14
14
  │ └── Wraps: codex app-server --listen stdio:// (long-running JSON-RPC; required for /goal)
15
15
  ├── engine: 'agy' → PersistentAgySession
16
16
  │ └── Wraps: agy -p (Google Antigravity CLI, per-message spawning, stream-json output)
17
- ├── engine: 'cursor' → PersistentCursorSession
17
+ ├── engine: 'grok' → PersistentGrokSession
18
+ │ └── Wraps: grok -p --output-format json (xAI Grok Build, per-message spawning)
19
+ ├── engine: 'cursor' → PersistentCursorSession (legacy)
18
20
  │ └── Wraps: agent -p --force --trust --output-format stream-json (per-message spawning)
19
21
  ├── engine: 'opencode' → PersistentOpencodeSession
20
22
  │ └── Wraps: opencode run --format json (per-message spawning)
@@ -173,9 +175,59 @@ await manager.startSession({
173
175
  > multi-model **proxy** still talks to the Gemini **API**; that is a different
174
176
  > subsystem and is unaffected.)
175
177
 
176
- ### Cursor Agent (`engine: 'cursor'`)
178
+ ### Grok Build (`engine: 'grok'`)
179
+
180
+ Wraps xAI's **Grok Build** CLI. Each `send()` spawns `grok -p <msg> --output-format json`, which
181
+ prints a single JSON object and exits. Verified against `grok` **1.0.5**.
182
+
183
+ - **Cost comes from the engine, not from our price table.** The result object carries
184
+ `total_cost_usd`, and the wrapper writes it straight into the session's spend. Every other engine
185
+ here multiplies tokens by a rate in `models.ts` — the metadata most prone to going stale — so on
186
+ this engine the run ledger and the `maxBudgetUsd` gate both read what xAI actually charged.
187
+ `grok-4.6` is still registered, for its context window and an indicative breakdown; its two price
188
+ tiers ($2/$0.50/$6 under a 200K prompt, $4/$1/$12 at or above, charged across the whole request)
189
+ therefore never have to be modelled here.
190
+ - **Real conversation continuity**: the `sessionId` from turn 1 is replayed as `--resume <id>`.
191
+ `--continue` is deliberately not used — it means "the most recent session for this cwd", which
192
+ collides between concurrent sessions. Confirmed with a two-turn recall test, not inferred.
193
+ - Real token counts from `usage` (`input_tokens`, `output_tokens`, `cache_read_input_tokens`).
194
+ These are **per-turn**, not cumulative over the thread — checked by resuming and reading turn 2,
195
+ because the same-looking field on codex is a running total.
196
+ - Permission modes pass straight through: grok's `--permission-mode` takes the same vocabulary we
197
+ use. The one exception is our `manual`, which grok spells `default`.
198
+ - Reasoning effort maps to `--effort`; grok accepts `low|medium|high`, so `max` and `xhigh` clamp.
199
+ - **`sandboxMode: 'read-only'` is refused, not approximated.** grok has `--permission-mode plan` and
200
+ `--deny` rules, but plan mode alone is model-cooperative — the shape that let an adversarial
201
+ prompt write through Cursor's plan mode — and the deny rules have not been through the
202
+ write × shell × subagent × resumed-turn matrix this project requires before claiming a boundary.
203
+ A read-only grok session throws rather than running writable under a read-only label.
204
+ - Binary: `grok` (set `GROK_BIN` to override). Not `agent`: xAI's installer claims that name too,
205
+ and so did Cursor's.
206
+ - Requires Grok Build: see `x.ai/cli`.
177
207
 
178
- Wraps the Cursor Agent CLI (`agent`) with `--print --output-format stream-json`. Write-enabled sessions use `--force`. Each `send()` spawns a new process.
208
+ ```typescript
209
+ await manager.startSession({
210
+ name: 'grok-task',
211
+ engine: 'grok',
212
+ model: 'grok-4.6',
213
+ cwd: '/project',
214
+ });
215
+ ```
216
+
217
+ ### Cursor Agent (`engine: 'cursor'`) — legacy
218
+
219
+ > **Legacy: `engine: 'cursor'`.** Superseded in this lineup by Grok Build (`engine: 'grok'`).
220
+ > The `cursor` engine still exists and still works — existing callers are not broken — but it is
221
+ > no longer a documented option, is not version-tracked, and gets no new work.
222
+ >
223
+ > Note what this is and is not: Cursor itself is **not** discontinued. Anysphere was acquired by
224
+ > SpaceX (closed 2026-08-15) and folded into the SpaceXAI team, and the CLI has shipped since. Two
225
+ > practical things pushed it out of the tracked set. Cursor never reports which model actually ran
226
+ > — its `system` init event says `"model": "Auto"` — so a router that spans Claude, GPT and Grok
227
+ > leaves every cost row attributed to a hardcoded proxy rate. And xAI's Grok installer now claims
228
+ > the bare `agent` name, so the binary that name resolves to depends on install order.
229
+
230
+ Wraps the Cursor Agent CLI with `--print --output-format stream-json`. Write-enabled sessions use `--force`. Each `send()` spawns a new process.
179
231
 
180
232
  - Conversation continuity: the chat id from the first turn's `system` event is captured and passed back as `--resume <chatId>` on later sends, so the model sees prior turns. `--continue` is deliberately not used: it resumes "the latest chat", which collides between concurrent sessions.
181
233
  - One-shot execution per message (no persistent subprocess)
@@ -185,7 +237,9 @@ Wraps the Cursor Agent CLI (`agent`) with `--print --output-format stream-json`.
185
237
  - `--trust` auto-trusts the workspace without prompting
186
238
  - Cursor uses its own model routing (e.g., `sonnet-4`, `gpt-5`, `auto`)
187
239
  - Requires Cursor Agent CLI: `curl https://cursor.com/install -fsSL | bash`
188
- - Binary: `agent` (set `CURSOR_BIN` env var to override)
240
+ - Binary: `cursor-agent` (set `CURSOR_BIN` env var to override). The generic `agent` name is
241
+ deliberately not used: xAI's Grok installer symlinks `agent` to its own binary, which rejects
242
+ `--force`/`--trust`/`--workspace` and fails the turn with "unexpected argument"
189
243
 
190
244
  ```typescript
191
245
  await manager.startSession({
@@ -263,6 +317,10 @@ interface ISession {
263
317
  }
264
318
  ```
265
319
 
320
+ `SessionStats` requires `turnsSucceeded` as well as `turns`, so an engine that
321
+ implements this interface has to say which of its turns succeeded — the exit code
322
+ is not the answer on every engine. See "Stats & Monitoring" in `sessions.md`.
323
+
266
324
  ## Team Tools Across Engines
267
325
 
268
326
  Team tools (`team_list`, `team_send`) operate on the same virtual-team layer for **every** engine: the "team" is the set of all active sessions managed by SessionManager.
@@ -34,7 +34,7 @@ every engine passes through.
34
34
  |---|---|
35
35
  | `ts` | ISO timestamp of turn completion |
36
36
  | `session` | SessionManager session name |
37
- | `engine` | `claude` / `codex` / `codex-app` / `cursor` / `opencode` / `agy` / `custom` |
37
+ | `engine` | `claude` / `codex` / `codex-app` / `grok` / `opencode` / `agy` / `custom` |
38
38
  | `model` | Configured model, or the engine's own reported model when none was set |
39
39
  | `cwd` | Working directory the turn ran in |
40
40
  | `turn` | 1-based turn index within the session |
@@ -43,8 +43,8 @@ every engine passes through.
43
43
  | `tokensEstimated` | `true` when the counts came from `estimateTokens()` (see below) |
44
44
  | `durationMs` | Wall-clock for the turn |
45
45
  | `toolCalls` / `toolErrors` | Per-turn deltas |
46
- | `ok` | `false` for a turn that threw or reported `is_error` |
47
- | `error` | Failure text, truncated to 500 chars |
46
+ | `ok` | `false` for a turn that threw, or that the session's own `turnsSucceeded` counter did not count (see `sessions.md`). Falls back to "nothing was thrown" when the counter cannot be read |
47
+ | `error` | Failure text, truncated to 500 chars. Absent when the turn resolved but the engine did not count it as succeeded (an interrupted or non-SUCCESS turn), so a failed row does not always carry one |
48
48
  | `parent` | council id / fanout id / autoloop run id, when the turn belongs to one |
49
49
 
50
50
  Deltas rather than totals means summing a query window gives that window's spend
@@ -116,7 +116,8 @@ flagged `tokensEstimated: true`; the CLI marks those costs with a trailing `~`.
116
116
  | `claude` | Engine-reported |
117
117
  | `codex` | Engine-reported |
118
118
  | `codex-app` | Engine-reported |
119
- | `cursor` | Engine-reported when the stream carries `usage`, else estimated |
119
+ | `grok` | Engine-reported — and so is the **cost**: this engine reports `total_cost_usd`, which the wrapper passes through instead of pricing tokens from the registry, so registry drift cannot affect a grok row |
120
+ | `cursor` (legacy) | Engine-reported when the stream carries `usage`, else estimated |
120
121
  | `opencode` | Engine-reported when the run JSON carries `tokens`, else estimated |
121
122
  | `agy` | Engine-reported when the result event carries usage, else estimated |
122
123
  | `custom` | Depends on the CLI; estimated when it emits no usage |
@@ -2,7 +2,7 @@
2
2
 
3
3
  > **Cost warning**: This bridge routes requests through the Claude Code CLI, which uses your Claude Max subscription's **extra usage** quota. When OpenClaw's agent loop sends its system prompt (with distinctive tool definitions and agent instructions), Anthropic's backend recognizes this as programmatic/agent traffic and bills it against extra usage — **not** the included allowance. This is by design: the bridge does NOT bypass Anthropic's billing or subscription enforcement. Using it as OpenClaw's primary model backend means every agent turn consumes extra usage credits at standard API rates ($15/M input, $75/M output for Opus). Monitor your usage at [claude.ai/settings/usage](https://claude.ai/settings/usage).
4
4
 
5
- The embedded server exposes a drop-in OpenAI-compatible endpoint so any client that speaks `/v1/chat/completions` can talk to a persistent Claude Code (or Codex / Antigravity / Cursor) session. The bridge is designed to serve **two kinds of clients as first-class citizens**:
5
+ The embedded server exposes a drop-in OpenAI-compatible endpoint so any client that speaks `/v1/chat/completions` can talk to a persistent Claude Code (or Codex / Antigravity / Grok) session. The bridge is designed to serve **two kinds of clients as first-class citizens**:
6
6
 
7
7
  1. **Upstream agents** that maintain their own conversation state and forward only the latest user turn — OpenClaw's main agent loop, cron jobs, subagents, programmatic clients.
8
8
  2. **OpenAI-compatible webchat / labeling tools** that re-send the full transcript on every turn — ChatGPT-Next-Web, Open WebUI, LobeChat, data-labeling pipelines.
@@ -81,7 +81,7 @@ mechanism is used depends on whether the engine keeps the conversation itself.
81
81
  | Engine | Turn 1 | Later turns |
82
82
  |---|---|---|
83
83
  | `claude` | Schemas go into the session system prompt (`--system-prompt`) | Nothing injected — the system prompt persists |
84
- | `codex`, `codex-app`, `agy`, `opencode`, `cursor` | Full schema block prepended to the message | A short reminder of the calling convention, no schemas — but only once the conversation id has been captured; until then the full block is sent again |
84
+ | `codex`, `codex-app`, `agy`, `opencode`, `grok` | Full schema block prepended to the message | A short reminder of the calling convention, no schemas — but only once the conversation id has been captured; until then the full block is sent again |
85
85
  | `gemini`, one-shot `custom` | Full schema block prepended to the message | Full schema block again — these have no resume surface, so nothing persists between sends |
86
86
 
87
87
  The middle row is the one worth understanding. Those engines resume a conversation by id, so
@@ -148,7 +148,8 @@ Sample response:
148
148
  "model": "claude-opus-4-6",
149
149
  "cwd": "/home/user/projects",
150
150
  "created": "2026-04-09T03:12:18.441Z",
151
- "turns": 14,
151
+ "turns": 68,
152
+ "turns_succeeded": 14,
152
153
  "tokens_in": 248312,
153
154
  "tokens_out": 38201,
154
155
  "cached_tokens": 198104,
@@ -159,6 +160,11 @@ Sample response:
159
160
  }
160
161
  ```
161
162
 
163
+ `turns_succeeded` is one per request; `turns` is not comparable to it on these
164
+ sessions — the Claude CLI emits a `user` event per tool-result batch and `turns`
165
+ counts those, so the gap above is tool use, not failures. Compare
166
+ `turns_succeeded` against your own request count.
167
+
162
168
  The single most important field is **`cached_tokens`**. If it grows turn-over-turn, the persistent CLI is being reused and Anthropic prompt caching is warming. If it stays at 0, something is killing the session every turn — check that no client is sending `X-Session-Reset` unintentionally and that `OPENAI_COMPAT_NEW_CONVO_HEURISTIC` is not set when it shouldn't be.
163
169
 
164
170
  ## Smoke tests
@@ -29,7 +29,7 @@ Key options:
29
29
 
30
30
  | Option | Description |
31
31
  |--------|-------------|
32
- | `engine` | `'claude'` (default), `'codex'`, `'codex-app'`, `'agy'`, `'cursor'`, `'opencode'`, or `'custom'` — see [Multi-Engine](./multi-engine.md) |
32
+ | `engine` | `'claude'` (default), `'codex'`, `'codex-app'`, `'agy'`, `'grok'`, `'opencode'`, or `'custom'` — see [Multi-Engine](./multi-engine.md) |
33
33
  | `model` | Model alias (`fable`, `opus`, `sonnet`, `haiku`, `agy-pro`) or full name |
34
34
  | `permissionMode` | `acceptEdits`, `bypassPermissions`, `plan`, `auto`, `manual`, `dontAsk` (`default` = legacy alias for `manual`) |
35
35
  | `effort` | `low`, `medium`, `high`, `max`, `auto` |
@@ -135,9 +135,32 @@ prompt over the window the engine actually enforces — so it rises and falls wi
135
135
  the conversation. `stats.tokensIn` is the different question of how many input
136
136
  tokens the session has been billed for in total, which only ever grows.
137
137
  `compactSession()` is a no-op on engines whose CLI has no compaction command
138
- (`codex`, `agy`, `cursor`, `opencode`); those sessions log a warning the first
138
+ (`codex`, `agy`, `grok`, `opencode`); those sessions log a warning the first
139
139
  time it is called.
140
140
 
141
+ `stats.turns` and `stats.turnsSucceeded` are the same kind of distinction.
142
+ `turns` counts turns that reached the engine, whatever their outcome, including
143
+ the ones that then failed. `turnsSucceeded` counts only the ones the engine
144
+ reported as successful, and it is one per send on every engine.
145
+
146
+ **The two are only comparable on the one-shot engines.** There `turns` is also
147
+ one per send, so the difference between them is the failure count. On `claude`
148
+ and a persistent `custom`, `turns` counts `user` events — and the CLI emits one
149
+ per tool-result batch as well as the prompt echo, so a send that used eight tools
150
+ counts nine. Compare `turnsSucceeded` against the number of sends there, never
151
+ against `turns`.
152
+
153
+ Which outcome counts as a success is the engine's own verdict, not the exit
154
+ code's: `codex` fails a turn that emits `turn.failed` while exiting 0, `agy`
155
+ requires a `SUCCESS` status *and* a zero exit (it can report success and then die
156
+ in cleanup), `gemini` succeeds on exit 53 because its turn limit resolves,
157
+ `codex-app` requires `status: 'completed'` so a turn cancelled through
158
+ `interrupt()` does not count, and `opencode` refuses a turn on purpose when
159
+ read-only enforcement did not load.
160
+
161
+ The run ledger's `ok` reads this same counter, so a turn cannot be a failure on
162
+ `/v1/sessions` and a success in `clawo runs`.
163
+
141
164
  ### Cost Tracking
142
165
 
143
166
  ```typescript
@@ -12,10 +12,10 @@ Start a persistent coding session with full CLI flag support.
12
12
  |-----------|------|-------------|
13
13
  | `name` | string | Session name (auto-generated if omitted) |
14
14
  | `cwd` | string | Working directory |
15
- | `engine` | `'claude'` \| `'codex'` \| `'codex-app'` \| `'agy'` \| `'cursor'` \| `'opencode'` \| `'custom'` | Engine to use (default: `claude`). `agy` wraps Google Antigravity CLI. `opencode` wraps sst/opencode (pass model as `provider/model`). Use `custom` with `customEngine` for any CLI. (`'gemini'` is still accepted for existing callers, but Gemini CLI is sunset — use `agy` for Google.) |
15
+ | `engine` | `'claude'` \| `'codex'` \| `'codex-app'` \| `'agy'` \| `'grok'` \| `'opencode'` \| `'custom'` | Engine to use (default: `claude`). `agy` wraps Google Antigravity CLI. `grok` wraps xAI Grok Build and reports its own per-turn cost. `opencode` wraps sst/opencode (pass model as `provider/model`). Use `custom` with `customEngine` for any CLI. (`'gemini'` and `'cursor'` are still accepted for existing callers but are legacy — not version-tracked; use `agy` for Google and `grok` in place of Cursor.) |
16
16
  | `model` | string | Model alias or full name |
17
17
  | `permissionMode` | string | `acceptEdits`, `bypassPermissions`, `plan`, `auto`, `manual`, `dontAsk` (`default` = legacy alias for `manual`) |
18
- | `sandboxMode` | `'read-only'` \| `'workspace-write'` \| `'danger-full-access'` | Sandbox policy. Codex supports all values. `read-only` is enforced on every other built-in engine too: Claude → plan mode; Antigravity / Cursor → their plan modes; OpenCode → a generated `clawo-readonly` agent denying `edit`/`bash`. A `custom` engine must map it via `permissionModes`, or the session refuses to start. Persisted across session resume. |
18
+ | `sandboxMode` | `'read-only'` \| `'workspace-write'` \| `'danger-full-access'` | Sandbox policy. Codex supports all values. `read-only` is enforced on every other built-in engine too: Claude → plan mode; Antigravity → its plan mode; OpenCode → a generated `clawo-readonly` agent denying `edit`/`bash`. **`grok` refuses a read-only session** rather than approximate one — its enforcement has not been adversarially verified. A `custom` engine must map it via `permissionModes`, or the session refuses to start. Persisted across session resume. |
19
19
  | `effort` | string | `low`, `medium`, `high`, `max`, `auto` |
20
20
  | `allowedTools` | string[] | Tools to auto-approve |
21
21
  | `disallowedTools` | string[] | Tools to deny |