@enderfga/claw-orchestrator 4.13.1 → 4.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +5 -4
  2. package/assets/banner.jpg +0 -0
  3. package/dist/bin/cli.js +29 -0
  4. package/dist/bin/cli.js.map +1 -1
  5. package/dist/src/autoloop/dispatcher.js +3 -0
  6. package/dist/src/autoloop/dispatcher.js.map +1 -1
  7. package/dist/src/base-oneshot-session.d.ts +8 -0
  8. package/dist/src/base-oneshot-session.js +15 -0
  9. package/dist/src/base-oneshot-session.js.map +1 -1
  10. package/dist/src/budget.d.ts +36 -0
  11. package/dist/src/budget.js +54 -0
  12. package/dist/src/budget.js.map +1 -0
  13. package/dist/src/council.js +2 -1
  14. package/dist/src/council.js.map +1 -1
  15. package/dist/src/dashboard/index.html +24 -0
  16. package/dist/src/embedded-server.js +16 -0
  17. package/dist/src/embedded-server.js.map +1 -1
  18. package/dist/src/fanout.js +2 -0
  19. package/dist/src/fanout.js.map +1 -1
  20. package/dist/src/openai-compat.d.ts +1 -0
  21. package/dist/src/openai-compat.js +17 -1
  22. package/dist/src/openai-compat.js.map +1 -1
  23. package/dist/src/persistent-agy-session.js +1 -0
  24. package/dist/src/persistent-agy-session.js.map +1 -1
  25. package/dist/src/persistent-cursor-session.js +1 -0
  26. package/dist/src/persistent-cursor-session.js.map +1 -1
  27. package/dist/src/persistent-custom-session.js +7 -0
  28. package/dist/src/persistent-custom-session.js.map +1 -1
  29. package/dist/src/persistent-gemini-session.js +1 -0
  30. package/dist/src/persistent-gemini-session.js.map +1 -1
  31. package/dist/src/persistent-opencode-session.js +1 -0
  32. package/dist/src/persistent-opencode-session.js.map +1 -1
  33. package/dist/src/persistent-session.js +10 -0
  34. package/dist/src/persistent-session.js.map +1 -1
  35. package/dist/src/run-ledger.d.ts +95 -0
  36. package/dist/src/run-ledger.js +216 -0
  37. package/dist/src/run-ledger.js.map +1 -0
  38. package/dist/src/session-manager.d.ts +27 -0
  39. package/dist/src/session-manager.js +153 -16
  40. package/dist/src/session-manager.js.map +1 -1
  41. package/dist/src/types.d.ts +17 -0
  42. package/openclaw.plugin.json +28 -4
  43. package/package.json +3 -2
  44. package/skills/SKILL.md +18 -2
  45. package/skills/references/cli.md +13 -0
  46. package/skills/references/observability.md +131 -0
  47. package/skills/references/tools.md +1 -1
@@ -293,6 +293,12 @@ export interface SessionStats {
293
293
  cursorChatId?: string;
294
294
  /** OpenCode session ID captured from the run's JSON output. Reused via `--session` for multi-turn context. */
295
295
  opencodeSessionId?: string;
296
+ /**
297
+ * True when the most recent turn's token counts came from estimateTokens()
298
+ * because the engine reported no usage. Cost derived from those counts is an
299
+ * estimate; the run ledger and budget docs surface it as such.
300
+ */
301
+ tokensEstimated?: boolean;
296
302
  }
297
303
  export interface HookConfig {
298
304
  onToolError?: string;
@@ -319,6 +325,11 @@ export interface SendOptions {
319
325
  stream?: boolean;
320
326
  onChunk?: (chunk: string) => void;
321
327
  onEvent?: (event: StreamEvent) => void;
328
+ /**
329
+ * council id / fanout id / autoloop run id. Stamped onto the run-ledger row
330
+ * so a multi-agent run can be reassembled from the ledger afterwards.
331
+ */
332
+ parentRunId?: string;
322
333
  }
323
334
  export interface StreamEvent {
324
335
  type: string;
@@ -349,6 +360,12 @@ export interface SessionInfo {
349
360
  model?: string;
350
361
  paused: boolean;
351
362
  stats: SessionStats;
363
+ /** Cumulative USD spent by this session, as reported by the engine's cost model. */
364
+ costUsd?: number;
365
+ /** The session's spend cap, when one was configured via `maxBudgetUsd`. */
366
+ budgetUsd?: number;
367
+ /** True once `costUsd` reached `budgetUsd` — further turns are refused. */
368
+ budgetExhausted?: boolean;
352
369
  }
353
370
  export interface SendResult {
354
371
  output: string;
@@ -17,12 +17,26 @@
17
17
  },
18
18
  "defaultPermissionMode": {
19
19
  "type": "string",
20
- "enum": ["acceptEdits", "bypassPermissions", "default", "delegate", "dontAsk", "plan", "auto"],
20
+ "enum": [
21
+ "acceptEdits",
22
+ "bypassPermissions",
23
+ "default",
24
+ "delegate",
25
+ "dontAsk",
26
+ "plan",
27
+ "auto"
28
+ ],
21
29
  "default": "acceptEdits"
22
30
  },
23
31
  "defaultEffort": {
24
32
  "type": "string",
25
- "enum": ["low", "medium", "high", "max", "auto"],
33
+ "enum": [
34
+ "low",
35
+ "medium",
36
+ "high",
37
+ "max",
38
+ "auto"
39
+ ],
26
40
  "default": "auto"
27
41
  },
28
42
  "maxConcurrentSessions": {
@@ -60,7 +74,9 @@
60
74
  },
61
75
  "capabilities": {
62
76
  "childProcess": true,
63
- "networkAccess": ["127.0.0.1"]
77
+ "networkAccess": [
78
+ "127.0.0.1"
79
+ ]
64
80
  },
65
81
  "contracts": {
66
82
  "tools": [
@@ -106,6 +122,12 @@
106
122
  "council_review",
107
123
  "council_accept",
108
124
  "council_reject",
125
+ "autoloop_start",
126
+ "autoloop_chat",
127
+ "autoloop_status",
128
+ "autoloop_list",
129
+ "autoloop_reset_agent",
130
+ "autoloop_stop",
109
131
  "session_send_to",
110
132
  "session_inbox",
111
133
  "session_deliver_inbox",
@@ -129,5 +151,7 @@
129
151
  "ultraapp_delete"
130
152
  ]
131
153
  },
132
- "skills": ["skills/SKILL.md"]
154
+ "skills": [
155
+ "skills/SKILL.md"
156
+ ]
133
157
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@enderfga/claw-orchestrator",
3
- "version": "4.13.1",
3
+ "version": "4.14.0",
4
4
  "description": "Claw Orchestrator — run Claude Code, Codex, Gemini, Cursor Agent, OpenCode and custom coding CLIs as one unified runtime. Drop into Hermes Agent, Claude Desktop, Cursor, Cline, Continue, Zed, Windsurf, Goose or any Model Context Protocol (MCP) host, install as an OpenClaw plugin, or run standalone. Persistent sessions, multi-agent council, ultraplan, ultrareview, autoloop, tool orchestration.",
5
5
  "type": "module",
6
6
  "main": "./dist/src/index.js",
@@ -83,7 +83,8 @@
83
83
  "jetbrains",
84
84
  "neovim",
85
85
  "antigravity",
86
- "coding-agent-protocol"
86
+ "coding-agent-protocol",
87
+ "deepseek-harness"
87
88
  ],
88
89
  "author": "enderfga",
89
90
  "repository": {
package/skills/SKILL.md CHANGED
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: claw-orchestrator
3
- description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Cursor, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's 65 tools as an MCP server to Hermes Agent / Claude Desktop / Cursor / Cline / Continue / Zed / Windsurf / Goose, or running as an Agent Client Protocol (ACP) agent that Zed / JetBrains / Neovim / Emacs / VS Code / dsh can drive directly. Triggers on "start a session", "send to session", "run council", "ultraplan", "ultrareview", "autoloop", "ultraapp", "Forge tab", "build a web app", "one-click app", "AppSpec", "autonomous iteration", "iterate until goal", "deep paper review", "auto research", "switch model", "multi-agent", "coding session", "session inbox", "cursor agent", "opencode", "mcp server", "clawo-mcp", "hermes mcp", "model context protocol", "ultracode", "dynamic workflow", "fanout", "fan-out", "best-of-N", "steer turn", "interrupt turn", "fork thread", "rollback turns", "acp", "agent client protocol", "clawo acp", "zed agent", "jetbrains agent", "external agent", "dsh subagent", "deepseek harness".
3
+ description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Cursor, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's 69 tools as an MCP server to Hermes Agent / Claude Desktop / Cursor / Cline / Continue / Zed / Windsurf / Goose, or running as an Agent Client Protocol (ACP) agent that Zed / JetBrains / Neovim / Emacs / VS Code / dsh can drive directly. Triggers on "start a session", "send to session", "run council", "ultraplan", "ultrareview", "autoloop", "ultraapp", "Forge tab", "build a web app", "one-click app", "AppSpec", "autonomous iteration", "iterate until goal", "deep paper review", "auto research", "switch model", "multi-agent", "coding session", "session inbox", "cursor agent", "opencode", "mcp server", "clawo-mcp", "hermes mcp", "model context protocol", "ultracode", "dynamic workflow", "fanout", "fan-out", "best-of-N", "steer turn", "interrupt turn", "fork thread", "rollback turns", "acp", "agent client protocol", "clawo acp", "zed agent", "jetbrains agent", "external agent", "dsh subagent", "deepseek harness", "clawo runs", "run ledger", "how much did it cost", "token usage", "spend cap", "budget limit", "maxBudgetUsd".
4
4
  metadata:
5
5
  {
6
6
  "openclaw":
@@ -36,7 +36,7 @@ metadata:
36
36
 
37
37
  # Claw Orchestrator Skill
38
38
 
39
- Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, and custom CLIs into headless agentic engines with 65 tools.
39
+ Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, and custom CLIs into headless agentic engines with 69 tools.
40
40
 
41
41
  ## Engine Quick Reference
42
42
 
@@ -208,6 +208,22 @@ engine mid-session.
208
208
 
209
209
  For setup, the dsh YAML block, and the cancellation/permission limits: see [references/acp.md](references/acp.md)
210
210
 
211
+ ## Cost & spend caps
212
+
213
+ Every turn on every engine is appended to a durable ledger at
214
+ `~/.claw-orchestrator/runs/YYYY-MM-DD.jsonl` — engine, model, per-turn tokens, cost, duration,
215
+ and the council / fanout / autoloop it belonged to. Query it with `clawo runs [--since 24h]
216
+ [--engine X] [--parent <run id>] [--json]`, `GET /runs`, or `manager.getRunLedger()`; it
217
+ survives restarts, so it answers "what did we run today and what did it cost" after the
218
+ sessions are gone.
219
+
220
+ `maxBudgetUsd` on a session (or per council / fanout agent) is enforced by the runtime, so it
221
+ holds on Codex, Cursor, agy, OpenCode and custom engines too — not just Claude Code. Once
222
+ cumulative spend reaches the cap, further sends are refused before the engine is spawned.
223
+
224
+ For the row schema, the query surfaces, and which engines report real token usage versus
225
+ estimating it: see [references/observability.md](references/observability.md)
226
+
211
227
  ## Authentication Prerequisites
212
228
 
213
229
  Each engine requires its own auth before use:
@@ -134,6 +134,19 @@ clawo session-grep <name> <pattern> [-n, --limit <n>]
134
134
  clawo session-compact <name> [--summary <text>]
135
135
  ```
136
136
 
137
+ ## Run Ledger
138
+
139
+ ```bash
140
+ clawo runs [--since <window>] [-n, --limit <n>] [--session <name>] [--engine <engine>] [--parent <id>] [--json]
141
+ ```
142
+
143
+ Show the durable per-turn record kept at `~/.claw-orchestrator/runs/`. Unlike
144
+ `session-status`, this survives restarts and covers sessions this process never
145
+ owned. `--since` takes `30m` / `24h` / `7d` / `2w` or an ISO timestamp (default
146
+ `24h`); `--parent` filters to one council / fanout / autoloop run. Costs marked
147
+ with a trailing `~` came from estimated token counts. See
148
+ [observability.md](observability.md).
149
+
137
150
  ## Agent Management
138
151
 
139
152
  ```bash
@@ -0,0 +1,131 @@
1
+ # Observability — Run Ledger & Spend Caps
2
+
3
+ Two related surfaces: a durable record of every turn this runtime executes, and a
4
+ spend cap that is enforced by the runtime rather than by whichever CLI happens to
5
+ support a budget flag.
6
+
7
+ ## Why
8
+
9
+ `getStats()` / `getCost()` describe a **live** session. They live in memory, and
10
+ per-session history is capped and evicted, so a restart erased everything except
11
+ the resume-id registry — there was no way to answer "what did we run today, on
12
+ which engine, for how much". The run ledger is that record.
13
+
14
+ The same gap made `maxBudgetUsd` a promise the runtime did not keep: it was only
15
+ ever translated into Claude Code's `--max-budget-usd` flag, so a council of Codex
16
+ agents ran with no cap at all. The cap is now applied in `SessionManager`, which
17
+ every engine passes through.
18
+
19
+ ## The ledger
20
+
21
+ - Location: `~/.claw-orchestrator/runs/YYYY-MM-DD.jsonl` (override with
22
+ `CLAWO_RUNS_DIR`).
23
+ - One JSON object per line, one line per completed turn — **successful or not**.
24
+ - Shards are one file per UTC day. Queries always filter on the row timestamp, so
25
+ the shard boundary is a storage detail; `--since 24h` spans midnight correctly.
26
+ - Writes are best-effort: a ledger failure is logged at `warn` and swallowed. It
27
+ can never break the turn it is describing.
28
+ - Nothing prunes old shards. A row is roughly 250 bytes; delete shards yourself if
29
+ you want the history gone.
30
+
31
+ ### Row schema
32
+
33
+ | Field | Meaning |
34
+ |---|---|
35
+ | `ts` | ISO timestamp of turn completion |
36
+ | `session` | SessionManager session name |
37
+ | `engine` | `claude` / `codex` / `codex-app` / `cursor` / `opencode` / `agy` / `custom` |
38
+ | `model` | Configured model, or the engine's own reported model when none was set |
39
+ | `cwd` | Working directory the turn ran in |
40
+ | `turn` | 1-based turn index within the session |
41
+ | `tokensIn` / `tokensOut` / `cachedTokens` | **Per-turn deltas**, not session totals |
42
+ | `costUsd` | Per-turn delta in USD |
43
+ | `tokensEstimated` | `true` when the counts came from `estimateTokens()` (see below) |
44
+ | `durationMs` | Wall-clock for the turn |
45
+ | `toolCalls` / `toolErrors` | Per-turn deltas |
46
+ | `ok` | `false` for a turn that threw or reported `is_error` |
47
+ | `error` | Failure text, truncated to 500 chars |
48
+ | `parent` | council id / fanout id / autoloop run id, when the turn belongs to one |
49
+
50
+ Deltas rather than totals means summing a query window gives that window's spend
51
+ without double-counting.
52
+
53
+ Every path funnels through `SessionManager.sendMessage`, so council, fanout,
54
+ autoloop, ACP, the OpenAI-compatible bridge, the MCP server and the CLI are all
55
+ covered by the same hook. `parent` is what lets you reassemble a multi-agent run:
56
+
57
+ ```bash
58
+ clawo runs --parent council-1a2b3c4d
59
+ ```
60
+
61
+ ### Reading it
62
+
63
+ ```bash
64
+ clawo runs # last 24h, table
65
+ clawo runs --since 7d --engine codex # one engine, one week
66
+ clawo runs --session my-session --json # raw rows + summary
67
+ clawo runs --parent fanout-b5f7c886 # every agent turn of one fan-out
68
+ ```
69
+
70
+ Over HTTP (GET query string or POST JSON body):
71
+
72
+ ```bash
73
+ curl "http://127.0.0.1:18796/runs?since=24h&limit=200" -H "Authorization: Bearer $TOKEN"
74
+ ```
75
+
76
+ Returns `{ ok, rows, summary }`, where `summary` carries `rows`, `costUsd`,
77
+ `tokensIn`, `tokensOut`, `estimatedRows` and a per-engine breakdown. The dashboard
78
+ header shows the 24-hour figure from the same endpoint.
79
+
80
+ Programmatically: `manager.getRunLedger({ since, session, engine, parent, limit })`.
81
+
82
+ ## Spend caps
83
+
84
+ Set `maxBudgetUsd` on a session (or on a council / fanout, which applies it per
85
+ agent). Before each turn the runtime compares the session's cumulative
86
+ `getCost().totalUsd` against the cap and refuses to send when it has been reached:
87
+
88
+ ```
89
+ Budget exceeded for session "my-session" (codex): spent $1.2400 of $1.0000 cap.
90
+ Raise maxBudgetUsd or start a new session to continue.
91
+ ```
92
+
93
+ The refusal is a typed `BudgetExceededError` carrying `session`, `engine`,
94
+ `spentUsd` and `capUsd`, and it happens **before** the engine is spawned.
95
+
96
+ Notes:
97
+
98
+ - The check is "has the cap been reached", not "would this turn exceed it" — a
99
+ turn's cost is unknown until it finishes, so the last allowed turn can overshoot.
100
+ Size the cap accordingly.
101
+ - A cap of `0` or a negative number means *unset*, not *refuse everything*.
102
+ - Claude Code still receives `--max-budget-usd` as well: an in-CLI stop happens
103
+ earlier and therefore costs less than an after-the-fact refusal.
104
+ - `session_list` / `GET /session/list` expose `costUsd`, `budgetUsd` and
105
+ `budgetExhausted` so a stalled session shows *why* it stopped taking turns.
106
+
107
+ ## Accuracy: which engines report real usage
108
+
109
+ `costUsd` is derived from token counts times the model's price in `models.ts`.
110
+ Where the engine reports usage, those counts are the engine's own. Where it does
111
+ not, the wrapper falls back to `estimateTokens()` (characters ÷ 4) and the row is
112
+ flagged `tokensEstimated: true`; the CLI marks those costs with a trailing `~`.
113
+
114
+ | Engine | Token counts |
115
+ |---|---|
116
+ | `claude` | Engine-reported |
117
+ | `codex` | Engine-reported |
118
+ | `codex-app` | Engine-reported |
119
+ | `cursor` | Engine-reported when the stream carries `usage`, else estimated |
120
+ | `opencode` | Engine-reported when the run JSON carries `tokens`, else estimated |
121
+ | `agy` | Engine-reported when the result event carries usage, else estimated |
122
+ | `custom` | Depends on the CLI; estimated when it emits no usage |
123
+
124
+ So on an estimating engine the cap is best-effort. It will stop a runaway session;
125
+ it is not an accounting guarantee, and it is not a substitute for the spend limits
126
+ your provider offers.
127
+
128
+ Cost figures are also only as good as the pricing table: a model missing from
129
+ `models.ts` prices at its family default, and subscription plans (Claude Max,
130
+ ChatGPT Pro) bill nothing per token while the ledger still reports the API-rate
131
+ equivalent. Read `costUsd` as "what this would cost at API rates".
@@ -20,7 +20,7 @@ Start a persistent coding session with full CLI flag support.
20
20
  | `allowedTools` | string[] | Tools to auto-approve |
21
21
  | `disallowedTools` | string[] | Tools to deny |
22
22
  | `maxTurns` | number | Max agent loop turns |
23
- | `maxBudgetUsd` | number | Max API spend (USD) |
23
+ | `maxBudgetUsd` | number | Max API spend (USD). Enforced by the runtime on every engine: once the session's cumulative cost reaches the cap, further sends are refused before the engine is spawned. See [observability.md](observability.md) for the accuracy caveat on engines that estimate token counts. |
24
24
  | `systemPrompt` | string | Replace system prompt |
25
25
  | `appendSystemPrompt` | string | Append to system prompt |
26
26
  | `agents` | object | Custom sub-agents JSON |