@enderfga/claw-orchestrator 4.13.1 → 4.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -4
- package/assets/banner.jpg +0 -0
- package/dist/bin/cli.js +29 -0
- package/dist/bin/cli.js.map +1 -1
- package/dist/src/autoloop/dispatcher.js +3 -0
- package/dist/src/autoloop/dispatcher.js.map +1 -1
- package/dist/src/base-oneshot-session.d.ts +8 -0
- package/dist/src/base-oneshot-session.js +15 -0
- package/dist/src/base-oneshot-session.js.map +1 -1
- package/dist/src/budget.d.ts +36 -0
- package/dist/src/budget.js +54 -0
- package/dist/src/budget.js.map +1 -0
- package/dist/src/council.js +2 -1
- package/dist/src/council.js.map +1 -1
- package/dist/src/dashboard/index.html +24 -0
- package/dist/src/embedded-server.js +16 -0
- package/dist/src/embedded-server.js.map +1 -1
- package/dist/src/fanout.js +2 -0
- package/dist/src/fanout.js.map +1 -1
- package/dist/src/openai-compat.d.ts +1 -0
- package/dist/src/openai-compat.js +17 -1
- package/dist/src/openai-compat.js.map +1 -1
- package/dist/src/persistent-agy-session.js +1 -0
- package/dist/src/persistent-agy-session.js.map +1 -1
- package/dist/src/persistent-cursor-session.js +1 -0
- package/dist/src/persistent-cursor-session.js.map +1 -1
- package/dist/src/persistent-custom-session.js +7 -0
- package/dist/src/persistent-custom-session.js.map +1 -1
- package/dist/src/persistent-gemini-session.js +1 -0
- package/dist/src/persistent-gemini-session.js.map +1 -1
- package/dist/src/persistent-opencode-session.js +1 -0
- package/dist/src/persistent-opencode-session.js.map +1 -1
- package/dist/src/persistent-session.js +10 -0
- package/dist/src/persistent-session.js.map +1 -1
- package/dist/src/run-ledger.d.ts +95 -0
- package/dist/src/run-ledger.js +216 -0
- package/dist/src/run-ledger.js.map +1 -0
- package/dist/src/session-manager.d.ts +27 -0
- package/dist/src/session-manager.js +153 -16
- package/dist/src/session-manager.js.map +1 -1
- package/dist/src/types.d.ts +17 -0
- package/openclaw.plugin.json +28 -4
- package/package.json +3 -2
- package/skills/SKILL.md +18 -2
- package/skills/references/cli.md +13 -0
- package/skills/references/observability.md +131 -0
- package/skills/references/tools.md +1 -1
package/dist/src/types.d.ts
CHANGED
|
@@ -293,6 +293,12 @@ export interface SessionStats {
|
|
|
293
293
|
cursorChatId?: string;
|
|
294
294
|
/** OpenCode session ID captured from the run's JSON output. Reused via `--session` for multi-turn context. */
|
|
295
295
|
opencodeSessionId?: string;
|
|
296
|
+
/**
|
|
297
|
+
* True when the most recent turn's token counts came from estimateTokens()
|
|
298
|
+
* because the engine reported no usage. Cost derived from those counts is an
|
|
299
|
+
* estimate; the run ledger and budget docs surface it as such.
|
|
300
|
+
*/
|
|
301
|
+
tokensEstimated?: boolean;
|
|
296
302
|
}
|
|
297
303
|
export interface HookConfig {
|
|
298
304
|
onToolError?: string;
|
|
@@ -319,6 +325,11 @@ export interface SendOptions {
|
|
|
319
325
|
stream?: boolean;
|
|
320
326
|
onChunk?: (chunk: string) => void;
|
|
321
327
|
onEvent?: (event: StreamEvent) => void;
|
|
328
|
+
/**
|
|
329
|
+
* council id / fanout id / autoloop run id. Stamped onto the run-ledger row
|
|
330
|
+
* so a multi-agent run can be reassembled from the ledger afterwards.
|
|
331
|
+
*/
|
|
332
|
+
parentRunId?: string;
|
|
322
333
|
}
|
|
323
334
|
export interface StreamEvent {
|
|
324
335
|
type: string;
|
|
@@ -349,6 +360,12 @@ export interface SessionInfo {
|
|
|
349
360
|
model?: string;
|
|
350
361
|
paused: boolean;
|
|
351
362
|
stats: SessionStats;
|
|
363
|
+
/** Cumulative USD spent by this session, as reported by the engine's cost model. */
|
|
364
|
+
costUsd?: number;
|
|
365
|
+
/** The session's spend cap, when one was configured via `maxBudgetUsd`. */
|
|
366
|
+
budgetUsd?: number;
|
|
367
|
+
/** True once `costUsd` reached `budgetUsd` — further turns are refused. */
|
|
368
|
+
budgetExhausted?: boolean;
|
|
352
369
|
}
|
|
353
370
|
export interface SendResult {
|
|
354
371
|
output: string;
|
package/openclaw.plugin.json
CHANGED
|
@@ -17,12 +17,26 @@
|
|
|
17
17
|
},
|
|
18
18
|
"defaultPermissionMode": {
|
|
19
19
|
"type": "string",
|
|
20
|
-
"enum": [
|
|
20
|
+
"enum": [
|
|
21
|
+
"acceptEdits",
|
|
22
|
+
"bypassPermissions",
|
|
23
|
+
"default",
|
|
24
|
+
"delegate",
|
|
25
|
+
"dontAsk",
|
|
26
|
+
"plan",
|
|
27
|
+
"auto"
|
|
28
|
+
],
|
|
21
29
|
"default": "acceptEdits"
|
|
22
30
|
},
|
|
23
31
|
"defaultEffort": {
|
|
24
32
|
"type": "string",
|
|
25
|
-
"enum": [
|
|
33
|
+
"enum": [
|
|
34
|
+
"low",
|
|
35
|
+
"medium",
|
|
36
|
+
"high",
|
|
37
|
+
"max",
|
|
38
|
+
"auto"
|
|
39
|
+
],
|
|
26
40
|
"default": "auto"
|
|
27
41
|
},
|
|
28
42
|
"maxConcurrentSessions": {
|
|
@@ -60,7 +74,9 @@
|
|
|
60
74
|
},
|
|
61
75
|
"capabilities": {
|
|
62
76
|
"childProcess": true,
|
|
63
|
-
"networkAccess": [
|
|
77
|
+
"networkAccess": [
|
|
78
|
+
"127.0.0.1"
|
|
79
|
+
]
|
|
64
80
|
},
|
|
65
81
|
"contracts": {
|
|
66
82
|
"tools": [
|
|
@@ -106,6 +122,12 @@
|
|
|
106
122
|
"council_review",
|
|
107
123
|
"council_accept",
|
|
108
124
|
"council_reject",
|
|
125
|
+
"autoloop_start",
|
|
126
|
+
"autoloop_chat",
|
|
127
|
+
"autoloop_status",
|
|
128
|
+
"autoloop_list",
|
|
129
|
+
"autoloop_reset_agent",
|
|
130
|
+
"autoloop_stop",
|
|
109
131
|
"session_send_to",
|
|
110
132
|
"session_inbox",
|
|
111
133
|
"session_deliver_inbox",
|
|
@@ -129,5 +151,7 @@
|
|
|
129
151
|
"ultraapp_delete"
|
|
130
152
|
]
|
|
131
153
|
},
|
|
132
|
-
"skills": [
|
|
154
|
+
"skills": [
|
|
155
|
+
"skills/SKILL.md"
|
|
156
|
+
]
|
|
133
157
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@enderfga/claw-orchestrator",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.14.0",
|
|
4
4
|
"description": "Claw Orchestrator — run Claude Code, Codex, Gemini, Cursor Agent, OpenCode and custom coding CLIs as one unified runtime. Drop into Hermes Agent, Claude Desktop, Cursor, Cline, Continue, Zed, Windsurf, Goose or any Model Context Protocol (MCP) host, install as an OpenClaw plugin, or run standalone. Persistent sessions, multi-agent council, ultraplan, ultrareview, autoloop, tool orchestration.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./dist/src/index.js",
|
|
@@ -83,7 +83,8 @@
|
|
|
83
83
|
"jetbrains",
|
|
84
84
|
"neovim",
|
|
85
85
|
"antigravity",
|
|
86
|
-
"coding-agent-protocol"
|
|
86
|
+
"coding-agent-protocol",
|
|
87
|
+
"deepseek-harness"
|
|
87
88
|
],
|
|
88
89
|
"author": "enderfga",
|
|
89
90
|
"repository": {
|
package/skills/SKILL.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: claw-orchestrator
|
|
3
|
-
description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Cursor, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's
|
|
3
|
+
description: Manage persistent coding sessions across Claude Code, Codex, Antigravity (agy), Cursor, and OpenCode engines. Use when orchestrating multi-engine coding agents, starting/sending/stopping sessions, running multi-agent council collaborations, cross-session messaging, ultraplan deep planning, ultrareview parallel code review, autoloop autonomous workspace iteration, ultraapp building deployable web apps from a structured Q&A interview, switching models/tools at runtime, exposing the orchestrator's 69 tools as an MCP server to Hermes Agent / Claude Desktop / Cursor / Cline / Continue / Zed / Windsurf / Goose, or running as an Agent Client Protocol (ACP) agent that Zed / JetBrains / Neovim / Emacs / VS Code / dsh can drive directly. Triggers on "start a session", "send to session", "run council", "ultraplan", "ultrareview", "autoloop", "ultraapp", "Forge tab", "build a web app", "one-click app", "AppSpec", "autonomous iteration", "iterate until goal", "deep paper review", "auto research", "switch model", "multi-agent", "coding session", "session inbox", "cursor agent", "opencode", "mcp server", "clawo-mcp", "hermes mcp", "model context protocol", "ultracode", "dynamic workflow", "fanout", "fan-out", "best-of-N", "steer turn", "interrupt turn", "fork thread", "rollback turns", "acp", "agent client protocol", "clawo acp", "zed agent", "jetbrains agent", "external agent", "dsh subagent", "deepseek harness", "clawo runs", "run ledger", "how much did it cost", "token usage", "spend cap", "budget limit", "maxBudgetUsd".
|
|
4
4
|
metadata:
|
|
5
5
|
{
|
|
6
6
|
"openclaw":
|
|
@@ -36,7 +36,7 @@ metadata:
|
|
|
36
36
|
|
|
37
37
|
# Claw Orchestrator Skill
|
|
38
38
|
|
|
39
|
-
Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, and custom CLIs into headless agentic engines with
|
|
39
|
+
Claw Orchestrator — persistent multi-engine coding session manager for claw-style agent systems. Runs as a standalone CLI/server, with first-class OpenClaw plugin support. Wraps Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, and custom CLIs into headless agentic engines with 69 tools.
|
|
40
40
|
|
|
41
41
|
## Engine Quick Reference
|
|
42
42
|
|
|
@@ -208,6 +208,22 @@ engine mid-session.
|
|
|
208
208
|
|
|
209
209
|
For setup, the dsh YAML block, and the cancellation/permission limits: see [references/acp.md](references/acp.md)
|
|
210
210
|
|
|
211
|
+
## Cost & spend caps
|
|
212
|
+
|
|
213
|
+
Every turn on every engine is appended to a durable ledger at
|
|
214
|
+
`~/.claw-orchestrator/runs/YYYY-MM-DD.jsonl` — engine, model, per-turn tokens, cost, duration,
|
|
215
|
+
and the council / fanout / autoloop it belonged to. Query it with `clawo runs [--since 24h]
|
|
216
|
+
[--engine X] [--parent <run id>] [--json]`, `GET /runs`, or `manager.getRunLedger()`; it
|
|
217
|
+
survives restarts, so it answers "what did we run today and what did it cost" after the
|
|
218
|
+
sessions are gone.
|
|
219
|
+
|
|
220
|
+
`maxBudgetUsd` on a session (or per council / fanout agent) is enforced by the runtime, so it
|
|
221
|
+
holds on Codex, Cursor, agy, OpenCode and custom engines too — not just Claude Code. Once
|
|
222
|
+
cumulative spend reaches the cap, further sends are refused before the engine is spawned.
|
|
223
|
+
|
|
224
|
+
For the row schema, the query surfaces, and which engines report real token usage versus
|
|
225
|
+
estimating it: see [references/observability.md](references/observability.md)
|
|
226
|
+
|
|
211
227
|
## Authentication Prerequisites
|
|
212
228
|
|
|
213
229
|
Each engine requires its own auth before use:
|
package/skills/references/cli.md
CHANGED
|
@@ -134,6 +134,19 @@ clawo session-grep <name> <pattern> [-n, --limit <n>]
|
|
|
134
134
|
clawo session-compact <name> [--summary <text>]
|
|
135
135
|
```
|
|
136
136
|
|
|
137
|
+
## Run Ledger
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
clawo runs [--since <window>] [-n, --limit <n>] [--session <name>] [--engine <engine>] [--parent <id>] [--json]
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Show the durable per-turn record kept at `~/.claw-orchestrator/runs/`. Unlike
|
|
144
|
+
`session-status`, this survives restarts and covers sessions this process never
|
|
145
|
+
owned. `--since` takes `30m` / `24h` / `7d` / `2w` or an ISO timestamp (default
|
|
146
|
+
`24h`); `--parent` filters to one council / fanout / autoloop run. Costs marked
|
|
147
|
+
with a trailing `~` came from estimated token counts. See
|
|
148
|
+
[observability.md](observability.md).
|
|
149
|
+
|
|
137
150
|
## Agent Management
|
|
138
151
|
|
|
139
152
|
```bash
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# Observability — Run Ledger & Spend Caps
|
|
2
|
+
|
|
3
|
+
Two related surfaces: a durable record of every turn this runtime executes, and a
|
|
4
|
+
spend cap that is enforced by the runtime rather than by whichever CLI happens to
|
|
5
|
+
support a budget flag.
|
|
6
|
+
|
|
7
|
+
## Why
|
|
8
|
+
|
|
9
|
+
`getStats()` / `getCost()` describe a **live** session. They live in memory, and
|
|
10
|
+
per-session history is capped and evicted, so a restart erased everything except
|
|
11
|
+
the resume-id registry — there was no way to answer "what did we run today, on
|
|
12
|
+
which engine, for how much". The run ledger is that record.
|
|
13
|
+
|
|
14
|
+
The same gap made `maxBudgetUsd` a promise the runtime did not keep: it was only
|
|
15
|
+
ever translated into Claude Code's `--max-budget-usd` flag, so a council of Codex
|
|
16
|
+
agents ran with no cap at all. The cap is now applied in `SessionManager`, which
|
|
17
|
+
every engine passes through.
|
|
18
|
+
|
|
19
|
+
## The ledger
|
|
20
|
+
|
|
21
|
+
- Location: `~/.claw-orchestrator/runs/YYYY-MM-DD.jsonl` (override with
|
|
22
|
+
`CLAWO_RUNS_DIR`).
|
|
23
|
+
- One JSON object per line, one line per completed turn — **successful or not**.
|
|
24
|
+
- Shards are one file per UTC day. Queries always filter on the row timestamp, so
|
|
25
|
+
the shard boundary is a storage detail; `--since 24h` spans midnight correctly.
|
|
26
|
+
- Writes are best-effort: a ledger failure is logged at `warn` and swallowed. It
|
|
27
|
+
can never break the turn it is describing.
|
|
28
|
+
- Nothing prunes old shards. A row is roughly 250 bytes; delete shards yourself if
|
|
29
|
+
you want the history gone.
|
|
30
|
+
|
|
31
|
+
### Row schema
|
|
32
|
+
|
|
33
|
+
| Field | Meaning |
|
|
34
|
+
|---|---|
|
|
35
|
+
| `ts` | ISO timestamp of turn completion |
|
|
36
|
+
| `session` | SessionManager session name |
|
|
37
|
+
| `engine` | `claude` / `codex` / `codex-app` / `cursor` / `opencode` / `agy` / `custom` |
|
|
38
|
+
| `model` | Configured model, or the engine's own reported model when none was set |
|
|
39
|
+
| `cwd` | Working directory the turn ran in |
|
|
40
|
+
| `turn` | 1-based turn index within the session |
|
|
41
|
+
| `tokensIn` / `tokensOut` / `cachedTokens` | **Per-turn deltas**, not session totals |
|
|
42
|
+
| `costUsd` | Per-turn delta in USD |
|
|
43
|
+
| `tokensEstimated` | `true` when the counts came from `estimateTokens()` (see below) |
|
|
44
|
+
| `durationMs` | Wall-clock for the turn |
|
|
45
|
+
| `toolCalls` / `toolErrors` | Per-turn deltas |
|
|
46
|
+
| `ok` | `false` for a turn that threw or reported `is_error` |
|
|
47
|
+
| `error` | Failure text, truncated to 500 chars |
|
|
48
|
+
| `parent` | council id / fanout id / autoloop run id, when the turn belongs to one |
|
|
49
|
+
|
|
50
|
+
Deltas rather than totals means summing a query window gives that window's spend
|
|
51
|
+
without double-counting.
|
|
52
|
+
|
|
53
|
+
Every path funnels through `SessionManager.sendMessage`, so council, fanout,
|
|
54
|
+
autoloop, ACP, the OpenAI-compatible bridge, the MCP server and the CLI are all
|
|
55
|
+
covered by the same hook. `parent` is what lets you reassemble a multi-agent run:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
clawo runs --parent council-1a2b3c4d
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
### Reading it
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
clawo runs # last 24h, table
|
|
65
|
+
clawo runs --since 7d --engine codex # one engine, one week
|
|
66
|
+
clawo runs --session my-session --json # raw rows + summary
|
|
67
|
+
clawo runs --parent fanout-b5f7c886 # every agent turn of one fan-out
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Over HTTP (GET query string or POST JSON body):
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
curl "http://127.0.0.1:18796/runs?since=24h&limit=200" -H "Authorization: Bearer $TOKEN"
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Returns `{ ok, rows, summary }`, where `summary` carries `rows`, `costUsd`,
|
|
77
|
+
`tokensIn`, `tokensOut`, `estimatedRows` and a per-engine breakdown. The dashboard
|
|
78
|
+
header shows the 24-hour figure from the same endpoint.
|
|
79
|
+
|
|
80
|
+
Programmatically: `manager.getRunLedger({ since, session, engine, parent, limit })`.
|
|
81
|
+
|
|
82
|
+
## Spend caps
|
|
83
|
+
|
|
84
|
+
Set `maxBudgetUsd` on a session (or on a council / fanout, which applies it per
|
|
85
|
+
agent). Before each turn the runtime compares the session's cumulative
|
|
86
|
+
`getCost().totalUsd` against the cap and refuses to send when it has been reached:
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
Budget exceeded for session "my-session" (codex): spent $1.2400 of $1.0000 cap.
|
|
90
|
+
Raise maxBudgetUsd or start a new session to continue.
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
The refusal is a typed `BudgetExceededError` carrying `session`, `engine`,
|
|
94
|
+
`spentUsd` and `capUsd`, and it happens **before** the engine is spawned.
|
|
95
|
+
|
|
96
|
+
Notes:
|
|
97
|
+
|
|
98
|
+
- The check is "has the cap been reached", not "would this turn exceed it" — a
|
|
99
|
+
turn's cost is unknown until it finishes, so the last allowed turn can overshoot.
|
|
100
|
+
Size the cap accordingly.
|
|
101
|
+
- A cap of `0` or a negative number means *unset*, not *refuse everything*.
|
|
102
|
+
- Claude Code still receives `--max-budget-usd` as well: an in-CLI stop happens
|
|
103
|
+
earlier and therefore costs less than an after-the-fact refusal.
|
|
104
|
+
- `session_list` / `GET /session/list` expose `costUsd`, `budgetUsd` and
|
|
105
|
+
`budgetExhausted` so a stalled session shows *why* it stopped taking turns.
|
|
106
|
+
|
|
107
|
+
## Accuracy: which engines report real usage
|
|
108
|
+
|
|
109
|
+
`costUsd` is derived from token counts times the model's price in `models.ts`.
|
|
110
|
+
Where the engine reports usage, those counts are the engine's own. Where it does
|
|
111
|
+
not, the wrapper falls back to `estimateTokens()` (characters ÷ 4) and the row is
|
|
112
|
+
flagged `tokensEstimated: true`; the CLI marks those costs with a trailing `~`.
|
|
113
|
+
|
|
114
|
+
| Engine | Token counts |
|
|
115
|
+
|---|---|
|
|
116
|
+
| `claude` | Engine-reported |
|
|
117
|
+
| `codex` | Engine-reported |
|
|
118
|
+
| `codex-app` | Engine-reported |
|
|
119
|
+
| `cursor` | Engine-reported when the stream carries `usage`, else estimated |
|
|
120
|
+
| `opencode` | Engine-reported when the run JSON carries `tokens`, else estimated |
|
|
121
|
+
| `agy` | Engine-reported when the result event carries usage, else estimated |
|
|
122
|
+
| `custom` | Depends on the CLI; estimated when it emits no usage |
|
|
123
|
+
|
|
124
|
+
So on an estimating engine the cap is best-effort. It will stop a runaway session;
|
|
125
|
+
it is not an accounting guarantee, and it is not a substitute for the spend limits
|
|
126
|
+
your provider offers.
|
|
127
|
+
|
|
128
|
+
Cost figures are also only as good as the pricing table: a model missing from
|
|
129
|
+
`models.ts` prices at its family default, and subscription plans (Claude Max,
|
|
130
|
+
ChatGPT Pro) bill nothing per token while the ledger still reports the API-rate
|
|
131
|
+
equivalent. Read `costUsd` as "what this would cost at API rates".
|
|
@@ -20,7 +20,7 @@ Start a persistent coding session with full CLI flag support.
|
|
|
20
20
|
| `allowedTools` | string[] | Tools to auto-approve |
|
|
21
21
|
| `disallowedTools` | string[] | Tools to deny |
|
|
22
22
|
| `maxTurns` | number | Max agent loop turns |
|
|
23
|
-
| `maxBudgetUsd` | number | Max API spend (USD) |
|
|
23
|
+
| `maxBudgetUsd` | number | Max API spend (USD). Enforced by the runtime on every engine: once the session's cumulative cost reaches the cap, further sends are refused before the engine is spawned. See [observability.md](observability.md) for the accuracy caveat on engines that estimate token counts. |
|
|
24
24
|
| `systemPrompt` | string | Replace system prompt |
|
|
25
25
|
| `appendSystemPrompt` | string | Append to system prompt |
|
|
26
26
|
| `agents` | object | Custom sub-agents JSON |
|