agent-dispatch 0.10.0__tar.gz → 0.11.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/AGENTS.md +11 -5
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/CHANGELOG.md +42 -1
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/PKG-INFO +5 -4
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/README.md +4 -3
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/agents.example.yaml +2 -1
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/pyproject.toml +1 -1
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/__init__.py +1 -1
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/cli.py +6 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/jobs.py +16 -4
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/models.py +1 -1
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/runner.py +147 -58
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_cli.py +18 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_jobs.py +58 -9
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_runner.py +500 -151
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/.github/dependabot.yml +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/.github/workflows/ci.yml +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/.github/workflows/publish.yml +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/.gitignore +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/LICENSE +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/SECURITY.md +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/assets/mascot.png +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/cache.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/config.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/src/agent_dispatch/server.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/__init__.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/conftest.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_cache.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_config.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_models.py +0 -0
- {agent_dispatch-0.10.0 → agent_dispatch-0.11.0}/tests/test_server.py +0 -0
|
@@ -28,7 +28,7 @@ pip install -e ".[dev]"
|
|
|
28
28
|
|
|
29
29
|
```bash
|
|
30
30
|
ruff check src/ tests/
|
|
31
|
-
python3 -m pytest tests/ -v #
|
|
31
|
+
python3 -m pytest tests/ -v # 495 tests, ~2s — all subprocess calls are mocked
|
|
32
32
|
```
|
|
33
33
|
|
|
34
34
|
Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.which` + `subprocess.run`/`Popen`; server tests mock `_get_config` + `runner.dispatch`.
|
|
@@ -36,14 +36,20 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
36
36
|
## Non-obvious invariants (violating these breaks real behavior)
|
|
37
37
|
|
|
38
38
|
- `allowed_tools` / `disallowed_tools` are **tri-state**: `None` = inherit settings defaults, `[]` = explicitly no tools, `[...]` = exactly these. Check with `is not None`, never `or` — `[]` is falsy but semantically distinct.
|
|
39
|
-
- `denied_tools` non-empty
|
|
39
|
+
- Error-type precedence on an `is_error` payload (`_build_error_result`): the CLI's own budget stop wins, then `denied_tools` non-empty ⇒ `error_type="permission"` regardless of the error text, then text classification.
|
|
40
40
|
- **Groups**: a group's `shared_context` is folded into the `context` *string* before the cache/runner calls (`_merge_group_context` in server.py) — runner.py and cache.py are untouched, the cache key disambiguates groups for free, and `group=""` is byte-identical to a plain dispatch. Membership is validated up front (`_validate_group_member`, separate from the pure merge so `dispatch_parallel`'s all-or-nothing pre-check holds). `DispatchConfig` validates only group *keys*, never member existence — a hard cross-ref check would brick config load when a shared gateway agent is removed; dangling refs are flagged (`unknown:true`) at read time instead.
|
|
41
41
|
- On failure, callers read `DispatchResult.error` + `error_type` — `result` holds the raw agent output even on errors.
|
|
42
42
|
- `--session-id` and `--resume` conflict — never pass both to `claude`.
|
|
43
43
|
- Valid permission modes: `default`, `plan`, `bypassPermissions` (`models.py: KNOWN_PERMISSION_MODES`).
|
|
44
|
-
- `JobStore.finish`/`fail` refuse already-terminal jobs (returns `None`) — this closes the race with force-cancel; never "fix" it by overwriting.
|
|
44
|
+
- `JobStore.finish`/`fail` refuse already-terminal jobs (returns `None`) — this closes the race with force-cancel; never "fix" it by overwriting. `mark_running` likewise refuses any job that isn't `pending`, so a stale or duplicate worker can't resurrect a finished one.
|
|
45
|
+
- "Is this group member missing?" has exactly one implementation: `DispatchConfig.unknown_group_members()`. Any new surface that lists or validates membership calls it instead of re-deriving the check.
|
|
45
46
|
- Cancelling a *running* job requires the in-memory `_running_procs` registry (server.py) — the job is marked `cancelled` **before** the subprocess is killed. Don't persist PIDs to disk (PID reuse after restart could kill an unrelated process).
|
|
46
|
-
- `max_budget_usd` is **
|
|
47
|
+
- `max_budget_usd` is enforced **by the claude CLI** (`_build_command` passes `--max-budget-usd`): a run stopped at the cap comes back `is_error` with no `result` text, and `_build_error_result` turns it into `error_type="budget"` + `budget_exceeded=True` + a resumable `session_id`. `_apply_budget` is the *secondary*, post-hoc signal for an overshoot that didn't stop the run; it never fails a dispatch.
|
|
48
|
+
- A CLI error payload can have no `result` field at all — the reason lives in `errors` / `subtype`. Read it via `_cli_error_details`, never assume `result` is populated on failure.
|
|
49
|
+
|
|
50
|
+
## Deliberately not built
|
|
51
|
+
|
|
52
|
+
These were considered — some fully implemented — and cut on purpose: an agent router / auto-dispatch (`recommend_agent` / `dispatch_auto`, removed before 0.8.0 — a keyword scorer adds little over the calling LLM at a handful of agents, and auto-dispatch can spend money or mutate a repo on a guess); groups as an execution engine (they are a descriptive layer — no routing, no per-group settings); an agent-dispatch-side budget ledger across dispatches (the CLI's own `--max-budget-usd` covers a single run; anything cumulative would need state we deliberately don't keep). Please open an issue with the use case before adding any of them.
|
|
47
53
|
|
|
48
54
|
## Conventions
|
|
49
55
|
|
|
@@ -55,4 +61,4 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
|
|
|
55
61
|
|
|
56
62
|
## More detail
|
|
57
63
|
|
|
58
|
-
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`,
|
|
64
|
+
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 495 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
|
|
@@ -7,6 +7,43 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.11.0] - 2026-07-27
|
|
11
|
+
|
|
12
|
+
The spend cap was already real — now the result says so.
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
- **`error_type: "budget"`.** The `claude` CLI enforces the `--max-budget-usd`
|
|
16
|
+
value agent-dispatch has always passed: when the cap is reached it ends the
|
|
17
|
+
session and reports the failure with no `result` text. That payload used to
|
|
18
|
+
surface as the useless `"reported an error with no details"` fallback with
|
|
19
|
+
`error_type: "cli_error"`. It is now classified as `budget`, carries the
|
|
20
|
+
CLI's own reason (`Reached maximum budget ($X)`), sets
|
|
21
|
+
`budget_exceeded: true`, and spells out the recovery — raise the cap, pick a
|
|
22
|
+
cheaper model, or resume the partial session via the returned `session_id`.
|
|
23
|
+
`agent-dispatch test` prints a matching diagnosis.
|
|
24
|
+
|
|
25
|
+
### Fixed
|
|
26
|
+
- **Error details are no longer dropped.** CLI-level failures (budget
|
|
27
|
+
exhausted, max turns, execution errors) put their reason in `errors` /
|
|
28
|
+
`subtype` rather than `result`. Both `dispatch` and `dispatch_stream` now
|
|
29
|
+
read those fields when `result` is empty, so the error text names the actual
|
|
30
|
+
problem. Values are capped (5 entries × 300 chars) like every other field
|
|
31
|
+
read back from the subprocess.
|
|
32
|
+
- **Job files can no longer be corrupted by a concurrent writer.** `JobStore`
|
|
33
|
+
wrote through a fixed `<id>.tmp` path; the in-process lock does not cover
|
|
34
|
+
the CLI (`cancel` / `gc`) and the MCP server writing the same job at the
|
|
35
|
+
same time, and the rename could publish interleaved JSON. Each write now
|
|
36
|
+
uses a unique temp name and cleans it up if the write fails.
|
|
37
|
+
|
|
38
|
+
### Changed
|
|
39
|
+
- The `is_error` handling in `dispatch` and `dispatch_stream` is now one
|
|
40
|
+
shared `_build_error_result()` with an explicit precedence: budget stop >
|
|
41
|
+
denied tools > text classification.
|
|
42
|
+
- Documentation corrected throughout: `max_budget_usd` is enforced per
|
|
43
|
+
dispatch by the CLI, with the `budget_exceeded` flag as the secondary
|
|
44
|
+
post-hoc signal for a run that overshot without being stopped. Previous
|
|
45
|
+
releases described it as post-hoc only.
|
|
46
|
+
|
|
10
47
|
## [0.10.0] - 2026-07-14
|
|
11
48
|
|
|
12
49
|
`doctor` learns to check groups; three correctness fixes found in review.
|
|
@@ -389,7 +426,11 @@ cache bounding, and stale-job recovery.
|
|
|
389
426
|
- Dependabot for `pip` + `github-actions`, GitHub Actions pinned to
|
|
390
427
|
commit SHAs for supply-chain integrity.
|
|
391
428
|
|
|
392
|
-
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.
|
|
429
|
+
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.11.0...HEAD
|
|
430
|
+
[0.11.0]: https://github.com/ginkida/agent-dispatch/compare/v0.10.0...v0.11.0
|
|
431
|
+
[0.10.0]: https://github.com/ginkida/agent-dispatch/compare/v0.9.0...v0.10.0
|
|
432
|
+
[0.9.0]: https://github.com/ginkida/agent-dispatch/compare/v0.8.0...v0.9.0
|
|
433
|
+
[0.8.0]: https://github.com/ginkida/agent-dispatch/compare/v0.6.0...v0.8.0
|
|
393
434
|
[0.6.0]: https://github.com/ginkida/agent-dispatch/compare/v0.5.0...v0.6.0
|
|
394
435
|
[0.5.0]: https://github.com/ginkida/agent-dispatch/compare/v0.4.0...v0.5.0
|
|
395
436
|
[0.4.0]: https://github.com/ginkida/agent-dispatch/compare/v0.3.0...v0.4.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: agent-dispatch
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.11.0
|
|
4
4
|
Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
|
|
5
5
|
Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
|
|
6
6
|
Project-URL: Repository, https://github.com/ginkida/agent-dispatch
|
|
@@ -233,7 +233,7 @@ dispatch(
|
|
|
233
233
|
}
|
|
234
234
|
```
|
|
235
235
|
|
|
236
|
-
**`error_type` values:** `permission` (tool/action denied), `timeout`, `recursion` (dispatch depth exceeded), `not_found` (missing directory or CLI), `cli_error` (other failures). Permission errors include an actionable hint.
|
|
236
|
+
**`error_type` values:** `permission` (tool/action denied), `timeout`, `recursion` (dispatch depth exceeded), `not_found` (missing directory or CLI), `budget` (the `claude` CLI stopped the session at `max_budget_usd`), `cli_error` (other failures). Permission and budget errors include an actionable hint.
|
|
237
237
|
|
|
238
238
|
**Resumable timeouts:** every fresh dispatch pre-assigns a session UUID (`--session-id`), so a timed-out dispatch still returns a `session_id` — the partial transcript survives the kill. The timeout error spells out the recovery: resume with `dispatch_session(agent, "Continue where you left off", session_id=...)`, retry with a bigger `timeout_seconds`, or use `dispatch_async`.
|
|
239
239
|
|
|
@@ -503,13 +503,14 @@ Failures are deterministic: check `success`, then branch on `error_type`.
|
|
|
503
503
|
| `timeout` | Process killed at the timeout | Resume the partial work: `dispatch_session(agent, "Continue where you left off", session_id=<from the error text>)`. Or retry with a bigger `timeout_seconds=`, or use `dispatch_async`. |
|
|
504
504
|
| `not_found` | Agent directory or `claude` CLI missing | `list_agents()` → check `healthy`. Re-add the agent with an existing path, or run `agent-dispatch doctor` to find what's missing. |
|
|
505
505
|
| `recursion` | Dispatch nesting exceeded `max_dispatch_depth` (default 3) | Don't dispatch from dispatched agents; if the nesting is intentional, raise `max_dispatch_depth` in settings. |
|
|
506
|
+
| `budget` | The `claude` CLI ended the session at the `max_budget_usd` spend cap — the answer is incomplete | Raise the cap (`update_agent(name, max_budget_usd=2.0)`), switch to a cheaper `model`, or split the task. The partial session is resumable: `dispatch_session(agent, "Continue where you left off", session_id=<from the result>)`. |
|
|
506
507
|
| `cli_error` | Anything else from the `claude` subprocess | Read the `error` text; run `agent-dispatch doctor` for environment issues; retry once if transient. |
|
|
507
508
|
|
|
508
509
|
Three soft signals that arrive with `success: true`:
|
|
509
510
|
|
|
510
511
|
- **`denied_tools` + `hint`** — the agent finished but some tool calls were blocked; the result may be incomplete. Grant access (see the `permission` row) and re-dispatch.
|
|
511
512
|
- **`parsed_result: null` with `response_format="json"`** — the reply wasn't valid JSON; the raw text is still in `result`. Caveat: an agent that *can't* comply returns `{"error": "<reason>"}` — which parses successfully — so also check `parsed_result` for an `"error"` key.
|
|
512
|
-
- **`budget_exceeded: true`** — `cost_usd`
|
|
513
|
+
- **`budget_exceeded: true`** — `cost_usd` came in over the agent's `max_budget_usd` (or the settings default) without the CLI stopping the run (the final turn can overshoot the cap). The dispatch is not failed — the money is already spent — but a runaway agent is now visible. Tighten the task, pick a cheaper model, or raise the budget. A run the CLI *did* stop fails with `error_type: "budget"` instead.
|
|
513
514
|
|
|
514
515
|
Tool-level errors (unknown agent, malformed input) return a plain envelope instead of a `DispatchResult`:
|
|
515
516
|
|
|
@@ -624,7 +625,7 @@ agent-dispatch MCP server
|
|
|
624
625
|
- **Argument-injection guard** — structured CLI fields (`session_id`, `model`, `permission_mode`, tool names) that start with `-` are rejected so they can't smuggle extra `claude` flags.
|
|
625
626
|
- **Path-traversal guard** — caller-supplied `job_id`/`ref` values are validated as 32-char hex before any filesystem access.
|
|
626
627
|
- **Owner-only state** — job files (`0o600`) and `agents.yaml` (`0o600`) are written for the owner only; their directories are `0o700`.
|
|
627
|
-
- **Cost
|
|
628
|
+
- **Cost control** — `max_budget_usd` per agent or globally is passed to the `claude` CLI as `--max-budget-usd`, so a runaway dispatch is stopped at the cap and comes back as `error_type: "budget"` with a resumable `session_id`. An overshoot that lands over budget without stopping is flagged post-hoc with `budget_exceeded: true` + a hint.
|
|
628
629
|
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
|
|
629
630
|
- **Timeout** — per-agent or global (default: 300s). Orphaned processes are cleaned up.
|
|
630
631
|
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
@@ -203,7 +203,7 @@ dispatch(
|
|
|
203
203
|
}
|
|
204
204
|
```
|
|
205
205
|
|
|
206
|
-
**`error_type` values:** `permission` (tool/action denied), `timeout`, `recursion` (dispatch depth exceeded), `not_found` (missing directory or CLI), `cli_error` (other failures). Permission errors include an actionable hint.
|
|
206
|
+
**`error_type` values:** `permission` (tool/action denied), `timeout`, `recursion` (dispatch depth exceeded), `not_found` (missing directory or CLI), `budget` (the `claude` CLI stopped the session at `max_budget_usd`), `cli_error` (other failures). Permission and budget errors include an actionable hint.
|
|
207
207
|
|
|
208
208
|
**Resumable timeouts:** every fresh dispatch pre-assigns a session UUID (`--session-id`), so a timed-out dispatch still returns a `session_id` — the partial transcript survives the kill. The timeout error spells out the recovery: resume with `dispatch_session(agent, "Continue where you left off", session_id=...)`, retry with a bigger `timeout_seconds`, or use `dispatch_async`.
|
|
209
209
|
|
|
@@ -473,13 +473,14 @@ Failures are deterministic: check `success`, then branch on `error_type`.
|
|
|
473
473
|
| `timeout` | Process killed at the timeout | Resume the partial work: `dispatch_session(agent, "Continue where you left off", session_id=<from the error text>)`. Or retry with a bigger `timeout_seconds=`, or use `dispatch_async`. |
|
|
474
474
|
| `not_found` | Agent directory or `claude` CLI missing | `list_agents()` → check `healthy`. Re-add the agent with an existing path, or run `agent-dispatch doctor` to find what's missing. |
|
|
475
475
|
| `recursion` | Dispatch nesting exceeded `max_dispatch_depth` (default 3) | Don't dispatch from dispatched agents; if the nesting is intentional, raise `max_dispatch_depth` in settings. |
|
|
476
|
+
| `budget` | The `claude` CLI ended the session at the `max_budget_usd` spend cap — the answer is incomplete | Raise the cap (`update_agent(name, max_budget_usd=2.0)`), switch to a cheaper `model`, or split the task. The partial session is resumable: `dispatch_session(agent, "Continue where you left off", session_id=<from the result>)`. |
|
|
476
477
|
| `cli_error` | Anything else from the `claude` subprocess | Read the `error` text; run `agent-dispatch doctor` for environment issues; retry once if transient. |
|
|
477
478
|
|
|
478
479
|
Three soft signals that arrive with `success: true`:
|
|
479
480
|
|
|
480
481
|
- **`denied_tools` + `hint`** — the agent finished but some tool calls were blocked; the result may be incomplete. Grant access (see the `permission` row) and re-dispatch.
|
|
481
482
|
- **`parsed_result: null` with `response_format="json"`** — the reply wasn't valid JSON; the raw text is still in `result`. Caveat: an agent that *can't* comply returns `{"error": "<reason>"}` — which parses successfully — so also check `parsed_result` for an `"error"` key.
|
|
482
|
-
- **`budget_exceeded: true`** — `cost_usd`
|
|
483
|
+
- **`budget_exceeded: true`** — `cost_usd` came in over the agent's `max_budget_usd` (or the settings default) without the CLI stopping the run (the final turn can overshoot the cap). The dispatch is not failed — the money is already spent — but a runaway agent is now visible. Tighten the task, pick a cheaper model, or raise the budget. A run the CLI *did* stop fails with `error_type: "budget"` instead.
|
|
483
484
|
|
|
484
485
|
Tool-level errors (unknown agent, malformed input) return a plain envelope instead of a `DispatchResult`:
|
|
485
486
|
|
|
@@ -594,7 +595,7 @@ agent-dispatch MCP server
|
|
|
594
595
|
- **Argument-injection guard** — structured CLI fields (`session_id`, `model`, `permission_mode`, tool names) that start with `-` are rejected so they can't smuggle extra `claude` flags.
|
|
595
596
|
- **Path-traversal guard** — caller-supplied `job_id`/`ref` values are validated as 32-char hex before any filesystem access.
|
|
596
597
|
- **Owner-only state** — job files (`0o600`) and `agents.yaml` (`0o600`) are written for the owner only; their directories are `0o700`.
|
|
597
|
-
- **Cost
|
|
598
|
+
- **Cost control** — `max_budget_usd` per agent or globally is passed to the `claude` CLI as `--max-budget-usd`, so a runaway dispatch is stopped at the cap and comes back as `error_type: "budget"` with a resumable `session_id`. An overshoot that lands over budget without stopping is flagged post-hoc with `budget_exceeded: true` + a hint.
|
|
598
599
|
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
|
|
599
600
|
- **Timeout** — per-agent or global (default: 300s). Orphaned processes are cleaned up.
|
|
600
601
|
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
@@ -93,7 +93,8 @@ settings:
|
|
|
93
93
|
default_timeout: 300
|
|
94
94
|
max_dispatch_depth: 3 # recursion protection: A -> B -> A
|
|
95
95
|
max_concurrency: 5 # max parallel claude -p processes
|
|
96
|
-
# default_max_budget_usd: 1.0 #
|
|
96
|
+
# default_max_budget_usd: 1.0 # spend cap per dispatch, passed to claude as --max-budget-usd
|
|
97
|
+
# (a run stopped at the cap fails with error_type: budget)
|
|
97
98
|
# default_permission_mode: bypassPermissions # inherited by agents without override
|
|
98
99
|
# default_allowed_tools: # inherited by agents without override
|
|
99
100
|
# - Bash
|
|
@@ -423,6 +423,12 @@ def test(name: str, task: str, stream: bool, timeout: int | None) -> None:
|
|
|
423
423
|
click.echo(click.style("Diagnosis: timeout", fg="yellow"))
|
|
424
424
|
click.echo(f" agent-dispatch test {name} --timeout 600 # one-off")
|
|
425
425
|
click.echo(f" agent-dispatch update {name} --timeout 600 # permanent")
|
|
426
|
+
elif result.error_type == "budget":
|
|
427
|
+
click.echo()
|
|
428
|
+
click.echo(click.style("Diagnosis: spend cap reached", fg="yellow"))
|
|
429
|
+
click.echo("The claude CLI stopped the session at --max-budget-usd.")
|
|
430
|
+
click.echo(f" agent-dispatch update {name} --max-budget-usd 2.0 # raise the cap")
|
|
431
|
+
click.echo(f" agent-dispatch update {name} --model haiku # cheaper model")
|
|
426
432
|
raise SystemExit(1)
|
|
427
433
|
|
|
428
434
|
|
|
@@ -99,10 +99,22 @@ class JobStore:
|
|
|
99
99
|
|
|
100
100
|
def _write(self, job: Job) -> None:
|
|
101
101
|
path = self._path(job.id)
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
102
|
+
# Unique temp name per write: `self._lock` only serializes writers inside
|
|
103
|
+
# one process, but the CLI (cancel/gc) and the MCP server touch the same
|
|
104
|
+
# files. A shared `<id>.tmp` could be written by both at once and the
|
|
105
|
+
# rename would publish interleaved, unparseable JSON.
|
|
106
|
+
tmp = path.with_name(f"{job.id}.{uuid.uuid4().hex}.tmp")
|
|
107
|
+
try:
|
|
108
|
+
tmp.write_text(job.model_dump_json(indent=2, exclude_none=True), encoding="utf-8")
|
|
109
|
+
_chmod_quiet(tmp, 0o600) # owner-only before it becomes visible
|
|
110
|
+
os.replace(tmp, path)
|
|
111
|
+
except OSError:
|
|
112
|
+
# Don't leave a half-written temp file behind on a failed write.
|
|
113
|
+
try:
|
|
114
|
+
tmp.unlink(missing_ok=True)
|
|
115
|
+
except OSError: # pragma: no cover - best effort
|
|
116
|
+
logger.debug("Failed to clean up temp file %s", tmp)
|
|
117
|
+
raise
|
|
106
118
|
|
|
107
119
|
def create(
|
|
108
120
|
self,
|
|
@@ -174,7 +174,7 @@ class DispatchResult(BaseModel):
|
|
|
174
174
|
duration_ms: int | None = None
|
|
175
175
|
num_turns: int | None = None
|
|
176
176
|
error: str | None = None
|
|
177
|
-
error_type: str | None = None # permission, timeout, recursion, not_found, cli_error
|
|
177
|
+
error_type: str | None = None # permission, timeout, recursion, not_found, budget, cli_error
|
|
178
178
|
# Set when response_format="json" was requested AND the agent's result
|
|
179
179
|
# parsed cleanly. None means: not requested, or requested but unparseable.
|
|
180
180
|
parsed_result: Any | None = None
|
|
@@ -138,16 +138,146 @@ def _denial_hint(agent_name: str, denied_tools: list[str]) -> str:
|
|
|
138
138
|
)
|
|
139
139
|
|
|
140
140
|
|
|
141
|
+
def _effective_budget(agent: AgentConfig, settings: Settings) -> float | None:
|
|
142
|
+
"""The spend cap for this dispatch: per-agent, else the settings default."""
|
|
143
|
+
return agent.max_budget_usd or settings.default_max_budget_usd
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# The claude CLI enforces `--max-budget-usd` itself: when the cap is reached it
|
|
147
|
+
# ends the session and emits a result payload with `is_error: true`, NO `result`
|
|
148
|
+
# field, `subtype: "error_max_budget_usd"`, `terminal_reason: "budget_exhausted"`
|
|
149
|
+
# and the human-readable reason in `errors`. Verified against claude 2.1.220.
|
|
150
|
+
_BUDGET_ERROR_SUBTYPE = "error_max_budget_usd"
|
|
151
|
+
_BUDGET_TERMINAL_REASON = "budget_exhausted"
|
|
152
|
+
|
|
153
|
+
# `errors` also comes from the untrusted subprocess — cap it like denied_tools.
|
|
154
|
+
_MAX_ERROR_DETAILS = 5
|
|
155
|
+
_MAX_ERROR_DETAIL_CHARS = 300
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _is_budget_error(data: dict) -> bool:
|
|
159
|
+
"""True when the CLI ended the session because the spend cap was reached."""
|
|
160
|
+
return (
|
|
161
|
+
data.get("subtype") == _BUDGET_ERROR_SUBTYPE
|
|
162
|
+
or data.get("terminal_reason") == _BUDGET_TERMINAL_REASON
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _cli_error_details(data: dict) -> str:
|
|
167
|
+
"""Best available failure text for an error payload that has no `result`.
|
|
168
|
+
|
|
169
|
+
CLI-level failures (budget exhausted, max turns, execution errors) carry no
|
|
170
|
+
`result` text — the reason lives in `errors` and `subtype`. Without this the
|
|
171
|
+
caller only ever saw "reported an error with no details".
|
|
172
|
+
"""
|
|
173
|
+
errors = data.get("errors")
|
|
174
|
+
if isinstance(errors, list):
|
|
175
|
+
details = [
|
|
176
|
+
text[:_MAX_ERROR_DETAIL_CHARS]
|
|
177
|
+
for text in (str(item).strip() for item in errors[:_MAX_ERROR_DETAILS])
|
|
178
|
+
if text
|
|
179
|
+
]
|
|
180
|
+
if details:
|
|
181
|
+
return "; ".join(details)
|
|
182
|
+
subtype = data.get("subtype")
|
|
183
|
+
if isinstance(subtype, str) and subtype.strip():
|
|
184
|
+
return f"claude CLI reported '{subtype.strip()[:_MAX_ERROR_DETAIL_CHARS]}'"
|
|
185
|
+
return ""
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _budget_error_hint(agent_name: str, budget: float | None, session_uuid: str | None) -> str:
|
|
189
|
+
"""Actionable hint for a dispatch the CLI stopped at the spend cap."""
|
|
190
|
+
cap = f"${budget:g}" if budget else "the configured cap"
|
|
191
|
+
msg = (
|
|
192
|
+
f"\n\nHint: the dispatch was stopped by the spend cap {cap} "
|
|
193
|
+
"(passed to claude as --max-budget-usd), so this answer is incomplete. "
|
|
194
|
+
f"Raise it (agent-dispatch update {agent_name} --max-budget-usd <amount>), "
|
|
195
|
+
"use a cheaper model, or split the task into smaller dispatches."
|
|
196
|
+
)
|
|
197
|
+
if session_uuid:
|
|
198
|
+
msg += (
|
|
199
|
+
f" Partial work may be resumable: dispatch_session(agent='{agent_name}', "
|
|
200
|
+
f"task='Continue where you left off', session_id='{session_uuid}')."
|
|
201
|
+
)
|
|
202
|
+
return msg
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _build_error_result(
|
|
206
|
+
agent_name: str,
|
|
207
|
+
data: dict,
|
|
208
|
+
denied: list[str] | None,
|
|
209
|
+
agent: AgentConfig,
|
|
210
|
+
settings: Settings,
|
|
211
|
+
*,
|
|
212
|
+
session_fallback: str | None,
|
|
213
|
+
exit_code: int | None = None,
|
|
214
|
+
) -> DispatchResult:
|
|
215
|
+
"""Build the DispatchResult for a CLI payload with ``is_error: true``.
|
|
216
|
+
|
|
217
|
+
Shared by ``dispatch`` and ``dispatch_stream`` — the two paths differ only
|
|
218
|
+
in which session id they fall back to and whether an exit code is known.
|
|
219
|
+
Error-type precedence: budget (the CLI's own terminal reason) > denied tools
|
|
220
|
+
(a deterministic signal) > text classification.
|
|
221
|
+
"""
|
|
222
|
+
raw_result = data.get("result", "")
|
|
223
|
+
result_text = str(raw_result) if raw_result else ""
|
|
224
|
+
detail = _cli_error_details(data)
|
|
225
|
+
if result_text:
|
|
226
|
+
error_text = result_text
|
|
227
|
+
elif detail:
|
|
228
|
+
error_text = f"Agent '{agent_name}' failed: {detail}"
|
|
229
|
+
else:
|
|
230
|
+
suffix = f" (exit code {exit_code})" if exit_code is not None else ""
|
|
231
|
+
error_text = f"Agent '{agent_name}' reported an error with no details{suffix}"
|
|
232
|
+
|
|
233
|
+
session_id = data.get("session_id") or session_fallback
|
|
234
|
+
budget_error = _is_budget_error(data)
|
|
235
|
+
if budget_error:
|
|
236
|
+
error_type = "budget"
|
|
237
|
+
error_text += _budget_error_hint(agent_name, _effective_budget(agent, settings), session_id)
|
|
238
|
+
elif denied:
|
|
239
|
+
# Denied tools are a stronger signal than substring matching.
|
|
240
|
+
error_type = "permission"
|
|
241
|
+
error_text += _permission_hint(agent_name)
|
|
242
|
+
else:
|
|
243
|
+
error_type = _classify_error(error_text)
|
|
244
|
+
if error_type == "permission":
|
|
245
|
+
error_text += _permission_hint(agent_name)
|
|
246
|
+
|
|
247
|
+
return _apply_budget(
|
|
248
|
+
DispatchResult(
|
|
249
|
+
agent=agent_name,
|
|
250
|
+
success=False,
|
|
251
|
+
result=result_text,
|
|
252
|
+
session_id=session_id,
|
|
253
|
+
cost_usd=data.get("total_cost_usd"),
|
|
254
|
+
duration_ms=data.get("duration_ms"),
|
|
255
|
+
num_turns=data.get("num_turns"),
|
|
256
|
+
error=error_text,
|
|
257
|
+
error_type=error_type,
|
|
258
|
+
denied_tools=denied,
|
|
259
|
+
# The cap was hit by definition — flag it even if the reported cost
|
|
260
|
+
# lands a hair under the budget after rounding.
|
|
261
|
+
budget_exceeded=True if budget_error else None,
|
|
262
|
+
),
|
|
263
|
+
agent,
|
|
264
|
+
settings,
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
|
|
141
268
|
def _apply_budget(result: DispatchResult, agent: AgentConfig, settings: Settings) -> DispatchResult:
|
|
142
269
|
"""Flag a result whose cost exceeded the configured budget (post-hoc).
|
|
143
270
|
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
271
|
+
The CLI's own ``--max-budget-usd`` stops a session once the cap is reached
|
|
272
|
+
(that path returns ``error_type="budget"``), but the last turn can still
|
|
273
|
+
overshoot, and the cap only covers a single dispatch. This adds the
|
|
274
|
+
post-hoc signal: ``budget_exceeded=True`` plus a hint, without failing the
|
|
275
|
+
result — the money is already spent. Applied to error results too.
|
|
149
276
|
"""
|
|
150
|
-
budget = agent
|
|
277
|
+
budget = _effective_budget(agent, settings)
|
|
278
|
+
if result.error_type == "budget":
|
|
279
|
+
# Already flagged with a far more actionable message — don't double up.
|
|
280
|
+
return result
|
|
151
281
|
if not budget or not result.cost_usd or result.cost_usd <= budget:
|
|
152
282
|
return result
|
|
153
283
|
result.budget_exceeded = True
|
|
@@ -510,36 +640,14 @@ def dispatch(
|
|
|
510
640
|
|
|
511
641
|
is_error = data.get("is_error", False)
|
|
512
642
|
if is_error:
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
else (
|
|
518
|
-
f"Agent '{agent_name}' reported an error with no details "
|
|
519
|
-
f"(exit code {proc.returncode})"
|
|
520
|
-
)
|
|
521
|
-
)
|
|
522
|
-
error_type = _classify_error(error_text)
|
|
523
|
-
if denied and error_type != "permission":
|
|
524
|
-
# Denied tools are a stronger signal than substring matching.
|
|
525
|
-
error_type = "permission"
|
|
526
|
-
if error_type == "permission":
|
|
527
|
-
error_text += _permission_hint(agent_name)
|
|
528
|
-
return _apply_budget(
|
|
529
|
-
DispatchResult(
|
|
530
|
-
agent=agent_name,
|
|
531
|
-
success=False,
|
|
532
|
-
result=str(raw_result) if raw_result else "",
|
|
533
|
-
session_id=data.get("session_id") or session_uuid,
|
|
534
|
-
cost_usd=data.get("total_cost_usd"),
|
|
535
|
-
duration_ms=data.get("duration_ms"),
|
|
536
|
-
num_turns=data.get("num_turns"),
|
|
537
|
-
error=error_text,
|
|
538
|
-
error_type=error_type,
|
|
539
|
-
denied_tools=denied,
|
|
540
|
-
),
|
|
643
|
+
return _build_error_result(
|
|
644
|
+
agent_name,
|
|
645
|
+
data,
|
|
646
|
+
denied,
|
|
541
647
|
agent,
|
|
542
648
|
settings,
|
|
649
|
+
session_fallback=session_uuid,
|
|
650
|
+
exit_code=proc.returncode,
|
|
543
651
|
)
|
|
544
652
|
|
|
545
653
|
result_text = data.get("result", "")
|
|
@@ -749,32 +857,13 @@ def dispatch_stream(
|
|
|
749
857
|
denied = _extract_denied_tools(result_data)
|
|
750
858
|
is_error = result_data.get("is_error", False)
|
|
751
859
|
if is_error:
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
else (f"Agent '{agent_name}' reported an error with no details")
|
|
757
|
-
)
|
|
758
|
-
error_type = _classify_error(error_text)
|
|
759
|
-
if denied and error_type != "permission":
|
|
760
|
-
error_type = "permission"
|
|
761
|
-
if error_type == "permission":
|
|
762
|
-
error_text += _permission_hint(agent_name)
|
|
763
|
-
return _apply_budget(
|
|
764
|
-
DispatchResult(
|
|
765
|
-
agent=agent_name,
|
|
766
|
-
success=False,
|
|
767
|
-
result=str(raw_result) if raw_result else "",
|
|
768
|
-
session_id=result_data.get("session_id") or new_session,
|
|
769
|
-
cost_usd=result_data.get("total_cost_usd"),
|
|
770
|
-
duration_ms=result_data.get("duration_ms"),
|
|
771
|
-
num_turns=result_data.get("num_turns"),
|
|
772
|
-
error=error_text,
|
|
773
|
-
error_type=error_type,
|
|
774
|
-
denied_tools=denied,
|
|
775
|
-
),
|
|
860
|
+
return _build_error_result(
|
|
861
|
+
agent_name,
|
|
862
|
+
result_data,
|
|
863
|
+
denied,
|
|
776
864
|
agent,
|
|
777
865
|
settings,
|
|
866
|
+
session_fallback=new_session,
|
|
778
867
|
)
|
|
779
868
|
result_text = result_data.get("result", "")
|
|
780
869
|
parsed = _parse_structured_response(result_text) if response_format == "json" else None
|
|
@@ -969,6 +969,24 @@ class TestTestCommand:
|
|
|
969
969
|
assert "Diagnosis: timeout" in result.output
|
|
970
970
|
assert "--timeout 600" in result.output
|
|
971
971
|
|
|
972
|
+
def test_budget_error_shows_diagnosis(self, tmp_path: Path):
|
|
973
|
+
agent_dir = tmp_path / "proj"
|
|
974
|
+
agent_dir.mkdir()
|
|
975
|
+
runner.invoke(cli, ["add", "proj", str(agent_dir), "-d", "Test"])
|
|
976
|
+
with patch("agent_dispatch.runner.dispatch") as mock_dispatch:
|
|
977
|
+
mock_dispatch.return_value = DispatchResult(
|
|
978
|
+
agent="proj",
|
|
979
|
+
success=False,
|
|
980
|
+
result="",
|
|
981
|
+
error="Agent 'proj' failed: Reached maximum budget ($0.10)",
|
|
982
|
+
error_type="budget",
|
|
983
|
+
budget_exceeded=True,
|
|
984
|
+
)
|
|
985
|
+
result = runner.invoke(cli, ["test", "proj"])
|
|
986
|
+
assert result.exit_code != 0
|
|
987
|
+
assert "Diagnosis: spend cap reached" in result.output
|
|
988
|
+
assert "--max-budget-usd" in result.output
|
|
989
|
+
|
|
972
990
|
def test_stream_uses_dispatch_stream(self, tmp_path: Path):
|
|
973
991
|
"""--stream should call dispatch_stream, not dispatch, and forward progress."""
|
|
974
992
|
agent_dir = tmp_path / "proj"
|