agent-dispatch 0.12.0__tar.gz → 0.13.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/workflows/publish.yml +8 -1
  2. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/AGENTS.md +13 -4
  3. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/CHANGELOG.md +130 -1
  4. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/PKG-INFO +32 -7
  5. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/README.md +30 -5
  6. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/agents.example.yaml +6 -0
  7. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/pyproject.toml +1 -1
  8. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/__init__.py +1 -1
  9. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/cli.py +65 -7
  10. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/config.py +39 -1
  11. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/jobs.py +40 -8
  12. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/models.py +10 -0
  13. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/runner.py +28 -0
  14. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/server.py +297 -143
  15. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_cli.py +83 -0
  16. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_config.py +56 -0
  17. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_jobs.py +33 -0
  18. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_runner.py +70 -0
  19. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_server.py +348 -0
  20. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/dependabot.yml +0 -0
  21. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/workflows/ci.yml +0 -0
  22. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.gitignore +0 -0
  23. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/LICENSE +0 -0
  24. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/SECURITY.md +0 -0
  25. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/assets/mascot.png +0 -0
  26. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/cache.py +0 -0
  27. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/__init__.py +0 -0
  28. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/conftest.py +0 -0
  29. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_cache.py +0 -0
  30. {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_models.py +0 -0
@@ -26,5 +26,12 @@ jobs:
26
26
  pip install build
27
27
  python -m build
28
28
 
29
+ # Keep this current with the build backend. The pinned action bundles its
30
+ # own twine, and twine rejects a Metadata-Version it does not know: the
31
+ # v0.13.0 publish failed because hatchling 1.32 emits `Metadata-Version:
32
+ # 2.5` while the previous pin (Feb 2026) shipped twine 6.1.0 / packaging
33
+ # 25.0. `python -m build` resolves the newest hatchling at build time, so
34
+ # this drifts on its own with no repo change — see the release checklist's
35
+ # `twine check` step, which catches it before a tag is cut.
29
36
  - name: Publish to PyPI
30
- uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # release/v1
37
+ uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
@@ -28,7 +28,7 @@ pip install -e ".[dev]"
28
28
 
29
29
  ```bash
30
30
  ruff check src/ tests/
31
- python3 -m pytest tests/ -v # 561 tests, ~4s
31
+ python3 -m pytest tests/ -v # 578 tests, ~5s
32
32
  ```
33
33
 
34
34
  Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.which` + `subprocess.run`/`Popen`; server tests mock `_get_config` + `runner.dispatch`. The one exception is `TestStreamPipeHandling`, which spawns a short-lived *python* subprocess: a pipe deadlock lives in the OS pipe buffer, so a mocked `Popen` structurally cannot reproduce it.
@@ -52,16 +52,25 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
52
52
  - **Never `close()` a pipe another thread may still be reading.** `close()` waits on the reader's buffer lock with *no timeout*, so it would hang the dispatch forever — the bounded `join()` before it buys nothing. `dispatch_stream` skips the stderr close while the drain thread is alive and lets the daemon reader + Popen finalizer release the fd. This is not hypothetical: a stdio MCP server inherits the child's `stderr`, so the pipe often has no EOF even after `claude` exits cleanly.
53
53
  - **No `await` inside `config_lock()`.** `ProcessLock`'s in-process guard is a `threading.RLock` — re-entrant per *thread* — and every MCP tool coroutine runs on the one event-loop thread. Suspending in the critical section lets a second coroutine re-enter the "held" lock and interleave its own load/mutate/save. Collect warnings as data, emit them after the `with` block (`test_no_await_inside_the_config_lock` enforces this by AST).
54
54
  - Cross-process locks are acquired with a **bounded** wait, never a blocking `flock`: the server takes them on its event-loop thread, so a wedged holder would freeze every tool. After the deadline it proceeds unlocked and logs — a possible lost update beats a permanent freeze.
55
- - `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore.
55
+ - `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore. For a *running* job, `started_at` alone does **not** prove abandonment — a dispatch may legitimately run to the 7200s timeout ceiling, so the file's own mtime is checked too (a live worker rewrites it on every progress flush). That check can only ever *skip* a recovery, never add one.
56
+ - Deleting job records is **opt-in** (`settings.job_retention_days`, default `0` = off) and only ever happens at server start, never inside a tool. They are the user's dispatch history and the deletion is irreversible, so an unreadable config is treated as "do nothing" rather than falling back to a default retention.
57
+ - `max_concurrency` must bound *subprocesses*, not coroutines. Cancelling a coroutine does not stop the thread behind `asyncio.to_thread`, so `async with sem:` gave the slot away while `claude` kept running and billing. Dispatches go through `_dispatch_guarded`, which releases from the future's done-callback and `shield`s the await. Never "simplify" it back to `async with`.
58
+ - Tool responses are serialized through `server._dumps`, never bare `json.dumps`: the stdlib default (`ensure_ascii=True`) turns every non-ASCII character into a `\uXXXX` escape, tripling the bytes and tokenizing badly, for no gain — the stdio transport emits raw UTF-8 via pydantic anyway. A real `list_groups()` carried 8520 escapes and weighed 59 KB instead of 25 KB.
59
+ - `agents.yaml` is parsed with libyaml's `CSafeLoader` when available (`config._YamlLoader`), falling back to the pure-Python **safe** loader — never `yaml.Loader`. The config is re-read on every tool call, so this parse is on the hot path of all 21 tools *and* blocks the event-loop thread: 9.80 ms → 0.76 ms on a real 38 KB config.
56
60
  - Pydantic does **not** validate on assignment. `Field(ge=...)` guards only the *load* path; every mutation surface (CLI `add`/`update`, MCP `add_agent`/`update_agent`) needs its own boundary check, or the bound escapes as a raw `ValidationError`.
57
61
  - Every state file (`agents.yaml`, job files) is written **temp file + `os.replace`**, never in place, and every load/mutate/save is wrapped in `config.ProcessLock` — the CLI and the MCP server are separate processes writing the same files, so a thread lock alone loses updates.
58
62
  - Anything that changes an agent's config must call `_invalidate_agent_cache` — the cache key holds the agent *name*, not its directory or permissions.
59
63
  - Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, and `budget_exceeded` results, so the documented "grant access, then re-dispatch" recovery is never short-circuited.
60
64
  - Remediation text is a contract: a hint that names a flag must name one that exists (`test_printed_budget_hint_is_a_runnable_command` feeds the printed flags back into the CLI). Run the command you print.
61
- - MCP tools that load config carry `@_config_guard` under `@mcp.tool()` so a broken `agents.yaml` returns the `{"error": ...}` envelope instead of a raw traceback.
65
+ - The config error sets are declared **once** and in two halves: `config.CONFIG_LOAD_ERRORS` (read) and `config.CONFIG_SAVE_ERRORS` (write — `yaml.dump`'s `RepresenterError` is a `yaml.YAMLError`, therefore neither `OSError` nor `ConfigLoadError`, and used to escape both the MCP guard and the CLI's `_save_or_exit`). Two halves, not one set, because the remediations differ: a failed write is atomic so the old config survives, while a failed read needs the YAML fixed.
66
+ - MCP tools that load config carry `@_config_guard` under `@mcp.tool()` so a broken `agents.yaml` — or a failed *write* — returns the `{"error": ...}` envelope instead of a raw traceback. The set of load errors lives in one place (`config.CONFIG_LOAD_ERRORS`) because three surfaces handle it: **`UnicodeDecodeError` is a `ValueError`, not an `OSError`**, and listing types per-site is exactly how a cp1251 config slipped past all three.
62
67
 
63
68
  - Tests must not touch anything outside `tmp_path`. `test_server.py`'s autouse `_reset_globals` and `test_cli.py`'s `_isolated_config` redirect **both** `AGENT_DISPATCH_CONFIG` and `AGENT_DISPATCH_JOBS_DIR`: a mutation tool that bails out early (unknown agent) still takes `config_lock()` first, which would otherwise create a lock file beside the developer's real config.
64
69
 
70
+ - `mcp` is pinned **`>=1.2.0,<2`** deliberately: 2.0 removed `mcp.server.fastmcp`, which `server.py` imports, so an unbounded range gives every fresh install a dead `agent-dispatch serve`. Lifting the cap means porting to `mcp.server.mcpserver.MCPServer` — it is not a dependency bump.
71
+ - Verify packaging in a **clean venv**, never the dev machine: build the wheel, install it fresh, import the server. A stale pin in local site-packages hides exactly the failure a new user hits first.
72
+ - Run **`twine check dist/*`** before cutting the tag, with a *current* twine. The clean-venv check does not cover this: it proves the wheel installs, not that PyPI's uploader will accept its metadata. `python -m build` pulls the newest hatchling at build time, so the emitted `Metadata-Version` climbs on its own, and `pypa/gh-action-pypi-publish` is pinned to a SHA that bundles a fixed twine — v0.13.0's publish failed on exactly that mismatch (hatchling 1.32 → metadata 2.5; pinned twine 6.1.0 → "not a valid metadata version"). When it happens, bump the action pin rather than pinning the backend down.
73
+
65
74
  ## Deliberately not built
66
75
 
67
76
  These were considered — some fully implemented — and cut on purpose: an agent router / auto-dispatch (`recommend_agent` / `dispatch_auto`, removed before 0.8.0 — a keyword scorer adds little over the calling LLM at a handful of agents, and auto-dispatch can spend money or mutate a repo on a guess); groups as an execution engine (they are a descriptive layer — no routing, no per-group settings); an agent-dispatch-side budget ledger across dispatches (the CLI's own `--max-budget-usd` covers a single run; anything cumulative would need state we deliberately don't keep). Please open an issue with the use case before adding any of them.
@@ -76,4 +85,4 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
76
85
 
77
86
  ## More detail
78
87
 
79
- [README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 561 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
88
+ [README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 578 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
@@ -7,6 +7,134 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.13.0] - 2026-08-13
11
+
12
+ An efficiency round, measured against a real 38 KB config (4 agents, 6 groups),
13
+ a 294-record jobs directory and 14 concurrently running servers — not against
14
+ synthetic fixtures. Findings came from a five-axis audit (caller context,
15
+ latency/IO, spend, CPU, concurrency) whose every claim was re-measured
16
+ adversarially before implementation; most were confirmed as real but too small
17
+ to matter next to a 10–120 s subprocess and were deliberately left alone.
18
+
19
+ ### Changed
20
+ - **Tool responses no longer escape non-ASCII, cutting discovery payloads
21
+ roughly in half.** `json.dumps` defaults to `ensure_ascii=True`, so every
22
+ Cyrillic character left as a `\uXXXX` escape — 6 bytes for what UTF-8 stores
23
+ in 2, and a far worse token sequence. One `list_groups()` carried **8520**
24
+ such escapes: 59 KB where 25 KB was enough. Measured over a full discovery
25
+ pass (`list_agents` + `list_groups` + `inspect_group` + `inspect_agent`):
26
+ 104 KB → 51 KB, about **13 000 tokens saved per pass**, on every call, in
27
+ every session. Nothing is lost: the stdio transport serializes the JSON-RPC
28
+ envelope with pydantic's `model_dump_json`, which already emits raw UTF-8 —
29
+ which is also why the `dispatch` family never had this problem while every
30
+ `json.dumps` tool did. All 57 call sites now go through one `_dumps` helper
31
+ so the two agree. `_dumps` probes encodability and falls back to the escaped
32
+ form for the one payload that needs it — a **lone surrogate**, which an agent
33
+ produces by printing a literal `\udXXX` escape in its JSON output (`json.loads`
34
+ manufactures it from perfectly ASCII input, so the subprocess-level
35
+ `errors="replace"` cannot help). Unescaped it would raise `UnicodeEncodeError`
36
+ inside the stdio transport *after* the tool returned — past every guard.
37
+ The saving is proportional to how much non-ASCII your config and results
38
+ carry: on an all-ASCII setup it is exactly zero.
39
+ - **`agents.yaml` parses ~13x faster.** The config is re-read on every single
40
+ tool call (deliberately — that is how a new agent is picked up without a
41
+ restart), so the parse sits on the hot path of all 21 tools *and* runs on the
42
+ event-loop thread, where it blocks every other tool. Switching to libyaml's
43
+ `CSafeLoader` takes the real config from **9.80 ms to 0.76 ms** per call.
44
+ Falls back to the pure-Python safe loader when PyYAML was built without
45
+ libyaml; both are the *safe* loader, so a config still cannot construct
46
+ arbitrary objects.
47
+
48
+ ### Added
49
+ - **`settings.job_retention_days`** — when > 0, terminal job records older than
50
+ N days are deleted at server start. Every async dispatch and every
51
+ `return_ref` dispatch leaves a job file that nothing removed on its own, so
52
+ the directory grew forever while `dispatch_jobs` and stale-job recovery
53
+ parsed every file in it. **Defaults to `0` (off)**: those records are the
54
+ user's own dispatch history and deleting them is irreversible, so it is an
55
+ explicit opt-in rather than something a version bump starts doing to an
56
+ existing install. An unreadable config is treated as `0`, never guessed.
57
+
58
+ ### Fixed
59
+ - **`max_concurrency` bounds live subprocesses again, not live coroutines.**
60
+ `async with sem:` released the slot when the *coroutine* unwound, but
61
+ cancelling a coroutine does not stop the thread behind `asyncio.to_thread` —
62
+ so an interrupted turn or a closed session handed the slot to the next
63
+ dispatch while the `claude` subprocess kept running and kept being billed. N
64
+ cancellations meant up to N extra concurrent subprocesses, and each abandoned
65
+ worker also held a thread out of the default executor (`min(32, cpu_count+4)`,
66
+ which `to_thread` needs for *every* dispatch) for the rest of its timeout —
67
+ up to 7200s. The slot is now tied to the worker's real lifetime via
68
+ `_dispatch_guarded`; cancellation still reaches the caller immediately.
69
+ - **A legitimately long dispatch is no longer flipped to `failed` by an
70
+ unrelated server starting up.** `recover_stale` judged abandonment from
71
+ `started_at` alone, but a dispatch may legitimately run up to the 7200s
72
+ timeout ceiling — so any of the (routinely 14+) other servers booting would
73
+ mark a live 2-hour job `failed`, after which `finish()` refuses the real,
74
+ already-paid-for result. A *running* job whose file was modified within the
75
+ threshold is now left alone: a live worker rewrites it on every progress
76
+ flush, so only the file's own age proves abandonment. This can only ever skip
77
+ a recovery — a genuinely abandoned job is picked up on a later start.
78
+ - **A config that cannot be *rendered* now returns an error envelope instead of
79
+ a traceback.** `_get_config` wraps read-side YAML failures into
80
+ `ConfigLoadError`, but `save_config` → `yaml.dump` raises `RepresenterError` —
81
+ a `yaml.YAMLError`, so neither an `OSError` nor a `ConfigLoadError`. It
82
+ therefore slipped past *both* arms of the MCP guard on
83
+ `add_agent`/`update_agent`/`remove_agent` **and** past the CLI's
84
+ `_save_or_exit`. Unreachable today (only JSON-native types are ever dumped)
85
+ and closed now because it is the same "an exception type escapes the handler
86
+ meant to catch it" class that produced the 0.12.0 and 0.12.1 rounds. The set
87
+ is declared once as `config.CONFIG_SAVE_ERRORS`, next to its read-side twin,
88
+ so the two surfaces cannot drift.
89
+ - **Stale-job recovery scans the jobs directory once instead of twice.** It
90
+ called `list()` per status, and `list()` reads and parses every file in the
91
+ directory (~85 ms at 294 records) — at the start of every server process.
92
+
93
+ ## [0.12.1] - 2026-07-30
94
+
95
+ Two holes in 0.12.0's own "tools always return a clean error" fix.
96
+
97
+ ### Fixed
98
+ - **A non-UTF-8 `agents.yaml` no longer crashes every surface.** `UnicodeDecodeError`
99
+ is a `ValueError`, not an `OSError`, so a config saved as cp1251/UTF-16 slipped
100
+ past all three handlers added in 0.12.0 and raised a bare traceback out of every
101
+ MCP tool, `agent-dispatch list` — and `doctor`, the command you run *because*
102
+ the config is broken. The set of load errors is now declared once
103
+ (`config.CONFIG_LOAD_ERRORS`) and shared by all three, and `doctor` reports the
104
+ encoding (and an unreadable file) as a normal FAIL with a fix. The shared set
105
+ lives in `config.CONFIG_LOAD_ERRORS`; the CLI still branches per type because
106
+ each one deserves a different remediation line.
107
+ - **An undecodable byte from the `claude` CLI no longer kills a paid-for
108
+ dispatch.** `text=True` decodes strictly, so one invalid byte on stdout raised
109
+ `UnicodeDecodeError` — a `ValueError`, caught by nothing in the runner — out of
110
+ `dispatch`, `dispatch_stream`, the MCP tools and `agent-dispatch test`. Both
111
+ spawn sites now decode with `errors="replace"`, so a mangled byte becomes
112
+ U+FFFD instead of discarding the run. (Present since the initial commit.)
113
+ - **`dispatch` classifies spawn failures like `dispatch_stream` already did.**
114
+ A directory that passes `is_dir()` but cannot be entered — or that vanishes
115
+ between the check and the spawn — made `subprocess.run` raise straight through
116
+ the `except subprocess.TimeoutExpired`. It now returns `not_found` /
117
+ `permission` / `cli_error` like the streaming path.
118
+ - **The six job tools got the I/O envelope too.** `dispatch_status`, `_wait`,
119
+ `_cancel`, `_jobs`, `fetch_result` and `dispatch_gc` never load config, so they
120
+ carried no guard — an unwritable or misconfigured jobs directory raised out of
121
+ them. The guard is now split (`_io_guard` / `_config_guard`) and both halves
122
+ are applied where each is needed.
123
+ - **`add_agent` reports an unresolvable path** (`~unknown-user`) as an error
124
+ envelope instead of raising `RuntimeError`.
125
+ - **A failed config *write* is reported, not raised.** The same contract had the
126
+ other half missing: a full disk or a read-only volume made `save_config`
127
+ raise OSError straight out of `add_agent`/`update_agent`/`remove_agent` and out
128
+ of the matching CLI commands. Both surfaces now report it — the CLI adds that
129
+ the previous config is intact, which the atomic write guarantees.
130
+
131
+ ### Changed
132
+ - Docs corrected against the code: stale-job recovery now documents the
133
+ `pending` sweep and its 24h threshold; the config-lock guarantee is stated as
134
+ best-effort (it proceeds unlocked after 10s rather than freezing the server);
135
+ `add_agent(timeout=0)` documents that it stores the literal 300, not
136
+ `settings.default_timeout`.
137
+
10
138
  ## [0.12.0] - 2026-07-29
11
139
 
12
140
  Reliability pass over the streaming path, config durability, and the tool
@@ -546,7 +674,8 @@ cache bounding, and stale-job recovery.
546
674
  - Dependabot for `pip` + `github-actions`, GitHub Actions pinned to
547
675
  commit SHAs for supply-chain integrity.
548
676
 
549
- [Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.12.0...HEAD
677
+ [Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.12.1...HEAD
678
+ [0.12.1]: https://github.com/ginkida/agent-dispatch/compare/v0.12.0...v0.12.1
550
679
  [0.12.0]: https://github.com/ginkida/agent-dispatch/compare/v0.11.0...v0.12.0
551
680
  [0.11.0]: https://github.com/ginkida/agent-dispatch/compare/v0.10.0...v0.11.0
552
681
  [0.10.0]: https://github.com/ginkida/agent-dispatch/compare/v0.9.0...v0.10.0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: agent-dispatch
3
- Version: 0.12.0
3
+ Version: 0.13.0
4
4
  Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
5
5
  Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
6
6
  Project-URL: Repository, https://github.com/ginkida/agent-dispatch
@@ -388,8 +388,8 @@ Register a new project directory as an agent. Description is auto-generated from
388
388
  | `name` | string | yes | Agent name (letters, digits, hyphens, underscores) |
389
389
  | `directory` | string | yes | Path to an existing project directory (`~` is expanded, relative paths resolved) |
390
390
  | `description` | string | no | What this agent can do — auto-generated if empty |
391
- | `timeout` | int | no | Timeout in seconds (0 = use global default) |
392
- | `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = no limit) |
391
+ | `timeout` | int | no | Timeout in seconds (0 = 300; this is a literal default, not `settings.default_timeout`) |
392
+ | `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = inherit `settings.default_max_budget_usd`; no cap only when that is unset too) |
393
393
  | `permission_mode` | string | no | Permission mode (e.g. `default`, `plan`, `bypassPermissions`) |
394
394
  | `allowed_tools` | string | no | Comma-separated allowed tools (e.g. `"Bash,Read,Edit"`) |
395
395
  | `disallowed_tools` | string | no | Comma-separated disallowed tools |
@@ -405,7 +405,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
405
405
  | `name` | string | yes | Agent name to update |
406
406
  | `description` | string | no | New description |
407
407
  | `timeout` | int | no | New timeout (0 = don't change) |
408
- | `max_budget_usd` | float | no | New budget limit (0 = don't change, negative = clear the limit) |
408
+ | `max_budget_usd` | float | no | New budget limit (0 = don't change; negative clears the *per-agent* cap, after which `settings.default_max_budget_usd` applies) |
409
409
  | `model` | string | no | Model override. `"none"` to clear |
410
410
  | `permission_mode` | string | no | Permission mode. `"none"` to clear |
411
411
  | `allowed_tools` | string | no | Comma-separated. `"none"` to clear |
@@ -474,7 +474,7 @@ Async workers run with streaming under the hood: the job file keeps a rolling ta
474
474
 
475
475
  `dispatch_jobs(status?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted.
476
476
 
477
- Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server marks jobs left in `running` by a crashed instance as `failed` once they are stale (stuck for over an hour).
477
+ Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server recovers jobs a crashed instance abandoned: `running` ones stuck over an hour, and `pending` ones over 24 hours, are marked `failed` so they stop being polled forever and become collectable by `dispatch_gc`. (The `pending` threshold is deliberately long — the jobs directory is shared by every running server, so a job queued behind another server's concurrency limit must not be swept.)
478
478
 
479
479
  | When to use async | When to use `dispatch` |
480
480
  |-------------------|------------------------|
@@ -574,6 +574,7 @@ settings:
574
574
  # - Edit
575
575
  max_dispatch_depth: 3 # recursion protection
576
576
  max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
577
+ # job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
577
578
  cache:
578
579
  enabled: true
579
580
  ttl: 300 # seconds
@@ -582,6 +583,30 @@ settings:
582
583
 
583
584
  Config is reloaded on every tool call — add agents without restarting.
584
585
 
586
+ ### Job retention
587
+
588
+ Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
589
+ record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
590
+ `dispatch_gc` has to be run by hand. The directory therefore grows without
591
+ bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
592
+ start read and parse *every* file in it.
593
+
594
+ Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
595
+ than N days when a server starts:
596
+
597
+ ```yaml
598
+ settings:
599
+ job_retention_days: 30
600
+ ```
601
+
602
+ It defaults to `0` — **off** — because those records are your own history of
603
+ past dispatches and deleting them cannot be undone. Pending and running jobs are
604
+ never touched.
605
+
606
+ `agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
607
+ one-off. Both *delete* immediately and report the count; neither previews, so
608
+ check what is there first with `agent-dispatch jobs`.
609
+
585
610
  ### Auto-Description
586
611
 
587
612
  `agent-dispatch add` without `--description` generates one from:
@@ -635,7 +660,7 @@ agent-dispatch MCP server
635
660
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
636
661
  - **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
637
662
  - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
638
- - **Durable config** — `agents.yaml` is written atomically (temp file + rename), and every mutation path (CLI and MCP server alike) holds a cross-process advisory lock, so concurrent edits cannot truncate the file or silently drop one another's agents.
663
+ - **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
639
664
 
640
665
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
641
666
 
@@ -358,8 +358,8 @@ Register a new project directory as an agent. Description is auto-generated from
358
358
  | `name` | string | yes | Agent name (letters, digits, hyphens, underscores) |
359
359
  | `directory` | string | yes | Path to an existing project directory (`~` is expanded, relative paths resolved) |
360
360
  | `description` | string | no | What this agent can do — auto-generated if empty |
361
- | `timeout` | int | no | Timeout in seconds (0 = use global default) |
362
- | `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = no limit) |
361
+ | `timeout` | int | no | Timeout in seconds (0 = 300; this is a literal default, not `settings.default_timeout`) |
362
+ | `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = inherit `settings.default_max_budget_usd`; no cap only when that is unset too) |
363
363
  | `permission_mode` | string | no | Permission mode (e.g. `default`, `plan`, `bypassPermissions`) |
364
364
  | `allowed_tools` | string | no | Comma-separated allowed tools (e.g. `"Bash,Read,Edit"`) |
365
365
  | `disallowed_tools` | string | no | Comma-separated disallowed tools |
@@ -375,7 +375,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
375
375
  | `name` | string | yes | Agent name to update |
376
376
  | `description` | string | no | New description |
377
377
  | `timeout` | int | no | New timeout (0 = don't change) |
378
- | `max_budget_usd` | float | no | New budget limit (0 = don't change, negative = clear the limit) |
378
+ | `max_budget_usd` | float | no | New budget limit (0 = don't change; negative clears the *per-agent* cap, after which `settings.default_max_budget_usd` applies) |
379
379
  | `model` | string | no | Model override. `"none"` to clear |
380
380
  | `permission_mode` | string | no | Permission mode. `"none"` to clear |
381
381
  | `allowed_tools` | string | no | Comma-separated. `"none"` to clear |
@@ -444,7 +444,7 @@ Async workers run with streaming under the hood: the job file keeps a rolling ta
444
444
 
445
445
  `dispatch_jobs(status?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted.
446
446
 
447
- Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server marks jobs left in `running` by a crashed instance as `failed` once they are stale (stuck for over an hour).
447
+ Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server recovers jobs a crashed instance abandoned: `running` ones stuck over an hour, and `pending` ones over 24 hours, are marked `failed` so they stop being polled forever and become collectable by `dispatch_gc`. (The `pending` threshold is deliberately long — the jobs directory is shared by every running server, so a job queued behind another server's concurrency limit must not be swept.)
448
448
 
449
449
  | When to use async | When to use `dispatch` |
450
450
  |-------------------|------------------------|
@@ -544,6 +544,7 @@ settings:
544
544
  # - Edit
545
545
  max_dispatch_depth: 3 # recursion protection
546
546
  max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
547
+ # job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
547
548
  cache:
548
549
  enabled: true
549
550
  ttl: 300 # seconds
@@ -552,6 +553,30 @@ settings:
552
553
 
553
554
  Config is reloaded on every tool call — add agents without restarting.
554
555
 
556
+ ### Job retention
557
+
558
+ Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
559
+ record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
560
+ `dispatch_gc` has to be run by hand. The directory therefore grows without
561
+ bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
562
+ start read and parse *every* file in it.
563
+
564
+ Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
565
+ than N days when a server starts:
566
+
567
+ ```yaml
568
+ settings:
569
+ job_retention_days: 30
570
+ ```
571
+
572
+ It defaults to `0` — **off** — because those records are your own history of
573
+ past dispatches and deleting them cannot be undone. Pending and running jobs are
574
+ never touched.
575
+
576
+ `agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
577
+ one-off. Both *delete* immediately and report the count; neither previews, so
578
+ check what is there first with `agent-dispatch jobs`.
579
+
555
580
  ### Auto-Description
556
581
 
557
582
  `agent-dispatch add` without `--description` generates one from:
@@ -605,7 +630,7 @@ agent-dispatch MCP server
605
630
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
606
631
  - **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
607
632
  - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
608
- - **Durable config** — `agents.yaml` is written atomically (temp file + rename), and every mutation path (CLI and MCP server alike) holds a cross-process advisory lock, so concurrent edits cannot truncate the file or silently drop one another's agents.
633
+ - **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
609
634
 
610
635
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
611
636
 
@@ -100,6 +100,12 @@ settings:
100
100
  # - Bash
101
101
  # - Read
102
102
  # - Edit
103
+ # job_retention_days: 30 # OFF by default (0). When > 0, terminal job records
104
+ # older than this are deleted at server start. Every
105
+ # async and every return_ref dispatch leaves a job
106
+ # file behind and nothing removes it otherwise, so
107
+ # the directory grows forever — but the records are
108
+ # your dispatch history, so pruning is opt-in.
103
109
  cache:
104
110
  enabled: true
105
111
  ttl: 300 # seconds; identical (agent, task, context) requests are cached
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agent-dispatch"
3
- version = "0.12.0"
3
+ version = "0.13.0"
4
4
  description = "MCP server that lets Claude Code agents delegate tasks to agents in other project directories"
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """agent-dispatch: Delegate tasks between Claude Code agents across projects."""
2
2
 
3
- __version__ = "0.12.0"
3
+ __version__ = "0.13.0"
@@ -13,7 +13,14 @@ import click
13
13
  import yaml
14
14
  from pydantic import ValidationError
15
15
 
16
- from .config import auto_describe, config_lock, config_path, load_config, save_config
16
+ from .config import (
17
+ CONFIG_SAVE_ERRORS,
18
+ auto_describe,
19
+ config_lock,
20
+ config_path,
21
+ load_config,
22
+ save_config,
23
+ )
17
24
  from .jobs import JobStore, default_jobs_dir, is_valid_job_id
18
25
  from .models import (
19
26
  AgentConfig,
@@ -29,6 +36,39 @@ def _parse_csv(value: str | None) -> list[str] | None:
29
36
  return [t.strip() for t in value.split(",") if t.strip()] if value else None
30
37
 
31
38
 
39
+ def _save_or_exit(config: DispatchConfig) -> None:
40
+ """Persist the config, turning an I/O failure into a message, not a traceback.
41
+
42
+ A full disk or a read-only volume makes `save_config` raise OSError. The
43
+ write is atomic, so the previous config survives — say so, since that is the
44
+ one thing the user needs to know.
45
+ """
46
+ try:
47
+ save_config(config)
48
+ except OSError as e:
49
+ click.echo(
50
+ click.style(
51
+ f"Error: could not write {config_path()}: {e}\n"
52
+ "Nothing was changed — the previous config is intact "
53
+ "(the write is atomic). Check free space and permissions.",
54
+ fg="red",
55
+ )
56
+ )
57
+ raise SystemExit(1) from None
58
+ except CONFIG_SAVE_ERRORS as e:
59
+ # A value YAML cannot represent — not an OSError, so the arm above never
60
+ # saw it and it escaped as a traceback. See config.CONFIG_SAVE_ERRORS.
61
+ click.echo(
62
+ click.style(
63
+ f"Error: could not serialize the config to YAML: {e}\n"
64
+ "Nothing was changed — the previous config is intact. "
65
+ "This is a bug: please report the field that failed.",
66
+ fg="red",
67
+ )
68
+ )
69
+ raise SystemExit(1) from None
70
+
71
+
32
72
  def _check_budget_or_exit(max_budget: float | None) -> None:
33
73
  """Reject a negative spend cap before it reaches `AgentConfig` (ge=0).
34
74
 
@@ -75,6 +115,15 @@ def _load_or_exit() -> DispatchConfig:
75
115
  click.echo(click.style(f"Error: config at {config_path()} is not valid YAML:", fg="red"))
76
116
  click.echo(str(e))
77
117
  raise SystemExit(1) from None
118
+ except UnicodeDecodeError as e:
119
+ click.echo(
120
+ click.style(
121
+ f"Error: config at {config_path()} is not valid UTF-8 (re-save it as UTF-8):",
122
+ fg="red",
123
+ )
124
+ )
125
+ click.echo(str(e))
126
+ raise SystemExit(1) from None
78
127
  except OSError as e:
79
128
  click.echo(click.style(f"Error: config at {config_path()} could not be read:", fg="red"))
80
129
  click.echo(str(e))
@@ -236,7 +285,7 @@ def add(
236
285
  if warning := check_permission_mode(permission_mode):
237
286
  click.echo(click.style(f"Warning: {warning}", fg="yellow"))
238
287
 
239
- save_config(config)
288
+ _save_or_exit(config)
240
289
  click.echo(f"Added agent '{name}' -> {dir_path}")
241
290
 
242
291
 
@@ -251,7 +300,7 @@ def remove(name: str) -> None:
251
300
  raise SystemExit(1)
252
301
 
253
302
  del config.agents[name]
254
- save_config(config)
303
+ _save_or_exit(config)
255
304
  click.echo(f"Removed agent '{name}'.")
256
305
 
257
306
 
@@ -381,7 +430,7 @@ def update(
381
430
  click.echo("Nothing to update. Pass at least one option (see --help).")
382
431
  raise SystemExit(1)
383
432
 
384
- save_config(config)
433
+ _save_or_exit(config)
385
434
  click.echo(f"Updated agent '{name}': {', '.join(updated)}")
386
435
 
387
436
 
@@ -639,7 +688,7 @@ def group_add(name: str, description: str, shared_context: str, members: tuple[s
639
688
  shared_context=shared_context,
640
689
  members=member_objs,
641
690
  )
642
- save_config(config)
691
+ _save_or_exit(config)
643
692
  click.echo(f"Added group '{name}' ({len(member_objs)} member(s)).")
644
693
  if not member_objs:
645
694
  click.echo(
@@ -737,7 +786,7 @@ def group_update(name: str, description: str | None, shared_context: str | None)
737
786
  click.echo("Nothing to update. Pass --description and/or --shared-context.")
738
787
  raise SystemExit(1)
739
788
 
740
- save_config(config)
789
+ _save_or_exit(config)
741
790
  click.echo(f"Updated group '{name}': {', '.join(updated)}")
742
791
 
743
792
 
@@ -752,7 +801,7 @@ def group_remove(name: str) -> None:
752
801
  raise SystemExit(1)
753
802
 
754
803
  del config.groups[name]
755
- save_config(config)
804
+ _save_or_exit(config)
756
805
  click.echo(f"Removed group '{name}'.")
757
806
 
758
807
 
@@ -807,6 +856,15 @@ def doctor() -> None:
807
856
  except yaml.YAMLError as e:
808
857
  fail(f"Config not valid YAML: {cp}")
809
858
  click.echo(f" {e}")
859
+ except UnicodeDecodeError as e:
860
+ # A ValueError, not an OSError — it slipped past both handlers above
861
+ # and crashed the one command meant to diagnose a broken config.
862
+ fail(f"Config is not valid UTF-8: {cp}")
863
+ click.echo(f" {e}")
864
+ click.echo(" Re-save the file as UTF-8.")
865
+ except OSError as e:
866
+ fail(f"Config could not be read: {cp}")
867
+ click.echo(f" {e}")
810
868
 
811
869
  section("MCP registration")
812
870
  if claude_path is None:
@@ -13,6 +13,7 @@ from collections.abc import Iterator
13
13
  from pathlib import Path
14
14
 
15
15
  import yaml
16
+ from pydantic import ValidationError
16
17
 
17
18
  from .models import DispatchConfig
18
19
 
@@ -21,6 +22,19 @@ try: # pragma: no cover - platform dependent
21
22
  except ImportError: # pragma: no cover - Windows has no fcntl
22
23
  fcntl = None # type: ignore[assignment]
23
24
 
25
+ # Parse with libyaml when PyYAML was built against it. Every MCP tool call
26
+ # reloads agents.yaml from scratch (deliberately — that is how a new agent is
27
+ # picked up without a restart), so this parse is on the hot path of all 21
28
+ # tools. On a real 38 KB config the pure-Python SafeLoader takes ~9.8 ms and
29
+ # CSafeLoader ~0.6 ms: a 16x saving repeated on every single call, and it runs
30
+ # on the event-loop thread where it blocks every other tool. The fallback is
31
+ # mandatory — a PyYAML installed from source without libyaml headers has no
32
+ # C extension. Same semantics either way: both are the *safe* loader.
33
+ try: # pragma: no cover - depends on how PyYAML was built
34
+ from yaml import CSafeLoader as _YamlLoader
35
+ except ImportError: # pragma: no cover - pure-Python PyYAML
36
+ from yaml import SafeLoader as _YamlLoader # type: ignore[assignment]
37
+
24
38
  logger = logging.getLogger(__name__)
25
39
 
26
40
  DEFAULT_CONFIG_DIR = Path.home() / ".config" / "agent-dispatch"
@@ -150,12 +164,36 @@ def file_lock(path: Path) -> Iterator[None]:
150
164
  _release_lock(fd)
151
165
 
152
166
 
167
+ # Everything load_config() can raise for a config a human can plausibly produce.
168
+ # Listed once because three surfaces handle it independently (server._get_config,
169
+ # cli._load_or_exit, cli.doctor) and they must not drift: UnicodeDecodeError, in
170
+ # particular, is a ValueError — NOT an OSError — so a file saved in cp1251 or
171
+ # UTF-16 used to escape every one of them as a raw traceback.
172
+ CONFIG_LOAD_ERRORS = (ValidationError, yaml.YAMLError, UnicodeDecodeError, OSError)
173
+
174
+ # The *write* half, and deliberately not a copy of the read half. `save_config`
175
+ # can fail two ways with two different remediations: OSError (full disk,
176
+ # read-only volume — handled separately, and the atomic rename means the old
177
+ # file survives), or a rendering failure. `yaml.dump` raises RepresenterError —
178
+ # a `yaml.YAMLError`, so neither an OSError nor a ValueError — for any value it
179
+ # cannot represent. That is unreachable today because save_config only ever
180
+ # feeds it `model_dump(mode="json")` output, i.e. JSON-native types. It becomes
181
+ # reachable the moment a field lands whose JSON dump is not one of those, and it
182
+ # would then escape *both* server guards and the CLI's `_save_or_exit` as a raw
183
+ # traceback — the exact "an exception type escapes the handler meant to catch
184
+ # it" class this codebase has been bitten by twice. Declared once, next to its
185
+ # read-side twin, so the two surfaces that handle it cannot drift.
186
+ CONFIG_SAVE_ERRORS = (yaml.YAMLError,)
187
+
188
+
153
189
  def load_config(path: Path | None = None) -> DispatchConfig:
154
190
  """Load config from YAML file. Returns empty config if file missing."""
155
191
  p = path or config_path()
156
192
  if not p.exists():
157
193
  return DispatchConfig()
158
- raw = yaml.safe_load(p.read_text(encoding="utf-8"))
194
+ # yaml.load with the safe loader == yaml.safe_load; _YamlLoader is the C
195
+ # one when available (see its definition for why this is worth doing).
196
+ raw = yaml.load(p.read_text(encoding="utf-8"), Loader=_YamlLoader) # noqa: S506
159
197
  if raw is None:
160
198
  return DispatchConfig()
161
199
  return DispatchConfig.model_validate(raw)