agent-dispatch 0.12.0__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/workflows/publish.yml +8 -1
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/AGENTS.md +13 -4
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/CHANGELOG.md +130 -1
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/PKG-INFO +32 -7
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/README.md +30 -5
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/agents.example.yaml +6 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/pyproject.toml +1 -1
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/__init__.py +1 -1
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/cli.py +65 -7
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/config.py +39 -1
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/jobs.py +40 -8
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/models.py +10 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/runner.py +28 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/server.py +297 -143
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_cli.py +83 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_config.py +56 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_jobs.py +33 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_runner.py +70 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_server.py +348 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/dependabot.yml +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.github/workflows/ci.yml +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/.gitignore +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/LICENSE +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/SECURITY.md +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/assets/mascot.png +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/src/agent_dispatch/cache.py +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/__init__.py +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/conftest.py +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_cache.py +0 -0
- {agent_dispatch-0.12.0 → agent_dispatch-0.13.0}/tests/test_models.py +0 -0
|
@@ -26,5 +26,12 @@ jobs:
|
|
|
26
26
|
pip install build
|
|
27
27
|
python -m build
|
|
28
28
|
|
|
29
|
+
# Keep this current with the build backend. The pinned action bundles its
|
|
30
|
+
# own twine, and twine rejects a Metadata-Version it does not know: the
|
|
31
|
+
# v0.13.0 publish failed because hatchling 1.32 emits `Metadata-Version:
|
|
32
|
+
# 2.5` while the previous pin (Feb 2026) shipped twine 6.1.0 / packaging
|
|
33
|
+
# 25.0. `python -m build` resolves the newest hatchling at build time, so
|
|
34
|
+
# this drifts on its own with no repo change — see the release checklist's
|
|
35
|
+
# `twine check` step, which catches it before a tag is cut.
|
|
29
36
|
- name: Publish to PyPI
|
|
30
|
-
uses: pypa/gh-action-pypi-publish@
|
|
37
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -28,7 +28,7 @@ pip install -e ".[dev]"
|
|
|
28
28
|
|
|
29
29
|
```bash
|
|
30
30
|
ruff check src/ tests/
|
|
31
|
-
python3 -m pytest tests/ -v #
|
|
31
|
+
python3 -m pytest tests/ -v # 578 tests, ~5s
|
|
32
32
|
```
|
|
33
33
|
|
|
34
34
|
Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.which` + `subprocess.run`/`Popen`; server tests mock `_get_config` + `runner.dispatch`. The one exception is `TestStreamPipeHandling`, which spawns a short-lived *python* subprocess: a pipe deadlock lives in the OS pipe buffer, so a mocked `Popen` structurally cannot reproduce it.
|
|
@@ -52,16 +52,25 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
52
52
|
- **Never `close()` a pipe another thread may still be reading.** `close()` waits on the reader's buffer lock with *no timeout*, so it would hang the dispatch forever — the bounded `join()` before it buys nothing. `dispatch_stream` skips the stderr close while the drain thread is alive and lets the daemon reader + Popen finalizer release the fd. This is not hypothetical: a stdio MCP server inherits the child's `stderr`, so the pipe often has no EOF even after `claude` exits cleanly.
|
|
53
53
|
- **No `await` inside `config_lock()`.** `ProcessLock`'s in-process guard is a `threading.RLock` — re-entrant per *thread* — and every MCP tool coroutine runs on the one event-loop thread. Suspending in the critical section lets a second coroutine re-enter the "held" lock and interleave its own load/mutate/save. Collect warnings as data, emit them after the `with` block (`test_no_await_inside_the_config_lock` enforces this by AST).
|
|
54
54
|
- Cross-process locks are acquired with a **bounded** wait, never a blocking `flock`: the server takes them on its event-loop thread, so a wedged holder would freeze every tool. After the deadline it proceeds unlocked and logs — a possible lost update beats a permanent freeze.
|
|
55
|
-
- `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore.
|
|
55
|
+
- `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore. For a *running* job, `started_at` alone does **not** prove abandonment — a dispatch may legitimately run to the 7200s timeout ceiling, so the file's own mtime is checked too (a live worker rewrites it on every progress flush). That check can only ever *skip* a recovery, never add one.
|
|
56
|
+
- Deleting job records is **opt-in** (`settings.job_retention_days`, default `0` = off) and only ever happens at server start, never inside a tool. They are the user's dispatch history and the deletion is irreversible, so an unreadable config is treated as "do nothing" rather than falling back to a default retention.
|
|
57
|
+
- `max_concurrency` must bound *subprocesses*, not coroutines. Cancelling a coroutine does not stop the thread behind `asyncio.to_thread`, so `async with sem:` gave the slot away while `claude` kept running and billing. Dispatches go through `_dispatch_guarded`, which releases from the future's done-callback and `shield`s the await. Never "simplify" it back to `async with`.
|
|
58
|
+
- Tool responses are serialized through `server._dumps`, never bare `json.dumps`: the stdlib default (`ensure_ascii=True`) turns every non-ASCII character into a `\uXXXX` escape, tripling the bytes and tokenizing badly, for no gain — the stdio transport emits raw UTF-8 via pydantic anyway. A real `list_groups()` carried 8520 escapes and weighed 59 KB instead of 25 KB.
|
|
59
|
+
- `agents.yaml` is parsed with libyaml's `CSafeLoader` when available (`config._YamlLoader`), falling back to the pure-Python **safe** loader — never `yaml.Loader`. The config is re-read on every tool call, so this parse is on the hot path of all 21 tools *and* blocks the event-loop thread: 9.80 ms → 0.76 ms on a real 38 KB config.
|
|
56
60
|
- Pydantic does **not** validate on assignment. `Field(ge=...)` guards only the *load* path; every mutation surface (CLI `add`/`update`, MCP `add_agent`/`update_agent`) needs its own boundary check, or the bound escapes as a raw `ValidationError`.
|
|
57
61
|
- Every state file (`agents.yaml`, job files) is written **temp file + `os.replace`**, never in place, and every load/mutate/save is wrapped in `config.ProcessLock` — the CLI and the MCP server are separate processes writing the same files, so a thread lock alone loses updates.
|
|
58
62
|
- Anything that changes an agent's config must call `_invalidate_agent_cache` — the cache key holds the agent *name*, not its directory or permissions.
|
|
59
63
|
- Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, and `budget_exceeded` results, so the documented "grant access, then re-dispatch" recovery is never short-circuited.
|
|
60
64
|
- Remediation text is a contract: a hint that names a flag must name one that exists (`test_printed_budget_hint_is_a_runnable_command` feeds the printed flags back into the CLI). Run the command you print.
|
|
61
|
-
-
|
|
65
|
+
- The config error sets are declared **once** and in two halves: `config.CONFIG_LOAD_ERRORS` (read) and `config.CONFIG_SAVE_ERRORS` (write — `yaml.dump`'s `RepresenterError` is a `yaml.YAMLError`, therefore neither `OSError` nor `ConfigLoadError`, and used to escape both the MCP guard and the CLI's `_save_or_exit`). Two halves, not one set, because the remediations differ: a failed write is atomic so the old config survives, while a failed read needs the YAML fixed.
|
|
66
|
+
- MCP tools that load config carry `@_config_guard` under `@mcp.tool()` so a broken `agents.yaml` — or a failed *write* — returns the `{"error": ...}` envelope instead of a raw traceback. The set of load errors lives in one place (`config.CONFIG_LOAD_ERRORS`) because three surfaces handle it: **`UnicodeDecodeError` is a `ValueError`, not an `OSError`**, and listing types per-site is exactly how a cp1251 config slipped past all three.
|
|
62
67
|
|
|
63
68
|
- Tests must not touch anything outside `tmp_path`. `test_server.py`'s autouse `_reset_globals` and `test_cli.py`'s `_isolated_config` redirect **both** `AGENT_DISPATCH_CONFIG` and `AGENT_DISPATCH_JOBS_DIR`: a mutation tool that bails out early (unknown agent) still takes `config_lock()` first, which would otherwise create a lock file beside the developer's real config.
|
|
64
69
|
|
|
70
|
+
- `mcp` is pinned **`>=1.2.0,<2`** deliberately: 2.0 removed `mcp.server.fastmcp`, which `server.py` imports, so an unbounded range gives every fresh install a dead `agent-dispatch serve`. Lifting the cap means porting to `mcp.server.mcpserver.MCPServer` — it is not a dependency bump.
|
|
71
|
+
- Verify packaging in a **clean venv**, never the dev machine: build the wheel, install it fresh, import the server. A stale pin in local site-packages hides exactly the failure a new user hits first.
|
|
72
|
+
- Run **`twine check dist/*`** before cutting the tag, with a *current* twine. The clean-venv check does not cover this: it proves the wheel installs, not that PyPI's uploader will accept its metadata. `python -m build` pulls the newest hatchling at build time, so the emitted `Metadata-Version` climbs on its own, and `pypa/gh-action-pypi-publish` is pinned to a SHA that bundles a fixed twine — v0.13.0's publish failed on exactly that mismatch (hatchling 1.32 → metadata 2.5; pinned twine 6.1.0 → "not a valid metadata version"). When it happens, bump the action pin rather than pinning the backend down.
|
|
73
|
+
|
|
65
74
|
## Deliberately not built
|
|
66
75
|
|
|
67
76
|
These were considered — some fully implemented — and cut on purpose: an agent router / auto-dispatch (`recommend_agent` / `dispatch_auto`, removed before 0.8.0 — a keyword scorer adds little over the calling LLM at a handful of agents, and auto-dispatch can spend money or mutate a repo on a guess); groups as an execution engine (they are a descriptive layer — no routing, no per-group settings); an agent-dispatch-side budget ledger across dispatches (the CLI's own `--max-budget-usd` covers a single run; anything cumulative would need state we deliberately don't keep). Please open an issue with the use case before adding any of them.
|
|
@@ -76,4 +85,4 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
|
|
|
76
85
|
|
|
77
86
|
## More detail
|
|
78
87
|
|
|
79
|
-
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`,
|
|
88
|
+
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 578 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
|
|
@@ -7,6 +7,134 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.13.0] - 2026-08-13
|
|
11
|
+
|
|
12
|
+
An efficiency round, measured against a real 38 KB config (4 agents, 6 groups),
|
|
13
|
+
a 294-record jobs directory and 14 concurrently running servers — not against
|
|
14
|
+
synthetic fixtures. Findings came from a five-axis audit (caller context,
|
|
15
|
+
latency/IO, spend, CPU, concurrency) whose every claim was re-measured
|
|
16
|
+
adversarially before implementation; most were confirmed as real but too small
|
|
17
|
+
to matter next to a 10–120 s subprocess and were deliberately left alone.
|
|
18
|
+
|
|
19
|
+
### Changed
|
|
20
|
+
- **Tool responses no longer escape non-ASCII, cutting discovery payloads
|
|
21
|
+
roughly in half.** `json.dumps` defaults to `ensure_ascii=True`, so every
|
|
22
|
+
Cyrillic character left as a `\uXXXX` escape — 6 bytes for what UTF-8 stores
|
|
23
|
+
in 2, and a far worse token sequence. One `list_groups()` carried **8520**
|
|
24
|
+
such escapes: 59 KB where 25 KB was enough. Measured over a full discovery
|
|
25
|
+
pass (`list_agents` + `list_groups` + `inspect_group` + `inspect_agent`):
|
|
26
|
+
104 KB → 51 KB, about **13 000 tokens saved per pass**, on every call, in
|
|
27
|
+
every session. Nothing is lost: the stdio transport serializes the JSON-RPC
|
|
28
|
+
envelope with pydantic's `model_dump_json`, which already emits raw UTF-8 —
|
|
29
|
+
which is also why the `dispatch` family never had this problem while every
|
|
30
|
+
`json.dumps` tool did. All 57 call sites now go through one `_dumps` helper
|
|
31
|
+
so the two agree. `_dumps` probes encodability and falls back to the escaped
|
|
32
|
+
form for the one payload that needs it — a **lone surrogate**, which an agent
|
|
33
|
+
produces by printing a literal `\udXXX` escape in its JSON output (`json.loads`
|
|
34
|
+
manufactures it from perfectly ASCII input, so the subprocess-level
|
|
35
|
+
`errors="replace"` cannot help). Unescaped it would raise `UnicodeEncodeError`
|
|
36
|
+
inside the stdio transport *after* the tool returned — past every guard.
|
|
37
|
+
The saving is proportional to how much non-ASCII your config and results
|
|
38
|
+
carry: on an all-ASCII setup it is exactly zero.
|
|
39
|
+
- **`agents.yaml` parses ~13x faster.** The config is re-read on every single
|
|
40
|
+
tool call (deliberately — that is how a new agent is picked up without a
|
|
41
|
+
restart), so the parse sits on the hot path of all 21 tools *and* runs on the
|
|
42
|
+
event-loop thread, where it blocks every other tool. Switching to libyaml's
|
|
43
|
+
`CSafeLoader` takes the real config from **9.80 ms to 0.76 ms** per call.
|
|
44
|
+
Falls back to the pure-Python safe loader when PyYAML was built without
|
|
45
|
+
libyaml; both are the *safe* loader, so a config still cannot construct
|
|
46
|
+
arbitrary objects.
|
|
47
|
+
|
|
48
|
+
### Added
|
|
49
|
+
- **`settings.job_retention_days`** — when > 0, terminal job records older than
|
|
50
|
+
N days are deleted at server start. Every async dispatch and every
|
|
51
|
+
`return_ref` dispatch leaves a job file that nothing removed on its own, so
|
|
52
|
+
the directory grew forever while `dispatch_jobs` and stale-job recovery
|
|
53
|
+
parsed every file in it. **Defaults to `0` (off)**: those records are the
|
|
54
|
+
user's own dispatch history and deleting them is irreversible, so it is an
|
|
55
|
+
explicit opt-in rather than something a version bump starts doing to an
|
|
56
|
+
existing install. An unreadable config is treated as `0`, never guessed.
|
|
57
|
+
|
|
58
|
+
### Fixed
|
|
59
|
+
- **`max_concurrency` bounds live subprocesses again, not live coroutines.**
|
|
60
|
+
`async with sem:` released the slot when the *coroutine* unwound, but
|
|
61
|
+
cancelling a coroutine does not stop the thread behind `asyncio.to_thread` —
|
|
62
|
+
so an interrupted turn or a closed session handed the slot to the next
|
|
63
|
+
dispatch while the `claude` subprocess kept running and kept being billed. N
|
|
64
|
+
cancellations meant up to N extra concurrent subprocesses, and each abandoned
|
|
65
|
+
worker also held a thread out of the default executor (`min(32, cpu_count+4)`,
|
|
66
|
+
which `to_thread` needs for *every* dispatch) for the rest of its timeout —
|
|
67
|
+
up to 7200s. The slot is now tied to the worker's real lifetime via
|
|
68
|
+
`_dispatch_guarded`; cancellation still reaches the caller immediately.
|
|
69
|
+
- **A legitimately long dispatch is no longer flipped to `failed` by an
|
|
70
|
+
unrelated server starting up.** `recover_stale` judged abandonment from
|
|
71
|
+
`started_at` alone, but a dispatch may legitimately run up to the 7200s
|
|
72
|
+
timeout ceiling — so any of the (routinely 14+) other servers booting would
|
|
73
|
+
mark a live 2-hour job `failed`, after which `finish()` refuses the real,
|
|
74
|
+
already-paid-for result. A *running* job whose file was modified within the
|
|
75
|
+
threshold is now left alone: a live worker rewrites it on every progress
|
|
76
|
+
flush, so only the file's own age proves abandonment. This can only ever skip
|
|
77
|
+
a recovery — a genuinely abandoned job is picked up on a later start.
|
|
78
|
+
- **A config that cannot be *rendered* now returns an error envelope instead of
|
|
79
|
+
a traceback.** `_get_config` wraps read-side YAML failures into
|
|
80
|
+
`ConfigLoadError`, but `save_config` → `yaml.dump` raises `RepresenterError` —
|
|
81
|
+
a `yaml.YAMLError`, so neither an `OSError` nor a `ConfigLoadError`. It
|
|
82
|
+
therefore slipped past *both* arms of the MCP guard on
|
|
83
|
+
`add_agent`/`update_agent`/`remove_agent` **and** past the CLI's
|
|
84
|
+
`_save_or_exit`. Unreachable today (only JSON-native types are ever dumped)
|
|
85
|
+
and closed now because it is the same "an exception type escapes the handler
|
|
86
|
+
meant to catch it" class that produced the 0.12.0 and 0.12.1 rounds. The set
|
|
87
|
+
is declared once as `config.CONFIG_SAVE_ERRORS`, next to its read-side twin,
|
|
88
|
+
so the two surfaces cannot drift.
|
|
89
|
+
- **Stale-job recovery scans the jobs directory once instead of twice.** It
|
|
90
|
+
called `list()` per status, and `list()` reads and parses every file in the
|
|
91
|
+
directory (~85 ms at 294 records) — at the start of every server process.
|
|
92
|
+
|
|
93
|
+
## [0.12.1] - 2026-07-30
|
|
94
|
+
|
|
95
|
+
Two holes in 0.12.0's own "tools always return a clean error" fix.
|
|
96
|
+
|
|
97
|
+
### Fixed
|
|
98
|
+
- **A non-UTF-8 `agents.yaml` no longer crashes every surface.** `UnicodeDecodeError`
|
|
99
|
+
is a `ValueError`, not an `OSError`, so a config saved as cp1251/UTF-16 slipped
|
|
100
|
+
past all three handlers added in 0.12.0 and raised a bare traceback out of every
|
|
101
|
+
MCP tool, `agent-dispatch list` — and `doctor`, the command you run *because*
|
|
102
|
+
the config is broken. The set of load errors is now declared once
|
|
103
|
+
(`config.CONFIG_LOAD_ERRORS`) and shared by all three, and `doctor` reports the
|
|
104
|
+
encoding (and an unreadable file) as a normal FAIL with a fix. The shared set
|
|
105
|
+
lives in `config.CONFIG_LOAD_ERRORS`; the CLI still branches per type because
|
|
106
|
+
each one deserves a different remediation line.
|
|
107
|
+
- **An undecodable byte from the `claude` CLI no longer kills a paid-for
|
|
108
|
+
dispatch.** `text=True` decodes strictly, so one invalid byte on stdout raised
|
|
109
|
+
`UnicodeDecodeError` — a `ValueError`, caught by nothing in the runner — out of
|
|
110
|
+
`dispatch`, `dispatch_stream`, the MCP tools and `agent-dispatch test`. Both
|
|
111
|
+
spawn sites now decode with `errors="replace"`, so a mangled byte becomes
|
|
112
|
+
U+FFFD instead of discarding the run. (Present since the initial commit.)
|
|
113
|
+
- **`dispatch` classifies spawn failures like `dispatch_stream` already did.**
|
|
114
|
+
A directory that passes `is_dir()` but cannot be entered — or that vanishes
|
|
115
|
+
between the check and the spawn — made `subprocess.run` raise straight through
|
|
116
|
+
the `except subprocess.TimeoutExpired`. It now returns `not_found` /
|
|
117
|
+
`permission` / `cli_error` like the streaming path.
|
|
118
|
+
- **The six job tools got the I/O envelope too.** `dispatch_status`, `_wait`,
|
|
119
|
+
`_cancel`, `_jobs`, `fetch_result` and `dispatch_gc` never load config, so they
|
|
120
|
+
carried no guard — an unwritable or misconfigured jobs directory raised out of
|
|
121
|
+
them. The guard is now split (`_io_guard` / `_config_guard`) and both halves
|
|
122
|
+
are applied where each is needed.
|
|
123
|
+
- **`add_agent` reports an unresolvable path** (`~unknown-user`) as an error
|
|
124
|
+
envelope instead of raising `RuntimeError`.
|
|
125
|
+
- **A failed config *write* is reported, not raised.** The same contract had the
|
|
126
|
+
other half missing: a full disk or a read-only volume made `save_config`
|
|
127
|
+
raise OSError straight out of `add_agent`/`update_agent`/`remove_agent` and out
|
|
128
|
+
of the matching CLI commands. Both surfaces now report it — the CLI adds that
|
|
129
|
+
the previous config is intact, which the atomic write guarantees.
|
|
130
|
+
|
|
131
|
+
### Changed
|
|
132
|
+
- Docs corrected against the code: stale-job recovery now documents the
|
|
133
|
+
`pending` sweep and its 24h threshold; the config-lock guarantee is stated as
|
|
134
|
+
best-effort (it proceeds unlocked after 10s rather than freezing the server);
|
|
135
|
+
`add_agent(timeout=0)` documents that it stores the literal 300, not
|
|
136
|
+
`settings.default_timeout`.
|
|
137
|
+
|
|
10
138
|
## [0.12.0] - 2026-07-29
|
|
11
139
|
|
|
12
140
|
Reliability pass over the streaming path, config durability, and the tool
|
|
@@ -546,7 +674,8 @@ cache bounding, and stale-job recovery.
|
|
|
546
674
|
- Dependabot for `pip` + `github-actions`, GitHub Actions pinned to
|
|
547
675
|
commit SHAs for supply-chain integrity.
|
|
548
676
|
|
|
549
|
-
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.12.
|
|
677
|
+
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.12.1...HEAD
|
|
678
|
+
[0.12.1]: https://github.com/ginkida/agent-dispatch/compare/v0.12.0...v0.12.1
|
|
550
679
|
[0.12.0]: https://github.com/ginkida/agent-dispatch/compare/v0.11.0...v0.12.0
|
|
551
680
|
[0.11.0]: https://github.com/ginkida/agent-dispatch/compare/v0.10.0...v0.11.0
|
|
552
681
|
[0.10.0]: https://github.com/ginkida/agent-dispatch/compare/v0.9.0...v0.10.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: agent-dispatch
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
|
|
5
5
|
Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
|
|
6
6
|
Project-URL: Repository, https://github.com/ginkida/agent-dispatch
|
|
@@ -388,8 +388,8 @@ Register a new project directory as an agent. Description is auto-generated from
|
|
|
388
388
|
| `name` | string | yes | Agent name (letters, digits, hyphens, underscores) |
|
|
389
389
|
| `directory` | string | yes | Path to an existing project directory (`~` is expanded, relative paths resolved) |
|
|
390
390
|
| `description` | string | no | What this agent can do — auto-generated if empty |
|
|
391
|
-
| `timeout` | int | no | Timeout in seconds (0 =
|
|
392
|
-
| `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = no
|
|
391
|
+
| `timeout` | int | no | Timeout in seconds (0 = 300; this is a literal default, not `settings.default_timeout`) |
|
|
392
|
+
| `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = inherit `settings.default_max_budget_usd`; no cap only when that is unset too) |
|
|
393
393
|
| `permission_mode` | string | no | Permission mode (e.g. `default`, `plan`, `bypassPermissions`) |
|
|
394
394
|
| `allowed_tools` | string | no | Comma-separated allowed tools (e.g. `"Bash,Read,Edit"`) |
|
|
395
395
|
| `disallowed_tools` | string | no | Comma-separated disallowed tools |
|
|
@@ -405,7 +405,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
|
|
|
405
405
|
| `name` | string | yes | Agent name to update |
|
|
406
406
|
| `description` | string | no | New description |
|
|
407
407
|
| `timeout` | int | no | New timeout (0 = don't change) |
|
|
408
|
-
| `max_budget_usd` | float | no | New budget limit (0 = don't change
|
|
408
|
+
| `max_budget_usd` | float | no | New budget limit (0 = don't change; negative clears the *per-agent* cap, after which `settings.default_max_budget_usd` applies) |
|
|
409
409
|
| `model` | string | no | Model override. `"none"` to clear |
|
|
410
410
|
| `permission_mode` | string | no | Permission mode. `"none"` to clear |
|
|
411
411
|
| `allowed_tools` | string | no | Comma-separated. `"none"` to clear |
|
|
@@ -474,7 +474,7 @@ Async workers run with streaming under the hood: the job file keeps a rolling ta
|
|
|
474
474
|
|
|
475
475
|
`dispatch_jobs(status?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted.
|
|
476
476
|
|
|
477
|
-
Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server
|
|
477
|
+
Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server recovers jobs a crashed instance abandoned: `running` ones stuck over an hour, and `pending` ones over 24 hours, are marked `failed` so they stop being polled forever and become collectable by `dispatch_gc`. (The `pending` threshold is deliberately long — the jobs directory is shared by every running server, so a job queued behind another server's concurrency limit must not be swept.)
|
|
478
478
|
|
|
479
479
|
| When to use async | When to use `dispatch` |
|
|
480
480
|
|-------------------|------------------------|
|
|
@@ -574,6 +574,7 @@ settings:
|
|
|
574
574
|
# - Edit
|
|
575
575
|
max_dispatch_depth: 3 # recursion protection
|
|
576
576
|
max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
|
|
577
|
+
# job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
|
|
577
578
|
cache:
|
|
578
579
|
enabled: true
|
|
579
580
|
ttl: 300 # seconds
|
|
@@ -582,6 +583,30 @@ settings:
|
|
|
582
583
|
|
|
583
584
|
Config is reloaded on every tool call — add agents without restarting.
|
|
584
585
|
|
|
586
|
+
### Job retention
|
|
587
|
+
|
|
588
|
+
Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
|
|
589
|
+
record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
|
|
590
|
+
`dispatch_gc` has to be run by hand. The directory therefore grows without
|
|
591
|
+
bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
|
|
592
|
+
start read and parse *every* file in it.
|
|
593
|
+
|
|
594
|
+
Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
|
|
595
|
+
than N days when a server starts:
|
|
596
|
+
|
|
597
|
+
```yaml
|
|
598
|
+
settings:
|
|
599
|
+
job_retention_days: 30
|
|
600
|
+
```
|
|
601
|
+
|
|
602
|
+
It defaults to `0` — **off** — because those records are your own history of
|
|
603
|
+
past dispatches and deleting them cannot be undone. Pending and running jobs are
|
|
604
|
+
never touched.
|
|
605
|
+
|
|
606
|
+
`agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
|
|
607
|
+
one-off. Both *delete* immediately and report the count; neither previews, so
|
|
608
|
+
check what is there first with `agent-dispatch jobs`.
|
|
609
|
+
|
|
585
610
|
### Auto-Description
|
|
586
611
|
|
|
587
612
|
`agent-dispatch add` without `--description` generates one from:
|
|
@@ -635,7 +660,7 @@ agent-dispatch MCP server
|
|
|
635
660
|
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
|
|
636
661
|
- **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
|
|
637
662
|
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
638
|
-
- **Durable config** — `agents.yaml` is written atomically (temp file + rename),
|
|
663
|
+
- **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
|
|
639
664
|
|
|
640
665
|
See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
|
|
641
666
|
|
|
@@ -358,8 +358,8 @@ Register a new project directory as an agent. Description is auto-generated from
|
|
|
358
358
|
| `name` | string | yes | Agent name (letters, digits, hyphens, underscores) |
|
|
359
359
|
| `directory` | string | yes | Path to an existing project directory (`~` is expanded, relative paths resolved) |
|
|
360
360
|
| `description` | string | no | What this agent can do — auto-generated if empty |
|
|
361
|
-
| `timeout` | int | no | Timeout in seconds (0 =
|
|
362
|
-
| `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = no
|
|
361
|
+
| `timeout` | int | no | Timeout in seconds (0 = 300; this is a literal default, not `settings.default_timeout`) |
|
|
362
|
+
| `max_budget_usd` | float | no | Max cost in USD per dispatch (0 = inherit `settings.default_max_budget_usd`; no cap only when that is unset too) |
|
|
363
363
|
| `permission_mode` | string | no | Permission mode (e.g. `default`, `plan`, `bypassPermissions`) |
|
|
364
364
|
| `allowed_tools` | string | no | Comma-separated allowed tools (e.g. `"Bash,Read,Edit"`) |
|
|
365
365
|
| `disallowed_tools` | string | no | Comma-separated disallowed tools |
|
|
@@ -375,7 +375,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
|
|
|
375
375
|
| `name` | string | yes | Agent name to update |
|
|
376
376
|
| `description` | string | no | New description |
|
|
377
377
|
| `timeout` | int | no | New timeout (0 = don't change) |
|
|
378
|
-
| `max_budget_usd` | float | no | New budget limit (0 = don't change
|
|
378
|
+
| `max_budget_usd` | float | no | New budget limit (0 = don't change; negative clears the *per-agent* cap, after which `settings.default_max_budget_usd` applies) |
|
|
379
379
|
| `model` | string | no | Model override. `"none"` to clear |
|
|
380
380
|
| `permission_mode` | string | no | Permission mode. `"none"` to clear |
|
|
381
381
|
| `allowed_tools` | string | no | Comma-separated. `"none"` to clear |
|
|
@@ -444,7 +444,7 @@ Async workers run with streaming under the hood: the job file keeps a rolling ta
|
|
|
444
444
|
|
|
445
445
|
`dispatch_jobs(status?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted.
|
|
446
446
|
|
|
447
|
-
Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server
|
|
447
|
+
Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server recovers jobs a crashed instance abandoned: `running` ones stuck over an hour, and `pending` ones over 24 hours, are marked `failed` so they stop being polled forever and become collectable by `dispatch_gc`. (The `pending` threshold is deliberately long — the jobs directory is shared by every running server, so a job queued behind another server's concurrency limit must not be swept.)
|
|
448
448
|
|
|
449
449
|
| When to use async | When to use `dispatch` |
|
|
450
450
|
|-------------------|------------------------|
|
|
@@ -544,6 +544,7 @@ settings:
|
|
|
544
544
|
# - Edit
|
|
545
545
|
max_dispatch_depth: 3 # recursion protection
|
|
546
546
|
max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
|
|
547
|
+
# job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
|
|
547
548
|
cache:
|
|
548
549
|
enabled: true
|
|
549
550
|
ttl: 300 # seconds
|
|
@@ -552,6 +553,30 @@ settings:
|
|
|
552
553
|
|
|
553
554
|
Config is reloaded on every tool call — add agents without restarting.
|
|
554
555
|
|
|
556
|
+
### Job retention
|
|
557
|
+
|
|
558
|
+
Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
|
|
559
|
+
record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
|
|
560
|
+
`dispatch_gc` has to be run by hand. The directory therefore grows without
|
|
561
|
+
bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
|
|
562
|
+
start read and parse *every* file in it.
|
|
563
|
+
|
|
564
|
+
Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
|
|
565
|
+
than N days when a server starts:
|
|
566
|
+
|
|
567
|
+
```yaml
|
|
568
|
+
settings:
|
|
569
|
+
job_retention_days: 30
|
|
570
|
+
```
|
|
571
|
+
|
|
572
|
+
It defaults to `0` — **off** — because those records are your own history of
|
|
573
|
+
past dispatches and deleting them cannot be undone. Pending and running jobs are
|
|
574
|
+
never touched.
|
|
575
|
+
|
|
576
|
+
`agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
|
|
577
|
+
one-off. Both *delete* immediately and report the count; neither previews, so
|
|
578
|
+
check what is there first with `agent-dispatch jobs`.
|
|
579
|
+
|
|
555
580
|
### Auto-Description
|
|
556
581
|
|
|
557
582
|
`agent-dispatch add` without `--description` generates one from:
|
|
@@ -605,7 +630,7 @@ agent-dispatch MCP server
|
|
|
605
630
|
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
|
|
606
631
|
- **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
|
|
607
632
|
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
608
|
-
- **Durable config** — `agents.yaml` is written atomically (temp file + rename),
|
|
633
|
+
- **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
|
|
609
634
|
|
|
610
635
|
See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
|
|
611
636
|
|
|
@@ -100,6 +100,12 @@ settings:
|
|
|
100
100
|
# - Bash
|
|
101
101
|
# - Read
|
|
102
102
|
# - Edit
|
|
103
|
+
# job_retention_days: 30 # OFF by default (0). When > 0, terminal job records
|
|
104
|
+
# older than this are deleted at server start. Every
|
|
105
|
+
# async and every return_ref dispatch leaves a job
|
|
106
|
+
# file behind and nothing removes it otherwise, so
|
|
107
|
+
# the directory grows forever — but the records are
|
|
108
|
+
# your dispatch history, so pruning is opt-in.
|
|
103
109
|
cache:
|
|
104
110
|
enabled: true
|
|
105
111
|
ttl: 300 # seconds; identical (agent, task, context) requests are cached
|
|
@@ -13,7 +13,14 @@ import click
|
|
|
13
13
|
import yaml
|
|
14
14
|
from pydantic import ValidationError
|
|
15
15
|
|
|
16
|
-
from .config import
|
|
16
|
+
from .config import (
|
|
17
|
+
CONFIG_SAVE_ERRORS,
|
|
18
|
+
auto_describe,
|
|
19
|
+
config_lock,
|
|
20
|
+
config_path,
|
|
21
|
+
load_config,
|
|
22
|
+
save_config,
|
|
23
|
+
)
|
|
17
24
|
from .jobs import JobStore, default_jobs_dir, is_valid_job_id
|
|
18
25
|
from .models import (
|
|
19
26
|
AgentConfig,
|
|
@@ -29,6 +36,39 @@ def _parse_csv(value: str | None) -> list[str] | None:
|
|
|
29
36
|
return [t.strip() for t in value.split(",") if t.strip()] if value else None
|
|
30
37
|
|
|
31
38
|
|
|
39
|
+
def _save_or_exit(config: DispatchConfig) -> None:
|
|
40
|
+
"""Persist the config, turning an I/O failure into a message, not a traceback.
|
|
41
|
+
|
|
42
|
+
A full disk or a read-only volume makes `save_config` raise OSError. The
|
|
43
|
+
write is atomic, so the previous config survives — say so, since that is the
|
|
44
|
+
one thing the user needs to know.
|
|
45
|
+
"""
|
|
46
|
+
try:
|
|
47
|
+
save_config(config)
|
|
48
|
+
except OSError as e:
|
|
49
|
+
click.echo(
|
|
50
|
+
click.style(
|
|
51
|
+
f"Error: could not write {config_path()}: {e}\n"
|
|
52
|
+
"Nothing was changed — the previous config is intact "
|
|
53
|
+
"(the write is atomic). Check free space and permissions.",
|
|
54
|
+
fg="red",
|
|
55
|
+
)
|
|
56
|
+
)
|
|
57
|
+
raise SystemExit(1) from None
|
|
58
|
+
except CONFIG_SAVE_ERRORS as e:
|
|
59
|
+
# A value YAML cannot represent — not an OSError, so the arm above never
|
|
60
|
+
# saw it and it escaped as a traceback. See config.CONFIG_SAVE_ERRORS.
|
|
61
|
+
click.echo(
|
|
62
|
+
click.style(
|
|
63
|
+
f"Error: could not serialize the config to YAML: {e}\n"
|
|
64
|
+
"Nothing was changed — the previous config is intact. "
|
|
65
|
+
"This is a bug: please report the field that failed.",
|
|
66
|
+
fg="red",
|
|
67
|
+
)
|
|
68
|
+
)
|
|
69
|
+
raise SystemExit(1) from None
|
|
70
|
+
|
|
71
|
+
|
|
32
72
|
def _check_budget_or_exit(max_budget: float | None) -> None:
|
|
33
73
|
"""Reject a negative spend cap before it reaches `AgentConfig` (ge=0).
|
|
34
74
|
|
|
@@ -75,6 +115,15 @@ def _load_or_exit() -> DispatchConfig:
|
|
|
75
115
|
click.echo(click.style(f"Error: config at {config_path()} is not valid YAML:", fg="red"))
|
|
76
116
|
click.echo(str(e))
|
|
77
117
|
raise SystemExit(1) from None
|
|
118
|
+
except UnicodeDecodeError as e:
|
|
119
|
+
click.echo(
|
|
120
|
+
click.style(
|
|
121
|
+
f"Error: config at {config_path()} is not valid UTF-8 (re-save it as UTF-8):",
|
|
122
|
+
fg="red",
|
|
123
|
+
)
|
|
124
|
+
)
|
|
125
|
+
click.echo(str(e))
|
|
126
|
+
raise SystemExit(1) from None
|
|
78
127
|
except OSError as e:
|
|
79
128
|
click.echo(click.style(f"Error: config at {config_path()} could not be read:", fg="red"))
|
|
80
129
|
click.echo(str(e))
|
|
@@ -236,7 +285,7 @@ def add(
|
|
|
236
285
|
if warning := check_permission_mode(permission_mode):
|
|
237
286
|
click.echo(click.style(f"Warning: {warning}", fg="yellow"))
|
|
238
287
|
|
|
239
|
-
|
|
288
|
+
_save_or_exit(config)
|
|
240
289
|
click.echo(f"Added agent '{name}' -> {dir_path}")
|
|
241
290
|
|
|
242
291
|
|
|
@@ -251,7 +300,7 @@ def remove(name: str) -> None:
|
|
|
251
300
|
raise SystemExit(1)
|
|
252
301
|
|
|
253
302
|
del config.agents[name]
|
|
254
|
-
|
|
303
|
+
_save_or_exit(config)
|
|
255
304
|
click.echo(f"Removed agent '{name}'.")
|
|
256
305
|
|
|
257
306
|
|
|
@@ -381,7 +430,7 @@ def update(
|
|
|
381
430
|
click.echo("Nothing to update. Pass at least one option (see --help).")
|
|
382
431
|
raise SystemExit(1)
|
|
383
432
|
|
|
384
|
-
|
|
433
|
+
_save_or_exit(config)
|
|
385
434
|
click.echo(f"Updated agent '{name}': {', '.join(updated)}")
|
|
386
435
|
|
|
387
436
|
|
|
@@ -639,7 +688,7 @@ def group_add(name: str, description: str, shared_context: str, members: tuple[s
|
|
|
639
688
|
shared_context=shared_context,
|
|
640
689
|
members=member_objs,
|
|
641
690
|
)
|
|
642
|
-
|
|
691
|
+
_save_or_exit(config)
|
|
643
692
|
click.echo(f"Added group '{name}' ({len(member_objs)} member(s)).")
|
|
644
693
|
if not member_objs:
|
|
645
694
|
click.echo(
|
|
@@ -737,7 +786,7 @@ def group_update(name: str, description: str | None, shared_context: str | None)
|
|
|
737
786
|
click.echo("Nothing to update. Pass --description and/or --shared-context.")
|
|
738
787
|
raise SystemExit(1)
|
|
739
788
|
|
|
740
|
-
|
|
789
|
+
_save_or_exit(config)
|
|
741
790
|
click.echo(f"Updated group '{name}': {', '.join(updated)}")
|
|
742
791
|
|
|
743
792
|
|
|
@@ -752,7 +801,7 @@ def group_remove(name: str) -> None:
|
|
|
752
801
|
raise SystemExit(1)
|
|
753
802
|
|
|
754
803
|
del config.groups[name]
|
|
755
|
-
|
|
804
|
+
_save_or_exit(config)
|
|
756
805
|
click.echo(f"Removed group '{name}'.")
|
|
757
806
|
|
|
758
807
|
|
|
@@ -807,6 +856,15 @@ def doctor() -> None:
|
|
|
807
856
|
except yaml.YAMLError as e:
|
|
808
857
|
fail(f"Config not valid YAML: {cp}")
|
|
809
858
|
click.echo(f" {e}")
|
|
859
|
+
except UnicodeDecodeError as e:
|
|
860
|
+
# A ValueError, not an OSError — it slipped past both handlers above
|
|
861
|
+
# and crashed the one command meant to diagnose a broken config.
|
|
862
|
+
fail(f"Config is not valid UTF-8: {cp}")
|
|
863
|
+
click.echo(f" {e}")
|
|
864
|
+
click.echo(" Re-save the file as UTF-8.")
|
|
865
|
+
except OSError as e:
|
|
866
|
+
fail(f"Config could not be read: {cp}")
|
|
867
|
+
click.echo(f" {e}")
|
|
810
868
|
|
|
811
869
|
section("MCP registration")
|
|
812
870
|
if claude_path is None:
|
|
@@ -13,6 +13,7 @@ from collections.abc import Iterator
|
|
|
13
13
|
from pathlib import Path
|
|
14
14
|
|
|
15
15
|
import yaml
|
|
16
|
+
from pydantic import ValidationError
|
|
16
17
|
|
|
17
18
|
from .models import DispatchConfig
|
|
18
19
|
|
|
@@ -21,6 +22,19 @@ try: # pragma: no cover - platform dependent
|
|
|
21
22
|
except ImportError: # pragma: no cover - Windows has no fcntl
|
|
22
23
|
fcntl = None # type: ignore[assignment]
|
|
23
24
|
|
|
25
|
+
# Parse with libyaml when PyYAML was built against it. Every MCP tool call
|
|
26
|
+
# reloads agents.yaml from scratch (deliberately — that is how a new agent is
|
|
27
|
+
# picked up without a restart), so this parse is on the hot path of all 21
|
|
28
|
+
# tools. On a real 38 KB config the pure-Python SafeLoader takes ~9.8 ms and
|
|
29
|
+
# CSafeLoader ~0.6 ms: a 16x saving repeated on every single call, and it runs
|
|
30
|
+
# on the event-loop thread where it blocks every other tool. The fallback is
|
|
31
|
+
# mandatory — a PyYAML installed from source without libyaml headers has no
|
|
32
|
+
# C extension. Same semantics either way: both are the *safe* loader.
|
|
33
|
+
try: # pragma: no cover - depends on how PyYAML was built
|
|
34
|
+
from yaml import CSafeLoader as _YamlLoader
|
|
35
|
+
except ImportError: # pragma: no cover - pure-Python PyYAML
|
|
36
|
+
from yaml import SafeLoader as _YamlLoader # type: ignore[assignment]
|
|
37
|
+
|
|
24
38
|
logger = logging.getLogger(__name__)
|
|
25
39
|
|
|
26
40
|
DEFAULT_CONFIG_DIR = Path.home() / ".config" / "agent-dispatch"
|
|
@@ -150,12 +164,36 @@ def file_lock(path: Path) -> Iterator[None]:
|
|
|
150
164
|
_release_lock(fd)
|
|
151
165
|
|
|
152
166
|
|
|
167
|
+
# Everything load_config() can raise for a config a human can plausibly produce.
|
|
168
|
+
# Listed once because three surfaces handle it independently (server._get_config,
|
|
169
|
+
# cli._load_or_exit, cli.doctor) and they must not drift: UnicodeDecodeError, in
|
|
170
|
+
# particular, is a ValueError — NOT an OSError — so a file saved in cp1251 or
|
|
171
|
+
# UTF-16 used to escape every one of them as a raw traceback.
|
|
172
|
+
CONFIG_LOAD_ERRORS = (ValidationError, yaml.YAMLError, UnicodeDecodeError, OSError)
|
|
173
|
+
|
|
174
|
+
# The *write* half, and deliberately not a copy of the read half. `save_config`
|
|
175
|
+
# can fail two ways with two different remediations: OSError (full disk,
|
|
176
|
+
# read-only volume — handled separately, and the atomic rename means the old
|
|
177
|
+
# file survives), or a rendering failure. `yaml.dump` raises RepresenterError —
|
|
178
|
+
# a `yaml.YAMLError`, so neither an OSError nor a ValueError — for any value it
|
|
179
|
+
# cannot represent. That is unreachable today because save_config only ever
|
|
180
|
+
# feeds it `model_dump(mode="json")` output, i.e. JSON-native types. It becomes
|
|
181
|
+
# reachable the moment a field lands whose JSON dump is not one of those, and it
|
|
182
|
+
# would then escape *both* server guards and the CLI's `_save_or_exit` as a raw
|
|
183
|
+
# traceback — the exact "an exception type escapes the handler meant to catch
|
|
184
|
+
# it" class this codebase has been bitten by twice. Declared once, next to its
|
|
185
|
+
# read-side twin, so the two surfaces that handle it cannot drift.
|
|
186
|
+
CONFIG_SAVE_ERRORS = (yaml.YAMLError,)
|
|
187
|
+
|
|
188
|
+
|
|
153
189
|
def load_config(path: Path | None = None) -> DispatchConfig:
|
|
154
190
|
"""Load config from YAML file. Returns empty config if file missing."""
|
|
155
191
|
p = path or config_path()
|
|
156
192
|
if not p.exists():
|
|
157
193
|
return DispatchConfig()
|
|
158
|
-
|
|
194
|
+
# yaml.load with the safe loader == yaml.safe_load; _YamlLoader is the C
|
|
195
|
+
# one when available (see its definition for why this is worth doing).
|
|
196
|
+
raw = yaml.load(p.read_text(encoding="utf-8"), Loader=_YamlLoader) # noqa: S506
|
|
159
197
|
if raw is None:
|
|
160
198
|
return DispatchConfig()
|
|
161
199
|
return DispatchConfig.model_validate(raw)
|