agent-dispatch 0.15.1__tar.gz → 0.16.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_dispatch-0.16.0/.github/workflows/ci.yml +58 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/.gitignore +3 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/AGENTS.md +16 -11
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/CHANGELOG.md +110 -1
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/PKG-INFO +73 -18
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/README.md +72 -17
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/agents.example.yaml +1 -1
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/pyproject.toml +7 -1
- agent_dispatch-0.16.0/scripts/check_install.py +116 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/__init__.py +1 -1
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/cache.py +20 -9
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/cli.py +221 -67
- agent_dispatch-0.16.0/src/agent_dispatch/concurrency.py +158 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/jobs.py +74 -31
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/models.py +33 -5
- agent_dispatch-0.16.0/src/agent_dispatch/progress.py +43 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/runner.py +428 -50
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/server.py +276 -159
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/servers.py +5 -3
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/usage.py +27 -7
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/conftest.py +61 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_cli.py +426 -7
- agent_dispatch-0.16.0/tests/test_concurrency.py +277 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_jobs.py +228 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_models.py +37 -1
- agent_dispatch-0.16.0/tests/test_progress.py +44 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_runner.py +647 -80
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_server.py +668 -11
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_usage.py +45 -0
- agent_dispatch-0.15.1/.coverage +0 -0
- agent_dispatch-0.15.1/.github/workflows/ci.yml +0 -34
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/.github/dependabot.yml +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/.github/workflows/publish.yml +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/LICENSE +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/SECURITY.md +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/assets/mascot.png +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/src/agent_dispatch/config.py +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/__init__.py +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_cache.py +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_config.py +0 -0
- {agent_dispatch-0.15.1 → agent_dispatch-0.16.0}/tests/test_servers.py +0 -0
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main]
|
|
8
|
+
|
|
9
|
+
# Least privilege: CI only needs to read the repo.
|
|
10
|
+
permissions:
|
|
11
|
+
contents: read
|
|
12
|
+
|
|
13
|
+
jobs:
|
|
14
|
+
test:
|
|
15
|
+
runs-on: ubuntu-latest
|
|
16
|
+
strategy:
|
|
17
|
+
matrix:
|
|
18
|
+
python-version: ["3.10", "3.11", "3.12", "3.13"]
|
|
19
|
+
|
|
20
|
+
steps:
|
|
21
|
+
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
22
|
+
|
|
23
|
+
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
|
24
|
+
with:
|
|
25
|
+
python-version: ${{ matrix.python-version }}
|
|
26
|
+
|
|
27
|
+
- name: Install dependencies
|
|
28
|
+
run: pip install -e ".[dev]"
|
|
29
|
+
|
|
30
|
+
- name: Lint
|
|
31
|
+
run: ruff check src/ tests/ scripts/
|
|
32
|
+
|
|
33
|
+
- name: Test
|
|
34
|
+
run: pytest tests/ -v
|
|
35
|
+
|
|
36
|
+
installed-package:
|
|
37
|
+
runs-on: ubuntu-latest
|
|
38
|
+
timeout-minutes: 10
|
|
39
|
+
strategy:
|
|
40
|
+
matrix:
|
|
41
|
+
python-version: ["3.10", "3.13"]
|
|
42
|
+
steps:
|
|
43
|
+
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
44
|
+
- uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6
|
|
45
|
+
with:
|
|
46
|
+
python-version: ${{ matrix.python-version }}
|
|
47
|
+
- name: Build wheel and source distribution
|
|
48
|
+
run: |
|
|
49
|
+
python -m pip install build twine
|
|
50
|
+
python -m build
|
|
51
|
+
python -m twine check dist/*
|
|
52
|
+
- name: Install wheel into a clean environment
|
|
53
|
+
run: |
|
|
54
|
+
python -m venv "$RUNNER_TEMP/package-check"
|
|
55
|
+
"$RUNNER_TEMP/package-check/bin/python" -m pip install dist/*.whl
|
|
56
|
+
"$RUNNER_TEMP/package-check/bin/python" -m pip check
|
|
57
|
+
- name: Verify CLI and MCP transport from installed wheel
|
|
58
|
+
run: '"$RUNNER_TEMP/package-check/bin/python" -I scripts/check_install.py'
|
|
@@ -11,13 +11,15 @@ MCP server + CLI that lets Claude Code agents delegate tasks to agents in other
|
|
|
11
11
|
| File | Role |
|
|
12
12
|
|------|------|
|
|
13
13
|
| `src/agent_dispatch/runner.py` | Sync subprocess wrapper around `claude -p` — the actual work |
|
|
14
|
-
| `src/agent_dispatch/server.py` | Async FastMCP interface (21 MCP tools), wraps runner in `asyncio.to_thread`
|
|
14
|
+
| `src/agent_dispatch/server.py` | Async FastMCP interface (21 MCP tools), wraps runner in `asyncio.to_thread` with a shared process limit |
|
|
15
15
|
| `src/agent_dispatch/usage.py` | Append-only dispatch journal (cost/duration/outcome) + aggregation for `stats` and the `typical` block |
|
|
16
16
|
| `src/agent_dispatch/servers.py` | Registry of live `serve` processes and the version each one runs (`doctor` reads it) |
|
|
17
17
|
| `src/agent_dispatch/cli.py` | Click CLI: `init`, `add`, `update`, `remove`, `list`, `describe`, `test`, `doctor`, `jobs`, `job`, `cancel`, `gc`, `group` (add/list/inspect/update/remove), `serve` |
|
|
18
18
|
| `src/agent_dispatch/models.py` | Pydantic v2 models (`AgentConfig`, `DispatchGroup`/`GroupMember`, `Settings`, `DispatchResult`) |
|
|
19
19
|
| `src/agent_dispatch/config.py` | YAML config load/save + project auto-description |
|
|
20
20
|
| `src/agent_dispatch/cache.py` | Thread-safe in-memory TTL cache |
|
|
21
|
+
| `src/agent_dispatch/concurrency.py` | Resizable process limit shared by async callers and background workers |
|
|
22
|
+
| `src/agent_dispatch/progress.py` | Bounded progress buffer, discarded when a streaming client disconnects |
|
|
21
23
|
| `src/agent_dispatch/jobs.py` | Persistent per-job JSON files for async dispatch |
|
|
22
24
|
|
|
23
25
|
## Dev setup
|
|
@@ -29,11 +31,13 @@ pip install -e ".[dev]"
|
|
|
29
31
|
## Gates — both must pass before a change is done (CI rejects otherwise)
|
|
30
32
|
|
|
31
33
|
```bash
|
|
32
|
-
ruff check src/ tests/
|
|
33
|
-
python3 -m pytest tests/ -v
|
|
34
|
+
ruff check src/ tests/ scripts/
|
|
35
|
+
python3 -m pytest tests/ -v
|
|
34
36
|
```
|
|
35
37
|
|
|
36
|
-
Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.which` + `
|
|
38
|
+
Tests must **never** invoke the real `claude` CLI. Runner unit tests mock `shutil.which` + `_run_captured`/`Popen`; server tests mock `_get_config` + `runner.dispatch`. Pipe and decoding integration tests spawn short-lived *python* subprocesses: a pipe deadlock lives in the OS pipe buffer, so a mocked `Popen` structurally cannot reproduce it.
|
|
39
|
+
|
|
40
|
+
The autouse guard in `tests/conftest.py` rejects real Claude process launches when a test forgets its mock. Diagnostic tests must mock discovery or CLI probes too.
|
|
37
41
|
|
|
38
42
|
## Non-obvious invariants (violating these breaks real behavior)
|
|
39
43
|
|
|
@@ -45,23 +49,23 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
45
49
|
- Valid permission modes: `default`, `plan`, `bypassPermissions` (`models.py: KNOWN_PERMISSION_MODES`).
|
|
46
50
|
- `JobStore.finish`/`fail` refuse already-terminal jobs (returns `None`) — this closes the race with force-cancel; never "fix" it by overwriting. `mark_running` likewise refuses any job that isn't `pending`, so a stale or duplicate worker can't resurrect a finished one.
|
|
47
51
|
- "Is this group member missing?" has exactly one implementation: `DispatchConfig.unknown_group_members()`. Any new surface that lists or validates membership calls it instead of re-deriving the check.
|
|
48
|
-
- Cancelling a *running* job requires the in-memory
|
|
52
|
+
- Cancelling a *running* job requires the in-memory registry (server.py): `_owned_jobs` records ownership **before** `mark_running`, `_running_procs` the Popen once spawned — a cancel landing between the two is still honored, and the late process is killed on registration. The job is marked `cancelled` **before** the subprocess is killed. A job record that is merely *unreadable* at spawn is not treated as terminal. Don't persist PIDs to disk (PID reuse after restart could kill an unrelated process).
|
|
49
53
|
- `max_budget_usd` is enforced **by the claude CLI** (`_build_command` passes `--max-budget-usd`): a run stopped at the cap comes back `is_error` with no `result` text, and `_build_error_result` turns it into `error_type="budget"` + `budget_exceeded=True` + a resumable `session_id`. `_apply_budget` is the *secondary*, post-hoc signal for an overshoot that didn't stop the run; it never fails a dispatch.
|
|
50
54
|
- A CLI error payload can have no `result` field at all — the reason lives in `errors` / `subtype`. Read it via `_cli_error_details`, never assume `result` is populated on failure.
|
|
51
|
-
- **Both subprocess pipes must be drained concurrently.** `dispatch_stream` reads stdout in a loop while a daemon thread drains stderr; reading stderr only after the loop deadlocks any child that writes more than ~64 KiB to it (the child blocks in `write(2)`, never emits its result, and the dispatch dies at the timeout). `dispatch
|
|
55
|
+
- **Both subprocess pipes must be drained concurrently.** `dispatch_stream` reads stdout in a loop while a daemon thread drains stderr; reading stderr only after the loop deadlocks any child that writes more than ~64 KiB to it (the child blocks in `write(2)`, never emits its result, and the dispatch dies at the timeout). Ordinary `dispatch` uses `_run_captured` with two readers; binary `read1` retains flushed output even without a newline. Readers own their pipes, and stop retaining output when the caller returns. **Completion is stdout EOF + leader exit, never stderr EOF**: a stdio MCP server started by `claude` inherits fd 2 and outlives it, so waiting for stderr EOF turned every such dispatch into a full-timeout wait and a real CLI error into `error_type="timeout"`. stderr gets a short bounded join afterwards, then the leftover group is killed. `dispatch_stream` reads stdout through a reader thread too, so a descendant that *detached* from the group (unreachable by `killpg`) cannot park it forever once the run was killed.
|
|
52
56
|
- **A received `result` event outranks the timeout flag.** The agent finished and was billed; a lingering process is a cleanup problem, reported as a `hint`, not a failure.
|
|
53
|
-
- **The timeout must kill the process *tree*, not the child.**
|
|
57
|
+
- **The timeout must kill the process *tree*, not the child.** Both runners spawn with `start_new_session=True` and `_kill_process_tree` sends SIGKILL to the group: a grandchild that inherited stdout keeps the read loop parked long past the deadline otherwise. `killpg` is guarded on a positive pid — `killpg(0)` would signal the dispatcher's own group — and on `proc.returncode is None`: a reaped leader's PID may already belong to an unrelated group. So nothing may reap the leader (`poll()`/`wait()`) before the kill decision; `_wait_leader_exit` observes exit without reaping (`waitid(WNOWAIT)` on Linux, a kqueue exit watch on macOS). `_run_captured` drains pipes before reaping the leader, so its PID cannot be recycled before a timeout kill. Cleanup waits are bounded even when a descendant deliberately leaves the process group. A full CLI `type=result` JSON payload outranks the ordinary timeout too; incomplete output does not.
|
|
54
58
|
- **Never `close()` a pipe another thread may still be reading.** `close()` waits on the reader's buffer lock with *no timeout*, so it would hang the dispatch forever — the bounded `join()` before it buys nothing. `dispatch_stream` skips the stderr close while the drain thread is alive and lets the daemon reader + Popen finalizer release the fd. This is not hypothetical: a stdio MCP server inherits the child's `stderr`, so the pipe often has no EOF even after `claude` exits cleanly.
|
|
55
59
|
- **No `await` inside `config_lock()`.** `ProcessLock`'s in-process guard is a `threading.RLock` — re-entrant per *thread* — and every MCP tool coroutine runs on the one event-loop thread. Suspending in the critical section lets a second coroutine re-enter the "held" lock and interleave its own load/mutate/save. Collect warnings as data, emit them after the `with` block (`test_no_await_inside_the_config_lock` enforces this by AST).
|
|
56
60
|
- Cross-process locks are acquired with a **bounded** wait, never a blocking `flock`: the server takes them on its event-loop thread, so a wedged holder would freeze every tool. After the deadline it proceeds unlocked and logs — a possible lost update beats a permanent freeze.
|
|
57
61
|
- `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore. For a *running* job, `started_at` alone does **not** prove abandonment — a dispatch may legitimately run to the 7200s timeout ceiling, so the file's own mtime is checked too (a live worker rewrites it on every progress flush). That check can only ever *skip* a recovery, never add one.
|
|
58
62
|
- Deleting job records is **opt-in** (`settings.job_retention_days`, default `0` = off) and only ever happens at server start, never inside a tool. They are the user's dispatch history and the deletion is irreversible, so an unreadable config is treated as "do nothing" rather than falling back to a default retention.
|
|
59
|
-
- `max_concurrency` must bound *subprocesses*, not coroutines.
|
|
63
|
+
- `max_concurrency` must bound *subprocesses*, not coroutines. All dispatch paths share one `DispatchLimiter` per server, resized in place so changing the limit cannot erase occupied slots. Ordinary and streaming dispatches go through `_start_guarded`, which releases from the future's done-callback; awaits on that future are shielded. Background jobs hold the same limiter in their worker thread. The limiter is **FIFO across threads and coroutines** — a freed slot is handed to the head waiter in `release()`; otherwise parked job threads, woken in microseconds, starved a queued synchronous `dispatch` indefinitely. A waiter cancelled after its grant passes the slot on exactly once. Cancelling a coroutine does not stop its thread: never release its slot from an `async with` or caller-side `finally`.
|
|
60
64
|
- Tool responses are serialized through `server._dumps`, never bare `json.dumps`: the stdlib default (`ensure_ascii=True`) turns every non-ASCII character into a `\uXXXX` escape, tripling the bytes and tokenizing badly, for no gain — the stdio transport emits raw UTF-8 via pydantic anyway. A real `list_groups()` carried 8520 escapes and weighed 59 KB instead of 25 KB.
|
|
61
65
|
- `agents.yaml` is parsed with libyaml's `CSafeLoader` when available (`config._YamlLoader`), falling back to the pure-Python **safe** loader — never `yaml.Loader`. The config is re-read on every tool call, so this parse is on the hot path of all 21 tools *and* blocks the event-loop thread: 9.80 ms → 0.76 ms on a real 38 KB config.
|
|
62
66
|
- Pydantic does **not** validate on assignment. `Field(ge=...)` guards only the *load* path; every mutation surface (CLI `add`/`update`, MCP `add_agent`/`update_agent`) needs its own boundary check, or the bound escapes as a raw `ValidationError`.
|
|
63
67
|
- Every state file (`agents.yaml`, job files) is written **temp file + `os.replace`**, never in place, and every load/mutate/save is wrapped in `config.ProcessLock` — the CLI and the MCP server are separate processes writing the same files, so a thread lock alone loses updates.
|
|
64
|
-
- Anything that changes an agent's config must call `_invalidate_agent_cache`
|
|
68
|
+
- Anything that changes an agent's config must call `_invalidate_agent_cache` to reclaim old entries. Cache lookups and writes also receive the same `_cache_config_key` snapshot, captured before awaiting work: this covers CLI/YAML/other-server edits and prevents an old worker from repopulating the current configuration's cache. Include new dispatch-affecting settings in this snapshot.
|
|
65
69
|
- Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, `budget_exceeded` results, and results whose `outcome` is `partial`/`blocked`, so the documented "grant access / fix the cause, then re-dispatch" recovery is never short-circuited.
|
|
66
70
|
- **Coerce every number read off disk through `usage.as_float`.** `float("abc")` raises **ValueError** — not `JSONDecodeError`, not `OSError` — so it escaped the per-line handler in `usage.load` and the sort key in `servers.live_servers`, turning `agent-dispatch stats --days 7` and `doctor` into tracebacks over one hand-edited character. Same escape route as the `UnicodeDecodeError`-is-a-`ValueError` round. A journal or registry field is untrusted input: skip the record or sort it last, never raise from the command someone ran *because* something is already wrong.
|
|
67
71
|
- **`--json` output must be valid JSON on the empty path too.** `stats --json` printed prose when the journal had no records — i.e. on a fresh install, which is exactly when a script first pipes it somewhere.
|
|
@@ -70,7 +74,7 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
70
74
|
- **Profiles are memoized on the journal's (path, size, mtime) and read a NARROWER tail than `stats`.** `list_agents`/`inspect_agent` run on the event-loop thread: a full 2 MB journal cost **24.8 ms** per discovery call at the 512 KB report window — 33x the config parse that 0.13.0 exists to have fixed. Now 3.3 ms cold, ~0 warm. Every profile consumer goes through `usage.profiles()`; never add a caller that re-parses per agent.
|
|
71
75
|
- **`usage_limit` is its own error type** (observed live: `You've hit your session limit · resets 6pm`). It is checked *before* the permission patterns and its hint says the limit is account-wide, so "try another agent" is not a workaround. Text-classified errors attach their advisory through the single `_classification_hint`, not per site.
|
|
72
76
|
- **Server liveness is an advisory lock, not a PID.** `servers.register` holds an exclusive `flock` on `<config dir>/servers/<pid>.json` for the process lifetime; a reader that *takes* that lock owns a dead entry and deletes it. `os.kill(pid, 0)` would believe a recycled PID and would never notice a SIGKILLed server. Never close that fd (re-registering closes the old one first).
|
|
73
|
-
- **The dispatch protocol goes in the system prompt, not the task.** `_build_system_prompt` (runner.py) renders the protocol (non-interactive, time/spend budget, lead with the outcome, trailing `STATUS:` line) plus the agent's `instructions`, and `_build_command` passes it as `--append-system-prompt` — on `--resume` too, since the CLI re-applies an appended prompt on every launch. `_build_prompt` (the `-p` text) is unchanged
|
|
77
|
+
- **The dispatch protocol goes in the system prompt, not the task.** `_build_system_prompt` (runner.py) renders the protocol (non-interactive, time/spend budget, lead with the outcome, trailing `STATUS:` line) plus the agent's `instructions`, and `_build_command` passes it as `--append-system-prompt` — on `--resume` too, since the CLI re-applies an appended prompt on every launch. `_build_prompt` (the `-p` text) is unchanged. Standing instructions and protocol settings are captured separately in `_cache_config_key`. The text always starts with a `##` header line so it can never be read as a flag. `settings.dispatch_protocol: false` turns the protocol off (for CLIs that predate the flag; `doctor` probes `claude --help` for it); per-agent `instructions` still go through when set.
|
|
74
78
|
- **`outcome` is lifted from the LAST line of the result** (`_split_outcome`), before JSON parsing, by the shared `_build_success_result` (the success-side twin of `_build_error_result` — both `dispatch` and `dispatch_stream` go through it) and by the plain-text fallback tier. A bare marker line is removed from `result`; a marker with a trailing reason stays. `outcome` never flips `success`; `partial`/`blocked` add a `hint` with the `dispatch_session(...)` continuation, after the denial hint. In JSON mode the protocol omits the STATUS instruction (the JSON footer governs), but a STATUS line that arrives anyway is still stripped so `parsed_result` survives.
|
|
75
79
|
- Remediation text is a contract: a hint that names a flag must name one that exists (`test_printed_budget_hint_is_a_runnable_command` feeds the printed flags back into the CLI). Run the command you print.
|
|
76
80
|
- The config error sets are declared **once** and in two halves: `config.CONFIG_LOAD_ERRORS` (read) and `config.CONFIG_SAVE_ERRORS` (write — `yaml.dump`'s `RepresenterError` is a `yaml.YAMLError`, therefore neither `OSError` nor `ConfigLoadError`, and used to escape both the MCP guard and the CLI's `_save_or_exit`). Two halves, not one set, because the remediations differ: a failed write is atomic so the old config survives, while a failed read needs the YAML fixed.
|
|
@@ -80,6 +84,7 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
80
84
|
|
|
81
85
|
- `mcp` is pinned **`>=1.2.0,<2`** deliberately: 2.0 removed `mcp.server.fastmcp`, which `server.py` imports, so an unbounded range gives every fresh install a dead `agent-dispatch serve`. Lifting the cap means porting to `mcp.server.mcpserver.MCPServer` — it is not a dependency bump.
|
|
82
86
|
- Verify packaging in a **clean venv**, never the dev machine: build the wheel, install it fresh, import the server. A stale pin in local site-packages hides exactly the failure a new user hits first.
|
|
87
|
+
- CI's `installed-package` job builds wheel and sdist, checks metadata with twine, then installs only the wheel and runtime dependencies into a fresh venv on Python 3.10 and 3.13. Run `python -I scripts/check_install.py` with that venv's Python to verify the installed CLI and a real MCP stdio connection. The check isolates all state in a temporary directory and cannot discover Claude on PATH.
|
|
83
88
|
- Run **`twine check dist/*`** before cutting the tag, with a *current* twine. The clean-venv check does not cover this: it proves the wheel installs, not that PyPI's uploader will accept its metadata. `python -m build` pulls the newest hatchling at build time, so the emitted `Metadata-Version` climbs on its own, and `pypa/gh-action-pypi-publish` is pinned to a SHA that bundles a fixed twine — v0.13.0's publish failed on exactly that mismatch (hatchling 1.32 → metadata 2.5; pinned twine 6.1.0 → "not a valid metadata version"). When it happens, bump the action pin rather than pinning the backend down.
|
|
84
89
|
|
|
85
90
|
## Deliberately not built
|
|
@@ -96,4 +101,4 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
|
|
|
96
101
|
|
|
97
102
|
## More detail
|
|
98
103
|
|
|
99
|
-
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests
|
|
104
|
+
[README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
|
|
@@ -7,6 +7,110 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.16.0] - 2026-09-30
|
|
11
|
+
|
|
12
|
+
The process-control round. One concurrency limit now covers every dispatch
|
|
13
|
+
path and serves callers in arrival order; ordinary dispatch runs on the same
|
|
14
|
+
bounded two-reader design as streaming; and a round of adversarial review
|
|
15
|
+
aimed at pipes, process groups and cancellation closed the cases where a
|
|
16
|
+
finished run waited out its whole timeout, a detached descendant hung a
|
|
17
|
+
dispatch forever, or a paid result was thrown away over optional metadata.
|
|
18
|
+
Scripts get machine-readable `doctor`, `jobs` and `job` output.
|
|
19
|
+
|
|
20
|
+
### Added
|
|
21
|
+
- CI verifies clean wheel installation on Python 3.10 and 3.13, distribution
|
|
22
|
+
metadata, and a real MCP stdio connection. `scripts/check_install.py` checks
|
|
23
|
+
CLI/MCP interoperability and full-result export with isolated temporary data.
|
|
24
|
+
- `doctor --json` emits a versioned diagnostic report with checks, severity,
|
|
25
|
+
counts, and remediation details. `--strict` makes warnings fail the command,
|
|
26
|
+
for installation checks in scripts.
|
|
27
|
+
- `agent-dispatch jobs --json` exports compact summaries; `job <id> --json`
|
|
28
|
+
exports the full record, including untruncated results and session metadata.
|
|
29
|
+
- Filter job history by exact agent name with `jobs --agent NAME` or
|
|
30
|
+
`dispatch_jobs(agent="NAME")`, including history for removed agents.
|
|
31
|
+
- Preview history cleanup with `agent-dispatch gc --dry-run` and
|
|
32
|
+
`dispatch_gc(dry_run=True)`. Both use the same eligibility rules as deletion.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
- Non-finite budgets are rejected before CLI/MCP mutations and during YAML
|
|
36
|
+
loading. Invalid numeric fields in parallel dispatch return an error before
|
|
37
|
+
any task starts, including values that overflow integer conversion.
|
|
38
|
+
- Ordinary dispatch now kills the process group on timeout, stopping descendants
|
|
39
|
+
that could continue work after the call failed. Pipe reads are concurrent and
|
|
40
|
+
cleanup is bounded even when a descendant detaches from the group.
|
|
41
|
+
- A complete CLI result from ordinary dispatch survives a cleanup timeout,
|
|
42
|
+
preserving the paid answer, cost, session and error classification. Partial
|
|
43
|
+
output still returns a resumable timeout.
|
|
44
|
+
- Running-job cancellation kills the process group, including descendants
|
|
45
|
+
holding output pipes open. A compatibility retry racing cancellation is
|
|
46
|
+
stopped too; the cancelled state is persisted before any kill.
|
|
47
|
+
- Streaming ignores malformed progress events without losing the final paid
|
|
48
|
+
result. Failed process registration cleans up the child and its pipes.
|
|
49
|
+
- Cache keys include the agent configuration and dispatch defaults. Edits from
|
|
50
|
+
the CLI, another server, or YAML cannot reuse an old answer, and workers
|
|
51
|
+
finishing under an older configuration cannot overwrite current answers.
|
|
52
|
+
- The process-tree timeout regression test synchronizes subprocess setup before
|
|
53
|
+
its one-second deadline, avoiding startup-related failures on busy machines.
|
|
54
|
+
- Non-finite and overflowing numeric metadata no longer poison usage reports
|
|
55
|
+
or crash server diagnostics. Invalid optional result costs are omitted while
|
|
56
|
+
preserving the dispatch output; journal records omit non-finite costs.
|
|
57
|
+
- Doctor's server counts and version drift now use one registry snapshot,
|
|
58
|
+
avoiding inconsistent totals when servers start or exit during a check.
|
|
59
|
+
- Two diagnostic tests now mock Claude discovery; a shared test guard rejects
|
|
60
|
+
accidental launches of the real Claude CLI.
|
|
61
|
+
- Streaming progress uses a bounded tail instead of an unlimited queue.
|
|
62
|
+
Slow clients receive an explicit omitted-message count; disconnected clients
|
|
63
|
+
stop retaining progress. Full dispatch results remain intact.
|
|
64
|
+
- All dispatch paths now share `max_concurrency` within a server process.
|
|
65
|
+
Mixing ordinary and background jobs no longer allows twice the configured
|
|
66
|
+
number of subprocesses. Changing the limit preserves occupied slots.
|
|
67
|
+
- Cancelling a streaming tool call keeps its concurrency reservation until the
|
|
68
|
+
worker finishes, just like an ordinary dispatch.
|
|
69
|
+
- Damaged job records cannot redirect status updates or stale recovery to a
|
|
70
|
+
different job: all reads verify the internal ID against the filename.
|
|
71
|
+
- Invalid and non-finite job timestamps are rejected at the read boundary.
|
|
72
|
+
Unreadable records are preserved, including during history cleanup.
|
|
73
|
+
- Job IDs with trailing newlines are rejected before file access.
|
|
74
|
+
- Agent and group name validation rejects trailing newlines instead of accepting
|
|
75
|
+
a partial regular-expression match.
|
|
76
|
+
- Ordinary dispatch completes when the CLI exits and its stdout closes, instead
|
|
77
|
+
of waiting for stderr. A stdio MCP server that inherited stderr no longer
|
|
78
|
+
turns a finished run into a full-timeout wait, and a real CLI error (for
|
|
79
|
+
example a usage limit) is no longer reported as a timeout.
|
|
80
|
+
- Streaming dispatch and running-job cancellation are bounded even when a
|
|
81
|
+
descendant leaves the process group while holding stdout; the concurrency
|
|
82
|
+
slot is released.
|
|
83
|
+
- A process group is never signalled after its leader was reaped, so a recycled
|
|
84
|
+
PID cannot direct the kill at an unrelated process group.
|
|
85
|
+
- The shared concurrency limit serves waiters in arrival order. Background jobs
|
|
86
|
+
queued later can no longer starve an ordinary, streaming or parallel call.
|
|
87
|
+
- Cancelling an owned job between its start and its process spawn now cancels
|
|
88
|
+
it instead of reporting it as belonging to another server. A job record that
|
|
89
|
+
cannot be read at spawn no longer makes the worker kill its own process.
|
|
90
|
+
- Malformed `duration_ms`, `num_turns` or `session_id` in CLI output no longer
|
|
91
|
+
discard the paid result; a bad session id falls back to the generated one.
|
|
92
|
+
- Usage reports ignore implausible journal values, so totals cannot overflow;
|
|
93
|
+
`stats --json` never prints `Infinity`.
|
|
94
|
+
- `doctor` survives non-UTF-8 `claude mcp list` output and an unreadable config
|
|
95
|
+
directory (JSON is still emitted). Its old-CLI warning accounts for
|
|
96
|
+
`dispatch_protocol` and per-agent instructions, so applying its remedy clears
|
|
97
|
+
it under `--strict`.
|
|
98
|
+
- `gc`, `gc --dry-run` and `cancel` report job-storage errors instead of
|
|
99
|
+
printing a traceback.
|
|
100
|
+
- The test guard against launching the real Claude CLI now fails the test even
|
|
101
|
+
when the launch happens in a background worker thread.
|
|
102
|
+
- A failing progress callback (for example a full disk while an async job
|
|
103
|
+
writes its progress) no longer kills the run or discards its paid result and
|
|
104
|
+
session. If the reader threads cannot be started, the spawned process is
|
|
105
|
+
killed instead of running unsupervised.
|
|
106
|
+
- `dispatch_cancel` reports `already_terminal` when this server's own job
|
|
107
|
+
finishes during the cancel, instead of claiming another server owns it.
|
|
108
|
+
- A lone surrogate escape in an agent's answer (typically a hand-escaped emoji
|
|
109
|
+
in a `response_format="json"` reply) no longer makes `dispatch`,
|
|
110
|
+
`dispatch_session`, `dispatch_stream` or `return_ref` raise instead of
|
|
111
|
+
answering, and no longer costs an async job its paid result. It is replaced
|
|
112
|
+
with U+FFFD where the CLI output is parsed.
|
|
113
|
+
|
|
10
114
|
## [0.15.1] - 2026-09-09
|
|
11
115
|
|
|
12
116
|
Three defects in 0.15.0's own new code, found by re-auditing it the way this
|
|
@@ -802,7 +906,12 @@ cache bounding, and stale-job recovery.
|
|
|
802
906
|
- Dependabot for `pip` + `github-actions`, GitHub Actions pinned to
|
|
803
907
|
commit SHAs for supply-chain integrity.
|
|
804
908
|
|
|
805
|
-
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.
|
|
909
|
+
[Unreleased]: https://github.com/ginkida/agent-dispatch/compare/v0.16.0...HEAD
|
|
910
|
+
[0.16.0]: https://github.com/ginkida/agent-dispatch/compare/v0.15.1...v0.16.0
|
|
911
|
+
[0.15.1]: https://github.com/ginkida/agent-dispatch/compare/v0.15.0...v0.15.1
|
|
912
|
+
[0.15.0]: https://github.com/ginkida/agent-dispatch/compare/v0.14.0...v0.15.0
|
|
913
|
+
[0.14.0]: https://github.com/ginkida/agent-dispatch/compare/v0.13.0...v0.14.0
|
|
914
|
+
[0.13.0]: https://github.com/ginkida/agent-dispatch/compare/v0.12.1...v0.13.0
|
|
806
915
|
[0.12.1]: https://github.com/ginkida/agent-dispatch/compare/v0.12.0...v0.12.1
|
|
807
916
|
[0.12.0]: https://github.com/ginkida/agent-dispatch/compare/v0.11.0...v0.12.0
|
|
808
917
|
[0.11.0]: https://github.com/ginkida/agent-dispatch/compare/v0.10.0...v0.11.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: agent-dispatch
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.16.0
|
|
4
4
|
Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
|
|
5
5
|
Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
|
|
6
6
|
Project-URL: Repository, https://github.com/ginkida/agent-dispatch
|
|
@@ -385,6 +385,13 @@ Run multiple tasks concurrently. Much faster than sequential `dispatch` calls.
|
|
|
385
385
|
|
|
386
386
|
Same as `dispatch` but shows live progress while the agent works. Use for long-running tasks. Not cached.
|
|
387
387
|
|
|
388
|
+
Progress notifications keep the latest 100 pending messages, capped at 300
|
|
389
|
+
characters each. If output arrives faster than the client can receive it,
|
|
390
|
+
older notifications are omitted with a count; the final result is unaffected.
|
|
391
|
+
Disconnecting or cancelling the tool call discards pending notifications and
|
|
392
|
+
stops buffering new ones. The worker still holds its concurrency slot until it
|
|
393
|
+
finishes.
|
|
394
|
+
|
|
388
395
|
Parameters are the same as `dispatch` except `return_ref`/`summary_chars` (streaming is incompatible with ref-mode) and `group` (group context injection is supported only on `dispatch` and per-item in `dispatch_parallel`).
|
|
389
396
|
|
|
390
397
|
### `dispatch_dialogue`
|
|
@@ -426,7 +433,7 @@ Register a new project directory as an agent. Description is auto-generated from
|
|
|
426
433
|
| `directory` | string | yes | Path to an existing project directory (`~` is expanded, relative paths resolved) |
|
|
427
434
|
| `description` | string | no | What this agent can do — auto-generated if empty |
|
|
428
435
|
| `timeout` | int | no | Timeout in seconds (0 = 300; this is a literal default, not `settings.default_timeout`) |
|
|
429
|
-
| `max_budget_usd` | float | no |
|
|
436
|
+
| `max_budget_usd` | float | no | Finite max cost in USD per dispatch (0 = inherit `settings.default_max_budget_usd`; no cap only when that is unset too) |
|
|
430
437
|
| `permission_mode` | string | no | Permission mode (e.g. `default`, `plan`, `bypassPermissions`) |
|
|
431
438
|
| `allowed_tools` | string | no | Comma-separated allowed tools (e.g. `"Bash,Read,Edit"`) |
|
|
432
439
|
| `disallowed_tools` | string | no | Comma-separated disallowed tools |
|
|
@@ -452,7 +459,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
|
|
|
452
459
|
| `risky_capabilities` | string | no | Comma-separated. `"none"` to clear |
|
|
453
460
|
| `instructions` | string | no | Standing orders for every dispatch (replaces the text). `"none"` to clear |
|
|
454
461
|
|
|
455
|
-
Changing an agent's config drops
|
|
462
|
+
Changing an agent's config through an MCP mutation tool drops its cached results immediately. Cache keys also include the agent configuration and dispatch defaults, so edits through the CLI, another server, or YAML take effect on the next request. A worker finishing with an older configuration cannot overwrite answers for the new configuration. Unchanged agents keep their cache.
|
|
456
463
|
|
|
457
464
|
### `remove_agent`
|
|
458
465
|
|
|
@@ -507,11 +514,11 @@ dispatch_wait(job_id="8f3a...e1", timeout_seconds=120)
|
|
|
507
514
|
-> {"id": "...", "status": "running", "timed_out_waiting": true}
|
|
508
515
|
```
|
|
509
516
|
|
|
510
|
-
`dispatch_cancel(job_id)` cancels a **pending** job, and also kills a **running** job's `claude`
|
|
517
|
+
`dispatch_cancel(job_id)` cancels a **pending** job, and also kills a **running** job's `claude` process group when the job was started by the same server instance (the job is marked `cancelled` first, so the worker's trailing write can't undo it; partial work is lost but the progress tail is preserved). On POSIX, descendants in that process group are killed too, releasing inherited pipes and the concurrency slot. A cancel that lands after the job was claimed but before its process spawned is honored too: the process is killed as soon as it starts. A compatibility retry that races cancellation is killed when it registers. A running job started by a *previous* server run can't be killed safely and is left to finish. The response carries an `outcome` of `cancelled`, `cancelled_running`, `running` (not owned by this server), `already_terminal`, or `not_found`.
|
|
511
518
|
|
|
512
519
|
Async workers run with streaming under the hood: the job file keeps a rolling tail (last 20 lines, ~1 write/sec) of assistant text and tool-use events. `dispatch_status` shows it as `progress` while the job runs and keeps it afterwards as a post-mortem trace; `dispatch_jobs` shows `last_progress` for running jobs.
|
|
513
520
|
|
|
514
|
-
`dispatch_jobs(status?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted.
|
|
521
|
+
`dispatch_jobs(status?, limit=50, agent?)` lists recent jobs as summaries (filter by `pending` / `running` / `done` / `failed` / `cancelled`). `agent` matches an exact name, including agents since removed from configuration; both filters apply before the limit. `dispatch_gc(max_age_days=7)` purges terminal jobs older than the threshold — pending and running jobs are never deleted. Use `dry_run=True` to preview eligible job IDs without deleting them.
|
|
515
522
|
|
|
516
523
|
Job state persists to disk at `~/.config/agent-dispatch/jobs/` (override with `AGENT_DISPATCH_JOBS_DIR`). One JSON file per job, written owner-only (`0o600`) with atomic writes — safe to read or `ls` while jobs are in flight. Caller-supplied `job_id`s are validated as 32-char hex before any file access (no path traversal). On startup the server recovers jobs a crashed instance abandoned: `running` ones stuck over an hour, and `pending` ones over 24 hours, are marked `failed` so they stop being polled forever and become collectable by `dispatch_gc`. (The `pending` threshold is deliberately long — the jobs directory is shared by every running server, so a job queued behind another server's concurrency limit must not be swept.)
|
|
517
524
|
|
|
@@ -545,7 +552,7 @@ Failures are deterministic: check `success`, then branch on `error_type`.
|
|
|
545
552
|
| `error_type` | Meaning | Recovery |
|
|
546
553
|
|--------------|---------|----------|
|
|
547
554
|
| `permission` | A tool call was denied | `update_agent(name, allowed_tools="Bash,Read")` (least privilege) or `update_agent(name, permission_mode="bypassPermissions")`, then re-dispatch. The `error` text includes a hint with the exact fix. |
|
|
548
|
-
| `timeout` | Process killed at the timeout | Resume the partial work: `dispatch_session(agent, "Continue where you left off", session_id=<from the error text>)`. Or retry with a bigger `timeout_seconds=` — once the journal has history the error names the value that would have covered this agent's p90 — or use `dispatch_async`.
|
|
555
|
+
| `timeout` | Process group killed at the timeout | Resume the partial work: `dispatch_session(agent, "Continue where you left off", session_id=<from the error text>)`. Or retry with a bigger `timeout_seconds=` — once the journal has history the error names the value that would have covered this agent's p90 — or use `dispatch_async`. Both ordinary and streaming dispatches preserve a complete CLI result received before termination; a successful result includes a cleanup `hint`. Partial output remains a timeout. |
|
|
549
556
|
| `not_found` | Agent directory or `claude` CLI missing | `list_agents()` → check `healthy`. Re-add the agent with an existing path, or run `agent-dispatch doctor` to find what's missing. |
|
|
550
557
|
| `recursion` | Dispatch nesting exceeded `max_dispatch_depth` (default 3) | Don't dispatch from dispatched agents; if the nesting is intentional, raise `max_dispatch_depth` in settings. |
|
|
551
558
|
| `budget` | The `claude` CLI ended the session at the `max_budget_usd` spend cap — the answer is incomplete | Raise the cap (`update_agent(name, max_budget_usd=2.0)`), switch to a cheaper `model`, or split the task. The partial session is resumable: `dispatch_session(agent, "Continue where you left off", session_id=<from the result>)`. |
|
|
@@ -617,7 +624,7 @@ settings:
|
|
|
617
624
|
# - Read
|
|
618
625
|
# - Edit
|
|
619
626
|
max_dispatch_depth: 3 # recursion protection
|
|
620
|
-
max_concurrency: 5 # max parallel claude -p processes
|
|
627
|
+
max_concurrency: 5 # max parallel claude -p processes per server, across all tools
|
|
621
628
|
# dispatch_protocol: true # send every agent the dispatch protocol (non-interactive,
|
|
622
629
|
# # time budget, STATUS line → `outcome`). false = raw claude -p.
|
|
623
630
|
# usage_log: true # record every dispatch in usage.jsonl (cost, duration, outcome).
|
|
@@ -689,8 +696,19 @@ past dispatches and deleting them cannot be undone. Pending and running jobs are
|
|
|
689
696
|
never touched.
|
|
690
697
|
|
|
691
698
|
`agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
|
|
692
|
-
one-off.
|
|
693
|
-
|
|
699
|
+
one-off. Preview eligible records before deleting them:
|
|
700
|
+
|
|
701
|
+
```bash
|
|
702
|
+
agent-dispatch gc --days 30 --dry-run
|
|
703
|
+
agent-dispatch gc --days 30
|
|
704
|
+
```
|
|
705
|
+
|
|
706
|
+
`--dry-run` also works with `--all`. The MCP equivalent is
|
|
707
|
+
`dispatch_gc(max_age_days=30, dry_run=True)`, returning `dry_run: true`,
|
|
708
|
+
`purged: 0`, `would_purge`, `job_ids`, and `max_age_days`.
|
|
709
|
+
A preview is a snapshot; deletion checks eligibility again, so its count can
|
|
710
|
+
change as jobs finish. Unreadable records, invalid timestamps, and records whose
|
|
711
|
+
internal ID differs from their filename are skipped and preserved for inspection.
|
|
694
712
|
|
|
695
713
|
### Auto-Description
|
|
696
714
|
|
|
@@ -725,8 +743,8 @@ Your Claude Code session
|
|
|
725
743
|
▼
|
|
726
744
|
agent-dispatch MCP server
|
|
727
745
|
├─ cache check → hit? return cached result
|
|
728
|
-
├─
|
|
729
|
-
└─ subprocess.
|
|
746
|
+
├─ shared process limit → bound concurrency across all dispatch tools
|
|
747
|
+
└─ subprocess.Popen(["claude", "-p", ...], cwd=~/projects/infra/)
|
|
730
748
|
│
|
|
731
749
|
▼
|
|
732
750
|
New Claude Code session in ~/projects/infra/
|
|
@@ -745,9 +763,9 @@ agent-dispatch MCP server
|
|
|
745
763
|
- **Path-traversal guard** — caller-supplied `job_id`/`ref` values are validated as 32-char hex before any filesystem access.
|
|
746
764
|
- **Owner-only state** — job files, `agents.yaml`, the usage journal and the server registry are all written `0o600`; their directories are `0o700`.
|
|
747
765
|
- **Cost control** — `max_budget_usd` per agent or globally is passed to the `claude` CLI as `--max-budget-usd`, so a runaway dispatch is stopped at the cap and comes back as `error_type: "budget"` with a resumable `session_id`. An overshoot that lands over budget without stopping is flagged post-hoc with `budget_exceeded: true` + a hint.
|
|
748
|
-
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes.
|
|
749
|
-
- **Timeout** — per-agent or global (default: 300s).
|
|
750
|
-
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, results flagged `budget_exceeded`, and results the agent itself reported as `partial` / `blocked` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
766
|
+
- **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes across all dispatch tools within one server process. Ordinary, streaming, and background dispatches share the same limit, and waiting calls are served in arrival order, so a queue of background jobs cannot starve an interactive call. Cancelling a waiting call never starts its process; cancelling an already-started ordinary or streaming call keeps its slot occupied until the worker ends. Reloading a changed limit preserves occupied slots: lowering it lets existing work finish and holds new launches until capacity is available. Separate server processes each have their own limit.
|
|
767
|
+
- **Timeout** — per-agent or global (default: 300s). Both ordinary and streaming dispatches run the agent in its own process group. On POSIX, the deadline kills that group, including descendants that inherited output pipes. Ordinary dispatch also bounds cleanup to one additional second if a descendant deliberately detached from the group; that detached process is outside the group's reach. A complete CLI result survives cleanup timeout, with a hint on successful results.
|
|
768
|
+
- **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests under the same agent configuration and dispatch defaults return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, results flagged `budget_exceeded`, and results the agent itself reported as `partial` / `blocked` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
|
|
751
769
|
- **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
|
|
752
770
|
|
|
753
771
|
See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
|
|
@@ -765,13 +783,50 @@ See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassP
|
|
|
765
783
|
| `agent-dispatch describe <name>` | Show full configuration for one agent (tri-state tools, project files) |
|
|
766
784
|
| `agent-dispatch test <name> [task] [--stream]` | Test an agent with a dispatch (`--stream` for live progress) |
|
|
767
785
|
| `agent-dispatch stats [--days N --agent X --json]` | What dispatches cost: spend, durations, outcomes and failures per agent |
|
|
768
|
-
| `agent-dispatch doctor` | Diagnose installation
|
|
769
|
-
| `agent-dispatch jobs [--status --limit]` | List
|
|
770
|
-
| `agent-dispatch job <id
|
|
786
|
+
| `agent-dispatch doctor [--json --strict]` | Diagnose installation, stale servers, MCP registration, agent health, and group membership; `--strict` also fails on warnings |
|
|
787
|
+
| `agent-dispatch jobs [--status STATUS --agent NAME --limit N --json]` | List recent job summaries, including results stored with `return_ref` |
|
|
788
|
+
| `agent-dispatch job <id> [--json]` | Show one job; `--json` exports the full record without truncation |
|
|
771
789
|
| `agent-dispatch cancel <id>` | Cancel a pending job (running jobs: use the `dispatch_cancel` MCP tool) |
|
|
772
|
-
| `agent-dispatch gc [--days N \| --all]` | Purge terminal jobs older than N days (default 7; `--all` purges every age) |
|
|
790
|
+
| `agent-dispatch gc [--days N \| --all] [--dry-run]` | Purge terminal jobs older than N days (default 7; `--all` purges every age); `--dry-run` previews without deleting |
|
|
773
791
|
| `agent-dispatch serve` | Start MCP server (stdio, used by Claude Code) |
|
|
774
792
|
|
|
793
|
+
To check an installation from a script:
|
|
794
|
+
|
|
795
|
+
```bash
|
|
796
|
+
agent-dispatch doctor --json --strict > diagnostics.json
|
|
797
|
+
```
|
|
798
|
+
|
|
799
|
+
The report contains `schema_version: 1`, overall `status` (`ok`, `warn`, or
|
|
800
|
+
`fail`), `issues` and `warnings` counts, and a `checks` array. Each check has a
|
|
801
|
+
`section`, `status`, `message`, and `details` array containing any remediation
|
|
802
|
+
steps or supporting information. JSON is emitted even when checks fail.
|
|
803
|
+
Failures exit with code 1; warnings exit with code 0 unless `--strict` is set.
|
|
804
|
+
The same checks and exit policy apply to the normal human-readable output.
|
|
805
|
+
This checks installation and configuration; it does not run a paid dispatch.
|
|
806
|
+
|
|
807
|
+
To inspect history from scripts or save a full result:
|
|
808
|
+
|
|
809
|
+
```bash
|
|
810
|
+
agent-dispatch jobs --agent infra --status failed --limit 10 --json
|
|
811
|
+
agent-dispatch job <job_id> --json > job.json
|
|
812
|
+
```
|
|
813
|
+
|
|
814
|
+
`jobs --json` returns an array of compact summaries in the same format as
|
|
815
|
+
`dispatch_jobs`, or `[]` if no jobs match. Each summary includes the ID, agent,
|
|
816
|
+
status, first 120 task characters, and timestamps; result cost, success, outcome,
|
|
817
|
+
error type, and latest running progress are included when available.
|
|
818
|
+
`job --json` returns the full stored job, including task/context, progress,
|
|
819
|
+
result text, structured result, session ID, and recovery hints when present.
|
|
820
|
+
Both formats preserve Unicode. Reading a failed job still exits successfully:
|
|
821
|
+
check its `status` and `result.success` to determine the dispatch outcome.
|
|
822
|
+
Missing/invalid job IDs and storage-open failures return `{"error": "..."}`
|
|
823
|
+
with exit code 1. Invalid command-line options are reported by Click on stderr.
|
|
824
|
+
|
|
825
|
+
Invalid or non-finite cost metadata is treated as unknown and omitted from
|
|
826
|
+
result exports; it does not hide the stored result text. Usage summaries ignore
|
|
827
|
+
unusable numeric measurements, and doctor omits uptime when a server's start
|
|
828
|
+
timestamp is unreadable.
|
|
829
|
+
|
|
775
830
|
## Requirements
|
|
776
831
|
|
|
777
832
|
- Python >= 3.10
|