agent-dispatch 0.13.0__tar.gz → 0.14.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/AGENTS.md +5 -3
  2. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/CHANGELOG.md +49 -0
  3. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/PKG-INFO +40 -6
  4. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/README.md +39 -5
  5. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/agents.example.yaml +13 -0
  6. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/pyproject.toml +1 -1
  7. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/__init__.py +1 -1
  8. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/cache.py +6 -0
  9. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/cli.py +57 -1
  10. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/config.py +1 -1
  11. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/models.py +17 -0
  12. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/runner.py +201 -42
  13. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/server.py +33 -0
  14. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_cache.py +15 -0
  15. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_cli.py +133 -18
  16. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_config.py +17 -0
  17. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_models.py +12 -3
  18. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_runner.py +202 -4
  19. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_server.py +133 -2
  20. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/.github/dependabot.yml +0 -0
  21. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/.github/workflows/ci.yml +0 -0
  22. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/.github/workflows/publish.yml +0 -0
  23. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/.gitignore +0 -0
  24. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/LICENSE +0 -0
  25. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/SECURITY.md +0 -0
  26. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/assets/mascot.png +0 -0
  27. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/src/agent_dispatch/jobs.py +0 -0
  28. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/__init__.py +0 -0
  29. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/conftest.py +0 -0
  30. {agent_dispatch-0.13.0 → agent_dispatch-0.14.0}/tests/test_jobs.py +0 -0
@@ -28,7 +28,7 @@ pip install -e ".[dev]"
28
28
 
29
29
  ```bash
30
30
  ruff check src/ tests/
31
- python3 -m pytest tests/ -v # 578 tests, ~5s
31
+ python3 -m pytest tests/ -v # 637 tests, ~15s
32
32
  ```
33
33
 
34
34
  Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.which` + `subprocess.run`/`Popen`; server tests mock `_get_config` + `runner.dispatch`. The one exception is `TestStreamPipeHandling`, which spawns a short-lived *python* subprocess: a pipe deadlock lives in the OS pipe buffer, so a mocked `Popen` structurally cannot reproduce it.
@@ -60,7 +60,9 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
60
60
  - Pydantic does **not** validate on assignment. `Field(ge=...)` guards only the *load* path; every mutation surface (CLI `add`/`update`, MCP `add_agent`/`update_agent`) needs its own boundary check, or the bound escapes as a raw `ValidationError`.
61
61
  - Every state file (`agents.yaml`, job files) is written **temp file + `os.replace`**, never in place, and every load/mutate/save is wrapped in `config.ProcessLock` — the CLI and the MCP server are separate processes writing the same files, so a thread lock alone loses updates.
62
62
  - Anything that changes an agent's config must call `_invalidate_agent_cache` — the cache key holds the agent *name*, not its directory or permissions.
63
- - Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, and `budget_exceeded` results, so the documented "grant access, then re-dispatch" recovery is never short-circuited.
63
+ - Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, `budget_exceeded` results, and results whose `outcome` is `partial`/`blocked`, so the documented "grant access / fix the cause, then re-dispatch" recovery is never short-circuited.
64
+ - **The dispatch protocol goes in the system prompt, not the task.** `_build_system_prompt` (runner.py) renders the protocol (non-interactive, time/spend budget, lead with the outcome, trailing `STATUS:` line) plus the agent's `instructions`, and `_build_command` passes it as `--append-system-prompt` — on `--resume` too, since the CLI re-applies an appended prompt on every launch. `_build_prompt` (the `-p` text) is unchanged, so the cache key is unchanged. The text always starts with a `##` header line so it can never be read as a flag. `settings.dispatch_protocol: false` turns the protocol off (for CLIs that predate the flag; `doctor` probes `claude --help` for it); per-agent `instructions` still go through when set.
65
+ - **`outcome` is lifted from the LAST line of the result** (`_split_outcome`), before JSON parsing, by the shared `_build_success_result` (the success-side twin of `_build_error_result` — both `dispatch` and `dispatch_stream` go through it) and by the plain-text fallback tier. A bare marker line is removed from `result`; a marker with a trailing reason stays. `outcome` never flips `success`; `partial`/`blocked` add a `hint` with the `dispatch_session(...)` continuation, after the denial hint. In JSON mode the protocol omits the STATUS instruction (the JSON footer governs), but a STATUS line that arrives anyway is still stripped so `parsed_result` survives.
64
66
  - Remediation text is a contract: a hint that names a flag must name one that exists (`test_printed_budget_hint_is_a_runnable_command` feeds the printed flags back into the CLI). Run the command you print.
65
67
  - The config error sets are declared **once** and in two halves: `config.CONFIG_LOAD_ERRORS` (read) and `config.CONFIG_SAVE_ERRORS` (write — `yaml.dump`'s `RepresenterError` is a `yaml.YAMLError`, therefore neither `OSError` nor `ConfigLoadError`, and used to escape both the MCP guard and the CLI's `_save_or_exit`). Two halves, not one set, because the remediations differ: a failed write is atomic so the old config survives, while a failed read needs the YAML fixed.
66
68
  - MCP tools that load config carry `@_config_guard` under `@mcp.tool()` so a broken `agents.yaml` — or a failed *write* — returns the `{"error": ...}` envelope instead of a raw traceback. The set of load errors lives in one place (`config.CONFIG_LOAD_ERRORS`) because three surfaces handle it: **`UnicodeDecodeError` is a `ValueError`, not an `OSError`**, and listing types per-site is exactly how a cp1251 config slipped past all three.
@@ -85,4 +87,4 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
85
87
 
86
88
  ## More detail
87
89
 
88
- [README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 578 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
90
+ [README.md](README.md) documents every MCP tool with parameter tables, response shapes, and the error-recovery map — it doubles as the behavioral spec. The test suite (`tests/`, 637 tests) encodes the exact expected behavior of every layer: when in doubt, read the tests for the module you're touching (`test_runner.py`, `test_server.py`, `test_cli.py`, ...).
@@ -7,6 +7,55 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.14.0] - 2026-09-08
11
+
12
+ The delegation round: what a dispatched agent is *told*, and what it tells
13
+ back. Until now `claude -p` was launched with the task and nothing else — it
14
+ did not know it was being driven by another agent, that nobody would answer a
15
+ question, how long it had, or how to report an unfinished job. The failure
16
+ mode was concrete and billed: an ambiguous task ended in *"Could you clarify
17
+ which service you mean?"*, `success: true`, cached for the whole TTL.
18
+
19
+ ### Added
20
+ - **The dispatch protocol.** Every dispatch now appends a short system prompt
21
+ (`--append-system-prompt`, never mixed into the task text) telling the agent
22
+ that it is non-interactive and dispatched by `caller`, that it must state
23
+ assumptions instead of asking, that a denied tool is something to report and
24
+ work around rather than stop on, what its time budget (and spend cap) is, to
25
+ lead with the outcome — which is what makes a `return_ref` summary, the
26
+ *head* of the text, worth reading — and to end with one line:
27
+ `STATUS: done | partial | blocked`. Passed on `--resume` too. Verified live
28
+ against claude 2.1.263 (`agent-dispatch test <agent> --stream`).
29
+ - **`DispatchResult.outcome`** — that trailing line, lifted out of `result`
30
+ into a field: `done`, `partial` or `blocked`, or absent when the agent did
31
+ not report one. It never flips `success`. `partial`/`blocked` come with a
32
+ `hint` carrying the exact `dispatch_session(..., session_id=...)` call to
33
+ continue, ride the `return_ref` payload and `dispatch_jobs` summaries, are
34
+ shown by `agent-dispatch test` / `job <id>`, are labelled for the
35
+ `dispatch_parallel` aggregator so a blocked member is not synthesized as a
36
+ finished one — and are **not cached**: whatever the agent was missing is not
37
+ in the cache key, so a retry after fixing it must run fresh.
38
+ - **Per-agent `instructions`** — standing orders appended after the protocol
39
+ on every dispatch ("read-only SQL only", "never restart a stack unless the
40
+ task says so"). `add_agent`/`update_agent` (`"none"` clears), CLI `add` /
41
+ `update --instructions` (`none` clears), shown by `inspect_agent` and
42
+ `describe`, pruned from YAML when empty. Changing them invalidates the
43
+ agent's cache entries like any other config change.
44
+ - **`settings.dispatch_protocol`** (default `true`) — set to `false` for a raw
45
+ `claude -p` (a CLI that predates `--append-system-prompt`, or an A/B). The
46
+ protocol and the instructions share one flag, so instructions still go
47
+ through when set. `doctor` now probes `claude --help` for the flag and warns
48
+ with the exact remediation when it is missing.
49
+
50
+ ### Changed
51
+ - In `response_format="json"` mode the protocol omits the "lead with the
52
+ outcome" and STATUS bullets — the JSON footer governs the reply shape. A
53
+ STATUS line that arrives anyway is still stripped before parsing, so
54
+ `parsed_result` survives it.
55
+ - `dispatch` and `dispatch_stream` build their success result through one
56
+ `_build_success_result`, the twin of `_build_error_result`.
57
+
58
+
10
59
  ## [0.13.0] - 2026-08-13
11
60
 
12
61
  An efficiency round, measured against a real 38 KB config (4 agents, 6 groups),
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: agent-dispatch
3
- Version: 0.13.0
3
+ Version: 0.14.0
4
4
  Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
5
5
  Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
6
6
  Project-URL: Repository, https://github.com/ginkida/agent-dispatch
@@ -220,7 +220,8 @@ dispatch(
220
220
  "session_id": "sess-abc-123",
221
221
  "cost_usd": 0.02,
222
222
  "duration_ms": 5000,
223
- "num_turns": 2
223
+ "num_turns": 2,
224
+ "outcome": "done"
224
225
  }
225
226
 
226
227
  // Response (failure — error_type helps you handle programmatically)
@@ -278,6 +279,30 @@ Error: TypeError at scheduler.py:42
278
279
  Check container logs for recent errors related to the scheduler service
279
280
  ```
280
281
 
282
+ **The dispatch protocol and `outcome`.** `claude -p` on its own does not know it is being driven by another agent: given an ambiguous task it will happily end with *"Could you clarify which service you mean?"* — a billed run that answered nothing, and one that a plain cache would then serve again for the whole TTL. So every dispatch also appends a short **dispatch protocol** to the agent's system prompt (`--append-system-prompt`, never mixed into your task text):
283
+
284
+ - it is running non-interactively, dispatched by `caller`, and nobody will answer a question or approve an action — state the assumption and proceed, take the safer reading when two differ;
285
+ - a denied tool or a missing thing is something to report, not something to stop on — finish everything else that is possible;
286
+ - its time budget is about the agent's timeout (and its spend cap, if one is set) — scope the work to fit; a complete partial answer beats an unfinished perfect one;
287
+ - lead with the outcome, then the evidence; list what could not be done and why (this is what makes a `return_ref` summary — the **head** of the text — worth reading);
288
+ - end with one line, `STATUS: done`, `STATUS: partial` or `STATUS: blocked`.
289
+
290
+ That last line is lifted out of `result` into **`outcome`** — the agent's own verdict, deterministic to check: `"done"` means complete; `"partial"` / `"blocked"` mean the agent says the work is unfinished, the result names what is missing, and a `hint` spells out the continuation (usually `dispatch_session(..., session_id=...)`). Read `outcome` before reading the text. It never flips `success`, `partial` / `blocked` results are **not cached**, and a `dispatch_parallel(..., aggregate=...)` labels such members for the aggregator so a blocked report is not synthesized as a finished one. Absent when the agent did not report one — the protocol is off (`settings.dispatch_protocol: false`), `response_format="json"` was requested (the JSON footer governs the reply shape there), or the agent simply skipped it.
291
+
292
+ **Standing orders.** Per-agent `instructions` (set via `add_agent` / `update_agent` or `agent-dispatch update <name> --instructions "..."`) follow the protocol in the same system prompt on **every** dispatch — "read-only SQL only", "never restart a stack unless the task says so", "answer with exact log lines". Unlike `context`, which is per call, and unlike the project's own `CLAUDE.md`, which is written for an interactive session, these are the rules for *being dispatched*.
293
+
294
+ ```json
295
+ // Response (success, but the agent says it did not finish)
296
+ {
297
+ "agent": "infra",
298
+ "success": true,
299
+ "result": "Restarted horizon. Could not verify the queue drained: the redis container is not reachable from here.",
300
+ "session_id": "sess-abc-123",
301
+ "outcome": "partial",
302
+ "hint": "The agent reports its work is PARTIAL — read the result for what is missing, then continue in the same session via dispatch_session(agent='infra', task='Continue where you left off', session_id='sess-abc-123') or re-dispatch with what it needed."
303
+ }
304
+ ```
305
+
281
306
  ### `dispatch_session`
282
307
 
283
308
  Multi-turn: continue a conversation with an agent. First call starts a session, pass `session_id` back to continue. Never cached.
@@ -395,6 +420,7 @@ Register a new project directory as an agent. Description is auto-generated from
395
420
  | `disallowed_tools` | string | no | Comma-separated disallowed tools |
396
421
  | `capabilities` | string | no | Comma-separated capability labels (e.g. `"docker_logs,deploy_debug"`) |
397
422
  | `risky_capabilities` | string | no | Comma-separated high-risk labels (e.g. `"restart_services"`) |
423
+ | `instructions` | string | no | Standing orders appended to the agent's system prompt on every dispatch (see [the dispatch protocol](#dispatch)) |
398
424
 
399
425
  ### `update_agent`
400
426
 
@@ -412,6 +438,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
412
438
  | `disallowed_tools` | string | no | Comma-separated. `"none"` to clear |
413
439
  | `capabilities` | string | no | Comma-separated. `"none"` to clear |
414
440
  | `risky_capabilities` | string | no | Comma-separated. `"none"` to clear |
441
+ | `instructions` | string | no | Standing orders for every dispatch (replaces the text). `"none"` to clear |
415
442
 
416
443
  Changing an agent's config drops that agent's cached results — the cache key holds the agent *name*, so a re-pointed or re-permissioned agent would otherwise keep answering from the previous config for the rest of the TTL. The same applies to `add_agent` and `remove_agent`.
417
444
 
@@ -515,6 +542,7 @@ Failures are deterministic: check `success`, then branch on `error_type`.
515
542
  Three soft signals that arrive with `success: true`:
516
543
 
517
544
  - **`denied_tools` + `hint`** — the agent finished but some tool calls were blocked; the result may be incomplete. Grant access (see the `permission` row) and re-dispatch.
545
+ - **`outcome: "partial"` / `"blocked"`** — the agent's own verdict that it did not finish; the result names what is missing and the `hint` carries the `dispatch_session(...)` call to continue in the same session. Not cached, so a re-dispatch after fixing the cause runs fresh.
518
546
  - **`parsed_result: null` with `response_format="json"`** — the reply wasn't valid JSON; the raw text is still in `result`. Caveat: an agent that *can't* comply returns `{"error": "<reason>"}` — which parses successfully — so also check `parsed_result` for an `"error"` key.
519
547
  - **`budget_exceeded: true`** — `cost_usd` came in over the agent's `max_budget_usd` (or the settings default) without the CLI stopping the run (the final turn can overshoot the cap). The dispatch is not failed — the money is already spent — but a runaway agent is now visible. Tighten the task, pick a cheaper model, or raise the budget. A run the CLI *did* stop fails with `error_type: "budget"` instead.
520
548
 
@@ -539,6 +567,9 @@ agents:
539
567
  - deploy_debug
540
568
  risky_capabilities: # high-risk labels, surfaced for visibility
541
569
  - restart_services
570
+ # instructions: | # standing orders, appended to the system prompt on every dispatch
571
+ # Read-only: never restart or redeploy unless the task says so.
572
+ # Quote exact log lines with timestamps.
542
573
  # model: sonnet # optional model override
543
574
  # max_budget_usd: 1.0 # cost limit per dispatch
544
575
  # permission_mode: bypassPermissions # one of: default | plan | bypassPermissions
@@ -574,6 +605,8 @@ settings:
574
605
  # - Edit
575
606
  max_dispatch_depth: 3 # recursion protection
576
607
  max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
608
+ # dispatch_protocol: true # send every agent the dispatch protocol (non-interactive,
609
+ # # time budget, STATUS line → `outcome`). false = raw claude -p.
577
610
  # job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
578
611
  cache:
579
612
  enabled: true
@@ -646,8 +679,9 @@ agent-dispatch MCP server
646
679
  ▼
647
680
  New Claude Code session in ~/projects/infra/
648
681
  ├─ Inherits: CLAUDE.md, .mcp.json, project tools
682
+ ├─ System prompt += dispatch protocol + the agent's standing `instructions`
649
683
  ├─ Receives structured prompt with goal/caller/context/task
650
- └─ Returns result → cached for future identical requests
684
+ └─ Returns result (+ its own STATUS → `outcome`) → cached when complete
651
685
  ```
652
686
 
653
687
  ## Safety
@@ -659,7 +693,7 @@ agent-dispatch MCP server
659
693
  - **Cost control** — `max_budget_usd` per agent or globally is passed to the `claude` CLI as `--max-budget-usd`, so a runaway dispatch is stopped at the cap and comes back as `error_type: "budget"` with a resumable `session_id`. An overshoot that lands over budget without stopping is flagged post-hoc with `budget_exceeded: true` + a hint.
660
694
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
661
695
  - **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
662
- - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
696
+ - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, results flagged `budget_exceeded`, and results the agent itself reported as `partial` / `blocked` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
663
697
  - **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
664
698
 
665
699
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
@@ -670,13 +704,13 @@ See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassP
670
704
  |---------|-------------|
671
705
  | `agent-dispatch init` | Create config + register MCP server with Claude Code |
672
706
  | `agent-dispatch add <name> <dir>` | Add an agent (auto-generates description) |
673
- | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, etc.) |
707
+ | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, `--instructions`, etc.) |
674
708
  | `agent-dispatch remove <name>` | Remove an agent |
675
709
  | `agent-dispatch list` | List agents with health status and permissions |
676
710
  | `agent-dispatch group <add\|list\|inspect\|update\|remove>` | Manage [groups](#groups) — cross-project working sets of agents |
677
711
  | `agent-dispatch describe <name>` | Show full configuration for one agent (tri-state tools, project files) |
678
712
  | `agent-dispatch test <name> [task] [--stream]` | Test an agent with a dispatch (`--stream` for live progress) |
679
- | `agent-dispatch doctor` | Diagnose installation: Claude CLI, MCP registration, agent health, and group membership |
713
+ | `agent-dispatch doctor` | Diagnose installation: Claude CLI (incl. `--append-system-prompt` support), MCP registration, agent health, and group membership |
680
714
  | `agent-dispatch jobs [--status --limit]` | List async dispatch jobs (most recent first) |
681
715
  | `agent-dispatch job <id>` | Show one job: status, progress tail, result preview |
682
716
  | `agent-dispatch cancel <id>` | Cancel a pending job (running jobs: use the `dispatch_cancel` MCP tool) |
@@ -190,7 +190,8 @@ dispatch(
190
190
  "session_id": "sess-abc-123",
191
191
  "cost_usd": 0.02,
192
192
  "duration_ms": 5000,
193
- "num_turns": 2
193
+ "num_turns": 2,
194
+ "outcome": "done"
194
195
  }
195
196
 
196
197
  // Response (failure — error_type helps you handle programmatically)
@@ -248,6 +249,30 @@ Error: TypeError at scheduler.py:42
248
249
  Check container logs for recent errors related to the scheduler service
249
250
  ```
250
251
 
252
+ **The dispatch protocol and `outcome`.** `claude -p` on its own does not know it is being driven by another agent: given an ambiguous task it will happily end with *"Could you clarify which service you mean?"* — a billed run that answered nothing, and one that a plain cache would then serve again for the whole TTL. So every dispatch also appends a short **dispatch protocol** to the agent's system prompt (`--append-system-prompt`, never mixed into your task text):
253
+
254
+ - it is running non-interactively, dispatched by `caller`, and nobody will answer a question or approve an action — state the assumption and proceed, take the safer reading when two differ;
255
+ - a denied tool or a missing thing is something to report, not something to stop on — finish everything else that is possible;
256
+ - its time budget is about the agent's timeout (and its spend cap, if one is set) — scope the work to fit; a complete partial answer beats an unfinished perfect one;
257
+ - lead with the outcome, then the evidence; list what could not be done and why (this is what makes a `return_ref` summary — the **head** of the text — worth reading);
258
+ - end with one line, `STATUS: done`, `STATUS: partial` or `STATUS: blocked`.
259
+
260
+ That last line is lifted out of `result` into **`outcome`** — the agent's own verdict, deterministic to check: `"done"` means complete; `"partial"` / `"blocked"` mean the agent says the work is unfinished, the result names what is missing, and a `hint` spells out the continuation (usually `dispatch_session(..., session_id=...)`). Read `outcome` before reading the text. It never flips `success`, `partial` / `blocked` results are **not cached**, and a `dispatch_parallel(..., aggregate=...)` labels such members for the aggregator so a blocked report is not synthesized as a finished one. Absent when the agent did not report one — the protocol is off (`settings.dispatch_protocol: false`), `response_format="json"` was requested (the JSON footer governs the reply shape there), or the agent simply skipped it.
261
+
262
+ **Standing orders.** Per-agent `instructions` (set via `add_agent` / `update_agent` or `agent-dispatch update <name> --instructions "..."`) follow the protocol in the same system prompt on **every** dispatch — "read-only SQL only", "never restart a stack unless the task says so", "answer with exact log lines". Unlike `context`, which is per call, and unlike the project's own `CLAUDE.md`, which is written for an interactive session, these are the rules for *being dispatched*.
263
+
264
+ ```json
265
+ // Response (success, but the agent says it did not finish)
266
+ {
267
+ "agent": "infra",
268
+ "success": true,
269
+ "result": "Restarted horizon. Could not verify the queue drained: the redis container is not reachable from here.",
270
+ "session_id": "sess-abc-123",
271
+ "outcome": "partial",
272
+ "hint": "The agent reports its work is PARTIAL — read the result for what is missing, then continue in the same session via dispatch_session(agent='infra', task='Continue where you left off', session_id='sess-abc-123') or re-dispatch with what it needed."
273
+ }
274
+ ```
275
+
251
276
  ### `dispatch_session`
252
277
 
253
278
  Multi-turn: continue a conversation with an agent. First call starts a session, pass `session_id` back to continue. Never cached.
@@ -365,6 +390,7 @@ Register a new project directory as an agent. Description is auto-generated from
365
390
  | `disallowed_tools` | string | no | Comma-separated disallowed tools |
366
391
  | `capabilities` | string | no | Comma-separated capability labels (e.g. `"docker_logs,deploy_debug"`) |
367
392
  | `risky_capabilities` | string | no | Comma-separated high-risk labels (e.g. `"restart_services"`) |
393
+ | `instructions` | string | no | Standing orders appended to the agent's system prompt on every dispatch (see [the dispatch protocol](#dispatch)) |
368
394
 
369
395
  ### `update_agent`
370
396
 
@@ -382,6 +408,7 @@ Update an existing agent's configuration. Only non-empty fields are changed. Pas
382
408
  | `disallowed_tools` | string | no | Comma-separated. `"none"` to clear |
383
409
  | `capabilities` | string | no | Comma-separated. `"none"` to clear |
384
410
  | `risky_capabilities` | string | no | Comma-separated. `"none"` to clear |
411
+ | `instructions` | string | no | Standing orders for every dispatch (replaces the text). `"none"` to clear |
385
412
 
386
413
  Changing an agent's config drops that agent's cached results — the cache key holds the agent *name*, so a re-pointed or re-permissioned agent would otherwise keep answering from the previous config for the rest of the TTL. The same applies to `add_agent` and `remove_agent`.
387
414
 
@@ -485,6 +512,7 @@ Failures are deterministic: check `success`, then branch on `error_type`.
485
512
  Three soft signals that arrive with `success: true`:
486
513
 
487
514
  - **`denied_tools` + `hint`** — the agent finished but some tool calls were blocked; the result may be incomplete. Grant access (see the `permission` row) and re-dispatch.
515
+ - **`outcome: "partial"` / `"blocked"`** — the agent's own verdict that it did not finish; the result names what is missing and the `hint` carries the `dispatch_session(...)` call to continue in the same session. Not cached, so a re-dispatch after fixing the cause runs fresh.
488
516
  - **`parsed_result: null` with `response_format="json"`** — the reply wasn't valid JSON; the raw text is still in `result`. Caveat: an agent that *can't* comply returns `{"error": "<reason>"}` — which parses successfully — so also check `parsed_result` for an `"error"` key.
489
517
  - **`budget_exceeded: true`** — `cost_usd` came in over the agent's `max_budget_usd` (or the settings default) without the CLI stopping the run (the final turn can overshoot the cap). The dispatch is not failed — the money is already spent — but a runaway agent is now visible. Tighten the task, pick a cheaper model, or raise the budget. A run the CLI *did* stop fails with `error_type: "budget"` instead.
490
518
 
@@ -509,6 +537,9 @@ agents:
509
537
  - deploy_debug
510
538
  risky_capabilities: # high-risk labels, surfaced for visibility
511
539
  - restart_services
540
+ # instructions: | # standing orders, appended to the system prompt on every dispatch
541
+ # Read-only: never restart or redeploy unless the task says so.
542
+ # Quote exact log lines with timestamps.
512
543
  # model: sonnet # optional model override
513
544
  # max_budget_usd: 1.0 # cost limit per dispatch
514
545
  # permission_mode: bypassPermissions # one of: default | plan | bypassPermissions
@@ -544,6 +575,8 @@ settings:
544
575
  # - Edit
545
576
  max_dispatch_depth: 3 # recursion protection
546
577
  max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
578
+ # dispatch_protocol: true # send every agent the dispatch protocol (non-interactive,
579
+ # # time budget, STATUS line → `outcome`). false = raw claude -p.
547
580
  # job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
548
581
  cache:
549
582
  enabled: true
@@ -616,8 +649,9 @@ agent-dispatch MCP server
616
649
  ▼
617
650
  New Claude Code session in ~/projects/infra/
618
651
  ├─ Inherits: CLAUDE.md, .mcp.json, project tools
652
+ ├─ System prompt += dispatch protocol + the agent's standing `instructions`
619
653
  ├─ Receives structured prompt with goal/caller/context/task
620
- └─ Returns result → cached for future identical requests
654
+ └─ Returns result (+ its own STATUS → `outcome`) → cached when complete
621
655
  ```
622
656
 
623
657
  ## Safety
@@ -629,7 +663,7 @@ agent-dispatch MCP server
629
663
  - **Cost control** — `max_budget_usd` per agent or globally is passed to the `claude` CLI as `--max-budget-usd`, so a runaway dispatch is stopped at the cap and comes back as `error_type: "budget"` with a resumable `session_id`. An overshoot that lands over budget without stopping is flagged post-hoc with `budget_exceeded: true` + a hint.
630
664
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
631
665
  - **Timeout** — per-agent or global (default: 300s). A streaming dispatch runs the agent in its own process group, so the deadline kills the whole tree: a process the agent left running in the background can't hold the dispatch (and its concurrency slot) open past the timeout.
632
- - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, and results flagged `budget_exceeded` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
666
+ - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only clean successes are cached: failures, results with `denied_tools`, results flagged `budget_exceeded`, and results the agent itself reported as `partial` / `blocked` are not, so the documented "grant access / raise the cap, then re-dispatch" recovery is never served a stale crippled answer. Changing an agent's config invalidates its entries. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
633
667
  - **Durable config** — `agents.yaml` is written atomically (temp file + rename), so an interrupted write can never truncate it. Every mutation path (CLI and MCP server alike) also takes a cross-process advisory lock, so concurrent edits don't drop one another's agents. The lock is best-effort by design: after waiting 10 seconds it logs a warning and proceeds anyway, because a wedged lock holder must not freeze the MCP server — so on a heavily contended config a lost update is possible, while a truncated one is not.
634
668
 
635
669
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
@@ -640,13 +674,13 @@ See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassP
640
674
  |---------|-------------|
641
675
  | `agent-dispatch init` | Create config + register MCP server with Claude Code |
642
676
  | `agent-dispatch add <name> <dir>` | Add an agent (auto-generates description) |
643
- | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, etc.) |
677
+ | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, `--instructions`, etc.) |
644
678
  | `agent-dispatch remove <name>` | Remove an agent |
645
679
  | `agent-dispatch list` | List agents with health status and permissions |
646
680
  | `agent-dispatch group <add\|list\|inspect\|update\|remove>` | Manage [groups](#groups) — cross-project working sets of agents |
647
681
  | `agent-dispatch describe <name>` | Show full configuration for one agent (tri-state tools, project files) |
648
682
  | `agent-dispatch test <name> [task] [--stream]` | Test an agent with a dispatch (`--stream` for live progress) |
649
- | `agent-dispatch doctor` | Diagnose installation: Claude CLI, MCP registration, agent health, and group membership |
683
+ | `agent-dispatch doctor` | Diagnose installation: Claude CLI (incl. `--append-system-prompt` support), MCP registration, agent health, and group membership |
650
684
  | `agent-dispatch jobs [--status --limit]` | List async dispatch jobs (most recent first) |
651
685
  | `agent-dispatch job <id>` | Show one job: status, progress tail, result preview |
652
686
  | `agent-dispatch cancel <id>` | Cancel a pending job (running jobs: use the `dispatch_cancel` MCP tool) |
@@ -9,6 +9,13 @@ agents:
9
9
  risky_capabilities:
10
10
  - restart_services
11
11
  timeout: 300
12
+ # Standing orders, appended to the agent's system prompt on EVERY dispatch
13
+ # (after the dispatch protocol). Unlike `context`, which is per call, and
14
+ # unlike the project's CLAUDE.md, which is written for an interactive
15
+ # session, these are the rules for being dispatched.
16
+ instructions: |
17
+ Read-only unless the task explicitly asks to restart or redeploy.
18
+ Quote exact log lines with timestamps; never paraphrase an error.
12
19
 
13
20
  # Analytics gateway — a browser + Yandex Metrica agent, no codebase of its own.
14
21
  # Its value is its MCP servers + access, not its source. Shared across groups.
@@ -93,6 +100,12 @@ settings:
93
100
  default_timeout: 300
94
101
  max_dispatch_depth: 3 # recursion protection: A -> B -> A
95
102
  max_concurrency: 5 # max parallel claude -p processes
103
+ # dispatch_protocol: true # Append the dispatch protocol to every agent's system
104
+ # prompt: it is non-interactive (never end with a
105
+ # question), its time/spend budget, lead with the
106
+ # outcome, end with `STATUS: done|partial|blocked`
107
+ # (→ DispatchResult.outcome). false = raw claude -p;
108
+ # needed only for CLIs without --append-system-prompt.
96
109
  # default_max_budget_usd: 1.0 # spend cap per dispatch, passed to claude as --max-budget-usd
97
110
  # (a run stopped at the cap fails with error_type: budget)
98
111
  # default_permission_mode: bypassPermissions # inherited by agents without override
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agent-dispatch"
3
- version = "0.13.0"
3
+ version = "0.14.0"
4
4
  description = "MCP server that lets Claude Code agents delegate tasks to agents in other project directories"
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """agent-dispatch: Delegate tasks between Claude Code agents across projects."""
2
2
 
3
- __version__ = "0.13.0"
3
+ __version__ = "0.14.0"
@@ -102,6 +102,12 @@ class DispatchCache:
102
102
  # serve the same crippled answer back for the whole TTL and make that
103
103
  # recovery a no-op (the permission config is not part of the key).
104
104
  return
105
+ if result.outcome in ("partial", "blocked"):
106
+ # The agent itself says the work is unfinished. Whatever it was
107
+ # missing (a service that was down, a file that did not exist yet)
108
+ # is not in the key either, so a retry after fixing it would be
109
+ # served the same unfinished answer for the whole TTL.
110
+ return
105
111
  key = self._make_key(agent, task, context, caller, goal, response_format)
106
112
  with self._lock:
107
113
  # Bound memory: when at capacity and inserting a new key, evict the
@@ -101,6 +101,22 @@ def _check_timeout_or_exit(timeout: int | None) -> None:
101
101
  raise SystemExit(1)
102
102
 
103
103
 
104
+ def _claude_supports_flag(claude_path: str, flag: str) -> bool | None:
105
+ """Whether `claude --help` lists *flag*; None when the probe itself failed."""
106
+ try:
107
+ proc = subprocess.run(
108
+ [claude_path, "--help"],
109
+ capture_output=True,
110
+ text=True,
111
+ encoding="utf-8",
112
+ errors="replace",
113
+ timeout=10,
114
+ )
115
+ except (OSError, subprocess.TimeoutExpired):
116
+ return None
117
+ return flag in (proc.stdout or "") + (proc.stderr or "")
118
+
119
+
104
120
  def _load_or_exit() -> DispatchConfig:
105
121
  """Load config, exiting with a friendly error on malformed YAML or schema."""
106
122
  try:
@@ -233,6 +249,11 @@ def init() -> None:
233
249
  default=None,
234
250
  help="Comma-separated risky capabilities (e.g. restart_services).",
235
251
  )
252
+ @click.option(
253
+ "--instructions",
254
+ default=None,
255
+ help="Standing orders appended to the agent's system prompt on every dispatch.",
256
+ )
236
257
  def add(
237
258
  name: str,
238
259
  directory: str,
@@ -245,6 +266,7 @@ def add(
245
266
  disallowed_tools: str | None,
246
267
  capabilities: str | None,
247
268
  risky_capabilities: str | None,
269
+ instructions: str | None,
248
270
  ) -> None:
249
271
  """Add an agent. Auto-generates description from project files if omitted."""
250
272
  try:
@@ -281,6 +303,7 @@ def add(
281
303
  disallowed_tools=_parse_csv(disallowed_tools),
282
304
  capabilities=_parse_csv(capabilities) or [],
283
305
  risky_capabilities=_parse_csv(risky_capabilities) or [],
306
+ instructions=(instructions or "").strip(),
284
307
  )
285
308
  if warning := check_permission_mode(permission_mode):
286
309
  click.echo(click.style(f"Warning: {warning}", fg="yellow"))
@@ -388,6 +411,11 @@ def list_agents() -> None:
388
411
  default=None,
389
412
  help="Comma-separated risky capabilities. Use 'none' to clear.",
390
413
  )
414
+ @click.option(
415
+ "--instructions",
416
+ default=None,
417
+ help="Standing orders for every dispatch (replaces the text). Use 'none' to clear.",
418
+ )
391
419
  @click.pass_context
392
420
  def update(
393
421
  ctx: click.Context,
@@ -401,6 +429,7 @@ def update(
401
429
  disallowed_tools: str | None,
402
430
  capabilities: str | None,
403
431
  risky_capabilities: str | None,
432
+ instructions: str | None,
404
433
  ) -> None:
405
434
  """Update an existing agent's configuration."""
406
435
  _check_timeout_or_exit(timeout)
@@ -424,6 +453,7 @@ def update(
424
453
  disallowed_tools=disallowed_tools,
425
454
  capabilities=capabilities,
426
455
  risky_capabilities=risky_capabilities,
456
+ instructions=instructions,
427
457
  )
428
458
 
429
459
  if not updated:
@@ -446,6 +476,7 @@ def _apply_cli_updates(
446
476
  disallowed_tools: str | None,
447
477
  capabilities: str | None,
448
478
  risky_capabilities: str | None,
479
+ instructions: str | None = None,
449
480
  ) -> list[str]:
450
481
  """Apply `update`'s explicitly-passed options in place; return the fields touched."""
451
482
  updated: list[str] = []
@@ -493,6 +524,10 @@ def _apply_cli_updates(
493
524
  else:
494
525
  agent.risky_capabilities = _parse_csv(risky_capabilities) or []
495
526
  updated.append("risky_capabilities")
527
+ if instructions is not None:
528
+ stripped = instructions.strip()
529
+ agent.instructions = "" if stripped.lower() == "none" else stripped
530
+ updated.append("instructions")
496
531
 
497
532
  return updated
498
533
 
@@ -551,7 +586,11 @@ def test(name: str, task: str, stream: bool, timeout: int | None) -> None:
551
586
  click.echo()
552
587
  click.echo(click.style(f"Note: {result.hint}", fg="yellow"))
553
588
  if result.cost_usd is not None:
554
- click.echo(f"\n--- Cost: ${result.cost_usd:.4f} | Turns: {result.num_turns}")
589
+ tail = f"\n--- Cost: ${result.cost_usd:.4f} | Turns: {result.num_turns}"
590
+ if result.outcome:
591
+ color = "green" if result.outcome == "done" else "yellow"
592
+ tail += f" | Outcome: {click.style(result.outcome, fg=color)}"
593
+ click.echo(tail)
555
594
  else:
556
595
  click.echo(click.style(f"Error: {result.error}", fg="red"))
557
596
  if result.error_type == "permission":
@@ -614,6 +653,11 @@ def describe(name: str) -> None:
614
653
  click.echo(f" capabilities: {', '.join(agent.capabilities)}")
615
654
  if agent.risky_capabilities:
616
655
  click.echo(f" risky_caps: {', '.join(agent.risky_capabilities)}")
656
+ if agent.instructions:
657
+ first, *rest = agent.instructions.splitlines()
658
+ click.echo(f" instructions: {first}")
659
+ for line in rest:
660
+ click.echo(f" {line}")
617
661
  click.echo(f" allowed_tools: {_render_tools(agent.allowed_tools)}")
618
662
  click.echo(f" disallowed_tools: {_render_tools(agent.disallowed_tools)}")
619
663
 
@@ -828,6 +872,15 @@ def doctor() -> None:
828
872
  claude_path = shutil.which("claude")
829
873
  if claude_path:
830
874
  ok(f"claude CLI: {claude_path}")
875
+ supported = _claude_supports_flag(claude_path, "--append-system-prompt")
876
+ if supported is True:
877
+ ok("claude CLI supports --append-system-prompt (dispatch protocol)")
878
+ elif supported is False:
879
+ warn("claude CLI predates --append-system-prompt: every dispatch will fail")
880
+ click.echo(" Upgrade claude, or set `settings.dispatch_protocol: false` and")
881
+ click.echo(" clear orders orders: agent-dispatch update <name> --instructions none")
882
+ else:
883
+ warn("Could not run `claude --help` to check --append-system-prompt support")
831
884
  else:
832
885
  fail("claude CLI not found on PATH")
833
886
  click.echo(" Install: https://docs.anthropic.com/en/docs/claude-code")
@@ -1048,6 +1101,9 @@ def job_show(job_id: str) -> None:
1048
1101
  click.echo(f" cost_usd: ${job.result.cost_usd:.4f}")
1049
1102
  if job.result.budget_exceeded:
1050
1103
  click.echo(click.style(" budget: EXCEEDED", fg="yellow"))
1104
+ if job.result.outcome:
1105
+ color = "green" if job.result.outcome == "done" else "yellow"
1106
+ click.echo(f" outcome: {click.style(job.result.outcome, fg=color)}")
1051
1107
  if job.result.result:
1052
1108
  preview = job.result.result[:2000]
1053
1109
  truncated = len(job.result.result) > 2000
@@ -222,7 +222,7 @@ def save_config(config: DispatchConfig, path: Path | None = None) -> None:
222
222
  # won't drop them — prune empties so agents that never declare capabilities
223
223
  # stay clean in YAML instead of growing two empty-list keys on every save.
224
224
  for agent_data in data.get("agents", {}).values():
225
- for key in ("capabilities", "risky_capabilities"):
225
+ for key in ("capabilities", "risky_capabilities", "instructions"):
226
226
  if not agent_data.get(key):
227
227
  agent_data.pop(key, None)
228
228
  # `groups` also defaults to {} (not None), so exclude_none keeps it — prune
@@ -53,6 +53,12 @@ class AgentConfig(BaseModel):
53
53
  disallowed_tools: list[str] | None = None
54
54
  capabilities: list[str] = Field(default_factory=list)
55
55
  risky_capabilities: list[str] = Field(default_factory=list)
56
+ # Standing orders for this agent whenever it is dispatched ("read-only SQL
57
+ # only", "never restart a stack unless the task says so"). Appended to the
58
+ # claude system prompt via --append-system-prompt, after the dispatch
59
+ # protocol — so they hold across every task, unlike `context`, which is
60
+ # per call. Empty = nothing appended. Pruned from YAML when empty.
61
+ instructions: str = ""
56
62
 
57
63
  @field_validator("directory", mode="before")
58
64
  @classmethod
@@ -107,6 +113,12 @@ class Settings(BaseModel):
107
113
  # must be an explicit choice rather than something a version bump starts
108
114
  # doing to an existing install.
109
115
  job_retention_days: int = Field(default=0, ge=0)
116
+ # Send every dispatched agent the dispatch protocol as an appended system
117
+ # prompt: it is running non-interactively, nobody will answer a question,
118
+ # its time/spend budget, lead with the outcome, end with a STATUS line
119
+ # (parsed into DispatchResult.outcome). Off = raw `claude -p` behaviour,
120
+ # for CLIs that predate --append-system-prompt or for A/B comparison.
121
+ dispatch_protocol: bool = True
110
122
 
111
123
 
112
124
  def validate_agent_name(name: str) -> str:
@@ -214,3 +226,8 @@ class DispatchResult(BaseModel):
214
226
  # default). Post-hoc only — the money is already spent, the dispatch is
215
227
  # NOT failed for it. None means: no budget configured, or within budget.
216
228
  budget_exceeded: bool | None = None
229
+ # The agent's own verdict on its work, from the trailing `STATUS:` line the
230
+ # dispatch protocol asks for: "done", "partial" or "blocked". None when the
231
+ # agent did not report one (protocol off, JSON mode, or it just didn't).
232
+ # Descriptive: never flips `success`. partial/blocked are not cached.
233
+ outcome: str | None = None