agent-dispatch 0.9.0__tar.gz → 0.10.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/AGENTS.md +1 -1
  2. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/CHANGELOG.md +28 -0
  3. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/PKG-INFO +23 -3
  4. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/README.md +22 -2
  5. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/agents.example.yaml +26 -1
  6. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/pyproject.toml +1 -1
  7. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/__init__.py +1 -1
  8. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/cli.py +32 -4
  9. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/config.py +34 -9
  10. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/jobs.py +7 -6
  11. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/models.py +4 -0
  12. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/server.py +4 -2
  13. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_cli.py +51 -0
  14. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_config.py +29 -1
  15. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_jobs.py +14 -0
  16. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/.github/dependabot.yml +0 -0
  17. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/.github/workflows/ci.yml +0 -0
  18. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/.github/workflows/publish.yml +0 -0
  19. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/.gitignore +0 -0
  20. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/LICENSE +0 -0
  21. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/SECURITY.md +0 -0
  22. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/assets/mascot.png +0 -0
  23. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/cache.py +0 -0
  24. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/src/agent_dispatch/runner.py +0 -0
  25. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/__init__.py +0 -0
  26. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/conftest.py +0 -0
  27. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_cache.py +0 -0
  28. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_models.py +0 -0
  29. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_runner.py +0 -0
  30. {agent_dispatch-0.9.0 → agent_dispatch-0.10.0}/tests/test_server.py +0 -0
@@ -51,7 +51,7 @@ Python ≥ 3.10 · `from __future__ import annotations` everywhere · Pydantic v
51
51
 
52
52
  ## When adding a feature, check every layer
53
53
 
54
- `models.py` (data shape) → `runner.py` (dispatch mechanics) → `server.py` (MCP tool) → `cli.py` (CLI flag) → tests for each → `README.md` + `agents.example.yaml` (user docs).
54
+ `models.py` (data shape) → `config.py` (YAML round-trip + empty-collection pruning) → `runner.py` (dispatch mechanics) → `server.py` (MCP tool) → `cli.py` (CLI flag) → tests for each → `README.md` + `agents.example.yaml` (user docs).
55
55
 
56
56
  ## More detail
57
57
 
@@ -7,6 +7,34 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.10.0] - 2026-07-14
11
+
12
+ `doctor` learns to check groups; three correctness fixes found in review.
13
+
14
+ ### Added
15
+ - **`agent-dispatch doctor` diagnoses group health.** A new "Groups" section
16
+ reports each group's member count, flags dangling members (agents removed
17
+ from config but still referenced) as a failure, and empty groups as a
18
+ warning — each with a concrete remediation command.
19
+
20
+ ### Fixed
21
+ - `auto_describe()`: an empty `"description"` field in `package.json` (a
22
+ common `npm init` placeholder) was no longer being filtered out, producing
23
+ malformed generated descriptions like `" | Stack: Node.js"`. Empty/blank
24
+ descriptions are ignored again.
25
+ - `doctor`'s group remediation hints pointed at `group update --members`,
26
+ a flag that doesn't exist (`group update` only edits `description` /
27
+ `shared_context`; membership is set via `group add --member`). Hints now
28
+ point at the working `group remove` + `group add --member` recreation path.
29
+ - `JobStore.mark_running()` now refuses to start any job that isn't
30
+ `pending` (previously only refused `cancelled`), closing a race where a
31
+ stale/duplicate worker could resurrect an already-finished or failed job.
32
+
33
+ ### Changed
34
+ - Consolidated the "unknown group member" check — previously duplicated
35
+ across five call sites in `cli.py` and `server.py` — into a single
36
+ `DispatchConfig.unknown_group_members()` helper.
37
+
10
38
  ## [0.9.0] - 2026-06-30
11
39
 
12
40
  Coordinate a group of related projects from one session.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agent-dispatch
3
- Version: 0.9.0
3
+ Version: 0.10.0
4
4
  Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
5
5
  Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
6
6
  Project-URL: Repository, https://github.com/ginkida/agent-dispatch
@@ -43,6 +43,8 @@ Description-Content-Type: text/markdown
43
43
 
44
44
  Each agent runs as a separate `claude -p` session in its own project directory — inheriting that project's MCP servers, CLAUDE.md, and tools. The calling agent just gets the result back.
45
45
 
46
+ Related projects can be bundled into a **[group](#groups)** — a shared brief plus a member list — so one session can coordinate work across them (e.g. code repos + an `infra`/Portainer gateway + an `analytics` gateway).
47
+
46
48
  Works with OAuth, API key, and Claude subscription authentication.
47
49
 
48
50
  > **AI agents:** this README is the canonical doc for *using* the tool — setup: [Quick Start](#quick-start) (every step has a deterministic verify), first call: [`dispatch`](#dispatch), tool selection: [Which Tool to Use](#which-tool-to-use), failure handling: [Error Recovery](#error-recovery). Working *on* this repo instead? See [AGENTS.md](AGENTS.md).
@@ -539,6 +541,23 @@ agents:
539
541
  # disallowed_tools: # block specific tools
540
542
  # - Write
541
543
 
544
+ # Optional: bundle related agents into a cross-project working set.
545
+ # A descriptive layer — no router; the orchestrating session coordinates
546
+ # with the normal dispatch tools. See the Groups section above.
547
+ groups:
548
+ shop:
549
+ # ORCHESTRATOR-facing: how to coordinate the group. Surfaced by
550
+ # list_groups/inspect_group, NEVER injected into a member's prompt.
551
+ description: "After a code change: deploy via infra, then verify via analytics."
552
+ # MEMBER-facing facts, auto-prepended to dispatch(..., group="shop").
553
+ shared_context: |
554
+ Prod runs in Portainer stack "shop". Metrica counter 12345.
555
+ members: # reference agents above (many-to-many)
556
+ - agent: infra
557
+ use_for: deploy, restart, container logs
558
+ # - agent: backend
559
+ # use_for: orders/payments endpoints
560
+
542
561
  settings:
543
562
  default_timeout: 300
544
563
  # default_permission_mode: bypassPermissions # inherited by all agents
@@ -608,7 +627,7 @@ agent-dispatch MCP server
608
627
  - **Cost visibility** — `max_budget_usd` per agent or globally; a dispatch whose cost exceeds it returns `budget_exceeded: true` + a hint (post-hoc — the `claude` CLI has no spend cap, so the overage can be flagged but not prevented).
609
628
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
610
629
  - **Timeout** — per-agent or global (default: 300s). Orphaned processes are cleaned up.
611
- - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached.
630
+ - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
612
631
 
613
632
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
614
633
 
@@ -621,9 +640,10 @@ See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassP
621
640
  | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, etc.) |
622
641
  | `agent-dispatch remove <name>` | Remove an agent |
623
642
  | `agent-dispatch list` | List agents with health status and permissions |
643
+ | `agent-dispatch group <add\|list\|inspect\|update\|remove>` | Manage [groups](#groups) — cross-project working sets of agents |
624
644
  | `agent-dispatch describe <name>` | Show full configuration for one agent (tri-state tools, project files) |
625
645
  | `agent-dispatch test <name> [task] [--stream]` | Test an agent with a dispatch (`--stream` for live progress) |
626
- | `agent-dispatch doctor` | Diagnose installation: claude CLI, MCP registration, agent health |
646
+ | `agent-dispatch doctor` | Diagnose installation: Claude CLI, MCP registration, agent health, and group membership |
627
647
  | `agent-dispatch jobs [--status --limit]` | List async dispatch jobs (most recent first) |
628
648
  | `agent-dispatch job <id>` | Show one job: status, progress tail, result preview |
629
649
  | `agent-dispatch cancel <id>` | Cancel a pending job (running jobs: use the `dispatch_cancel` MCP tool) |
@@ -13,6 +13,8 @@
13
13
 
14
14
  Each agent runs as a separate `claude -p` session in its own project directory — inheriting that project's MCP servers, CLAUDE.md, and tools. The calling agent just gets the result back.
15
15
 
16
+ Related projects can be bundled into a **[group](#groups)** — a shared brief plus a member list — so one session can coordinate work across them (e.g. code repos + an `infra`/Portainer gateway + an `analytics` gateway).
17
+
16
18
  Works with OAuth, API key, and Claude subscription authentication.
17
19
 
18
20
  > **AI agents:** this README is the canonical doc for *using* the tool — setup: [Quick Start](#quick-start) (every step has a deterministic verify), first call: [`dispatch`](#dispatch), tool selection: [Which Tool to Use](#which-tool-to-use), failure handling: [Error Recovery](#error-recovery). Working *on* this repo instead? See [AGENTS.md](AGENTS.md).
@@ -509,6 +511,23 @@ agents:
509
511
  # disallowed_tools: # block specific tools
510
512
  # - Write
511
513
 
514
+ # Optional: bundle related agents into a cross-project working set.
515
+ # A descriptive layer — no router; the orchestrating session coordinates
516
+ # with the normal dispatch tools. See the Groups section above.
517
+ groups:
518
+ shop:
519
+ # ORCHESTRATOR-facing: how to coordinate the group. Surfaced by
520
+ # list_groups/inspect_group, NEVER injected into a member's prompt.
521
+ description: "After a code change: deploy via infra, then verify via analytics."
522
+ # MEMBER-facing facts, auto-prepended to dispatch(..., group="shop").
523
+ shared_context: |
524
+ Prod runs in Portainer stack "shop". Metrica counter 12345.
525
+ members: # reference agents above (many-to-many)
526
+ - agent: infra
527
+ use_for: deploy, restart, container logs
528
+ # - agent: backend
529
+ # use_for: orders/payments endpoints
530
+
512
531
  settings:
513
532
  default_timeout: 300
514
533
  # default_permission_mode: bypassPermissions # inherited by all agents
@@ -578,7 +597,7 @@ agent-dispatch MCP server
578
597
  - **Cost visibility** — `max_budget_usd` per agent or globally; a dispatch whose cost exceeds it returns `budget_exceeded: true` + a hint (post-hoc — the `claude` CLI has no spend cap, so the overage can be flagged but not prevented).
579
598
  - **Concurrency** — `max_concurrency` (default: 5) caps parallel `claude -p` processes. Note: the sync and async dispatch paths use separate semaphores, so the worst-case total is `2 × max_concurrency`.
580
599
  - **Timeout** — per-agent or global (default: 300s). Orphaned processes are cleaned up.
581
- - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached.
600
+ - **Caching** — identical `(agent, task, context, caller, goal, response_format)` requests return cached results, bounded by `cache.max_size` (oldest entry evicted first). Only successes are cached. Sessions and dialogues are never cached. A `group=` dispatch folds the group's `shared_context` into `context`, so different groups cache separately and a plain dispatch is unaffected.
582
601
 
583
602
  See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassPermissions` escalation risk and on-disk job files).
584
603
 
@@ -591,9 +610,10 @@ See [SECURITY.md](SECURITY.md) for the full threat model (including the `bypassP
591
610
  | `agent-dispatch update <name>` | Update agent config (permissions, timeout, model, etc.) |
592
611
  | `agent-dispatch remove <name>` | Remove an agent |
593
612
  | `agent-dispatch list` | List agents with health status and permissions |
613
+ | `agent-dispatch group <add\|list\|inspect\|update\|remove>` | Manage [groups](#groups) — cross-project working sets of agents |
594
614
  | `agent-dispatch describe <name>` | Show full configuration for one agent (tri-state tools, project files) |
595
615
  | `agent-dispatch test <name> [task] [--stream]` | Test an agent with a dispatch (`--stream` for live progress) |
596
- | `agent-dispatch doctor` | Diagnose installation: claude CLI, MCP registration, agent health |
616
+ | `agent-dispatch doctor` | Diagnose installation: Claude CLI, MCP registration, agent health, and group membership |
597
617
  | `agent-dispatch jobs [--status --limit]` | List async dispatch jobs (most recent first) |
598
618
  | `agent-dispatch job <id>` | Show one job: status, progress tail, result preview |
599
619
  | `agent-dispatch cancel <id>` | Cancel a pending job (running jobs: use the `dispatch_cancel` MCP tool) |
@@ -10,6 +10,17 @@ agents:
10
10
  - restart_services
11
11
  timeout: 300
12
12
 
13
+ # Analytics gateway — a browser + Yandex Metrica agent, no codebase of its own.
14
+ # Its value is its MCP servers + access, not its source. Shared across groups.
15
+ analytics:
16
+ directory: ~/projects/analytics
17
+ description: "Analytics gateway. MCP servers: browser, yandex-metrica. Pulls funnels, conversion, traffic. Read-only."
18
+ capabilities:
19
+ - funnel_report
20
+ - conversion_metrics
21
+ permission_mode: bypassPermissions # runs non-interactively against read-only sources
22
+ timeout: 300
23
+
13
24
  # Backend agent — source code, tests, database
14
25
  backend:
15
26
  directory: ~/projects/backend
@@ -62,7 +73,21 @@ groups:
62
73
  use_for: orders/payments endpoints, migrations
63
74
  - agent: infra
64
75
  use_for: deploy, restart, container logs
65
- # - agent: analytics # a shared gateway agent could also live here
76
+ - agent: analytics
77
+ use_for: funnel + conversion verification
78
+
79
+ # A second product reusing the SHARED infra/analytics gateways. Membership is
80
+ # many-to-many — gateways are referenced, not owned, so one analytics/infra
81
+ # agent serves every product.
82
+ blog:
83
+ description: "Content site. Same deploy-then-verify loop via the shared gateways."
84
+ shared_context: |
85
+ Production runs in Portainer stack "blog". Metrica counter 67890.
86
+ members:
87
+ - agent: infra
88
+ use_for: deploy, logs
89
+ - agent: analytics
90
+ use_for: pageview + bounce-rate checks
66
91
 
67
92
  settings:
68
93
  default_timeout: 300
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "agent-dispatch"
3
- version = "0.9.0"
3
+ version = "0.10.0"
4
4
  description = "MCP server that lets Claude Code agents delegate tasks to agents in other project directories"
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -1,3 +1,3 @@
1
1
  """agent-dispatch: Delegate tasks between Claude Code agents across projects."""
2
2
 
3
- __version__ = "0.9.0"
3
+ __version__ = "0.10.0"
@@ -563,12 +563,13 @@ def group_list() -> None:
563
563
  click.echo(f" {click.style(name, bold=True)} ({len(grp.members)} member(s))")
564
564
  if grp.description:
565
565
  click.echo(f" desc: {grp.description}")
566
+ unknown_members = set(config.unknown_group_members(grp))
566
567
  rendered: list[str] = []
567
568
  for m in grp.members:
568
- if m.agent in config.agents:
569
- rendered.append(m.agent)
570
- else:
569
+ if m.agent in unknown_members:
571
570
  rendered.append(click.style(f"{m.agent}(unknown)", fg="red"))
571
+ else:
572
+ rendered.append(m.agent)
572
573
  if rendered:
573
574
  click.echo(f" members: {', '.join(rendered)}")
574
575
  click.echo(f" shared context: {'yes' if grp.shared_context.strip() else 'no'}")
@@ -593,8 +594,9 @@ def group_inspect(name: str) -> None:
593
594
  for line in grp.shared_context.splitlines():
594
595
  click.echo(f" {line}")
595
596
  click.echo(f" members ({len(grp.members)}):")
597
+ unknown_members = set(config.unknown_group_members(grp))
596
598
  for m in grp.members:
597
- marker = "" if m.agent in config.agents else click.style(" (unknown)", fg="red")
599
+ marker = click.style(" (unknown)", fg="red") if m.agent in unknown_members else ""
598
600
  hint = f" — {m.use_for}" if m.use_for else ""
599
601
  click.echo(f" - {m.agent}{marker}{hint}")
600
602
  if not grp.members:
@@ -750,6 +752,32 @@ def doctor() -> None:
750
752
  except OSError as e:
751
753
  fail(f"{name}: directory unreadable - {e}")
752
754
 
755
+ section("Groups")
756
+ if config is None:
757
+ warn("Skipped (config could not be loaded)")
758
+ elif not config.groups:
759
+ ok("No groups configured")
760
+ else:
761
+ for name, group in config.groups.items():
762
+ unknown = config.unknown_group_members(group)
763
+ if unknown:
764
+ missing = ", ".join(unknown)
765
+ fail(f"{name}: unknown member(s): {missing}")
766
+ click.echo(
767
+ f" Fix by recreating: agent-dispatch group remove {name} && "
768
+ f"agent-dispatch group add {name} --member ... (or edit agents.yaml)"
769
+ )
770
+ elif not group.members:
771
+ warn(f"{name}: no members configured")
772
+ click.echo(
773
+ f" Add members by recreating: agent-dispatch group remove {name} && "
774
+ f"agent-dispatch group add {name} --member agent1 --member agent2"
775
+ )
776
+ else:
777
+ count = len(group.members)
778
+ suffix = "member" if count == 1 else "members"
779
+ ok(f"{name}: {count} {suffix}")
780
+
753
781
  section("Summary")
754
782
  issues = counters["issues"]
755
783
  warnings = counters["warnings"]
@@ -80,8 +80,13 @@ def _collect_mcp_servers(directory: Path) -> list[str]:
80
80
  if path.exists():
81
81
  try:
82
82
  data = json.loads(path.read_text(encoding="utf-8"))
83
- servers.extend(data.get("mcpServers", {}).keys())
84
- except (json.JSONDecodeError, KeyError):
83
+ if not isinstance(data, dict):
84
+ raise ValueError("top-level JSON value is not an object")
85
+ configured = data.get("mcpServers", {})
86
+ if not isinstance(configured, dict):
87
+ raise ValueError("mcpServers is not an object")
88
+ servers.extend(str(name) for name in configured)
89
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError, ValueError):
85
90
  logger.debug("Failed to parse MCP config: %s", path)
86
91
  return list(dict.fromkeys(servers)) # deduplicate, preserve order
87
92
 
@@ -137,7 +142,12 @@ def auto_describe(directory: Path) -> str:
137
142
  claude_md = directory / "CLAUDE.md"
138
143
  if claude_md.exists():
139
144
  sentences: list[str] = []
140
- for line in claude_md.read_text(encoding="utf-8").strip().splitlines()[:40]:
145
+ try:
146
+ lines = claude_md.read_text(encoding="utf-8").strip().splitlines()[:40]
147
+ except (OSError, UnicodeDecodeError):
148
+ logger.debug("Failed to read CLAUDE.md: %s", claude_md)
149
+ lines = []
150
+ for line in lines:
141
151
  stripped = line.strip()
142
152
  if stripped and not stripped.startswith("#") and not stripped.startswith("--"):
143
153
  sentences.append(stripped)
@@ -150,7 +160,12 @@ def auto_describe(directory: Path) -> str:
150
160
  if not parts:
151
161
  readme = directory / "README.md"
152
162
  if readme.exists():
153
- for line in readme.read_text(encoding="utf-8").strip().splitlines()[:20]:
163
+ try:
164
+ lines = readme.read_text(encoding="utf-8").strip().splitlines()[:20]
165
+ except (OSError, UnicodeDecodeError):
166
+ logger.debug("Failed to read README.md: %s", readme)
167
+ lines = []
168
+ for line in lines:
154
169
  stripped = line.strip()
155
170
  if (
156
171
  stripped
@@ -165,9 +180,17 @@ def auto_describe(directory: Path) -> str:
165
180
  # pyproject.toml — project description
166
181
  pyproject = directory / "pyproject.toml"
167
182
  if pyproject.exists():
168
- for line in pyproject.read_text(encoding="utf-8").splitlines():
183
+ try:
184
+ lines = pyproject.read_text(encoding="utf-8").splitlines()
185
+ except (OSError, UnicodeDecodeError):
186
+ logger.debug("Failed to read pyproject.toml: %s", pyproject)
187
+ lines = []
188
+ for line in lines:
169
189
  if line.strip().startswith("description"):
170
- desc = line.split("=", 1)[1].strip().strip('"').strip("'")
190
+ _, separator, value = line.partition("=")
191
+ if not separator:
192
+ continue
193
+ desc = value.strip().strip('"').strip("'")
171
194
  if desc:
172
195
  parts.append(desc)
173
196
  break
@@ -177,9 +200,11 @@ def auto_describe(directory: Path) -> str:
177
200
  if pkg_json.exists():
178
201
  try:
179
202
  pkg = json.loads(pkg_json.read_text(encoding="utf-8"))
180
- if pkg.get("description"):
181
- parts.append(pkg["description"])
182
- except (json.JSONDecodeError, KeyError):
203
+ if isinstance(pkg, dict):
204
+ desc = pkg.get("description")
205
+ if isinstance(desc, str) and desc.strip():
206
+ parts.append(desc)
207
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError):
183
208
  logger.debug("Failed to parse package.json: %s", pkg_json)
184
209
 
185
210
  # MCP servers — critical for understanding what tools this agent has
@@ -192,17 +192,18 @@ class JobStore:
192
192
  def mark_running(self, job_id: str) -> Job | None:
193
193
  """Mark a pending job as running.
194
194
 
195
- Returns the updated job, or None if the job is missing OR has already
196
- been cancelled. Refusing to run a cancelled job closes the race with
197
- ``cancel()``: both take ``self._lock``, so whichever wins, the worker
198
- either sees ``cancelled`` (and skips) or sets ``running`` first (and
199
- cancel then refuses).
195
+ Returns the updated job, or None unless the job exists and is still
196
+ pending. Refusing every non-pending state prevents a duplicate worker
197
+ from resurrecting a completed/failed/cancelled job. It also closes the
198
+ race with ``cancel()``: both take ``self._lock``, so whichever wins,
199
+ the worker either sees ``cancelled`` (and skips) or sets ``running``
200
+ first (and cancel then refuses).
200
201
  """
201
202
  with self._lock:
202
203
  job = self.get(job_id)
203
204
  if job is None:
204
205
  return None
205
- if job.status == "cancelled":
206
+ if job.status != "pending":
206
207
  return None
207
208
  job.status = "running"
208
209
  job.started_at = time.time()
@@ -158,6 +158,10 @@ class DispatchConfig(BaseModel):
158
158
  validate_agent_name(name)
159
159
  return self
160
160
 
161
+ def unknown_group_members(self, group: DispatchGroup) -> list[str]:
162
+ """Member agent names in `group` that aren't in `self.agents` (sorted, deduped)."""
163
+ return sorted({m.agent for m in group.members if m.agent not in self.agents})
164
+
161
165
 
162
166
  class DispatchResult(BaseModel):
163
167
  """Result of a dispatch call."""
@@ -500,6 +500,7 @@ async def list_groups(ctx: Context | None = None) -> str:
500
500
 
501
501
  groups = []
502
502
  for name, grp in config.groups.items():
503
+ unknown_members = set(config.unknown_group_members(grp))
503
504
  members = []
504
505
  for m in grp.members:
505
506
  entry: dict = {"agent": m.agent}
@@ -507,7 +508,7 @@ async def list_groups(ctx: Context | None = None) -> str:
507
508
  entry["use_for"] = m.use_for
508
509
  # Flag dangling refs (agent removed) instead of crashing — never
509
510
  # touch config.agents[m.agent] when it's unknown.
510
- if m.agent not in config.agents:
511
+ if m.agent in unknown_members:
511
512
  entry["unknown"] = True
512
513
  else:
513
514
  try:
@@ -548,12 +549,13 @@ async def inspect_group(name: str, ctx: Context | None = None) -> str:
548
549
  return err
549
550
 
550
551
  grp = config.groups[name]
552
+ unknown_members = set(config.unknown_group_members(grp))
551
553
  members = []
552
554
  for m in grp.members:
553
555
  entry: dict = {"agent": m.agent}
554
556
  if m.use_for:
555
557
  entry["use_for"] = m.use_for
556
- if m.agent not in config.agents:
558
+ if m.agent in unknown_members:
557
559
  entry["unknown"] = True
558
560
  else:
559
561
  agent = config.agents[m.agent]
@@ -852,6 +852,57 @@ class TestDoctor:
852
852
  assert "CLAUDE.md" in result.output
853
853
  assert ".mcp.json" in result.output
854
854
 
855
+ def test_healthy_group_passes(self, tmp_path: Path, _isolated_config: Path):
856
+ agent_dir = tmp_path / "proj"
857
+ agent_dir.mkdir()
858
+ _isolated_config.parent.mkdir(parents=True, exist_ok=True)
859
+ _isolated_config.write_text(
860
+ f"agents:\n proj:\n directory: {agent_dir}\n"
861
+ "groups:\n product:\n members:\n - agent: proj\n"
862
+ )
863
+ with (
864
+ patch("agent_dispatch.cli.shutil.which", side_effect=lambda x: f"/usr/bin/{x}"),
865
+ self._patch_claude_mcp_list(registered=True),
866
+ ):
867
+ result = runner.invoke(cli, ["doctor"])
868
+ assert result.exit_code == 0, result.output
869
+ assert "product: 1 member" in result.output
870
+ assert "All checks passed" in result.output
871
+
872
+ def test_dangling_group_member_fails(self, tmp_path: Path, _isolated_config: Path):
873
+ agent_dir = tmp_path / "proj"
874
+ agent_dir.mkdir()
875
+ _isolated_config.parent.mkdir(parents=True, exist_ok=True)
876
+ _isolated_config.write_text(
877
+ f"agents:\n proj:\n directory: {agent_dir}\n"
878
+ "groups:\n product:\n members:\n"
879
+ " - agent: proj\n - agent: removed_gateway\n"
880
+ )
881
+ with (
882
+ patch("agent_dispatch.cli.shutil.which", side_effect=lambda x: f"/usr/bin/{x}"),
883
+ self._patch_claude_mcp_list(registered=True),
884
+ ):
885
+ result = runner.invoke(cli, ["doctor"])
886
+ assert result.exit_code != 0
887
+ assert "product: unknown member(s): removed_gateway" in result.output
888
+ assert "group remove product && agent-dispatch group add product --member" in result.output
889
+
890
+ def test_empty_group_warns(self, tmp_path: Path, _isolated_config: Path):
891
+ agent_dir = tmp_path / "proj"
892
+ agent_dir.mkdir()
893
+ _isolated_config.parent.mkdir(parents=True, exist_ok=True)
894
+ _isolated_config.write_text(
895
+ f"agents:\n proj:\n directory: {agent_dir}\ngroups:\n product:\n members: []\n"
896
+ )
897
+ with (
898
+ patch("agent_dispatch.cli.shutil.which", side_effect=lambda x: f"/usr/bin/{x}"),
899
+ self._patch_claude_mcp_list(registered=True),
900
+ ):
901
+ result = runner.invoke(cli, ["doctor"])
902
+ assert result.exit_code == 0
903
+ assert "product: no members configured" in result.output
904
+ assert "1 warning" in result.output
905
+
855
906
  def test_summary_singular_plural(self, tmp_path: Path):
856
907
  """One issue should say 'issue' not 'issues'."""
857
908
  with patch("agent_dispatch.cli.shutil.which", return_value=None):
@@ -10,7 +10,7 @@ from pathlib import Path
10
10
  import pytest
11
11
  import yaml
12
12
 
13
- from agent_dispatch.config import auto_describe, load_config, save_config
13
+ from agent_dispatch.config import auto_describe, collect_mcp_servers, load_config, save_config
14
14
  from agent_dispatch.models import (
15
15
  AgentConfig,
16
16
  DispatchConfig,
@@ -176,6 +176,34 @@ def test_auto_describe_fallback(tmp_path: Path):
176
176
  assert tmp_path.name in desc
177
177
 
178
178
 
179
+ def test_collect_mcp_servers_ignores_wrong_json_shapes(tmp_path: Path):
180
+ (tmp_path / ".mcp.json").write_text("[]")
181
+ claude_dir = tmp_path / ".claude"
182
+ claude_dir.mkdir()
183
+ (claude_dir / "settings.local.json").write_text('{"mcpServers": []}')
184
+
185
+ assert collect_mcp_servers(tmp_path) == []
186
+
187
+
188
+ def test_auto_describe_tolerates_malformed_metadata_shapes(tmp_path: Path):
189
+ (tmp_path / "CLAUDE.md").write_bytes(b"\xff\xfe")
190
+ (tmp_path / "README.md").write_bytes(b"\xff\xfe")
191
+ (tmp_path / "pyproject.toml").write_text("description\n")
192
+ (tmp_path / "package.json").write_text("[]")
193
+ (tmp_path / ".mcp.json").write_text('{"mcpServers": []}')
194
+
195
+ assert auto_describe(tmp_path) == "Stack: Python, Node.js"
196
+
197
+
198
+ def test_auto_describe_ignores_empty_package_json_description(tmp_path: Path):
199
+ (tmp_path / "package.json").write_text('{"description": ""}')
200
+
201
+ desc = auto_describe(tmp_path)
202
+
203
+ assert not desc.startswith(" | ")
204
+ assert "Stack: Node.js" == desc
205
+
206
+
179
207
  def test_save_and_load_groups_roundtrip(tmp_path: Path):
180
208
  f = tmp_path / "test.yaml"
181
209
  config = DispatchConfig(
@@ -315,6 +315,20 @@ class TestJobCancel:
315
315
  assert store.mark_running(job.id) is None
316
316
  assert store.get(job.id).status == "cancelled"
317
317
 
318
+ def test_mark_running_refuses_completed_job(self, store: JobStore):
319
+ job = store.create("agent", "task")
320
+ store.finish(job.id, DispatchResult(agent="agent", success=True, result="done"))
321
+
322
+ assert store.mark_running(job.id) is None
323
+ assert store.get(job.id).status == "done"
324
+
325
+ def test_mark_running_refuses_failed_job(self, store: JobStore):
326
+ job = store.create("agent", "task")
327
+ store.fail(job.id, "boom")
328
+
329
+ assert store.mark_running(job.id) is None
330
+ assert store.get(job.id).status == "failed"
331
+
318
332
 
319
333
  class TestRecoverStale:
320
334
  def test_recovers_old_running_job(self, store: JobStore):
File without changes