agent-dispatch 0.12.1__tar.gz → 0.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/.github/workflows/publish.yml +8 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/AGENTS.md +10 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/CHANGELOG.md +83 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/PKG-INFO +27 -2
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/README.md +25 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/agents.example.yaml +6 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/pyproject.toml +1 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/__init__.py +1 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/cli.py +20 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/config.py +30 -1
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/jobs.py +40 -8
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/models.py +10 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/server.py +246 -137
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_cli.py +25 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_config.py +56 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_jobs.py +33 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_server.py +250 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/.github/dependabot.yml +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/.github/workflows/ci.yml +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/.gitignore +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/LICENSE +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/SECURITY.md +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/assets/mascot.png +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/cache.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/src/agent_dispatch/runner.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/__init__.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/conftest.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_cache.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_models.py +0 -0
- {agent_dispatch-0.12.1 → agent_dispatch-0.13.0}/tests/test_runner.py +0 -0
|
@@ -26,5 +26,12 @@ jobs:
|
|
|
26
26
|
pip install build
|
|
27
27
|
python -m build
|
|
28
28
|
|
|
29
|
+
# Keep this current with the build backend. The pinned action bundles its
|
|
30
|
+
# own twine, and twine rejects a Metadata-Version it does not know: the
|
|
31
|
+
# v0.13.0 publish failed because hatchling 1.32 emits `Metadata-Version:
|
|
32
|
+
# 2.5` while the previous pin (Feb 2026) shipped twine 6.1.0 / packaging
|
|
33
|
+
# 25.0. `python -m build` resolves the newest hatchling at build time, so
|
|
34
|
+
# this drifts on its own with no repo change — see the release checklist's
|
|
35
|
+
# `twine check` step, which catches it before a tag is cut.
|
|
29
36
|
- name: Publish to PyPI
|
|
30
|
-
uses: pypa/gh-action-pypi-publish@
|
|
37
|
+
uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
|
|
@@ -52,16 +52,25 @@ Tests must **never** invoke the real `claude` CLI. Runner tests mock `shutil.whi
|
|
|
52
52
|
- **Never `close()` a pipe another thread may still be reading.** `close()` waits on the reader's buffer lock with *no timeout*, so it would hang the dispatch forever — the bounded `join()` before it buys nothing. `dispatch_stream` skips the stderr close while the drain thread is alive and lets the daemon reader + Popen finalizer release the fd. This is not hypothetical: a stdio MCP server inherits the child's `stderr`, so the pipe often has no EOF even after `claude` exits cleanly.
|
|
53
53
|
- **No `await` inside `config_lock()`.** `ProcessLock`'s in-process guard is a `threading.RLock` — re-entrant per *thread* — and every MCP tool coroutine runs on the one event-loop thread. Suspending in the critical section lets a second coroutine re-enter the "held" lock and interleave its own load/mutate/save. Collect warnings as data, emit them after the `with` block (`test_no_await_inside_the_config_lock` enforces this by AST).
|
|
54
54
|
- Cross-process locks are acquired with a **bounded** wait, never a blocking `flock`: the server takes them on its event-loop thread, so a wedged holder would freeze every tool. After the deadline it proceeds unlocked and logs — a possible lost update beats a permanent freeze.
|
|
55
|
-
- `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore.
|
|
55
|
+
- `recover_stale` sweeps `pending` on a **much longer** threshold than `running`: the jobs directory is shared by every `agent-dispatch serve`, so an hours-old pending job may still be queued behind another live server's semaphore. For a *running* job, `started_at` alone does **not** prove abandonment — a dispatch may legitimately run to the 7200s timeout ceiling, so the file's own mtime is checked too (a live worker rewrites it on every progress flush). That check can only ever *skip* a recovery, never add one.
|
|
56
|
+
- Deleting job records is **opt-in** (`settings.job_retention_days`, default `0` = off) and only ever happens at server start, never inside a tool. They are the user's dispatch history and the deletion is irreversible, so an unreadable config is treated as "do nothing" rather than falling back to a default retention.
|
|
57
|
+
- `max_concurrency` must bound *subprocesses*, not coroutines. Cancelling a coroutine does not stop the thread behind `asyncio.to_thread`, so `async with sem:` gave the slot away while `claude` kept running and billing. Dispatches go through `_dispatch_guarded`, which releases from the future's done-callback and `shield`s the await. Never "simplify" it back to `async with`.
|
|
58
|
+
- Tool responses are serialized through `server._dumps`, never bare `json.dumps`: the stdlib default (`ensure_ascii=True`) turns every non-ASCII character into a `\uXXXX` escape, tripling the bytes and tokenizing badly, for no gain — the stdio transport emits raw UTF-8 via pydantic anyway. A real `list_groups()` carried 8520 escapes and weighed 59 KB instead of 25 KB.
|
|
59
|
+
- `agents.yaml` is parsed with libyaml's `CSafeLoader` when available (`config._YamlLoader`), falling back to the pure-Python **safe** loader — never `yaml.Loader`. The config is re-read on every tool call, so this parse is on the hot path of all 21 tools *and* blocks the event-loop thread: 9.80 ms → 0.76 ms on a real 38 KB config.
|
|
56
60
|
- Pydantic does **not** validate on assignment. `Field(ge=...)` guards only the *load* path; every mutation surface (CLI `add`/`update`, MCP `add_agent`/`update_agent`) needs its own boundary check, or the bound escapes as a raw `ValidationError`.
|
|
57
61
|
- Every state file (`agents.yaml`, job files) is written **temp file + `os.replace`**, never in place, and every load/mutate/save is wrapped in `config.ProcessLock` — the CLI and the MCP server are separate processes writing the same files, so a thread lock alone loses updates.
|
|
58
62
|
- Anything that changes an agent's config must call `_invalidate_agent_cache` — the cache key holds the agent *name*, not its directory or permissions.
|
|
59
63
|
- Only *clean* successes are cached: `cache.put` refuses failures, `denied_tools` results, and `budget_exceeded` results, so the documented "grant access, then re-dispatch" recovery is never short-circuited.
|
|
60
64
|
- Remediation text is a contract: a hint that names a flag must name one that exists (`test_printed_budget_hint_is_a_runnable_command` feeds the printed flags back into the CLI). Run the command you print.
|
|
65
|
+
- The config error sets are declared **once** and in two halves: `config.CONFIG_LOAD_ERRORS` (read) and `config.CONFIG_SAVE_ERRORS` (write — `yaml.dump`'s `RepresenterError` is a `yaml.YAMLError`, therefore neither `OSError` nor `ConfigLoadError`, and used to escape both the MCP guard and the CLI's `_save_or_exit`). Two halves, not one set, because the remediations differ: a failed write is atomic so the old config survives, while a failed read needs the YAML fixed.
|
|
61
66
|
- MCP tools that load config carry `@_config_guard` under `@mcp.tool()` so a broken `agents.yaml` — or a failed *write* — returns the `{"error": ...}` envelope instead of a raw traceback. The set of load errors lives in one place (`config.CONFIG_LOAD_ERRORS`) because three surfaces handle it: **`UnicodeDecodeError` is a `ValueError`, not an `OSError`**, and listing types per-site is exactly how a cp1251 config slipped past all three.
|
|
62
67
|
|
|
63
68
|
- Tests must not touch anything outside `tmp_path`. `test_server.py`'s autouse `_reset_globals` and `test_cli.py`'s `_isolated_config` redirect **both** `AGENT_DISPATCH_CONFIG` and `AGENT_DISPATCH_JOBS_DIR`: a mutation tool that bails out early (unknown agent) still takes `config_lock()` first, which would otherwise create a lock file beside the developer's real config.
|
|
64
69
|
|
|
70
|
+
- `mcp` is pinned **`>=1.2.0,<2`** deliberately: 2.0 removed `mcp.server.fastmcp`, which `server.py` imports, so an unbounded range gives every fresh install a dead `agent-dispatch serve`. Lifting the cap means porting to `mcp.server.mcpserver.MCPServer` — it is not a dependency bump.
|
|
71
|
+
- Verify packaging in a **clean venv**, never the dev machine: build the wheel, install it fresh, import the server. A stale pin in local site-packages hides exactly the failure a new user hits first.
|
|
72
|
+
- Run **`twine check dist/*`** before cutting the tag, with a *current* twine. The clean-venv check does not cover this: it proves the wheel installs, not that PyPI's uploader will accept its metadata. `python -m build` pulls the newest hatchling at build time, so the emitted `Metadata-Version` climbs on its own, and `pypa/gh-action-pypi-publish` is pinned to a SHA that bundles a fixed twine — v0.13.0's publish failed on exactly that mismatch (hatchling 1.32 → metadata 2.5; pinned twine 6.1.0 → "not a valid metadata version"). When it happens, bump the action pin rather than pinning the backend down.
|
|
73
|
+
|
|
65
74
|
## Deliberately not built
|
|
66
75
|
|
|
67
76
|
These were considered — some fully implemented — and cut on purpose: an agent router / auto-dispatch (`recommend_agent` / `dispatch_auto`, removed before 0.8.0 — a keyword scorer adds little over the calling LLM at a handful of agents, and auto-dispatch can spend money or mutate a repo on a guess); groups as an execution engine (they are a descriptive layer — no routing, no per-group settings); an agent-dispatch-side budget ledger across dispatches (the CLI's own `--max-budget-usd` covers a single run; anything cumulative would need state we deliberately don't keep). Please open an issue with the use case before adding any of them.
|
|
@@ -7,6 +7,89 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.13.0] - 2026-08-13
|
|
11
|
+
|
|
12
|
+
An efficiency round, measured against a real 38 KB config (4 agents, 6 groups),
|
|
13
|
+
a 294-record jobs directory and 14 concurrently running servers — not against
|
|
14
|
+
synthetic fixtures. Findings came from a five-axis audit (caller context,
|
|
15
|
+
latency/IO, spend, CPU, concurrency) whose every claim was re-measured
|
|
16
|
+
adversarially before implementation; most were confirmed as real but too small
|
|
17
|
+
to matter next to a 10–120 s subprocess and were deliberately left alone.
|
|
18
|
+
|
|
19
|
+
### Changed
|
|
20
|
+
- **Tool responses no longer escape non-ASCII, cutting discovery payloads
|
|
21
|
+
roughly in half.** `json.dumps` defaults to `ensure_ascii=True`, so every
|
|
22
|
+
Cyrillic character left as a `\uXXXX` escape — 6 bytes for what UTF-8 stores
|
|
23
|
+
in 2, and a far worse token sequence. One `list_groups()` carried **8520**
|
|
24
|
+
such escapes: 59 KB where 25 KB was enough. Measured over a full discovery
|
|
25
|
+
pass (`list_agents` + `list_groups` + `inspect_group` + `inspect_agent`):
|
|
26
|
+
104 KB → 51 KB, about **13 000 tokens saved per pass**, on every call, in
|
|
27
|
+
every session. Nothing is lost: the stdio transport serializes the JSON-RPC
|
|
28
|
+
envelope with pydantic's `model_dump_json`, which already emits raw UTF-8 —
|
|
29
|
+
which is also why the `dispatch` family never had this problem while every
|
|
30
|
+
`json.dumps` tool did. All 57 call sites now go through one `_dumps` helper
|
|
31
|
+
so the two agree. `_dumps` probes encodability and falls back to the escaped
|
|
32
|
+
form for the one payload that needs it — a **lone surrogate**, which an agent
|
|
33
|
+
produces by printing a literal `\udXXX` escape in its JSON output (`json.loads`
|
|
34
|
+
manufactures it from perfectly ASCII input, so the subprocess-level
|
|
35
|
+
`errors="replace"` cannot help). Unescaped it would raise `UnicodeEncodeError`
|
|
36
|
+
inside the stdio transport *after* the tool returned — past every guard.
|
|
37
|
+
The saving is proportional to how much non-ASCII your config and results
|
|
38
|
+
carry: on an all-ASCII setup it is exactly zero.
|
|
39
|
+
- **`agents.yaml` parses ~13x faster.** The config is re-read on every single
|
|
40
|
+
tool call (deliberately — that is how a new agent is picked up without a
|
|
41
|
+
restart), so the parse sits on the hot path of all 21 tools *and* runs on the
|
|
42
|
+
event-loop thread, where it blocks every other tool. Switching to libyaml's
|
|
43
|
+
`CSafeLoader` takes the real config from **9.80 ms to 0.76 ms** per call.
|
|
44
|
+
Falls back to the pure-Python safe loader when PyYAML was built without
|
|
45
|
+
libyaml; both are the *safe* loader, so a config still cannot construct
|
|
46
|
+
arbitrary objects.
|
|
47
|
+
|
|
48
|
+
### Added
|
|
49
|
+
- **`settings.job_retention_days`** — when > 0, terminal job records older than
|
|
50
|
+
N days are deleted at server start. Every async dispatch and every
|
|
51
|
+
`return_ref` dispatch leaves a job file that nothing removed on its own, so
|
|
52
|
+
the directory grew forever while `dispatch_jobs` and stale-job recovery
|
|
53
|
+
parsed every file in it. **Defaults to `0` (off)**: those records are the
|
|
54
|
+
user's own dispatch history and deleting them is irreversible, so it is an
|
|
55
|
+
explicit opt-in rather than something a version bump starts doing to an
|
|
56
|
+
existing install. An unreadable config is treated as `0`, never guessed.
|
|
57
|
+
|
|
58
|
+
### Fixed
|
|
59
|
+
- **`max_concurrency` bounds live subprocesses again, not live coroutines.**
|
|
60
|
+
`async with sem:` released the slot when the *coroutine* unwound, but
|
|
61
|
+
cancelling a coroutine does not stop the thread behind `asyncio.to_thread` —
|
|
62
|
+
so an interrupted turn or a closed session handed the slot to the next
|
|
63
|
+
dispatch while the `claude` subprocess kept running and kept being billed. N
|
|
64
|
+
cancellations meant up to N extra concurrent subprocesses, and each abandoned
|
|
65
|
+
worker also held a thread out of the default executor (`min(32, cpu_count+4)`,
|
|
66
|
+
which `to_thread` needs for *every* dispatch) for the rest of its timeout —
|
|
67
|
+
up to 7200s. The slot is now tied to the worker's real lifetime via
|
|
68
|
+
`_dispatch_guarded`; cancellation still reaches the caller immediately.
|
|
69
|
+
- **A legitimately long dispatch is no longer flipped to `failed` by an
|
|
70
|
+
unrelated server starting up.** `recover_stale` judged abandonment from
|
|
71
|
+
`started_at` alone, but a dispatch may legitimately run up to the 7200s
|
|
72
|
+
timeout ceiling — so any of the (routinely 14+) other servers booting would
|
|
73
|
+
mark a live 2-hour job `failed`, after which `finish()` refuses the real,
|
|
74
|
+
already-paid-for result. A *running* job whose file was modified within the
|
|
75
|
+
threshold is now left alone: a live worker rewrites it on every progress
|
|
76
|
+
flush, so only the file's own age proves abandonment. This can only ever skip
|
|
77
|
+
a recovery — a genuinely abandoned job is picked up on a later start.
|
|
78
|
+
- **A config that cannot be *rendered* now returns an error envelope instead of
|
|
79
|
+
a traceback.** `_get_config` wraps read-side YAML failures into
|
|
80
|
+
`ConfigLoadError`, but `save_config` → `yaml.dump` raises `RepresenterError` —
|
|
81
|
+
a `yaml.YAMLError`, so neither an `OSError` nor a `ConfigLoadError`. It
|
|
82
|
+
therefore slipped past *both* arms of the MCP guard on
|
|
83
|
+
`add_agent`/`update_agent`/`remove_agent` **and** past the CLI's
|
|
84
|
+
`_save_or_exit`. Unreachable today (only JSON-native types are ever dumped)
|
|
85
|
+
and closed now because it is the same "an exception type escapes the handler
|
|
86
|
+
meant to catch it" class that produced the 0.12.0 and 0.12.1 rounds. The set
|
|
87
|
+
is declared once as `config.CONFIG_SAVE_ERRORS`, next to its read-side twin,
|
|
88
|
+
so the two surfaces cannot drift.
|
|
89
|
+
- **Stale-job recovery scans the jobs directory once instead of twice.** It
|
|
90
|
+
called `list()` per status, and `list()` reads and parses every file in the
|
|
91
|
+
directory (~85 ms at 294 records) — at the start of every server process.
|
|
92
|
+
|
|
10
93
|
## [0.12.1] - 2026-07-30
|
|
11
94
|
|
|
12
95
|
Two holes in 0.12.0's own "tools always return a clean error" fix.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: agent-dispatch
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.13.0
|
|
4
4
|
Summary: MCP server that lets Claude Code agents delegate tasks to agents in other project directories
|
|
5
5
|
Project-URL: Homepage, https://github.com/ginkida/agent-dispatch
|
|
6
6
|
Project-URL: Repository, https://github.com/ginkida/agent-dispatch
|
|
@@ -574,6 +574,7 @@ settings:
|
|
|
574
574
|
# - Edit
|
|
575
575
|
max_dispatch_depth: 3 # recursion protection
|
|
576
576
|
max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
|
|
577
|
+
# job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
|
|
577
578
|
cache:
|
|
578
579
|
enabled: true
|
|
579
580
|
ttl: 300 # seconds
|
|
@@ -582,6 +583,30 @@ settings:
|
|
|
582
583
|
|
|
583
584
|
Config is reloaded on every tool call — add agents without restarting.
|
|
584
585
|
|
|
586
|
+
### Job retention
|
|
587
|
+
|
|
588
|
+
Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
|
|
589
|
+
record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
|
|
590
|
+
`dispatch_gc` has to be run by hand. The directory therefore grows without
|
|
591
|
+
bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
|
|
592
|
+
start read and parse *every* file in it.
|
|
593
|
+
|
|
594
|
+
Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
|
|
595
|
+
than N days when a server starts:
|
|
596
|
+
|
|
597
|
+
```yaml
|
|
598
|
+
settings:
|
|
599
|
+
job_retention_days: 30
|
|
600
|
+
```
|
|
601
|
+
|
|
602
|
+
It defaults to `0` — **off** — because those records are your own history of
|
|
603
|
+
past dispatches and deleting them cannot be undone. Pending and running jobs are
|
|
604
|
+
never touched.
|
|
605
|
+
|
|
606
|
+
`agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
|
|
607
|
+
one-off. Both *delete* immediately and report the count; neither previews, so
|
|
608
|
+
check what is there first with `agent-dispatch jobs`.
|
|
609
|
+
|
|
585
610
|
### Auto-Description
|
|
586
611
|
|
|
587
612
|
`agent-dispatch add` without `--description` generates one from:
|
|
@@ -544,6 +544,7 @@ settings:
|
|
|
544
544
|
# - Edit
|
|
545
545
|
max_dispatch_depth: 3 # recursion protection
|
|
546
546
|
max_concurrency: 5 # max parallel claude -p processes (per dispatch path)
|
|
547
|
+
# job_retention_days: 30 # 0 (default) = never prune. See "Job retention" below.
|
|
547
548
|
cache:
|
|
548
549
|
enabled: true
|
|
549
550
|
ttl: 300 # seconds
|
|
@@ -552,6 +553,30 @@ settings:
|
|
|
552
553
|
|
|
553
554
|
Config is reloaded on every tool call — add agents without restarting.
|
|
554
555
|
|
|
556
|
+
### Job retention
|
|
557
|
+
|
|
558
|
+
Every `dispatch_async` **and** every `dispatch(..., return_ref=True)` writes a
|
|
559
|
+
record to `~/.config/agent-dispatch/jobs/`, and nothing deletes it on its own —
|
|
560
|
+
`dispatch_gc` has to be run by hand. The directory therefore grows without
|
|
561
|
+
bound, and `dispatch_jobs` plus the stale-job recovery that runs at every server
|
|
562
|
+
start read and parse *every* file in it.
|
|
563
|
+
|
|
564
|
+
Set `job_retention_days` to prune terminal (done/failed/cancelled) records older
|
|
565
|
+
than N days when a server starts:
|
|
566
|
+
|
|
567
|
+
```yaml
|
|
568
|
+
settings:
|
|
569
|
+
job_retention_days: 30
|
|
570
|
+
```
|
|
571
|
+
|
|
572
|
+
It defaults to `0` — **off** — because those records are your own history of
|
|
573
|
+
past dispatches and deleting them cannot be undone. Pending and running jobs are
|
|
574
|
+
never touched.
|
|
575
|
+
|
|
576
|
+
`agent-dispatch gc --days N` and the `dispatch_gc` tool apply the same rule as a
|
|
577
|
+
one-off. Both *delete* immediately and report the count; neither previews, so
|
|
578
|
+
check what is there first with `agent-dispatch jobs`.
|
|
579
|
+
|
|
555
580
|
### Auto-Description
|
|
556
581
|
|
|
557
582
|
`agent-dispatch add` without `--description` generates one from:
|
|
@@ -100,6 +100,12 @@ settings:
|
|
|
100
100
|
# - Bash
|
|
101
101
|
# - Read
|
|
102
102
|
# - Edit
|
|
103
|
+
# job_retention_days: 30 # OFF by default (0). When > 0, terminal job records
|
|
104
|
+
# older than this are deleted at server start. Every
|
|
105
|
+
# async and every return_ref dispatch leaves a job
|
|
106
|
+
# file behind and nothing removes it otherwise, so
|
|
107
|
+
# the directory grows forever — but the records are
|
|
108
|
+
# your dispatch history, so pruning is opt-in.
|
|
103
109
|
cache:
|
|
104
110
|
enabled: true
|
|
105
111
|
ttl: 300 # seconds; identical (agent, task, context) requests are cached
|
|
@@ -13,7 +13,14 @@ import click
|
|
|
13
13
|
import yaml
|
|
14
14
|
from pydantic import ValidationError
|
|
15
15
|
|
|
16
|
-
from .config import
|
|
16
|
+
from .config import (
|
|
17
|
+
CONFIG_SAVE_ERRORS,
|
|
18
|
+
auto_describe,
|
|
19
|
+
config_lock,
|
|
20
|
+
config_path,
|
|
21
|
+
load_config,
|
|
22
|
+
save_config,
|
|
23
|
+
)
|
|
17
24
|
from .jobs import JobStore, default_jobs_dir, is_valid_job_id
|
|
18
25
|
from .models import (
|
|
19
26
|
AgentConfig,
|
|
@@ -48,6 +55,18 @@ def _save_or_exit(config: DispatchConfig) -> None:
|
|
|
48
55
|
)
|
|
49
56
|
)
|
|
50
57
|
raise SystemExit(1) from None
|
|
58
|
+
except CONFIG_SAVE_ERRORS as e:
|
|
59
|
+
# A value YAML cannot represent — not an OSError, so the arm above never
|
|
60
|
+
# saw it and it escaped as a traceback. See config.CONFIG_SAVE_ERRORS.
|
|
61
|
+
click.echo(
|
|
62
|
+
click.style(
|
|
63
|
+
f"Error: could not serialize the config to YAML: {e}\n"
|
|
64
|
+
"Nothing was changed — the previous config is intact. "
|
|
65
|
+
"This is a bug: please report the field that failed.",
|
|
66
|
+
fg="red",
|
|
67
|
+
)
|
|
68
|
+
)
|
|
69
|
+
raise SystemExit(1) from None
|
|
51
70
|
|
|
52
71
|
|
|
53
72
|
def _check_budget_or_exit(max_budget: float | None) -> None:
|
|
@@ -22,6 +22,19 @@ try: # pragma: no cover - platform dependent
|
|
|
22
22
|
except ImportError: # pragma: no cover - Windows has no fcntl
|
|
23
23
|
fcntl = None # type: ignore[assignment]
|
|
24
24
|
|
|
25
|
+
# Parse with libyaml when PyYAML was built against it. Every MCP tool call
|
|
26
|
+
# reloads agents.yaml from scratch (deliberately — that is how a new agent is
|
|
27
|
+
# picked up without a restart), so this parse is on the hot path of all 21
|
|
28
|
+
# tools. On a real 38 KB config the pure-Python SafeLoader takes ~9.8 ms and
|
|
29
|
+
# CSafeLoader ~0.6 ms: a 16x saving repeated on every single call, and it runs
|
|
30
|
+
# on the event-loop thread where it blocks every other tool. The fallback is
|
|
31
|
+
# mandatory — a PyYAML installed from source without libyaml headers has no
|
|
32
|
+
# C extension. Same semantics either way: both are the *safe* loader.
|
|
33
|
+
try: # pragma: no cover - depends on how PyYAML was built
|
|
34
|
+
from yaml import CSafeLoader as _YamlLoader
|
|
35
|
+
except ImportError: # pragma: no cover - pure-Python PyYAML
|
|
36
|
+
from yaml import SafeLoader as _YamlLoader # type: ignore[assignment]
|
|
37
|
+
|
|
25
38
|
logger = logging.getLogger(__name__)
|
|
26
39
|
|
|
27
40
|
DEFAULT_CONFIG_DIR = Path.home() / ".config" / "agent-dispatch"
|
|
@@ -158,13 +171,29 @@ def file_lock(path: Path) -> Iterator[None]:
|
|
|
158
171
|
# UTF-16 used to escape every one of them as a raw traceback.
|
|
159
172
|
CONFIG_LOAD_ERRORS = (ValidationError, yaml.YAMLError, UnicodeDecodeError, OSError)
|
|
160
173
|
|
|
174
|
+
# The *write* half, and deliberately not a copy of the read half. `save_config`
|
|
175
|
+
# can fail two ways with two different remediations: OSError (full disk,
|
|
176
|
+
# read-only volume — handled separately, and the atomic rename means the old
|
|
177
|
+
# file survives), or a rendering failure. `yaml.dump` raises RepresenterError —
|
|
178
|
+
# a `yaml.YAMLError`, so neither an OSError nor a ValueError — for any value it
|
|
179
|
+
# cannot represent. That is unreachable today because save_config only ever
|
|
180
|
+
# feeds it `model_dump(mode="json")` output, i.e. JSON-native types. It becomes
|
|
181
|
+
# reachable the moment a field lands whose JSON dump is not one of those, and it
|
|
182
|
+
# would then escape *both* server guards and the CLI's `_save_or_exit` as a raw
|
|
183
|
+
# traceback — the exact "an exception type escapes the handler meant to catch
|
|
184
|
+
# it" class this codebase has been bitten by twice. Declared once, next to its
|
|
185
|
+
# read-side twin, so the two surfaces that handle it cannot drift.
|
|
186
|
+
CONFIG_SAVE_ERRORS = (yaml.YAMLError,)
|
|
187
|
+
|
|
161
188
|
|
|
162
189
|
def load_config(path: Path | None = None) -> DispatchConfig:
|
|
163
190
|
"""Load config from YAML file. Returns empty config if file missing."""
|
|
164
191
|
p = path or config_path()
|
|
165
192
|
if not p.exists():
|
|
166
193
|
return DispatchConfig()
|
|
167
|
-
|
|
194
|
+
# yaml.load with the safe loader == yaml.safe_load; _YamlLoader is the C
|
|
195
|
+
# one when available (see its definition for why this is worth doing).
|
|
196
|
+
raw = yaml.load(p.read_text(encoding="utf-8"), Loader=_YamlLoader) # noqa: S506
|
|
168
197
|
if raw is None:
|
|
169
198
|
return DispatchConfig()
|
|
170
199
|
return DispatchConfig.model_validate(raw)
|
|
@@ -286,6 +286,17 @@ class JobStore:
|
|
|
286
286
|
# real queue wait reaches instead of the running-job one.
|
|
287
287
|
_PENDING_STALE_MULTIPLIER = 24
|
|
288
288
|
|
|
289
|
+
def _touched_after(self, job_id: str, cutoff: float) -> bool:
|
|
290
|
+
"""True if the job file's mtime is newer than *cutoff* (someone is live).
|
|
291
|
+
|
|
292
|
+
Best-effort: an unreadable or vanished file reports False so recovery
|
|
293
|
+
falls back to the timestamp check rather than skipping silently.
|
|
294
|
+
"""
|
|
295
|
+
try:
|
|
296
|
+
return self._path(job_id).stat().st_mtime > cutoff
|
|
297
|
+
except (OSError, ValueError): # pragma: no cover - racing unlink
|
|
298
|
+
return False
|
|
299
|
+
|
|
289
300
|
def recover_stale(self, stale_threshold_seconds: float = 3600) -> int:
|
|
290
301
|
"""Mark jobs abandoned in 'running' or 'pending' beyond the threshold as failed.
|
|
291
302
|
|
|
@@ -307,14 +318,35 @@ class JobStore:
|
|
|
307
318
|
recovered = 0
|
|
308
319
|
stale: list[tuple[Job, float, str]] = []
|
|
309
320
|
with self._lock:
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
for job in self.list(
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
321
|
+
# ONE scan, not one per status. list() reads and json-parses every
|
|
322
|
+
# file in the directory (~85ms for 300 jobs), and this runs at the
|
|
323
|
+
# start of every `serve` process — of which there is one per open
|
|
324
|
+
# Claude Code session, routinely 14+.
|
|
325
|
+
for job in self.list():
|
|
326
|
+
if job.status == "running":
|
|
327
|
+
age = now - (job.started_at or job.created_at)
|
|
328
|
+
if age <= stale_threshold_seconds:
|
|
329
|
+
continue
|
|
330
|
+
# `started_at` alone does not prove abandonment: a dispatch
|
|
331
|
+
# may legitimately run up to the 7200s timeout ceiling, so
|
|
332
|
+
# any other server starting up would flip a live 2-hour job
|
|
333
|
+
# to 'failed' — and finish() then refuses it, losing an
|
|
334
|
+
# already-paid-for result. A live worker rewrites the file
|
|
335
|
+
# at least once per progress interval, so a recently
|
|
336
|
+
# *modified* file means someone is still on it. This can
|
|
337
|
+
# only ever skip a recovery (a genuinely abandoned job is
|
|
338
|
+
# picked up on a later start), never add one. Only 'running'
|
|
339
|
+
# gets this check: nothing rewrites a pending job's file, so
|
|
340
|
+
# there its mtime is just created_at by another name.
|
|
341
|
+
if self._touched_after(job.id, now - stale_threshold_seconds):
|
|
342
|
+
continue
|
|
343
|
+
elif job.status == "pending":
|
|
344
|
+
age = now - job.created_at
|
|
345
|
+
if age <= pending_threshold:
|
|
346
|
+
continue
|
|
347
|
+
else:
|
|
348
|
+
continue
|
|
349
|
+
stale.append((job, age, job.status))
|
|
318
350
|
for job, age, state in stale:
|
|
319
351
|
# Count only jobs we actually transitioned (fail() returns
|
|
320
352
|
# None for a missing/malformed id, e.g. a planted file).
|
|
@@ -97,6 +97,16 @@ class Settings(BaseModel):
|
|
|
97
97
|
max_dispatch_depth: int = Field(default=3, ge=1)
|
|
98
98
|
max_concurrency: int = Field(default=5, ge=1)
|
|
99
99
|
cache: CacheSettings = Field(default_factory=CacheSettings)
|
|
100
|
+
# Opt-in retention sweep: when > 0, terminal jobs older than this many days
|
|
101
|
+
# are deleted at server start. Every async dispatch AND every return_ref
|
|
102
|
+
# dispatch writes a job file that nothing removes on its own — dispatch_gc
|
|
103
|
+
# has to be called by hand — so the directory only ever grows, and
|
|
104
|
+
# JobStore.list() (dispatch_jobs, stale-job recovery) parses every file in
|
|
105
|
+
# it. Defaults to 0 (**off**) on purpose: job records are the user's own
|
|
106
|
+
# history of past dispatches, and deleting them is not reversible, so it
|
|
107
|
+
# must be an explicit choice rather than something a version bump starts
|
|
108
|
+
# doing to an existing install.
|
|
109
|
+
job_retention_days: int = Field(default=0, ge=0)
|
|
100
110
|
|
|
101
111
|
|
|
102
112
|
def validate_agent_name(name: str) -> str:
|