playmaker-cli 0.12.1__tar.gz → 0.14.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/CHANGELOG.md +106 -0
  2. playmaker_cli-0.12.1/README.md → playmaker_cli-0.14.0/PKG-INFO +60 -7
  3. playmaker_cli-0.12.1/PKG-INFO → playmaker_cli-0.14.0/README.md +33 -34
  4. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/pyproject.toml +4 -2
  5. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/SKILL.md +83 -25
  6. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md +0 -16
  7. playmaker_cli-0.14.0/skills/playmaker-coach/references/lanes.md +117 -0
  8. playmaker_cli-0.14.0/skills/playmaker-coach/references/ledger.md +77 -0
  9. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/prompt-templates.md +15 -4
  10. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/quotas.md +14 -14
  11. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/review-board.md +45 -8
  12. playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger-hook.py +571 -0
  13. playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger.py +264 -0
  14. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/scripts/review-board.sh +89 -27
  15. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/claude.py +40 -0
  16. playmaker_cli-0.14.0/src/playmaker/agents/muse.py +376 -0
  17. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/cli.py +59 -1
  18. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/quotas.py +261 -86
  19. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/registry.py +2 -0
  20. playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_inventory.json +7 -0
  21. playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_usage.json +19 -0
  22. playmaker_cli-0.14.0/tests/fixtures/muse_exec.jsonl +4 -0
  23. playmaker_cli-0.14.0/tests/fixtures/muse_session.jsonl +13 -0
  24. playmaker_cli-0.14.0/tests/test_claude_effort.py +294 -0
  25. playmaker_cli-0.14.0/tests/test_ledger_concurrency.py +112 -0
  26. playmaker_cli-0.14.0/tests/test_muse.py +182 -0
  27. playmaker_cli-0.14.0/tests/test_quotas_antigravity.py +489 -0
  28. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_claude.py +120 -0
  29. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_codex.py +84 -0
  30. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_kimi.py +21 -0
  31. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_registry.py +2 -2
  32. playmaker_cli-0.12.1/skills/playmaker-coach/references/lanes.md +0 -101
  33. playmaker_cli-0.12.1/tests/test_quotas_antigravity.py +0 -209
  34. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/.gitignore +0 -0
  35. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/LICENSE +0 -0
  36. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/commands.md +0 -0
  37. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/__init__.py +0 -0
  38. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/__main__.py +0 -0
  39. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/__init__.py +0 -0
  40. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/agy.py +0 -0
  41. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/base.py +0 -0
  42. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/codex.py +0 -0
  43. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/gemini.py +0 -0
  44. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/kimi.py +0 -0
  45. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/opencode.py +0 -0
  46. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/config.py +0 -0
  47. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/notify.py +0 -0
  48. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/state.py +0 -0
  49. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/watcher.py +0 -0
  50. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/__init__.py +0 -0
  51. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/fixtures/kimi_wire.jsonl +0 -0
  52. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_agy.py +0 -0
  53. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_batch.py +0 -0
  54. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_binary.py +0 -0
  55. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_claude.py +0 -0
  56. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_codex.py +0 -0
  57. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_kimi.py +0 -0
  58. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_no_changes.py +0 -0
  59. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_opencode.py +0 -0
  60. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_permissions.py +0 -0
  61. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_freshness.py +0 -0
  62. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_zai.py +0 -0
  63. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_skill.py +0 -0
  64. {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_state.py +0 -0
@@ -5,6 +5,112 @@ All notable changes to this project will be documented in this file.
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [0.14.0] - 2026-09-30
9
+
10
+ ### Added
11
+
12
+ - **`--effort` for the `claude` lane.** `[agents.claude] effort = "xhigh"` or
13
+ `playmaker dispatch|continue … --effort <low|medium|high|xhigh|max>` is forwarded to
14
+ Claude Code's own `--effort`, right after `--model`, on dispatch and resume. The flag
15
+ overrides the config for one run (carried through `PLAYMAKER_CLAUDE_EFFORT`, never
16
+ persisted); an invalid value fails before any process starts; other lanes accept and
17
+ ignore it with a note.
18
+ - **Banked limit resets in `playmaker quotas`.** The Codex block shows
19
+ `Banked resets N (M usable now)` and each reset's expiry (from
20
+ `wham/rate-limit-reset-credits`, same bearer token). Claude's reset is not reachable
21
+ with the CLI token (CodexBar reads it from the web session) and stays out.
22
+ - **The coach's ledger.** `scripts/ledger.py` keeps one JSON line per landed work package
23
+ in `~/.playmaker/ledger.jsonl` (`add`, `fix`, `escape`, `stats`, `tail`), and
24
+ `scripts/ledger-hook.py` — a Claude Code `PostToolUse` hook on `Bash` — writes the row
25
+ on every `git commit` from the WP's review directory, with an optional junior model
26
+ (`[ledger] junior = "codex:<model>"`) filling the soft fields. Fields, mapping and the
27
+ settings snippet: `references/ledger.md`.
28
+ - **`--risk seams` in `review-board.sh`.** A one-seat board over the joints between the
29
+ work packages of one fan-out, after each passed its own board and the global gate.
30
+
31
+ ### Changed
32
+
33
+ - **The coach skill, 2026-09-30 revision** (`skills/playmaker-coach`): routing is
34
+ «fit before headroom» (class the WP, then the ledger, then the pool floors, then
35
+ headroom); map vs. semantic recon; an integration step with a seam review before
36
+ landing; fixed board columns; a first-minute check on every detached dispatch and no
37
+ waiting on closed lanes; policy load order is personal, then repo. Reviewer roster
38
+ lines are an ordered preference list — the script seats exactly 1/2/3/1 for
39
+ routine/normal/high/seams after skipping the implementer and holds the rest in
40
+ reserve; fewer than required stops it (`PM_REVIEW_ALLOW_SHORT=1` overrides).
41
+ - **`review-board.sh` protects its artefacts.** Previous verdicts are archived on a new
42
+ dispatch and ignored by `--collect` when older than the current patch; `--dry-run`
43
+ no longer truncates `sessions.txt`; untracked files under the review paths are listed
44
+ with the `git add -N` hint; `board.env` records risk, round and required count;
45
+ `--collect` marks `failed`/`killed` seats DEAD, parses the last object carrying the
46
+ contract, and prints `verdicts: k/N` (`BOARD INCOMPLETE` while short). The prompt
47
+ interpolates the base ref, forbids edits and scratch files inside the tree, tells a
48
+ seat that cannot run the gate to report it under `unverifiable`, and asks the `risk`
49
+ lens to falsify the acceptance criteria.
50
+
51
+ ### Fixed
52
+
53
+ - **`playmaker quotas` no longer refreshes the Claude OAuth token.** The probe shared
54
+ the keychain entry and the single-use refresh token with the Claude Code CLI, and the
55
+ race logged the CLI out. The probe now only reads; when the token is expired it lets
56
+ the CLI refresh it (`[quotas] claude_refresh_via_cli`, default on) or shows
57
+ `login expired — run: claude auth login`. An outage after a successful refresh is
58
+ reported as an outage, not as an expired login.
59
+ - **The reset-credits inventory is best-effort**: a failing secondary request no
60
+ longer sinks the whole Codex block.
61
+ - **The ledger's writers take a lock and rewrites are atomic.** `ledger.py add` (and so
62
+ the hook) appends under an `flock`; `fix` and `escape` hold the same lock across their
63
+ read-modify-write and publish through a temp file and `os.replace`, so two coach sessions
64
+ committing at once cannot lose a row or tear the file; a trailing partial line is skipped
65
+ by the readers instead of stopping them. `--effort` is validated in the CLI before a
66
+ detached run spawns.
67
+
68
+ ## [0.13.0] - 2026-09-30
69
+
70
+ ### Added
71
+
72
+ - **`muse` lane — Meta's Muse Code CLI as a first-class agent.**
73
+ `playmaker dispatch muse` runs `muse exec --json …`, and the session id
74
+ arrives in the **first** stdout line — unlike kimi, whose id comes last — so
75
+ `playmaker list` shows it at once. Resume re-runs `muse exec` with
76
+ `--session-id` in the same cwd, and `summary` reads
77
+ `${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<uuid>/session.jsonl`.
78
+ Permissions default to Muse's sandbox with approval off and the workspace
79
+ trusted (`--disable-approval --trust-workspace` — without trust Muse ignores
80
+ the repo's AGENTS.md), and `[agents.muse]` takes `binary`, `model`,
81
+ `reasoning_effort`, `yolo`, `trust_workspace`, `sandbox_network` and
82
+ `permission_profile`. Muse Spark is a senior-tier peer of codex, GLM and K3,
83
+ on its own `muse login`. No usage/quota API exists, so `playmaker quotas`
84
+ has no Muse block.
85
+
86
+ ### Fixed
87
+
88
+ - **`playmaker quotas` printed `unsupported` for Kimi Code.** By 2026-09-29
89
+ the `/usages` endpoint stopped returning the `user` object — and with it
90
+ `membership.level` — and the probe required it, so a payload whose `usage`
91
+ and `limits[]` buckets were unchanged was rejected as unrecognised. `user`
92
+ is now optional (when present it must still be an object), so the Session
93
+ and Weekly rows render again; the `Kimi Code` header goes without the tier
94
+ the API no longer reports. The new `usages` block (`limit_5h`/`limit_7d`
95
+ with `used_ratio`) is not read: on the live account it said 0 while `usage`
96
+ counted 21 used.
97
+ - **`playmaker quotas` printed an error for Antigravity while agy worked.**
98
+ Two things broke at once. agy 1.2's embedded language server now refuses a
99
+ request without its CSRF token (401 `missing CSRF token`), and an agy someone
100
+ else started keeps that token to itself, so the local path found nothing it
101
+ could read. The remote fallback (`retrieveUserQuota`, ideType ANTIGRAVITY)
102
+ meanwhile began answering 403 "You do not have a valid license of this
103
+ product". The probe now starts a short-lived agy of its own when no running
104
+ one answers — headless (`--input-format stream-json -p=` waits on stdin, so
105
+ no TTY and no request spent), with a token it chose, in an empty scratch
106
+ directory, stopped as a process group once the summary is in (about 1.5 s;
107
+ capped at 12 s) — and reads the full Gemini and Claude/GPT 5h/weekly
108
+ windows off it. Process discovery also matches `agy-real`, the binary
109
+ behind an `agy` wrapper. If both sources still fail and the remote says "no
110
+ valid license", the block reads `unavailable` with a hint and keeps its last
111
+ success instead of `error`; any other remote failure is still an error. A
112
+ remote fallback now says why the daemon was missed.
113
+
8
114
  ## [0.12.1] - 2026-09-08
9
115
 
10
116
  ### Fixed
@@ -1,3 +1,30 @@
1
+ Metadata-Version: 2.5
2
+ Name: playmaker-cli
3
+ Version: 0.14.0
4
+ Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel.
5
+ Project-URL: Homepage, https://github.com/vladsafedev/playmaker
6
+ Project-URL: Repository, https://github.com/vladsafedev/playmaker
7
+ Project-URL: Issues, https://github.com/vladsafedev/playmaker/issues
8
+ Author-email: Vladislav Shulyugin <vladislav.shulyugin@gmail.com>
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: agents,ai,antigravity,claude,claude-code,cli,codex,gemini,glm,kimi,kimi-code,muse,muse-code,opencode,orchestration,subagents
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Operating System :: MacOS
16
+ Classifier: Operating System :: POSIX :: Linux
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development
22
+ Classifier: Topic :: Utilities
23
+ Requires-Python: >=3.11
24
+ Requires-Dist: rich>=13.7
25
+ Requires-Dist: typer>=0.27.0
26
+ Description-Content-Type: text/markdown
27
+
1
28
  # playmaker
2
29
 
3
30
  [![CI](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml/badge.svg)](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml)
@@ -5,7 +32,7 @@
5
32
  [![Python](https://img.shields.io/pypi/pyversions/playmaker-cli.svg?cacheSeconds=3600)](https://pypi.org/project/playmaker-cli/)
6
33
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
7
34
 
8
- **Run Claude Code, Codex, Antigravity, Kimi Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
35
+ **Run Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
9
36
 
10
37
  You stay in your Claude Code session doing the part only you can do. `playmaker`
11
38
  fans the rest out to other agent CLIs as detached processes, tracks them,
@@ -47,11 +74,13 @@ in one serial session:
47
74
  1. **Wall-clock speed.** A task that decomposes into 3–5 independent
48
75
  work-streams (schema, backend, frontend, tests, docs) finishes 2–4× faster
49
76
  when each stream runs as its own parallel agent.
50
- 2. **Provider arbitrage.** Codex, Antigravity, Kimi Code and opencode quotas
51
- are entirely separate pools from your Anthropic plan. Every slice you hand
52
- them is capacity your main session never spends — and Antigravity's roster
53
- includes Claude Sonnet/Opus, so even Claude-quality work can run on Google's
54
- pool. Kimi Code brings its own subscription with the senior-tier K3.
77
+ 2. **Provider arbitrage.** Codex, Antigravity, Kimi Code, Muse Code and
78
+ opencode quotas are entirely separate pools from your Anthropic plan. Every
79
+ slice you hand them is capacity your main session never spends — and
80
+ Antigravity's roster includes Claude Sonnet/Opus, so even Claude-quality
81
+ work can run on Google's pool. Kimi Code brings its own subscription with
82
+ the senior-tier K3; Muse Code adds Meta's senior-tier Muse Spark on its own
83
+ login.
55
84
  `opencode` widens this the most: one CLI fronting ~75 providers, from a z.ai
56
85
  GLM coding plan to models running locally on your own machine.
57
86
  3. **Bucket arbitrage inside one plan.** Headless `claude -p` draws on the same
@@ -124,6 +153,11 @@ The other agents differ, because their CLIs do:
124
153
  `--auto`, `--yolo` and `--plan`, and already runs with auto-approval, so
125
154
  there is no read-only mode below the prompt.
126
155
 
156
+ - **muse** runs sandboxed by default with approval off and the workspace
157
+ trusted (`--disable-approval --trust-workspace`). The sandbox denies writes
158
+ outside the repo, so `uv run`, npm and similar caches fail inside it — set
159
+ `yolo = true` if workers must run such gates.
160
+
127
161
  - **gemini** (legacy) runs with `--yolo`.
128
162
 
129
163
  ## Install
@@ -162,6 +196,7 @@ playmaker agents # which agent CLIs are reachable
162
196
  | **Antigravity (`agy`)** | bundled with [Antigravity](https://antigravity.google) | `--model claude-opus-4-6-thinking` — the roster moves, so read it from `agy models` |
163
197
  | **opencode** | `brew install sst/tap/opencode` (or see [opencode.ai](https://opencode.ai)) | `--model provider/model`, e.g. `zai-coding-plan/glm-5.2`; roster from `opencode models`, providers from `opencode auth login` |
164
198
  | **Kimi Code CLI** | `npm i -g @moonshot-ai/kimi-code` | needs Node ≥ 22.19 (a wrapper that pins a newer Node is fine — point `[agents.kimi] binary` at it); `--model kimi-code/k3-256k` — 256k window at half the quota cost of `k3`; log in once with `kimi login --region global`; model ids from `~/.kimi-code/config.toml` |
199
+ | **Muse Code CLI** | `curl -fsSL https://dev.meta.ai/install.sh \| sh` | self-updating launcher in `~/.local/bin` (point `[agents.muse] binary` at it if a detached dispatch can't find it); log in once with `muse login` or set `META_API_KEY`; `--model muse-spark-1.3` — omit it and Muse's own default from `~/.config/muse/settings.json` applies |
165
200
  | **Gemini CLI** (legacy) | `npm i -g @google/gemini-cli` | still supported, superseded by `agy` |
166
201
 
167
202
  At least one is required; `playmaker agents` tells you which it can see.
@@ -297,6 +332,7 @@ and locates the session file the tool writes locally. Empirically:
297
332
  | Antigravity | `~/.gemini/antigravity-cli/brain/<conversation-id>/.system_generated/logs/transcript_full.jsonl` |
298
333
  | opencode | SQLite — `~/.local/share/opencode/opencode.db` (`session` / `message` / `part`); playmaker keeps a pointer at `~/.playmaker/opencode/<id>.session` |
299
334
  | kimi | `~/.kimi-code/sessions/wd_<cwd-basename>_<hash>/session_<id>/agents/main/wire.jsonl` (per-cwd; `KIMI_CODE_HOME` overrides the root) |
335
+ | muse | `${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<session-uuid>/session.jsonl` (the date directory is the session's creation date; a resume appends to the same file) |
300
336
  | Gemini | `~/.gemini/tmp/<cwd-basename>/chats/session-<ts>-<short_id>.{json,jsonl}` |
301
337
 
302
338
  `thread` and `summary` normalize all of them into the same turn list, so every
@@ -339,6 +375,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
339
375
  `.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
340
376
  otherwise, and playmaker would report the agent as unavailable.
341
377
 
378
+ **claude** also accepts an effort level — Claude Code's `--effort
379
+ low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
380
+ motivating case: running Opus review boards at `xhigh`). Set the lane default
381
+ once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
382
+ claude --effort <level>` and `playmaker continue <id> --effort <level>`
383
+ override it for one run without persisting anything. The flag is accepted and
384
+ ignored on every other lane so scripts can pass it uniformly, and when it is
385
+ unset everywhere no `--effort` flag is sent at all — Claude Code's own default
386
+ applies.
387
+
342
388
  ## Notifications
343
389
 
344
390
  Every detached dispatch pings when it finishes. With
@@ -400,9 +446,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
400
446
  **separate buckets**. So is the `Codex — Spark` block, every agy row, and the
401
447
  whole Z.ai block. Routing a subtask is choosing which of them to spend.
402
448
 
403
- - **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
449
+ - **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
450
+ it expires, Claude Code itself refreshes it; the quota probe never rotates the
451
+ shared refresh token.
404
452
  Model-scoped weekly buckets come from the usage API's `limits[]` array and
405
453
  print as `Weekly · <model>`.
454
+ `[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
455
+ report `login expired — run: claude auth login` without spawning `claude`.
406
456
  - **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
407
457
  Spark model's own 5-hour and weekly windows come from
408
458
  `additional_rate_limits[]` and print as their own `Codex — Spark` block.
@@ -423,6 +473,9 @@ whole Z.ai block. Routing a subtask is choosing which of them to spend.
423
473
  `~/.kimi-code/credentials/kimi-code-env-*.json` (`$KIMI_CODE_HOME` overrides
424
474
  the root). Its 5-hour `Session` and weekly rows are separate percentage
425
475
  buckets; no credential reads as *unsupported* rather than a failed probe.
476
+ - **Muse Code** — no usage or quota API exists (billing is pay-as-you-go or a
477
+ subscription via accountscenter.meta.com), so `playmaker quotas` has no Muse
478
+ block.
426
479
 
427
480
  Reading these at *model* granularity is the point: they are the load-balancing
428
481
  input the coach skill uses to route each subtask.
@@ -1,30 +1,3 @@
1
- Metadata-Version: 2.5
2
- Name: playmaker-cli
3
- Version: 0.12.1
4
- Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code and opencode sub-agents in parallel.
5
- Project-URL: Homepage, https://github.com/vladsafedev/playmaker
6
- Project-URL: Repository, https://github.com/vladsafedev/playmaker
7
- Project-URL: Issues, https://github.com/vladsafedev/playmaker/issues
8
- Author-email: Vladislav Shulyugin <vladislav.shulyugin@gmail.com>
9
- License-Expression: MIT
10
- License-File: LICENSE
11
- Keywords: agents,ai,antigravity,claude,claude-code,cli,codex,gemini,glm,kimi,kimi-code,opencode,orchestration,subagents
12
- Classifier: Development Status :: 3 - Alpha
13
- Classifier: Environment :: Console
14
- Classifier: Intended Audience :: Developers
15
- Classifier: Operating System :: MacOS
16
- Classifier: Operating System :: POSIX :: Linux
17
- Classifier: Programming Language :: Python :: 3
18
- Classifier: Programming Language :: Python :: 3.11
19
- Classifier: Programming Language :: Python :: 3.12
20
- Classifier: Programming Language :: Python :: 3.13
21
- Classifier: Topic :: Software Development
22
- Classifier: Topic :: Utilities
23
- Requires-Python: >=3.11
24
- Requires-Dist: rich>=13.7
25
- Requires-Dist: typer>=0.27.0
26
- Description-Content-Type: text/markdown
27
-
28
1
  # playmaker
29
2
 
30
3
  [![CI](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml/badge.svg)](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml)
@@ -32,7 +5,7 @@ Description-Content-Type: text/markdown
32
5
  [![Python](https://img.shields.io/pypi/pyversions/playmaker-cli.svg?cacheSeconds=3600)](https://pypi.org/project/playmaker-cli/)
33
6
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](LICENSE)
34
7
 
35
- **Run Claude Code, Codex, Antigravity, Kimi Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
8
+ **Run Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
36
9
 
37
10
  You stay in your Claude Code session doing the part only you can do. `playmaker`
38
11
  fans the rest out to other agent CLIs as detached processes, tracks them,
@@ -74,11 +47,13 @@ in one serial session:
74
47
  1. **Wall-clock speed.** A task that decomposes into 3–5 independent
75
48
  work-streams (schema, backend, frontend, tests, docs) finishes 2–4× faster
76
49
  when each stream runs as its own parallel agent.
77
- 2. **Provider arbitrage.** Codex, Antigravity, Kimi Code and opencode quotas
78
- are entirely separate pools from your Anthropic plan. Every slice you hand
79
- them is capacity your main session never spends — and Antigravity's roster
80
- includes Claude Sonnet/Opus, so even Claude-quality work can run on Google's
81
- pool. Kimi Code brings its own subscription with the senior-tier K3.
50
+ 2. **Provider arbitrage.** Codex, Antigravity, Kimi Code, Muse Code and
51
+ opencode quotas are entirely separate pools from your Anthropic plan. Every
52
+ slice you hand them is capacity your main session never spends — and
53
+ Antigravity's roster includes Claude Sonnet/Opus, so even Claude-quality
54
+ work can run on Google's pool. Kimi Code brings its own subscription with
55
+ the senior-tier K3; Muse Code adds Meta's senior-tier Muse Spark on its own
56
+ login.
82
57
  `opencode` widens this the most: one CLI fronting ~75 providers, from a z.ai
83
58
  GLM coding plan to models running locally on your own machine.
84
59
  3. **Bucket arbitrage inside one plan.** Headless `claude -p` draws on the same
@@ -151,6 +126,11 @@ The other agents differ, because their CLIs do:
151
126
  `--auto`, `--yolo` and `--plan`, and already runs with auto-approval, so
152
127
  there is no read-only mode below the prompt.
153
128
 
129
+ - **muse** runs sandboxed by default with approval off and the workspace
130
+ trusted (`--disable-approval --trust-workspace`). The sandbox denies writes
131
+ outside the repo, so `uv run`, npm and similar caches fail inside it — set
132
+ `yolo = true` if workers must run such gates.
133
+
154
134
  - **gemini** (legacy) runs with `--yolo`.
155
135
 
156
136
  ## Install
@@ -189,6 +169,7 @@ playmaker agents # which agent CLIs are reachable
189
169
  | **Antigravity (`agy`)** | bundled with [Antigravity](https://antigravity.google) | `--model claude-opus-4-6-thinking` — the roster moves, so read it from `agy models` |
190
170
  | **opencode** | `brew install sst/tap/opencode` (or see [opencode.ai](https://opencode.ai)) | `--model provider/model`, e.g. `zai-coding-plan/glm-5.2`; roster from `opencode models`, providers from `opencode auth login` |
191
171
  | **Kimi Code CLI** | `npm i -g @moonshot-ai/kimi-code` | needs Node ≥ 22.19 (a wrapper that pins a newer Node is fine — point `[agents.kimi] binary` at it); `--model kimi-code/k3-256k` — 256k window at half the quota cost of `k3`; log in once with `kimi login --region global`; model ids from `~/.kimi-code/config.toml` |
172
+ | **Muse Code CLI** | `curl -fsSL https://dev.meta.ai/install.sh \| sh` | self-updating launcher in `~/.local/bin` (point `[agents.muse] binary` at it if a detached dispatch can't find it); log in once with `muse login` or set `META_API_KEY`; `--model muse-spark-1.3` — omit it and Muse's own default from `~/.config/muse/settings.json` applies |
192
173
  | **Gemini CLI** (legacy) | `npm i -g @google/gemini-cli` | still supported, superseded by `agy` |
193
174
 
194
175
  At least one is required; `playmaker agents` tells you which it can see.
@@ -324,6 +305,7 @@ and locates the session file the tool writes locally. Empirically:
324
305
  | Antigravity | `~/.gemini/antigravity-cli/brain/<conversation-id>/.system_generated/logs/transcript_full.jsonl` |
325
306
  | opencode | SQLite — `~/.local/share/opencode/opencode.db` (`session` / `message` / `part`); playmaker keeps a pointer at `~/.playmaker/opencode/<id>.session` |
326
307
  | kimi | `~/.kimi-code/sessions/wd_<cwd-basename>_<hash>/session_<id>/agents/main/wire.jsonl` (per-cwd; `KIMI_CODE_HOME` overrides the root) |
308
+ | muse | `${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<session-uuid>/session.jsonl` (the date directory is the session's creation date; a resume appends to the same file) |
327
309
  | Gemini | `~/.gemini/tmp/<cwd-basename>/chats/session-<ts>-<short_id>.{json,jsonl}` |
328
310
 
329
311
  `thread` and `summary` normalize all of them into the same turn list, so every
@@ -366,6 +348,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
366
348
  `.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
367
349
  otherwise, and playmaker would report the agent as unavailable.
368
350
 
351
+ **claude** also accepts an effort level — Claude Code's `--effort
352
+ low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
353
+ motivating case: running Opus review boards at `xhigh`). Set the lane default
354
+ once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
355
+ claude --effort <level>` and `playmaker continue <id> --effort <level>`
356
+ override it for one run without persisting anything. The flag is accepted and
357
+ ignored on every other lane so scripts can pass it uniformly, and when it is
358
+ unset everywhere no `--effort` flag is sent at all — Claude Code's own default
359
+ applies.
360
+
369
361
  ## Notifications
370
362
 
371
363
  Every detached dispatch pings when it finishes. With
@@ -427,9 +419,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
427
419
  **separate buckets**. So is the `Codex — Spark` block, every agy row, and the
428
420
  whole Z.ai block. Routing a subtask is choosing which of them to spend.
429
421
 
430
- - **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
422
+ - **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
423
+ it expires, Claude Code itself refreshes it; the quota probe never rotates the
424
+ shared refresh token.
431
425
  Model-scoped weekly buckets come from the usage API's `limits[]` array and
432
426
  print as `Weekly · <model>`.
427
+ `[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
428
+ report `login expired — run: claude auth login` without spawning `claude`.
433
429
  - **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
434
430
  Spark model's own 5-hour and weekly windows come from
435
431
  `additional_rate_limits[]` and print as their own `Codex — Spark` block.
@@ -450,6 +446,9 @@ whole Z.ai block. Routing a subtask is choosing which of them to spend.
450
446
  `~/.kimi-code/credentials/kimi-code-env-*.json` (`$KIMI_CODE_HOME` overrides
451
447
  the root). Its 5-hour `Session` and weekly rows are separate percentage
452
448
  buckets; no credential reads as *unsupported* rather than a failed probe.
449
+ - **Muse Code** — no usage or quota API exists (billing is pay-as-you-go or a
450
+ subscription via accountscenter.meta.com), so `playmaker quotas` has no Muse
451
+ block.
453
452
 
454
453
  Reading these at *model* granularity is the point: they are the load-balancing
455
454
  input the coach skill uses to route each subtask.
@@ -1,7 +1,7 @@
1
1
  [project]
2
2
  name = "playmaker-cli"
3
- version = "0.12.1"
4
- description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code and opencode sub-agents in parallel."
3
+ version = "0.14.0"
4
+ description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
7
7
  license = "MIT"
@@ -21,6 +21,8 @@ keywords = [
21
21
  "gemini",
22
22
  "kimi",
23
23
  "kimi-code",
24
+ "muse",
25
+ "muse-code",
24
26
  "opencode",
25
27
  "glm",
26
28
  "subagents",
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: playmaker-coach
3
- description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode/Kimi Code(kimi) workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
3
+ description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
4
4
  ---
5
5
 
6
6
  # playmaker-coach — you are the tech lead, not the typist
7
7
 
8
- `playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / Kimi Code (`kimi`) / a
9
- sibling Claude, tracks them, and returns their threads. This skill is the judgment on top: what to
10
- split, who gets which slice, how to size it so verifying is cheap, and **how the review board runs**.
8
+ `playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / a sibling Claude,
9
+ tracks them, and returns their threads. This skill is the judgment on top: what to split,
10
+ who gets which slice, how to size it so verifying is cheap, and **how the review board runs**.
11
11
 
12
12
  **The coach produces plans, prompts, verdicts and integration — not feature diffs.** Your context
13
13
  window is the most expensive resource on the table. Every file you read yourself and every line you
@@ -17,9 +17,9 @@ summarization, and even the drafting of worker prompts when the task is big enou
17
17
  Two loops run under your hand, always both:
18
18
 
19
19
  ```
20
- decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue → land
21
- ↑______________________|
22
- max 2 cycles
20
+ decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
21
+ ↑______________________| max 2 cycles
22
+ all WPs green → integrate → GLOBAL GATE → SEAM REVIEW (≥2 interacting WPs) → land → ledger row
23
23
  ```
24
24
 
25
25
  ## 1. Activation gate
@@ -47,21 +47,29 @@ outside the skill so it survives `playmaker skill install --force`. Read, in thi
47
47
  later files override earlier ones:
48
48
 
49
49
  ```bash
50
- cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions)
51
- cat ~/.playmaker/policy.md 2>/dev/null # personal policy (quota economics, lane defaults)
50
+ cat ~/.playmaker/policy.md 2>/dev/null # personal policy (tiers, quota economics, lane defaults)
51
+ cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions) — wins
52
52
  ls ./.playmaker/agents/*.md 2>/dev/null || ls ~/.playmaker/agents/*.md 2>/dev/null
53
53
  ```
54
54
 
55
- Agent profiles describe each lane's strengths, ceiling, and quota position — trust a profile over
56
- the generic defaults here. If no policy file exists, say so once in the plan and use these defaults.
55
+ Agent profiles describe each lane's traps, mechanics and quota position — trust a profile over the
56
+ generic defaults here, never over the policy's tier table or a dated owner order: profiles say how a
57
+ lane behaves, policy says what it may be given. A newer dated order beats an older line anywhere. If no policy file exists, say so once in the plan and use these defaults.
57
58
 
58
59
  ## 3. Protocol
59
60
 
60
61
  ### 3.1 Recon — delegate it
61
62
 
62
- Codebase exploration is the highest-leverage thing to delegate: raw reading is exactly what the
63
- cheapest model does as well as you do. Before your own `grep`/`Read` sweep, dispatch a read-only
64
- recon with an explicit deliverable and `--sync`, and read a 200-word report instead of ten files:
63
+ Codebase exploration is the highest-leverage thing to delegate. Two kinds, two tiers:
64
+
65
+ - **Map recon** — locate files, symbols, call sites, line ranges, inventories. Raw reading: the
66
+ cheapest lane (Gemini Flash) does it as well as you do.
67
+ - **Semantic recon** — invariants, ownership, contracts, side effects, impact radius, «why is it like
68
+ this». Judgment: a senior lane (Kimi K3 for many-file sweeps, Gemini Pro, Codex). A cheap model
69
+ returns a confident wrong map, and you plan on it.
70
+
71
+ Before your own `grep`/`Read` sweep, dispatch the matching kind with an explicit deliverable and
72
+ `--sync`, and read a 200-word report instead of ten files:
65
73
 
66
74
  ```bash
67
75
  playmaker dispatch agy --model <cheap-tier> --cwd "$(pwd)" --sync --read-only \
@@ -77,14 +85,17 @@ playmaker quotas # capacity per MODEL, not per agent; re-probes it
77
85
  ```
78
86
 
79
87
  Read at model granularity: separate buckets inside one provider are separate capacity. The aim is
80
- **level-loading** — finish the week with every pool drawn down evenly, except the one reserved for
81
- the coach — not hoarding the pools other people also use. Details and per-provider quirks:
88
+ **accepted work per point of scarce quota**, with prepaid capacity spent before it resets.
89
+ Level-loading — every pool drawn down evenly by week's end, except the coach's own — is the
90
+ tie-breaker between lanes that fit a WP equally, never a reason to queue a WP behind a closed lane. Details and per-provider quirks:
82
91
  `references/quotas.md`.
83
92
 
84
93
  ### 3.3 Decompose into work packages
85
94
 
86
- 2–5 WPs on the first pass. A WP is dispatchable only when all five hold — this is what makes review
87
- cheap and re-prompting rare:
95
+ 2–5 WPs on the first pass. Name each WP's **class** first — `mechanical | feature | terminal-heavy |
96
+ repo-recon | architecture | high-risk` — it decides which senior lanes are eligible (policy: fit before
97
+ headroom), which lenses the board gets, and it is the first field of the ledger row. Then a WP is
98
+ dispatchable only when all five hold — this is what makes review cheap and re-prompting rare:
88
99
 
89
100
  1. **Hard file boundary.** "Edit only `x.ts` and its spec" — never "do the backend part".
90
101
  2. **A gate the worker runs itself** — `tsc --noEmit`, a named spec file, a lint pass. It must exit 0
@@ -104,7 +115,7 @@ policy first, then `references/lanes.md`.
104
115
 
105
116
  ### 3.4 Propose, then wait
106
117
 
107
- Post the plan as: WP → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
118
+ Post the plan as: WP → class → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
108
119
  the current per-model capacity. **Dispatch nothing until the user approves.** Approval may be
109
120
  partial ("go but reroute tests"); restate the modified plan in one line, then dispatch.
110
121
 
@@ -118,19 +129,27 @@ playmaker dispatch <agent> --model <name> --batch "$B" --cwd "$(pwd)" --prompt "
118
129
  Detached by default — that is the point; never `--sync` a whole fan-out. Always pass `--cwd`,
119
130
  `--batch` (one summary ping for the batch), and `--model` unless the profile says otherwise.
120
131
  Prompt shape: `references/prompt-templates.md`. Per-agent traps (agy scratch dir, opencode relative
121
- paths, codex model roster, kimi's K2.7 default and exit codes): `references/agent-gotchas.md`.
132
+ paths, codex model roster): `references/agent-gotchas.md`.
122
133
 
123
134
  Parallel WPs that touch the same files go in **git worktrees**, one per WP, or they will collide.
124
135
 
136
+ Within a minute of every detached dispatch, `playmaker get <id> --json`: a quota or auth refusal
137
+ finishes in seconds as `failed`, pings nobody, and looks exactly like a slow worker. Give every
138
+ dispatch a deadline (recon 15 minutes; a WP by its size) and on expiry check the process is alive
139
+ (`playmaker logs <id>`, its RSS) before waiting longer. Never wait for a closed lane.
140
+
125
141
  ### 3.6 Prove it on disk before you believe it
126
142
 
127
143
  `done` means "the process exited cleanly with text", not "code changed". playmaker ≥0.9 marks a
128
144
  zero-change write task `no_changes` — treat it exactly as a failure. On older builds, check yourself:
129
145
 
130
146
  ```bash
131
- git -C "<cwd>" status --short # empty tree on a write WP = NOT done
147
+ git -C "<cwd>" status --short --untracked-files=all # empty tree on a write WP = NOT done
132
148
  ```
133
149
 
150
+ New files are invisible to `git diff`: `git add -N <new files>` before the board, and compare
151
+ `git diff --stat` with what the worker claims it changed.
152
+
134
153
  Then run the WP's own gate yourself, once, cheaply. A WP that fails its gate never reaches the
135
154
  review board — it goes straight back to the worker.
136
155
 
@@ -156,10 +175,18 @@ Non-negotiables:
156
175
  - Every finding carries `file:line` + a concrete failure scenario. **No evidence → dropped.** You
157
176
  arbitrate, and a reviewer's confidence is an input, not a verdict.
158
177
  - Fixes go back via `playmaker continue <impl-id>` as a numbered list — the worker still has its
159
- context. Re-review runs on the **delta only**.
178
+ context. Re-review runs on the **delta only** — which means passing the round-1 state as the
179
+ base-ref (a wip commit in the worktree, squashed before landing); with the original base the
180
+ patch is cumulative and reviewers re-open settled code.
181
+ - A ruling that changes the scope goes into `spec.md` before the next round — reviewers refute
182
+ against the spec, and a stale spec produces false blockers. Before charging a finding to the
183
+ worker, check it is not already present on the base ref.
160
184
  - **Two cycles maximum.** Still blocking after two → stop and escalate to the user with both
161
185
  verdicts. A third round at the same lane is the most expensive way to use a cheap model.
162
186
  - Land only at **zero blocking findings**.
187
+ - On `high`, the `risk` seat is told to **falsify** the acceptance criteria — a command, input or repro
188
+ per criterion it attacked, not only an argument; its `pass` lists what it tried. A seat that cannot
189
+ execute (the `claude` lane runs with binaries forbidden) never holds `correctness` — give it `contracts`.
163
190
 
164
191
  ### 3.8 Keep a board file
165
192
 
@@ -172,11 +199,39 @@ Long fan-outs outlive your context. Maintain `./.playmaker/board.md` — one row
172
199
  Update it at dispatch, at gate, after each review round. On resume, read the board before anything
173
200
  else. It is also what you paste back to the user as the status report.
174
201
 
202
+ The columns are fixed — one layout, not one per board: a resumed session and the ledger both parse
203
+ them. At landing, the ledger hook writes the WP's row to the cross-repo ledger on `git commit`; read its
204
+ `[ledger]` line and correct the soft fields with `python3 ~/.playmaker/scripts/ledger.py fix wp=… k=v`.
205
+ When the hook matched nothing, append the row by hand: `python3 ~/.playmaker/scripts/ledger.py add
206
+ wp=… class=… impl_lane=… …` (fields: `~/.playmaker/policy.md`, «Ledger»). It keeps what board prose loses: class, first-gate pass, blocking findings that survived
207
+ adjudication, cycles, wall time — the evidence the routing step reads.
208
+
175
209
  ### 3.9 Failures
176
210
 
177
- Surface them; never silently retry. A failed dispatch (missing binary, bad auth, rejected model)
178
- gets a re-routed plan proposed to the user, not a second attempt at the same string. Diagnosis per
179
- agent: `references/agent-gotchas.md`.
211
+ Surface them; never silently retry — and never wait. A failed dispatch (missing binary, bad auth,
212
+ rejected model, quota) is neither a fix cycle nor a worker failure: re-route the WP to the next live
213
+ senior lane now, with the full context in a fresh prompt, and record the substitution in the board.
214
+ Ask the user only when no eligible lane is left. Diagnosis per agent: `references/agent-gotchas.md`.
215
+
216
+ ### 3.10 Integrate before landing — when the fan-out had two or more WPs
217
+
218
+ Two green WPs can be wrong together. Per-WP gates and boards prove the parts; nothing above proves the
219
+ whole. Once every WP is green:
220
+
221
+ ```
222
+ integrate on one tree (worktrees merged in dependency order)
223
+ → GLOBAL GATE — the repo's minimum bar plus the union of the WPs' own gates, run by you
224
+ → if ≥2 WPs touched interacting modules:
225
+ SEAM REVIEW — pm-review <batch>-seams <base-before-first-WP> --risk seams --impl-agent <main lane>
226
+ → commit, one WP per commit, in integration order → ledger rows
227
+ ```
228
+
229
+ "Interacting" = shared types or packages, the same service or screen, a migration and its consumers,
230
+ event or config names, shared state. The seam reviewer gets a spec that lists the seams and reads only
231
+ those: changed interfaces, duplicated or conflicting assumptions, incompatible types, migration order,
232
+ naming, config, shared state. The packages' internals already passed their own boards — re-reviewing
233
+ them is the third round the protocol forbids. Roster: the `seams` block in `reviewers.conf`;
234
+ `--impl-agent` is the lane that implemented most of the fan-out.
180
235
 
181
236
  ## 4. What the coach may still type by hand
182
237
 
@@ -209,3 +264,6 @@ editor on product code, ask whether that is a WP you failed to write.
209
264
  - **Dispatching a WP you cannot verify in one command and one paragraph.**
210
265
  - **Omitting `--model`** and letting a CLI default drain a top-tier bucket.
211
266
  - **Reading whole agent threads** when `summary` answers the question.
267
+ - **Landing two green WPs without running the tree they make together.** Per-WP gates prove the
268
+ parts; the global gate and the seam review prove the whole.
269
+ - **Routing by headroom alone.** Class first, then the ledger's floor, then the pool floors, then headroom.
@@ -65,22 +65,6 @@ line from `agy models` / `opencode models` rather than typing it.
65
65
  - Neither trap applies to **review** dispatches, which write nothing — which makes opencode a
66
66
  perfectly good reviewer even where it is a shaky implementer.
67
67
 
68
- ## kimi (Kimi Code CLI)
69
-
70
- - There is **no read-only mode below the prompt**: `-p` refuses `--auto`, `--yolo` and `--plan`
71
- (exit 1) and already runs with auto-approval. Only dispatch work you would run unattended anyway.
72
- - The session id arrives **only in the trailing `session.resume_hint` line**, so `playmaker list`
73
- shows the agent session late — do not conclude a dispatch failed just because the id has not
74
- appeared yet.
75
- - Sessions are **per-cwd**: `kimi session list` from another directory shows nothing. Track the
76
- session through playmaker, not through the CLI.
77
- - The stream carries **no token/cost fields** — there is nothing to budget against mid-run.
78
- - Exit codes: **exit 1 is non-retryable** (auth, quota, unknown model — "is not configured in
79
- config.toml"); **exit 75 is retryable**. Fix the cause on 1, re-dispatch on 75.
80
- - K3 is **slow on real tickets** — never `--sync` a big WP; dispatch detached and poll.
81
- - Needs **Node ≥ 22.19** — hence the wrapper binary; point `[agents.kimi] binary` at it.
82
- - **Always pass `-m kimi-code/k3-256k`** — the CLI default is the weaker K2.7 `kimi-for-coding`.
83
-
84
68
  ## Worktrees
85
69
 
86
70
  Parallel WPs that touch the same files collide. Give each its own git worktree and dispatch with