playmaker-cli 0.12.1__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/CHANGELOG.md +106 -0
- playmaker_cli-0.12.1/README.md → playmaker_cli-0.14.0/PKG-INFO +60 -7
- playmaker_cli-0.12.1/PKG-INFO → playmaker_cli-0.14.0/README.md +33 -34
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/pyproject.toml +4 -2
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/SKILL.md +83 -25
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md +0 -16
- playmaker_cli-0.14.0/skills/playmaker-coach/references/lanes.md +117 -0
- playmaker_cli-0.14.0/skills/playmaker-coach/references/ledger.md +77 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/prompt-templates.md +15 -4
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/quotas.md +14 -14
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/review-board.md +45 -8
- playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger-hook.py +571 -0
- playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger.py +264 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/scripts/review-board.sh +89 -27
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/claude.py +40 -0
- playmaker_cli-0.14.0/src/playmaker/agents/muse.py +376 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/cli.py +59 -1
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/quotas.py +261 -86
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/registry.py +2 -0
- playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_inventory.json +7 -0
- playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_usage.json +19 -0
- playmaker_cli-0.14.0/tests/fixtures/muse_exec.jsonl +4 -0
- playmaker_cli-0.14.0/tests/fixtures/muse_session.jsonl +13 -0
- playmaker_cli-0.14.0/tests/test_claude_effort.py +294 -0
- playmaker_cli-0.14.0/tests/test_ledger_concurrency.py +112 -0
- playmaker_cli-0.14.0/tests/test_muse.py +182 -0
- playmaker_cli-0.14.0/tests/test_quotas_antigravity.py +489 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_claude.py +120 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_codex.py +84 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_kimi.py +21 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_registry.py +2 -2
- playmaker_cli-0.12.1/skills/playmaker-coach/references/lanes.md +0 -101
- playmaker_cli-0.12.1/tests/test_quotas_antigravity.py +0 -209
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/.gitignore +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/LICENSE +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/commands.md +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/__init__.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/__main__.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/__init__.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/agy.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/base.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/codex.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/gemini.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/kimi.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/agents/opencode.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/config.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/notify.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/state.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/src/playmaker/watcher.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/__init__.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/fixtures/kimi_wire.jsonl +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_agy.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_batch.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_binary.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_claude.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_codex.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_kimi.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_no_changes.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_opencode.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_permissions.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_freshness.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_quotas_zai.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_skill.py +0 -0
- {playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/tests/test_state.py +0 -0
|
@@ -5,6 +5,112 @@ All notable changes to this project will be documented in this file.
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
+
## [0.14.0] - 2026-09-30
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- **`--effort` for the `claude` lane.** `[agents.claude] effort = "xhigh"` or
|
|
13
|
+
`playmaker dispatch|continue … --effort <low|medium|high|xhigh|max>` is forwarded to
|
|
14
|
+
Claude Code's own `--effort`, right after `--model`, on dispatch and resume. The flag
|
|
15
|
+
overrides the config for one run (carried through `PLAYMAKER_CLAUDE_EFFORT`, never
|
|
16
|
+
persisted); an invalid value fails before any process starts; other lanes accept and
|
|
17
|
+
ignore it with a note.
|
|
18
|
+
- **Banked limit resets in `playmaker quotas`.** The Codex block shows
|
|
19
|
+
`Banked resets N (M usable now)` and each reset's expiry (from
|
|
20
|
+
`wham/rate-limit-reset-credits`, same bearer token). Claude's reset is not reachable
|
|
21
|
+
with the CLI token (CodexBar reads it from the web session) and stays out.
|
|
22
|
+
- **The coach's ledger.** `scripts/ledger.py` keeps one JSON line per landed work package
|
|
23
|
+
in `~/.playmaker/ledger.jsonl` (`add`, `fix`, `escape`, `stats`, `tail`), and
|
|
24
|
+
`scripts/ledger-hook.py` — a Claude Code `PostToolUse` hook on `Bash` — writes the row
|
|
25
|
+
on every `git commit` from the WP's review directory, with an optional junior model
|
|
26
|
+
(`[ledger] junior = "codex:<model>"`) filling the soft fields. Fields, mapping and the
|
|
27
|
+
settings snippet: `references/ledger.md`.
|
|
28
|
+
- **`--risk seams` in `review-board.sh`.** A one-seat board over the joints between the
|
|
29
|
+
work packages of one fan-out, after each passed its own board and the global gate.
|
|
30
|
+
|
|
31
|
+
### Changed
|
|
32
|
+
|
|
33
|
+
- **The coach skill, 2026-09-30 revision** (`skills/playmaker-coach`): routing is
|
|
34
|
+
«fit before headroom» (class the WP, then the ledger, then the pool floors, then
|
|
35
|
+
headroom); map vs. semantic recon; an integration step with a seam review before
|
|
36
|
+
landing; fixed board columns; a first-minute check on every detached dispatch and no
|
|
37
|
+
waiting on closed lanes; policy load order is personal, then repo. Reviewer roster
|
|
38
|
+
lines are an ordered preference list — the script seats exactly 1/2/3/1 for
|
|
39
|
+
routine/normal/high/seams after skipping the implementer and holds the rest in
|
|
40
|
+
reserve; fewer than required stops it (`PM_REVIEW_ALLOW_SHORT=1` overrides).
|
|
41
|
+
- **`review-board.sh` protects its artefacts.** Previous verdicts are archived on a new
|
|
42
|
+
dispatch and ignored by `--collect` when older than the current patch; `--dry-run`
|
|
43
|
+
no longer truncates `sessions.txt`; untracked files under the review paths are listed
|
|
44
|
+
with the `git add -N` hint; `board.env` records risk, round and required count;
|
|
45
|
+
`--collect` marks `failed`/`killed` seats DEAD, parses the last object carrying the
|
|
46
|
+
contract, and prints `verdicts: k/N` (`BOARD INCOMPLETE` while short). The prompt
|
|
47
|
+
interpolates the base ref, forbids edits and scratch files inside the tree, tells a
|
|
48
|
+
seat that cannot run the gate to report it under `unverifiable`, and asks the `risk`
|
|
49
|
+
lens to falsify the acceptance criteria.
|
|
50
|
+
|
|
51
|
+
### Fixed
|
|
52
|
+
|
|
53
|
+
- **`playmaker quotas` no longer refreshes the Claude OAuth token.** The probe shared
|
|
54
|
+
the keychain entry and the single-use refresh token with the Claude Code CLI, and the
|
|
55
|
+
race logged the CLI out. The probe now only reads; when the token is expired it lets
|
|
56
|
+
the CLI refresh it (`[quotas] claude_refresh_via_cli`, default on) or shows
|
|
57
|
+
`login expired — run: claude auth login`. An outage after a successful refresh is
|
|
58
|
+
reported as an outage, not as an expired login.
|
|
59
|
+
- **The reset-credits inventory is best-effort**: a failing secondary request no
|
|
60
|
+
longer sinks the whole Codex block.
|
|
61
|
+
- **The ledger's writers take a lock and rewrites are atomic.** `ledger.py add` (and so
|
|
62
|
+
the hook) appends under an `flock`; `fix` and `escape` hold the same lock across their
|
|
63
|
+
read-modify-write and publish through a temp file and `os.replace`, so two coach sessions
|
|
64
|
+
committing at once cannot lose a row or tear the file; a trailing partial line is skipped
|
|
65
|
+
by the readers instead of stopping them. `--effort` is validated in the CLI before a
|
|
66
|
+
detached run spawns.
|
|
67
|
+
|
|
68
|
+
## [0.13.0] - 2026-09-30
|
|
69
|
+
|
|
70
|
+
### Added
|
|
71
|
+
|
|
72
|
+
- **`muse` lane — Meta's Muse Code CLI as a first-class agent.**
|
|
73
|
+
`playmaker dispatch muse` runs `muse exec --json …`, and the session id
|
|
74
|
+
arrives in the **first** stdout line — unlike kimi, whose id comes last — so
|
|
75
|
+
`playmaker list` shows it at once. Resume re-runs `muse exec` with
|
|
76
|
+
`--session-id` in the same cwd, and `summary` reads
|
|
77
|
+
`${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<uuid>/session.jsonl`.
|
|
78
|
+
Permissions default to Muse's sandbox with approval off and the workspace
|
|
79
|
+
trusted (`--disable-approval --trust-workspace` — without trust Muse ignores
|
|
80
|
+
the repo's AGENTS.md), and `[agents.muse]` takes `binary`, `model`,
|
|
81
|
+
`reasoning_effort`, `yolo`, `trust_workspace`, `sandbox_network` and
|
|
82
|
+
`permission_profile`. Muse Spark is a senior-tier peer of codex, GLM and K3,
|
|
83
|
+
on its own `muse login`. No usage/quota API exists, so `playmaker quotas`
|
|
84
|
+
has no Muse block.
|
|
85
|
+
|
|
86
|
+
### Fixed
|
|
87
|
+
|
|
88
|
+
- **`playmaker quotas` printed `unsupported` for Kimi Code.** By 2026-09-29
|
|
89
|
+
the `/usages` endpoint stopped returning the `user` object — and with it
|
|
90
|
+
`membership.level` — and the probe required it, so a payload whose `usage`
|
|
91
|
+
and `limits[]` buckets were unchanged was rejected as unrecognised. `user`
|
|
92
|
+
is now optional (when present it must still be an object), so the Session
|
|
93
|
+
and Weekly rows render again; the `Kimi Code` header goes without the tier
|
|
94
|
+
the API no longer reports. The new `usages` block (`limit_5h`/`limit_7d`
|
|
95
|
+
with `used_ratio`) is not read: on the live account it said 0 while `usage`
|
|
96
|
+
counted 21 used.
|
|
97
|
+
- **`playmaker quotas` printed an error for Antigravity while agy worked.**
|
|
98
|
+
Two things broke at once. agy 1.2's embedded language server now refuses a
|
|
99
|
+
request without its CSRF token (401 `missing CSRF token`), and an agy someone
|
|
100
|
+
else started keeps that token to itself, so the local path found nothing it
|
|
101
|
+
could read. The remote fallback (`retrieveUserQuota`, ideType ANTIGRAVITY)
|
|
102
|
+
meanwhile began answering 403 "You do not have a valid license of this
|
|
103
|
+
product". The probe now starts a short-lived agy of its own when no running
|
|
104
|
+
one answers — headless (`--input-format stream-json -p=` waits on stdin, so
|
|
105
|
+
no TTY and no request spent), with a token it chose, in an empty scratch
|
|
106
|
+
directory, stopped as a process group once the summary is in (about 1.5 s;
|
|
107
|
+
capped at 12 s) — and reads the full Gemini and Claude/GPT 5h/weekly
|
|
108
|
+
windows off it. Process discovery also matches `agy-real`, the binary
|
|
109
|
+
behind an `agy` wrapper. If both sources still fail and the remote says "no
|
|
110
|
+
valid license", the block reads `unavailable` with a hint and keeps its last
|
|
111
|
+
success instead of `error`; any other remote failure is still an error. A
|
|
112
|
+
remote fallback now says why the daemon was missed.
|
|
113
|
+
|
|
8
114
|
## [0.12.1] - 2026-09-08
|
|
9
115
|
|
|
10
116
|
### Fixed
|
|
@@ -1,3 +1,30 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: playmaker-cli
|
|
3
|
+
Version: 0.14.0
|
|
4
|
+
Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel.
|
|
5
|
+
Project-URL: Homepage, https://github.com/vladsafedev/playmaker
|
|
6
|
+
Project-URL: Repository, https://github.com/vladsafedev/playmaker
|
|
7
|
+
Project-URL: Issues, https://github.com/vladsafedev/playmaker/issues
|
|
8
|
+
Author-email: Vladislav Shulyugin <vladislav.shulyugin@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: agents,ai,antigravity,claude,claude-code,cli,codex,gemini,glm,kimi,kimi-code,muse,muse-code,opencode,orchestration,subagents
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: MacOS
|
|
16
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development
|
|
22
|
+
Classifier: Topic :: Utilities
|
|
23
|
+
Requires-Python: >=3.11
|
|
24
|
+
Requires-Dist: rich>=13.7
|
|
25
|
+
Requires-Dist: typer>=0.27.0
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
|
|
1
28
|
# playmaker
|
|
2
29
|
|
|
3
30
|
[](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml)
|
|
@@ -5,7 +32,7 @@
|
|
|
5
32
|
[](https://pypi.org/project/playmaker-cli/)
|
|
6
33
|
[](LICENSE)
|
|
7
34
|
|
|
8
|
-
**Run Claude Code, Codex, Antigravity, Kimi Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
|
|
35
|
+
**Run Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
|
|
9
36
|
|
|
10
37
|
You stay in your Claude Code session doing the part only you can do. `playmaker`
|
|
11
38
|
fans the rest out to other agent CLIs as detached processes, tracks them,
|
|
@@ -47,11 +74,13 @@ in one serial session:
|
|
|
47
74
|
1. **Wall-clock speed.** A task that decomposes into 3–5 independent
|
|
48
75
|
work-streams (schema, backend, frontend, tests, docs) finishes 2–4× faster
|
|
49
76
|
when each stream runs as its own parallel agent.
|
|
50
|
-
2. **Provider arbitrage.** Codex, Antigravity, Kimi Code
|
|
51
|
-
are entirely separate pools from your Anthropic plan. Every
|
|
52
|
-
them is capacity your main session never spends — and
|
|
53
|
-
includes Claude Sonnet/Opus, so even Claude-quality
|
|
54
|
-
pool. Kimi Code brings its own subscription with
|
|
77
|
+
2. **Provider arbitrage.** Codex, Antigravity, Kimi Code, Muse Code and
|
|
78
|
+
opencode quotas are entirely separate pools from your Anthropic plan. Every
|
|
79
|
+
slice you hand them is capacity your main session never spends — and
|
|
80
|
+
Antigravity's roster includes Claude Sonnet/Opus, so even Claude-quality
|
|
81
|
+
work can run on Google's pool. Kimi Code brings its own subscription with
|
|
82
|
+
the senior-tier K3; Muse Code adds Meta's senior-tier Muse Spark on its own
|
|
83
|
+
login.
|
|
55
84
|
`opencode` widens this the most: one CLI fronting ~75 providers, from a z.ai
|
|
56
85
|
GLM coding plan to models running locally on your own machine.
|
|
57
86
|
3. **Bucket arbitrage inside one plan.** Headless `claude -p` draws on the same
|
|
@@ -124,6 +153,11 @@ The other agents differ, because their CLIs do:
|
|
|
124
153
|
`--auto`, `--yolo` and `--plan`, and already runs with auto-approval, so
|
|
125
154
|
there is no read-only mode below the prompt.
|
|
126
155
|
|
|
156
|
+
- **muse** runs sandboxed by default with approval off and the workspace
|
|
157
|
+
trusted (`--disable-approval --trust-workspace`). The sandbox denies writes
|
|
158
|
+
outside the repo, so `uv run`, npm and similar caches fail inside it — set
|
|
159
|
+
`yolo = true` if workers must run such gates.
|
|
160
|
+
|
|
127
161
|
- **gemini** (legacy) runs with `--yolo`.
|
|
128
162
|
|
|
129
163
|
## Install
|
|
@@ -162,6 +196,7 @@ playmaker agents # which agent CLIs are reachable
|
|
|
162
196
|
| **Antigravity (`agy`)** | bundled with [Antigravity](https://antigravity.google) | `--model claude-opus-4-6-thinking` — the roster moves, so read it from `agy models` |
|
|
163
197
|
| **opencode** | `brew install sst/tap/opencode` (or see [opencode.ai](https://opencode.ai)) | `--model provider/model`, e.g. `zai-coding-plan/glm-5.2`; roster from `opencode models`, providers from `opencode auth login` |
|
|
164
198
|
| **Kimi Code CLI** | `npm i -g @moonshot-ai/kimi-code` | needs Node ≥ 22.19 (a wrapper that pins a newer Node is fine — point `[agents.kimi] binary` at it); `--model kimi-code/k3-256k` — 256k window at half the quota cost of `k3`; log in once with `kimi login --region global`; model ids from `~/.kimi-code/config.toml` |
|
|
199
|
+
| **Muse Code CLI** | `curl -fsSL https://dev.meta.ai/install.sh \| sh` | self-updating launcher in `~/.local/bin` (point `[agents.muse] binary` at it if a detached dispatch can't find it); log in once with `muse login` or set `META_API_KEY`; `--model muse-spark-1.3` — omit it and Muse's own default from `~/.config/muse/settings.json` applies |
|
|
165
200
|
| **Gemini CLI** (legacy) | `npm i -g @google/gemini-cli` | still supported, superseded by `agy` |
|
|
166
201
|
|
|
167
202
|
At least one is required; `playmaker agents` tells you which it can see.
|
|
@@ -297,6 +332,7 @@ and locates the session file the tool writes locally. Empirically:
|
|
|
297
332
|
| Antigravity | `~/.gemini/antigravity-cli/brain/<conversation-id>/.system_generated/logs/transcript_full.jsonl` |
|
|
298
333
|
| opencode | SQLite — `~/.local/share/opencode/opencode.db` (`session` / `message` / `part`); playmaker keeps a pointer at `~/.playmaker/opencode/<id>.session` |
|
|
299
334
|
| kimi | `~/.kimi-code/sessions/wd_<cwd-basename>_<hash>/session_<id>/agents/main/wire.jsonl` (per-cwd; `KIMI_CODE_HOME` overrides the root) |
|
|
335
|
+
| muse | `${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<session-uuid>/session.jsonl` (the date directory is the session's creation date; a resume appends to the same file) |
|
|
300
336
|
| Gemini | `~/.gemini/tmp/<cwd-basename>/chats/session-<ts>-<short_id>.{json,jsonl}` |
|
|
301
337
|
|
|
302
338
|
`thread` and `summary` normalize all of them into the same turn list, so every
|
|
@@ -339,6 +375,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
|
|
|
339
375
|
`.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
|
|
340
376
|
otherwise, and playmaker would report the agent as unavailable.
|
|
341
377
|
|
|
378
|
+
**claude** also accepts an effort level — Claude Code's `--effort
|
|
379
|
+
low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
|
|
380
|
+
motivating case: running Opus review boards at `xhigh`). Set the lane default
|
|
381
|
+
once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
|
|
382
|
+
claude --effort <level>` and `playmaker continue <id> --effort <level>`
|
|
383
|
+
override it for one run without persisting anything. The flag is accepted and
|
|
384
|
+
ignored on every other lane so scripts can pass it uniformly, and when it is
|
|
385
|
+
unset everywhere no `--effort` flag is sent at all — Claude Code's own default
|
|
386
|
+
applies.
|
|
387
|
+
|
|
342
388
|
## Notifications
|
|
343
389
|
|
|
344
390
|
Every detached dispatch pings when it finishes. With
|
|
@@ -400,9 +446,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
|
|
|
400
446
|
**separate buckets**. So is the `Codex — Spark` block, every agy row, and the
|
|
401
447
|
whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
402
448
|
|
|
403
|
-
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
|
|
449
|
+
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
|
|
450
|
+
it expires, Claude Code itself refreshes it; the quota probe never rotates the
|
|
451
|
+
shared refresh token.
|
|
404
452
|
Model-scoped weekly buckets come from the usage API's `limits[]` array and
|
|
405
453
|
print as `Weekly · <model>`.
|
|
454
|
+
`[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
|
|
455
|
+
report `login expired — run: claude auth login` without spawning `claude`.
|
|
406
456
|
- **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
|
|
407
457
|
Spark model's own 5-hour and weekly windows come from
|
|
408
458
|
`additional_rate_limits[]` and print as their own `Codex — Spark` block.
|
|
@@ -423,6 +473,9 @@ whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
|
423
473
|
`~/.kimi-code/credentials/kimi-code-env-*.json` (`$KIMI_CODE_HOME` overrides
|
|
424
474
|
the root). Its 5-hour `Session` and weekly rows are separate percentage
|
|
425
475
|
buckets; no credential reads as *unsupported* rather than a failed probe.
|
|
476
|
+
- **Muse Code** — no usage or quota API exists (billing is pay-as-you-go or a
|
|
477
|
+
subscription via accountscenter.meta.com), so `playmaker quotas` has no Muse
|
|
478
|
+
block.
|
|
426
479
|
|
|
427
480
|
Reading these at *model* granularity is the point: they are the load-balancing
|
|
428
481
|
input the coach skill uses to route each subtask.
|
|
@@ -1,30 +1,3 @@
|
|
|
1
|
-
Metadata-Version: 2.5
|
|
2
|
-
Name: playmaker-cli
|
|
3
|
-
Version: 0.12.1
|
|
4
|
-
Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code and opencode sub-agents in parallel.
|
|
5
|
-
Project-URL: Homepage, https://github.com/vladsafedev/playmaker
|
|
6
|
-
Project-URL: Repository, https://github.com/vladsafedev/playmaker
|
|
7
|
-
Project-URL: Issues, https://github.com/vladsafedev/playmaker/issues
|
|
8
|
-
Author-email: Vladislav Shulyugin <vladislav.shulyugin@gmail.com>
|
|
9
|
-
License-Expression: MIT
|
|
10
|
-
License-File: LICENSE
|
|
11
|
-
Keywords: agents,ai,antigravity,claude,claude-code,cli,codex,gemini,glm,kimi,kimi-code,opencode,orchestration,subagents
|
|
12
|
-
Classifier: Development Status :: 3 - Alpha
|
|
13
|
-
Classifier: Environment :: Console
|
|
14
|
-
Classifier: Intended Audience :: Developers
|
|
15
|
-
Classifier: Operating System :: MacOS
|
|
16
|
-
Classifier: Operating System :: POSIX :: Linux
|
|
17
|
-
Classifier: Programming Language :: Python :: 3
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
-
Classifier: Topic :: Software Development
|
|
22
|
-
Classifier: Topic :: Utilities
|
|
23
|
-
Requires-Python: >=3.11
|
|
24
|
-
Requires-Dist: rich>=13.7
|
|
25
|
-
Requires-Dist: typer>=0.27.0
|
|
26
|
-
Description-Content-Type: text/markdown
|
|
27
|
-
|
|
28
1
|
# playmaker
|
|
29
2
|
|
|
30
3
|
[](https://github.com/vladsafedev/playmaker/actions/workflows/ci.yml)
|
|
@@ -32,7 +5,7 @@ Description-Content-Type: text/markdown
|
|
|
32
5
|
[](https://pypi.org/project/playmaker-cli/)
|
|
33
6
|
[](LICENSE)
|
|
34
7
|
|
|
35
|
-
**Run Claude Code, Codex, Antigravity, Kimi Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
|
|
8
|
+
**Run Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode as parallel sub-agents from one terminal — and spend separate quotas instead of one.**
|
|
36
9
|
|
|
37
10
|
You stay in your Claude Code session doing the part only you can do. `playmaker`
|
|
38
11
|
fans the rest out to other agent CLIs as detached processes, tracks them,
|
|
@@ -74,11 +47,13 @@ in one serial session:
|
|
|
74
47
|
1. **Wall-clock speed.** A task that decomposes into 3–5 independent
|
|
75
48
|
work-streams (schema, backend, frontend, tests, docs) finishes 2–4× faster
|
|
76
49
|
when each stream runs as its own parallel agent.
|
|
77
|
-
2. **Provider arbitrage.** Codex, Antigravity, Kimi Code
|
|
78
|
-
are entirely separate pools from your Anthropic plan. Every
|
|
79
|
-
them is capacity your main session never spends — and
|
|
80
|
-
includes Claude Sonnet/Opus, so even Claude-quality
|
|
81
|
-
pool. Kimi Code brings its own subscription with
|
|
50
|
+
2. **Provider arbitrage.** Codex, Antigravity, Kimi Code, Muse Code and
|
|
51
|
+
opencode quotas are entirely separate pools from your Anthropic plan. Every
|
|
52
|
+
slice you hand them is capacity your main session never spends — and
|
|
53
|
+
Antigravity's roster includes Claude Sonnet/Opus, so even Claude-quality
|
|
54
|
+
work can run on Google's pool. Kimi Code brings its own subscription with
|
|
55
|
+
the senior-tier K3; Muse Code adds Meta's senior-tier Muse Spark on its own
|
|
56
|
+
login.
|
|
82
57
|
`opencode` widens this the most: one CLI fronting ~75 providers, from a z.ai
|
|
83
58
|
GLM coding plan to models running locally on your own machine.
|
|
84
59
|
3. **Bucket arbitrage inside one plan.** Headless `claude -p` draws on the same
|
|
@@ -151,6 +126,11 @@ The other agents differ, because their CLIs do:
|
|
|
151
126
|
`--auto`, `--yolo` and `--plan`, and already runs with auto-approval, so
|
|
152
127
|
there is no read-only mode below the prompt.
|
|
153
128
|
|
|
129
|
+
- **muse** runs sandboxed by default with approval off and the workspace
|
|
130
|
+
trusted (`--disable-approval --trust-workspace`). The sandbox denies writes
|
|
131
|
+
outside the repo, so `uv run`, npm and similar caches fail inside it — set
|
|
132
|
+
`yolo = true` if workers must run such gates.
|
|
133
|
+
|
|
154
134
|
- **gemini** (legacy) runs with `--yolo`.
|
|
155
135
|
|
|
156
136
|
## Install
|
|
@@ -189,6 +169,7 @@ playmaker agents # which agent CLIs are reachable
|
|
|
189
169
|
| **Antigravity (`agy`)** | bundled with [Antigravity](https://antigravity.google) | `--model claude-opus-4-6-thinking` — the roster moves, so read it from `agy models` |
|
|
190
170
|
| **opencode** | `brew install sst/tap/opencode` (or see [opencode.ai](https://opencode.ai)) | `--model provider/model`, e.g. `zai-coding-plan/glm-5.2`; roster from `opencode models`, providers from `opencode auth login` |
|
|
191
171
|
| **Kimi Code CLI** | `npm i -g @moonshot-ai/kimi-code` | needs Node ≥ 22.19 (a wrapper that pins a newer Node is fine — point `[agents.kimi] binary` at it); `--model kimi-code/k3-256k` — 256k window at half the quota cost of `k3`; log in once with `kimi login --region global`; model ids from `~/.kimi-code/config.toml` |
|
|
172
|
+
| **Muse Code CLI** | `curl -fsSL https://dev.meta.ai/install.sh \| sh` | self-updating launcher in `~/.local/bin` (point `[agents.muse] binary` at it if a detached dispatch can't find it); log in once with `muse login` or set `META_API_KEY`; `--model muse-spark-1.3` — omit it and Muse's own default from `~/.config/muse/settings.json` applies |
|
|
192
173
|
| **Gemini CLI** (legacy) | `npm i -g @google/gemini-cli` | still supported, superseded by `agy` |
|
|
193
174
|
|
|
194
175
|
At least one is required; `playmaker agents` tells you which it can see.
|
|
@@ -324,6 +305,7 @@ and locates the session file the tool writes locally. Empirically:
|
|
|
324
305
|
| Antigravity | `~/.gemini/antigravity-cli/brain/<conversation-id>/.system_generated/logs/transcript_full.jsonl` |
|
|
325
306
|
| opencode | SQLite — `~/.local/share/opencode/opencode.db` (`session` / `message` / `part`); playmaker keeps a pointer at `~/.playmaker/opencode/<id>.session` |
|
|
326
307
|
| kimi | `~/.kimi-code/sessions/wd_<cwd-basename>_<hash>/session_<id>/agents/main/wire.jsonl` (per-cwd; `KIMI_CODE_HOME` overrides the root) |
|
|
308
|
+
| muse | `${XDG_DATA_HOME:-~/.local/share}/muse/sessions/YYYY/MM/DD/<session-uuid>/session.jsonl` (the date directory is the session's creation date; a resume appends to the same file) |
|
|
327
309
|
| Gemini | `~/.gemini/tmp/<cwd-basename>/chats/session-<ts>-<short_id>.{json,jsonl}` |
|
|
328
310
|
|
|
329
311
|
`thread` and `summary` normalize all of them into the same turn list, so every
|
|
@@ -366,6 +348,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
|
|
|
366
348
|
`.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
|
|
367
349
|
otherwise, and playmaker would report the agent as unavailable.
|
|
368
350
|
|
|
351
|
+
**claude** also accepts an effort level — Claude Code's `--effort
|
|
352
|
+
low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
|
|
353
|
+
motivating case: running Opus review boards at `xhigh`). Set the lane default
|
|
354
|
+
once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
|
|
355
|
+
claude --effort <level>` and `playmaker continue <id> --effort <level>`
|
|
356
|
+
override it for one run without persisting anything. The flag is accepted and
|
|
357
|
+
ignored on every other lane so scripts can pass it uniformly, and when it is
|
|
358
|
+
unset everywhere no `--effort` flag is sent at all — Claude Code's own default
|
|
359
|
+
applies.
|
|
360
|
+
|
|
369
361
|
## Notifications
|
|
370
362
|
|
|
371
363
|
Every detached dispatch pings when it finishes. With
|
|
@@ -427,9 +419,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
|
|
|
427
419
|
**separate buckets**. So is the `Codex — Spark` block, every agy row, and the
|
|
428
420
|
whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
429
421
|
|
|
430
|
-
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
|
|
422
|
+
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
|
|
423
|
+
it expires, Claude Code itself refreshes it; the quota probe never rotates the
|
|
424
|
+
shared refresh token.
|
|
431
425
|
Model-scoped weekly buckets come from the usage API's `limits[]` array and
|
|
432
426
|
print as `Weekly · <model>`.
|
|
427
|
+
`[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
|
|
428
|
+
report `login expired — run: claude auth login` without spawning `claude`.
|
|
433
429
|
- **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
|
|
434
430
|
Spark model's own 5-hour and weekly windows come from
|
|
435
431
|
`additional_rate_limits[]` and print as their own `Codex — Spark` block.
|
|
@@ -450,6 +446,9 @@ whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
|
450
446
|
`~/.kimi-code/credentials/kimi-code-env-*.json` (`$KIMI_CODE_HOME` overrides
|
|
451
447
|
the root). Its 5-hour `Session` and weekly rows are separate percentage
|
|
452
448
|
buckets; no credential reads as *unsupported* rather than a failed probe.
|
|
449
|
+
- **Muse Code** — no usage or quota API exists (billing is pay-as-you-go or a
|
|
450
|
+
subscription via accountscenter.meta.com), so `playmaker quotas` has no Muse
|
|
451
|
+
block.
|
|
453
452
|
|
|
454
453
|
Reading these at *model* granularity is the point: they are the load-balancing
|
|
455
454
|
input the coach skill uses to route each subtask.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "playmaker-cli"
|
|
3
|
-
version = "0.
|
|
4
|
-
description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code and opencode sub-agents in parallel."
|
|
3
|
+
version = "0.14.0"
|
|
4
|
+
description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
7
7
|
license = "MIT"
|
|
@@ -21,6 +21,8 @@ keywords = [
|
|
|
21
21
|
"gemini",
|
|
22
22
|
"kimi",
|
|
23
23
|
"kimi-code",
|
|
24
|
+
"muse",
|
|
25
|
+
"muse-code",
|
|
24
26
|
"opencode",
|
|
25
27
|
"glm",
|
|
26
28
|
"subagents",
|
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: playmaker-coach
|
|
3
|
-
description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode
|
|
3
|
+
description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# playmaker-coach — you are the tech lead, not the typist
|
|
7
7
|
|
|
8
|
-
`playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode /
|
|
9
|
-
|
|
10
|
-
|
|
8
|
+
`playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / a sibling Claude,
|
|
9
|
+
tracks them, and returns their threads. This skill is the judgment on top: what to split,
|
|
10
|
+
who gets which slice, how to size it so verifying is cheap, and **how the review board runs**.
|
|
11
11
|
|
|
12
12
|
**The coach produces plans, prompts, verdicts and integration — not feature diffs.** Your context
|
|
13
13
|
window is the most expensive resource on the table. Every file you read yourself and every line you
|
|
@@ -17,9 +17,9 @@ summarization, and even the drafting of worker prompts when the task is big enou
|
|
|
17
17
|
Two loops run under your hand, always both:
|
|
18
18
|
|
|
19
19
|
```
|
|
20
|
-
decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
|
|
21
|
-
↑______________________|
|
|
22
|
-
|
|
20
|
+
decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
|
|
21
|
+
↑______________________| max 2 cycles
|
|
22
|
+
all WPs green → integrate → GLOBAL GATE → SEAM REVIEW (≥2 interacting WPs) → land → ledger row
|
|
23
23
|
```
|
|
24
24
|
|
|
25
25
|
## 1. Activation gate
|
|
@@ -47,21 +47,29 @@ outside the skill so it survives `playmaker skill install --force`. Read, in thi
|
|
|
47
47
|
later files override earlier ones:
|
|
48
48
|
|
|
49
49
|
```bash
|
|
50
|
-
cat
|
|
51
|
-
cat
|
|
50
|
+
cat ~/.playmaker/policy.md 2>/dev/null # personal policy (tiers, quota economics, lane defaults)
|
|
51
|
+
cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions) — wins
|
|
52
52
|
ls ./.playmaker/agents/*.md 2>/dev/null || ls ~/.playmaker/agents/*.md 2>/dev/null
|
|
53
53
|
```
|
|
54
54
|
|
|
55
|
-
Agent profiles describe each lane's
|
|
56
|
-
|
|
55
|
+
Agent profiles describe each lane's traps, mechanics and quota position — trust a profile over the
|
|
56
|
+
generic defaults here, never over the policy's tier table or a dated owner order: profiles say how a
|
|
57
|
+
lane behaves, policy says what it may be given. A newer dated order beats an older line anywhere. If no policy file exists, say so once in the plan and use these defaults.
|
|
57
58
|
|
|
58
59
|
## 3. Protocol
|
|
59
60
|
|
|
60
61
|
### 3.1 Recon — delegate it
|
|
61
62
|
|
|
62
|
-
Codebase exploration is the highest-leverage thing to delegate
|
|
63
|
-
|
|
64
|
-
|
|
63
|
+
Codebase exploration is the highest-leverage thing to delegate. Two kinds, two tiers:
|
|
64
|
+
|
|
65
|
+
- **Map recon** — locate files, symbols, call sites, line ranges, inventories. Raw reading: the
|
|
66
|
+
cheapest lane (Gemini Flash) does it as well as you do.
|
|
67
|
+
- **Semantic recon** — invariants, ownership, contracts, side effects, impact radius, «why is it like
|
|
68
|
+
this». Judgment: a senior lane (Kimi K3 for many-file sweeps, Gemini Pro, Codex). A cheap model
|
|
69
|
+
returns a confident wrong map, and you plan on it.
|
|
70
|
+
|
|
71
|
+
Before your own `grep`/`Read` sweep, dispatch the matching kind with an explicit deliverable and
|
|
72
|
+
`--sync`, and read a 200-word report instead of ten files:
|
|
65
73
|
|
|
66
74
|
```bash
|
|
67
75
|
playmaker dispatch agy --model <cheap-tier> --cwd "$(pwd)" --sync --read-only \
|
|
@@ -77,14 +85,17 @@ playmaker quotas # capacity per MODEL, not per agent; re-probes it
|
|
|
77
85
|
```
|
|
78
86
|
|
|
79
87
|
Read at model granularity: separate buckets inside one provider are separate capacity. The aim is
|
|
80
|
-
**
|
|
81
|
-
|
|
88
|
+
**accepted work per point of scarce quota**, with prepaid capacity spent before it resets.
|
|
89
|
+
Level-loading — every pool drawn down evenly by week's end, except the coach's own — is the
|
|
90
|
+
tie-breaker between lanes that fit a WP equally, never a reason to queue a WP behind a closed lane. Details and per-provider quirks:
|
|
82
91
|
`references/quotas.md`.
|
|
83
92
|
|
|
84
93
|
### 3.3 Decompose into work packages
|
|
85
94
|
|
|
86
|
-
2–5 WPs on the first pass.
|
|
87
|
-
|
|
95
|
+
2–5 WPs on the first pass. Name each WP's **class** first — `mechanical | feature | terminal-heavy |
|
|
96
|
+
repo-recon | architecture | high-risk` — it decides which senior lanes are eligible (policy: fit before
|
|
97
|
+
headroom), which lenses the board gets, and it is the first field of the ledger row. Then a WP is
|
|
98
|
+
dispatchable only when all five hold — this is what makes review cheap and re-prompting rare:
|
|
88
99
|
|
|
89
100
|
1. **Hard file boundary.** "Edit only `x.ts` and its spec" — never "do the backend part".
|
|
90
101
|
2. **A gate the worker runs itself** — `tsc --noEmit`, a named spec file, a lint pass. It must exit 0
|
|
@@ -104,7 +115,7 @@ policy first, then `references/lanes.md`.
|
|
|
104
115
|
|
|
105
116
|
### 3.4 Propose, then wait
|
|
106
117
|
|
|
107
|
-
Post the plan as: WP → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
|
|
118
|
+
Post the plan as: WP → class → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
|
|
108
119
|
the current per-model capacity. **Dispatch nothing until the user approves.** Approval may be
|
|
109
120
|
partial ("go but reroute tests"); restate the modified plan in one line, then dispatch.
|
|
110
121
|
|
|
@@ -118,19 +129,27 @@ playmaker dispatch <agent> --model <name> --batch "$B" --cwd "$(pwd)" --prompt "
|
|
|
118
129
|
Detached by default — that is the point; never `--sync` a whole fan-out. Always pass `--cwd`,
|
|
119
130
|
`--batch` (one summary ping for the batch), and `--model` unless the profile says otherwise.
|
|
120
131
|
Prompt shape: `references/prompt-templates.md`. Per-agent traps (agy scratch dir, opencode relative
|
|
121
|
-
paths, codex model roster
|
|
132
|
+
paths, codex model roster): `references/agent-gotchas.md`.
|
|
122
133
|
|
|
123
134
|
Parallel WPs that touch the same files go in **git worktrees**, one per WP, or they will collide.
|
|
124
135
|
|
|
136
|
+
Within a minute of every detached dispatch, `playmaker get <id> --json`: a quota or auth refusal
|
|
137
|
+
finishes in seconds as `failed`, pings nobody, and looks exactly like a slow worker. Give every
|
|
138
|
+
dispatch a deadline (recon 15 minutes; a WP by its size) and on expiry check the process is alive
|
|
139
|
+
(`playmaker logs <id>`, its RSS) before waiting longer. Never wait for a closed lane.
|
|
140
|
+
|
|
125
141
|
### 3.6 Prove it on disk before you believe it
|
|
126
142
|
|
|
127
143
|
`done` means "the process exited cleanly with text", not "code changed". playmaker ≥0.9 marks a
|
|
128
144
|
zero-change write task `no_changes` — treat it exactly as a failure. On older builds, check yourself:
|
|
129
145
|
|
|
130
146
|
```bash
|
|
131
|
-
git -C "<cwd>" status --short
|
|
147
|
+
git -C "<cwd>" status --short --untracked-files=all # empty tree on a write WP = NOT done
|
|
132
148
|
```
|
|
133
149
|
|
|
150
|
+
New files are invisible to `git diff`: `git add -N <new files>` before the board, and compare
|
|
151
|
+
`git diff --stat` with what the worker claims it changed.
|
|
152
|
+
|
|
134
153
|
Then run the WP's own gate yourself, once, cheaply. A WP that fails its gate never reaches the
|
|
135
154
|
review board — it goes straight back to the worker.
|
|
136
155
|
|
|
@@ -156,10 +175,18 @@ Non-negotiables:
|
|
|
156
175
|
- Every finding carries `file:line` + a concrete failure scenario. **No evidence → dropped.** You
|
|
157
176
|
arbitrate, and a reviewer's confidence is an input, not a verdict.
|
|
158
177
|
- Fixes go back via `playmaker continue <impl-id>` as a numbered list — the worker still has its
|
|
159
|
-
context. Re-review runs on the **delta only
|
|
178
|
+
context. Re-review runs on the **delta only** — which means passing the round-1 state as the
|
|
179
|
+
base-ref (a wip commit in the worktree, squashed before landing); with the original base the
|
|
180
|
+
patch is cumulative and reviewers re-open settled code.
|
|
181
|
+
- A ruling that changes the scope goes into `spec.md` before the next round — reviewers refute
|
|
182
|
+
against the spec, and a stale spec produces false blockers. Before charging a finding to the
|
|
183
|
+
worker, check it is not already present on the base ref.
|
|
160
184
|
- **Two cycles maximum.** Still blocking after two → stop and escalate to the user with both
|
|
161
185
|
verdicts. A third round at the same lane is the most expensive way to use a cheap model.
|
|
162
186
|
- Land only at **zero blocking findings**.
|
|
187
|
+
- On `high`, the `risk` seat is told to **falsify** the acceptance criteria — a command, input or repro
|
|
188
|
+
per criterion it attacked, not only an argument; its `pass` lists what it tried. A seat that cannot
|
|
189
|
+
execute (the `claude` lane runs with binaries forbidden) never holds `correctness` — give it `contracts`.
|
|
163
190
|
|
|
164
191
|
### 3.8 Keep a board file
|
|
165
192
|
|
|
@@ -172,11 +199,39 @@ Long fan-outs outlive your context. Maintain `./.playmaker/board.md` — one row
|
|
|
172
199
|
Update it at dispatch, at gate, after each review round. On resume, read the board before anything
|
|
173
200
|
else. It is also what you paste back to the user as the status report.
|
|
174
201
|
|
|
202
|
+
The columns are fixed — one layout, not one per board: a resumed session and the ledger both parse
|
|
203
|
+
them. At landing, the ledger hook writes the WP's row to the cross-repo ledger on `git commit`; read its
|
|
204
|
+
`[ledger]` line and correct the soft fields with `python3 ~/.playmaker/scripts/ledger.py fix wp=… k=v`.
|
|
205
|
+
When the hook matched nothing, append the row by hand: `python3 ~/.playmaker/scripts/ledger.py add
|
|
206
|
+
wp=… class=… impl_lane=… …` (fields: `~/.playmaker/policy.md`, «Ledger»). It keeps what board prose loses: class, first-gate pass, blocking findings that survived
|
|
207
|
+
adjudication, cycles, wall time — the evidence the routing step reads.
|
|
208
|
+
|
|
175
209
|
### 3.9 Failures
|
|
176
210
|
|
|
177
|
-
Surface them; never silently retry. A failed dispatch (missing binary, bad auth,
|
|
178
|
-
|
|
179
|
-
|
|
211
|
+
Surface them; never silently retry — and never wait. A failed dispatch (missing binary, bad auth,
|
|
212
|
+
rejected model, quota) is neither a fix cycle nor a worker failure: re-route the WP to the next live
|
|
213
|
+
senior lane now, with the full context in a fresh prompt, and record the substitution in the board.
|
|
214
|
+
Ask the user only when no eligible lane is left. Diagnosis per agent: `references/agent-gotchas.md`.
|
|
215
|
+
|
|
216
|
+
### 3.10 Integrate before landing — when the fan-out had two or more WPs
|
|
217
|
+
|
|
218
|
+
Two green WPs can be wrong together. Per-WP gates and boards prove the parts; nothing above proves the
|
|
219
|
+
whole. Once every WP is green:
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
integrate on one tree (worktrees merged in dependency order)
|
|
223
|
+
→ GLOBAL GATE — the repo's minimum bar plus the union of the WPs' own gates, run by you
|
|
224
|
+
→ if ≥2 WPs touched interacting modules:
|
|
225
|
+
SEAM REVIEW — pm-review <batch>-seams <base-before-first-WP> --risk seams --impl-agent <main lane>
|
|
226
|
+
→ commit, one WP per commit, in integration order → ledger rows
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
"Interacting" = shared types or packages, the same service or screen, a migration and its consumers,
|
|
230
|
+
event or config names, shared state. The seam reviewer gets a spec that lists the seams and reads only
|
|
231
|
+
those: changed interfaces, duplicated or conflicting assumptions, incompatible types, migration order,
|
|
232
|
+
naming, config, shared state. The packages' internals already passed their own boards — re-reviewing
|
|
233
|
+
them is the third round the protocol forbids. Roster: the `seams` block in `reviewers.conf`;
|
|
234
|
+
`--impl-agent` is the lane that implemented most of the fan-out.
|
|
180
235
|
|
|
181
236
|
## 4. What the coach may still type by hand
|
|
182
237
|
|
|
@@ -209,3 +264,6 @@ editor on product code, ask whether that is a WP you failed to write.
|
|
|
209
264
|
- **Dispatching a WP you cannot verify in one command and one paragraph.**
|
|
210
265
|
- **Omitting `--model`** and letting a CLI default drain a top-tier bucket.
|
|
211
266
|
- **Reading whole agent threads** when `summary` answers the question.
|
|
267
|
+
- **Landing two green WPs without running the tree they make together.** Per-WP gates prove the
|
|
268
|
+
parts; the global gate and the seam review prove the whole.
|
|
269
|
+
- **Routing by headroom alone.** Class first, then the ledger's floor, then the pool floors, then headroom.
|
{playmaker_cli-0.12.1 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md
RENAMED
|
@@ -65,22 +65,6 @@ line from `agy models` / `opencode models` rather than typing it.
|
|
|
65
65
|
- Neither trap applies to **review** dispatches, which write nothing — which makes opencode a
|
|
66
66
|
perfectly good reviewer even where it is a shaky implementer.
|
|
67
67
|
|
|
68
|
-
## kimi (Kimi Code CLI)
|
|
69
|
-
|
|
70
|
-
- There is **no read-only mode below the prompt**: `-p` refuses `--auto`, `--yolo` and `--plan`
|
|
71
|
-
(exit 1) and already runs with auto-approval. Only dispatch work you would run unattended anyway.
|
|
72
|
-
- The session id arrives **only in the trailing `session.resume_hint` line**, so `playmaker list`
|
|
73
|
-
shows the agent session late — do not conclude a dispatch failed just because the id has not
|
|
74
|
-
appeared yet.
|
|
75
|
-
- Sessions are **per-cwd**: `kimi session list` from another directory shows nothing. Track the
|
|
76
|
-
session through playmaker, not through the CLI.
|
|
77
|
-
- The stream carries **no token/cost fields** — there is nothing to budget against mid-run.
|
|
78
|
-
- Exit codes: **exit 1 is non-retryable** (auth, quota, unknown model — "is not configured in
|
|
79
|
-
config.toml"); **exit 75 is retryable**. Fix the cause on 1, re-dispatch on 75.
|
|
80
|
-
- K3 is **slow on real tickets** — never `--sync` a big WP; dispatch detached and poll.
|
|
81
|
-
- Needs **Node ≥ 22.19** — hence the wrapper binary; point `[agents.kimi] binary` at it.
|
|
82
|
-
- **Always pass `-m kimi-code/k3-256k`** — the CLI default is the weaker K2.7 `kimi-for-coding`.
|
|
83
|
-
|
|
84
68
|
## Worktrees
|
|
85
69
|
|
|
86
70
|
Parallel WPs that touch the same files collide. Give each its own git worktree and dispatch with
|