playmaker-cli 0.13.0__tar.gz → 0.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/CHANGELOG.md +60 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/PKG-INFO +16 -2
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/README.md +15 -1
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/pyproject.toml +1 -1
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/SKILL.md +83 -26
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md +0 -33
- playmaker_cli-0.14.0/skills/playmaker-coach/references/lanes.md +117 -0
- playmaker_cli-0.14.0/skills/playmaker-coach/references/ledger.md +77 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/prompt-templates.md +15 -4
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/quotas.md +16 -18
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/review-board.md +45 -8
- playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger-hook.py +571 -0
- playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger.py +264 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/scripts/review-board.sh +89 -27
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/claude.py +40 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/cli.py +46 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/quotas.py +98 -58
- playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_inventory.json +7 -0
- playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_usage.json +19 -0
- playmaker_cli-0.14.0/tests/test_claude_effort.py +294 -0
- playmaker_cli-0.14.0/tests/test_ledger_concurrency.py +112 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_claude.py +120 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_codex.py +84 -0
- playmaker_cli-0.13.0/skills/playmaker-coach/references/lanes.md +0 -104
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/.gitignore +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/LICENSE +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/commands.md +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/__init__.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/__main__.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/__init__.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/agy.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/base.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/codex.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/gemini.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/kimi.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/muse.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/opencode.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/config.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/notify.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/registry.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/state.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/watcher.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/__init__.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/kimi_wire.jsonl +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/muse_exec.jsonl +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/muse_session.jsonl +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_agy.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_batch.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_binary.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_claude.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_codex.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_kimi.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_muse.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_no_changes.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_opencode.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_permissions.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_antigravity.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_freshness.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_kimi.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_zai.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_registry.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_skill.py +0 -0
- {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_state.py +0 -0
|
@@ -5,6 +5,66 @@ All notable changes to this project will be documented in this file.
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
+
## [0.14.0] - 2026-09-30
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- **`--effort` for the `claude` lane.** `[agents.claude] effort = "xhigh"` or
|
|
13
|
+
`playmaker dispatch|continue … --effort <low|medium|high|xhigh|max>` is forwarded to
|
|
14
|
+
Claude Code's own `--effort`, right after `--model`, on dispatch and resume. The flag
|
|
15
|
+
overrides the config for one run (carried through `PLAYMAKER_CLAUDE_EFFORT`, never
|
|
16
|
+
persisted); an invalid value fails before any process starts; other lanes accept and
|
|
17
|
+
ignore it with a note.
|
|
18
|
+
- **Banked limit resets in `playmaker quotas`.** The Codex block shows
|
|
19
|
+
`Banked resets N (M usable now)` and each reset's expiry (from
|
|
20
|
+
`wham/rate-limit-reset-credits`, same bearer token). Claude's reset is not reachable
|
|
21
|
+
with the CLI token (CodexBar reads it from the web session) and stays out.
|
|
22
|
+
- **The coach's ledger.** `scripts/ledger.py` keeps one JSON line per landed work package
|
|
23
|
+
in `~/.playmaker/ledger.jsonl` (`add`, `fix`, `escape`, `stats`, `tail`), and
|
|
24
|
+
`scripts/ledger-hook.py` — a Claude Code `PostToolUse` hook on `Bash` — writes the row
|
|
25
|
+
on every `git commit` from the WP's review directory, with an optional junior model
|
|
26
|
+
(`[ledger] junior = "codex:<model>"`) filling the soft fields. Fields, mapping and the
|
|
27
|
+
settings snippet: `references/ledger.md`.
|
|
28
|
+
- **`--risk seams` in `review-board.sh`.** A one-seat board over the joints between the
|
|
29
|
+
work packages of one fan-out, after each passed its own board and the global gate.
|
|
30
|
+
|
|
31
|
+
### Changed
|
|
32
|
+
|
|
33
|
+
- **The coach skill, 2026-09-30 revision** (`skills/playmaker-coach`): routing is
|
|
34
|
+
«fit before headroom» (class the WP, then the ledger, then the pool floors, then
|
|
35
|
+
headroom); map vs. semantic recon; an integration step with a seam review before
|
|
36
|
+
landing; fixed board columns; a first-minute check on every detached dispatch and no
|
|
37
|
+
waiting on closed lanes; policy load order is personal, then repo. Reviewer roster
|
|
38
|
+
lines are an ordered preference list — the script seats exactly 1/2/3/1 for
|
|
39
|
+
routine/normal/high/seams after skipping the implementer and holds the rest in
|
|
40
|
+
reserve; fewer than required stops it (`PM_REVIEW_ALLOW_SHORT=1` overrides).
|
|
41
|
+
- **`review-board.sh` protects its artefacts.** Previous verdicts are archived on a new
|
|
42
|
+
dispatch and ignored by `--collect` when older than the current patch; `--dry-run`
|
|
43
|
+
no longer truncates `sessions.txt`; untracked files under the review paths are listed
|
|
44
|
+
with the `git add -N` hint; `board.env` records risk, round and required count;
|
|
45
|
+
`--collect` marks `failed`/`killed` seats DEAD, parses the last object carrying the
|
|
46
|
+
contract, and prints `verdicts: k/N` (`BOARD INCOMPLETE` while short). The prompt
|
|
47
|
+
interpolates the base ref, forbids edits and scratch files inside the tree, tells a
|
|
48
|
+
seat that cannot run the gate to report it under `unverifiable`, and asks the `risk`
|
|
49
|
+
lens to falsify the acceptance criteria.
|
|
50
|
+
|
|
51
|
+
### Fixed
|
|
52
|
+
|
|
53
|
+
- **`playmaker quotas` no longer refreshes the Claude OAuth token.** The probe shared
|
|
54
|
+
the keychain entry and the single-use refresh token with the Claude Code CLI, and the
|
|
55
|
+
race logged the CLI out. The probe now only reads; when the token is expired it lets
|
|
56
|
+
the CLI refresh it (`[quotas] claude_refresh_via_cli`, default on) or shows
|
|
57
|
+
`login expired — run: claude auth login`. An outage after a successful refresh is
|
|
58
|
+
reported as an outage, not as an expired login.
|
|
59
|
+
- **The reset-credits inventory is best-effort**: a failing secondary request no
|
|
60
|
+
longer sinks the whole Codex block.
|
|
61
|
+
- **The ledger's writers take a lock and rewrites are atomic.** `ledger.py add` (and so
|
|
62
|
+
the hook) appends under an `flock`; `fix` and `escape` hold the same lock across their
|
|
63
|
+
read-modify-write and publish through a temp file and `os.replace`, so two coach sessions
|
|
64
|
+
committing at once cannot lose a row or tear the file; a trailing partial line is skipped
|
|
65
|
+
by the readers instead of stopping them. `--effort` is validated in the CLI before a
|
|
66
|
+
detached run spawns.
|
|
67
|
+
|
|
8
68
|
## [0.13.0] - 2026-09-30
|
|
9
69
|
|
|
10
70
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: playmaker-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.14.0
|
|
4
4
|
Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel.
|
|
5
5
|
Project-URL: Homepage, https://github.com/vladsafedev/playmaker
|
|
6
6
|
Project-URL: Repository, https://github.com/vladsafedev/playmaker
|
|
@@ -375,6 +375,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
|
|
|
375
375
|
`.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
|
|
376
376
|
otherwise, and playmaker would report the agent as unavailable.
|
|
377
377
|
|
|
378
|
+
**claude** also accepts an effort level — Claude Code's `--effort
|
|
379
|
+
low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
|
|
380
|
+
motivating case: running Opus review boards at `xhigh`). Set the lane default
|
|
381
|
+
once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
|
|
382
|
+
claude --effort <level>` and `playmaker continue <id> --effort <level>`
|
|
383
|
+
override it for one run without persisting anything. The flag is accepted and
|
|
384
|
+
ignored on every other lane so scripts can pass it uniformly, and when it is
|
|
385
|
+
unset everywhere no `--effort` flag is sent at all — Claude Code's own default
|
|
386
|
+
applies.
|
|
387
|
+
|
|
378
388
|
## Notifications
|
|
379
389
|
|
|
380
390
|
Every detached dispatch pings when it finishes. With
|
|
@@ -436,9 +446,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
|
|
|
436
446
|
**separate buckets**. So is the `Codex — Spark` block, every agy row, and the
|
|
437
447
|
whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
438
448
|
|
|
439
|
-
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
|
|
449
|
+
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
|
|
450
|
+
it expires, Claude Code itself refreshes it; the quota probe never rotates the
|
|
451
|
+
shared refresh token.
|
|
440
452
|
Model-scoped weekly buckets come from the usage API's `limits[]` array and
|
|
441
453
|
print as `Weekly · <model>`.
|
|
454
|
+
`[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
|
|
455
|
+
report `login expired — run: claude auth login` without spawning `claude`.
|
|
442
456
|
- **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
|
|
443
457
|
Spark model's own 5-hour and weekly windows come from
|
|
444
458
|
`additional_rate_limits[]` and print as their own `Codex — Spark` block.
|
|
@@ -348,6 +348,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
|
|
|
348
348
|
`.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
|
|
349
349
|
otherwise, and playmaker would report the agent as unavailable.
|
|
350
350
|
|
|
351
|
+
**claude** also accepts an effort level — Claude Code's `--effort
|
|
352
|
+
low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
|
|
353
|
+
motivating case: running Opus review boards at `xhigh`). Set the lane default
|
|
354
|
+
once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
|
|
355
|
+
claude --effort <level>` and `playmaker continue <id> --effort <level>`
|
|
356
|
+
override it for one run without persisting anything. The flag is accepted and
|
|
357
|
+
ignored on every other lane so scripts can pass it uniformly, and when it is
|
|
358
|
+
unset everywhere no `--effort` flag is sent at all — Claude Code's own default
|
|
359
|
+
applies.
|
|
360
|
+
|
|
351
361
|
## Notifications
|
|
352
362
|
|
|
353
363
|
Every detached dispatch pings when it finishes. With
|
|
@@ -409,9 +419,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
|
|
|
409
419
|
**separate buckets**. So is the `Codex — Spark` block, every agy row, and the
|
|
410
420
|
whole Z.ai block. Routing a subtask is choosing which of them to spend.
|
|
411
421
|
|
|
412
|
-
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
|
|
422
|
+
- **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
|
|
423
|
+
it expires, Claude Code itself refreshes it; the quota probe never rotates the
|
|
424
|
+
shared refresh token.
|
|
413
425
|
Model-scoped weekly buckets come from the usage API's `limits[]` array and
|
|
414
426
|
print as `Weekly · <model>`.
|
|
427
|
+
`[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
|
|
428
|
+
report `login expired — run: claude auth login` without spawning `claude`.
|
|
415
429
|
- **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
|
|
416
430
|
Spark model's own 5-hour and weekly windows come from
|
|
417
431
|
`additional_rate_limits[]` and print as their own `Codex — Spark` block.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "playmaker-cli"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.14.0"
|
|
4
4
|
description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
@@ -1,14 +1,13 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: playmaker-coach
|
|
3
|
-
description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode
|
|
3
|
+
description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
# playmaker-coach — you are the tech lead, not the typist
|
|
7
7
|
|
|
8
|
-
`playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode /
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
**how the review board runs**.
|
|
8
|
+
`playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / a sibling Claude,
|
|
9
|
+
tracks them, and returns their threads. This skill is the judgment on top: what to split,
|
|
10
|
+
who gets which slice, how to size it so verifying is cheap, and **how the review board runs**.
|
|
12
11
|
|
|
13
12
|
**The coach produces plans, prompts, verdicts and integration — not feature diffs.** Your context
|
|
14
13
|
window is the most expensive resource on the table. Every file you read yourself and every line you
|
|
@@ -18,9 +17,9 @@ summarization, and even the drafting of worker prompts when the task is big enou
|
|
|
18
17
|
Two loops run under your hand, always both:
|
|
19
18
|
|
|
20
19
|
```
|
|
21
|
-
decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
|
|
22
|
-
↑______________________|
|
|
23
|
-
|
|
20
|
+
decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
|
|
21
|
+
↑______________________| max 2 cycles
|
|
22
|
+
all WPs green → integrate → GLOBAL GATE → SEAM REVIEW (≥2 interacting WPs) → land → ledger row
|
|
24
23
|
```
|
|
25
24
|
|
|
26
25
|
## 1. Activation gate
|
|
@@ -48,21 +47,29 @@ outside the skill so it survives `playmaker skill install --force`. Read, in thi
|
|
|
48
47
|
later files override earlier ones:
|
|
49
48
|
|
|
50
49
|
```bash
|
|
51
|
-
cat
|
|
52
|
-
cat
|
|
50
|
+
cat ~/.playmaker/policy.md 2>/dev/null # personal policy (tiers, quota economics, lane defaults)
|
|
51
|
+
cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions) — wins
|
|
53
52
|
ls ./.playmaker/agents/*.md 2>/dev/null || ls ~/.playmaker/agents/*.md 2>/dev/null
|
|
54
53
|
```
|
|
55
54
|
|
|
56
|
-
Agent profiles describe each lane's
|
|
57
|
-
|
|
55
|
+
Agent profiles describe each lane's traps, mechanics and quota position — trust a profile over the
|
|
56
|
+
generic defaults here, never over the policy's tier table or a dated owner order: profiles say how a
|
|
57
|
+
lane behaves, policy says what it may be given. A newer dated order beats an older line anywhere. If no policy file exists, say so once in the plan and use these defaults.
|
|
58
58
|
|
|
59
59
|
## 3. Protocol
|
|
60
60
|
|
|
61
61
|
### 3.1 Recon — delegate it
|
|
62
62
|
|
|
63
|
-
Codebase exploration is the highest-leverage thing to delegate
|
|
64
|
-
|
|
65
|
-
|
|
63
|
+
Codebase exploration is the highest-leverage thing to delegate. Two kinds, two tiers:
|
|
64
|
+
|
|
65
|
+
- **Map recon** — locate files, symbols, call sites, line ranges, inventories. Raw reading: the
|
|
66
|
+
cheapest lane (Gemini Flash) does it as well as you do.
|
|
67
|
+
- **Semantic recon** — invariants, ownership, contracts, side effects, impact radius, «why is it like
|
|
68
|
+
this». Judgment: a senior lane (Kimi K3 for many-file sweeps, Gemini Pro, Codex). A cheap model
|
|
69
|
+
returns a confident wrong map, and you plan on it.
|
|
70
|
+
|
|
71
|
+
Before your own `grep`/`Read` sweep, dispatch the matching kind with an explicit deliverable and
|
|
72
|
+
`--sync`, and read a 200-word report instead of ten files:
|
|
66
73
|
|
|
67
74
|
```bash
|
|
68
75
|
playmaker dispatch agy --model <cheap-tier> --cwd "$(pwd)" --sync --read-only \
|
|
@@ -78,14 +85,17 @@ playmaker quotas # capacity per MODEL, not per agent; re-probes it
|
|
|
78
85
|
```
|
|
79
86
|
|
|
80
87
|
Read at model granularity: separate buckets inside one provider are separate capacity. The aim is
|
|
81
|
-
**
|
|
82
|
-
|
|
88
|
+
**accepted work per point of scarce quota**, with prepaid capacity spent before it resets.
|
|
89
|
+
Level-loading — every pool drawn down evenly by week's end, except the coach's own — is the
|
|
90
|
+
tie-breaker between lanes that fit a WP equally, never a reason to queue a WP behind a closed lane. Details and per-provider quirks:
|
|
83
91
|
`references/quotas.md`.
|
|
84
92
|
|
|
85
93
|
### 3.3 Decompose into work packages
|
|
86
94
|
|
|
87
|
-
2–5 WPs on the first pass.
|
|
88
|
-
|
|
95
|
+
2–5 WPs on the first pass. Name each WP's **class** first — `mechanical | feature | terminal-heavy |
|
|
96
|
+
repo-recon | architecture | high-risk` — it decides which senior lanes are eligible (policy: fit before
|
|
97
|
+
headroom), which lenses the board gets, and it is the first field of the ledger row. Then a WP is
|
|
98
|
+
dispatchable only when all five hold — this is what makes review cheap and re-prompting rare:
|
|
89
99
|
|
|
90
100
|
1. **Hard file boundary.** "Edit only `x.ts` and its spec" — never "do the backend part".
|
|
91
101
|
2. **A gate the worker runs itself** — `tsc --noEmit`, a named spec file, a lint pass. It must exit 0
|
|
@@ -105,7 +115,7 @@ policy first, then `references/lanes.md`.
|
|
|
105
115
|
|
|
106
116
|
### 3.4 Propose, then wait
|
|
107
117
|
|
|
108
|
-
Post the plan as: WP → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
|
|
118
|
+
Post the plan as: WP → class → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
|
|
109
119
|
the current per-model capacity. **Dispatch nothing until the user approves.** Approval may be
|
|
110
120
|
partial ("go but reroute tests"); restate the modified plan in one line, then dispatch.
|
|
111
121
|
|
|
@@ -119,19 +129,27 @@ playmaker dispatch <agent> --model <name> --batch "$B" --cwd "$(pwd)" --prompt "
|
|
|
119
129
|
Detached by default — that is the point; never `--sync` a whole fan-out. Always pass `--cwd`,
|
|
120
130
|
`--batch` (one summary ping for the batch), and `--model` unless the profile says otherwise.
|
|
121
131
|
Prompt shape: `references/prompt-templates.md`. Per-agent traps (agy scratch dir, opencode relative
|
|
122
|
-
paths, codex model roster
|
|
132
|
+
paths, codex model roster): `references/agent-gotchas.md`.
|
|
123
133
|
|
|
124
134
|
Parallel WPs that touch the same files go in **git worktrees**, one per WP, or they will collide.
|
|
125
135
|
|
|
136
|
+
Within a minute of every detached dispatch, `playmaker get <id> --json`: a quota or auth refusal
|
|
137
|
+
finishes in seconds as `failed`, pings nobody, and looks exactly like a slow worker. Give every
|
|
138
|
+
dispatch a deadline (recon 15 minutes; a WP by its size) and on expiry check the process is alive
|
|
139
|
+
(`playmaker logs <id>`, its RSS) before waiting longer. Never wait for a closed lane.
|
|
140
|
+
|
|
126
141
|
### 3.6 Prove it on disk before you believe it
|
|
127
142
|
|
|
128
143
|
`done` means "the process exited cleanly with text", not "code changed". playmaker ≥0.9 marks a
|
|
129
144
|
zero-change write task `no_changes` — treat it exactly as a failure. On older builds, check yourself:
|
|
130
145
|
|
|
131
146
|
```bash
|
|
132
|
-
git -C "<cwd>" status --short
|
|
147
|
+
git -C "<cwd>" status --short --untracked-files=all # empty tree on a write WP = NOT done
|
|
133
148
|
```
|
|
134
149
|
|
|
150
|
+
New files are invisible to `git diff`: `git add -N <new files>` before the board, and compare
|
|
151
|
+
`git diff --stat` with what the worker claims it changed.
|
|
152
|
+
|
|
135
153
|
Then run the WP's own gate yourself, once, cheaply. A WP that fails its gate never reaches the
|
|
136
154
|
review board — it goes straight back to the worker.
|
|
137
155
|
|
|
@@ -157,10 +175,18 @@ Non-negotiables:
|
|
|
157
175
|
- Every finding carries `file:line` + a concrete failure scenario. **No evidence → dropped.** You
|
|
158
176
|
arbitrate, and a reviewer's confidence is an input, not a verdict.
|
|
159
177
|
- Fixes go back via `playmaker continue <impl-id>` as a numbered list — the worker still has its
|
|
160
|
-
context. Re-review runs on the **delta only
|
|
178
|
+
context. Re-review runs on the **delta only** — which means passing the round-1 state as the
|
|
179
|
+
base-ref (a wip commit in the worktree, squashed before landing); with the original base the
|
|
180
|
+
patch is cumulative and reviewers re-open settled code.
|
|
181
|
+
- A ruling that changes the scope goes into `spec.md` before the next round — reviewers refute
|
|
182
|
+
against the spec, and a stale spec produces false blockers. Before charging a finding to the
|
|
183
|
+
worker, check it is not already present on the base ref.
|
|
161
184
|
- **Two cycles maximum.** Still blocking after two → stop and escalate to the user with both
|
|
162
185
|
verdicts. A third round at the same lane is the most expensive way to use a cheap model.
|
|
163
186
|
- Land only at **zero blocking findings**.
|
|
187
|
+
- On `high`, the `risk` seat is told to **falsify** the acceptance criteria — a command, input or repro
|
|
188
|
+
per criterion it attacked, not only an argument; its `pass` lists what it tried. A seat that cannot
|
|
189
|
+
execute (the `claude` lane runs with binaries forbidden) never holds `correctness` — give it `contracts`.
|
|
164
190
|
|
|
165
191
|
### 3.8 Keep a board file
|
|
166
192
|
|
|
@@ -173,11 +199,39 @@ Long fan-outs outlive your context. Maintain `./.playmaker/board.md` — one row
|
|
|
173
199
|
Update it at dispatch, at gate, after each review round. On resume, read the board before anything
|
|
174
200
|
else. It is also what you paste back to the user as the status report.
|
|
175
201
|
|
|
202
|
+
The columns are fixed — one layout, not one per board: a resumed session and the ledger both parse
|
|
203
|
+
them. At landing, the ledger hook writes the WP's row to the cross-repo ledger on `git commit`; read its
|
|
204
|
+
`[ledger]` line and correct the soft fields with `python3 ~/.playmaker/scripts/ledger.py fix wp=… k=v`.
|
|
205
|
+
When the hook matched nothing, append the row by hand: `python3 ~/.playmaker/scripts/ledger.py add
|
|
206
|
+
wp=… class=… impl_lane=… …` (fields: `~/.playmaker/policy.md`, «Ledger»). It keeps what board prose loses: class, first-gate pass, blocking findings that survived
|
|
207
|
+
adjudication, cycles, wall time — the evidence the routing step reads.
|
|
208
|
+
|
|
176
209
|
### 3.9 Failures
|
|
177
210
|
|
|
178
|
-
Surface them; never silently retry. A failed dispatch (missing binary, bad auth,
|
|
179
|
-
|
|
180
|
-
|
|
211
|
+
Surface them; never silently retry — and never wait. A failed dispatch (missing binary, bad auth,
|
|
212
|
+
rejected model, quota) is neither a fix cycle nor a worker failure: re-route the WP to the next live
|
|
213
|
+
senior lane now, with the full context in a fresh prompt, and record the substitution in the board.
|
|
214
|
+
Ask the user only when no eligible lane is left. Diagnosis per agent: `references/agent-gotchas.md`.
|
|
215
|
+
|
|
216
|
+
### 3.10 Integrate before landing — when the fan-out had two or more WPs
|
|
217
|
+
|
|
218
|
+
Two green WPs can be wrong together. Per-WP gates and boards prove the parts; nothing above proves the
|
|
219
|
+
whole. Once every WP is green:
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
integrate on one tree (worktrees merged in dependency order)
|
|
223
|
+
→ GLOBAL GATE — the repo's minimum bar plus the union of the WPs' own gates, run by you
|
|
224
|
+
→ if ≥2 WPs touched interacting modules:
|
|
225
|
+
SEAM REVIEW — pm-review <batch>-seams <base-before-first-WP> --risk seams --impl-agent <main lane>
|
|
226
|
+
→ commit, one WP per commit, in integration order → ledger rows
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
"Interacting" = shared types or packages, the same service or screen, a migration and its consumers,
|
|
230
|
+
event or config names, shared state. The seam reviewer gets a spec that lists the seams and reads only
|
|
231
|
+
those: changed interfaces, duplicated or conflicting assumptions, incompatible types, migration order,
|
|
232
|
+
naming, config, shared state. The packages' internals already passed their own boards — re-reviewing
|
|
233
|
+
them is the third round the protocol forbids. Roster: the `seams` block in `reviewers.conf`;
|
|
234
|
+
`--impl-agent` is the lane that implemented most of the fan-out.
|
|
181
235
|
|
|
182
236
|
## 4. What the coach may still type by hand
|
|
183
237
|
|
|
@@ -210,3 +264,6 @@ editor on product code, ask whether that is a WP you failed to write.
|
|
|
210
264
|
- **Dispatching a WP you cannot verify in one command and one paragraph.**
|
|
211
265
|
- **Omitting `--model`** and letting a CLI default drain a top-tier bucket.
|
|
212
266
|
- **Reading whole agent threads** when `summary` answers the question.
|
|
267
|
+
- **Landing two green WPs without running the tree they make together.** Per-WP gates prove the
|
|
268
|
+
parts; the global gate and the seam review prove the whole.
|
|
269
|
+
- **Routing by headroom alone.** Class first, then the ledger's floor, then the pool floors, then headroom.
|
{playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md
RENAMED
|
@@ -65,39 +65,6 @@ line from `agy models` / `opencode models` rather than typing it.
|
|
|
65
65
|
- Neither trap applies to **review** dispatches, which write nothing — which makes opencode a
|
|
66
66
|
perfectly good reviewer even where it is a shaky implementer.
|
|
67
67
|
|
|
68
|
-
## kimi (Kimi Code CLI)
|
|
69
|
-
|
|
70
|
-
- There is **no read-only mode below the prompt**: `-p` refuses `--auto`, `--yolo` and `--plan`
|
|
71
|
-
(exit 1) and already runs with auto-approval. Only dispatch work you would run unattended anyway.
|
|
72
|
-
- The session id arrives **only in the trailing `session.resume_hint` line**, so `playmaker list`
|
|
73
|
-
shows the agent session late — do not conclude a dispatch failed just because the id has not
|
|
74
|
-
appeared yet.
|
|
75
|
-
- Sessions are **per-cwd**: `kimi session list` from another directory shows nothing. Track the
|
|
76
|
-
session through playmaker, not through the CLI.
|
|
77
|
-
- The stream carries **no token/cost fields** — there is nothing to budget against mid-run.
|
|
78
|
-
- Exit codes: **exit 1 is non-retryable** (auth, quota, unknown model — "is not configured in
|
|
79
|
-
config.toml"); **exit 75 is retryable**. Fix the cause on 1, re-dispatch on 75.
|
|
80
|
-
- K3 is **slow on real tickets** — never `--sync` a big WP; dispatch detached and poll.
|
|
81
|
-
- Needs **Node ≥ 22.19** — hence the wrapper binary; point `[agents.kimi] binary` at it.
|
|
82
|
-
- **Always pass `-m kimi-code/k3-256k`** — the CLI default is the weaker K2.7 `kimi-for-coding`.
|
|
83
|
-
|
|
84
|
-
## muse (Muse Code CLI)
|
|
85
|
-
|
|
86
|
-
- The default run keeps Muse's **sandbox**: approval off, workspace trusted, writes inside the
|
|
87
|
-
repo fine — but writes **outside** it are denied, so `uv run pytest` fails initializing
|
|
88
|
-
`~/.cache/uv` (npm/pnpm caches alike). A WP whose gate needs those caches needs
|
|
89
|
-
`[agents.muse] yolo = true`.
|
|
90
|
-
- **Without workspace trust Muse ignores the repo's AGENTS.md.** playmaker passes
|
|
91
|
-
`--trust-workspace` by default; `trust_workspace = false` drops it, and repo rules stop
|
|
92
|
-
applying.
|
|
93
|
-
- Exit codes: **0** turn completed; **1** run failed (bad model, auth) — the reason is in the
|
|
94
|
-
terminal event and stderr's last line; **2** usage error; **130/143** SIGINT/SIGTERM.
|
|
95
|
-
- **Resume only from the same cwd** — a `--session-id` elsewhere would need
|
|
96
|
-
`--allow-workspace-switch`, which playmaker never passes.
|
|
97
|
-
- stderr **always** carries informational lines such as `muse: workspace root: …`, even on
|
|
98
|
-
success — they are not errors.
|
|
99
|
-
- The stream has **no cost fields** — token usage exists only in the session log on disk.
|
|
100
|
-
|
|
101
68
|
## Worktrees
|
|
102
69
|
|
|
103
70
|
Parallel WPs that touch the same files collide. Give each its own git worktree and dispatch with
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
> **Opus 5 lane — OUT of the review board (owner order 2026-09-08):** `playmaker dispatch claude -m opus` draws the Anthropic all-models weekly — the bucket the coach session falls back to (as Opus) when the Fable weekly runs out — and was eating it fast. It no longer holds the `high` risk seat by default (that was the 2026-09-02 order); `codex` does. Opus is weighted by the Fable weekly instead: while `Weekly · Fable` ≥ 50% it fills the risk seat on `high` boards that Codex implemented, ¼ of that at 25–50%, nothing below 25% or when the all-models weekly is under 50%. Never for implementation or recon, never on `normal`. Table and flip: `~/.playmaker/policy.md`, `~/.playmaker/reviewers.conf`.
|
|
2
|
+
|
|
3
|
+
> **GEMINI ONLY on agy (owner order 2026-09-02):** never dispatch `claude-*` or `gpt-oss-*` models on the `agy` lane. Want a Claude? Don't dispatch one: the coach session is the Claude, and the Anthropic bucket is out of the board (2026-09-08). See `~/.playmaker/policy.md`.
|
|
4
|
+
|
|
5
|
+
# Execution lanes — where work runs
|
|
6
|
+
|
|
7
|
+
All Claude work — coach, in-session sub-agents, external `claude -p` — draws from the **same Claude
|
|
8
|
+
subscription**. So routing is about *which bucket* you spend and *where the result lands*, not who
|
|
9
|
+
pays. Four lanes:
|
|
10
|
+
|
|
11
|
+
1. **Coach (this thread).** Interactive Claude Code on the top-tier weekly bucket — the scarcest,
|
|
12
|
+
hardest to replenish. Serial, and the most expensive in *context*. Reserve for orchestration,
|
|
13
|
+
architecture judgment, adjudication, final integration.
|
|
14
|
+
|
|
15
|
+
2. **In-session sub-agents (the Task/Agent tool).** Run inside this session, write files (they
|
|
16
|
+
inherit the coach's permission mode), return straight into the coach's context, run in parallel.
|
|
17
|
+
Same subscription. Use when the coach folds the result in directly.
|
|
18
|
+
|
|
19
|
+
3. **External dispatch — `playmaker dispatch claude`.** A tracked, detached stream you can monitor
|
|
20
|
+
and `continue` independently. Same subscription, and since September 2026 the same bucket: Sonnet has no
|
|
21
|
+
separate weekly (`seven_day_sonnet: null`), so this lane spends the coach's own all-models weekly.
|
|
22
|
+
Default `--model sonnet` (`haiku` for trivial mechanical work), and only when policy allows it. playmaker runs it in `acceptEdits`: it writes freely
|
|
23
|
+
inside `--cwd` and is refused outside it, so keep every path in the prompt inside `--cwd`.
|
|
24
|
+
|
|
25
|
+
4. **External dispatch — `codex` / `agy` / `opencode`.** Each on its own subscription or plan — the
|
|
26
|
+
home for write-heavy parallel implementation that can leave the Anthropic subscription.
|
|
27
|
+
- **`agy` (Antigravity)** carries more than Google models: alongside Gemini Flash and Pro tiers it
|
|
28
|
+
serves **Claude Sonnet/Opus (Thinking)** and a GPT-OSS tier. Its Claude runs on *Google's* pool —
|
|
29
|
+
capable judgment that spends none of the Anthropic bucket, but a generation behind the
|
|
30
|
+
frontier: treat the whole lane as **middle** (see Model tiers below). Note the internal
|
|
31
|
+
split: all Gemini models share one bucket, Claude and GPT-OSS share another.
|
|
32
|
+
- **`opencode`** is the widest lane: one CLI over ~75 providers addressed as `provider/model` — a
|
|
33
|
+
GLM coding plan, or a model running locally on this machine, which spends no subscription quota
|
|
34
|
+
at all.
|
|
35
|
+
|
|
36
|
+
**Never write an agy or opencode model name from memory** — run `agy models` / `opencode models` and
|
|
37
|
+
copy a line. Both rosters and their spelling move with releases, and playmaker validates `--model`
|
|
38
|
+
against the live roster, failing the dispatch on a stale name.
|
|
39
|
+
|
|
40
|
+
## Routing cheat-sheet
|
|
41
|
+
|
|
42
|
+
| The work… | Lane | Why |
|
|
43
|
+
|---|---|---|
|
|
44
|
+
| writes files, coach integrates the result directly | in-session sub-agent | write-capable, returns into context |
|
|
45
|
+
| is an independent stream to monitor separately | `dispatch claude --model sonnet` | tracked, detached — but on the coach's own subscription (Sonnet has no separate bucket): policy says when, and it is rarely |
|
|
46
|
+
| is heavy reasoning only the coach can do | coach | top tier, serial |
|
|
47
|
+
| is write-heavy and can leave Claude | codex / agy / opencode | their own quotas |
|
|
48
|
+
| wants Claude-flavoured judgment without spending the Anthropic bucket | `dispatch agy --model <claude-opus-thinking>` | middle tier — a generation behind; never the only senior eye |
|
|
49
|
+
| is bulk work with every subscription low | `dispatch opencode --model <plan>/<model>` | a separate plan, untouched by the others |
|
|
50
|
+
| is mechanical and privacy-sensitive, or all quotas spent | `dispatch opencode --model <local>/<model>` | runs on this machine, costs wall-clock only |
|
|
51
|
+
|
|
52
|
+
## Model tiers — senior / middle / junior
|
|
53
|
+
|
|
54
|
+
> **Superseded on this machine by the tier table in `~/.playmaker/policy.md` (owner order
|
|
55
|
+
> 2026-09-04): there is no middle tier there, and `agy` Claude models are junior. The rows below are
|
|
56
|
+
> the skill's generic defaults for a machine without a policy file.
|
|
57
|
+
|
|
58
|
+
**Judge a lane by the model actually serving it, never by the lane's name.** Model rosters move
|
|
59
|
+
faster than this file; re-derive the table whenever a lane surprises you, and correct it here.
|
|
60
|
+
|
|
61
|
+
| Tier | What is actually there | Give it |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| **Senior** | the coach's own session (Fable 5.1, falling back to Opus 5.5); **`codex`** — the real Codex CLI on the ChatGPT plan (`--model` omitted; verify with `codex --version`); **`opencode` / GLM** (`zai-coding-plan/glm-*`) | architecture, spec interpretation, cross-module integration, adjudication, anything irreversible |
|
|
64
|
+
| **Near-senior** | **`agy gemini-3.1-pro-high`** — not quite the three above, but close enough to carry a review lens or a hard WP on its own | the always-on review seat, demanding implementation, deep recon |
|
|
65
|
+
| **Middle** | the rest of `agy` / Antigravity — its **Claude 4.6 / Sonnet 4.6** are a generation behind the frontier despite the name — plus GPT-OSS | well-specified implementation against a ready plan, refactors, CRUD by convention, a second opinion |
|
|
66
|
+
| **Reserve** | `dispatch claude --model sonnet` / `--model opus` — Sonnet 5.5 and Opus 5.5, senior capability on the coach's own subscription | nothing by default: tier is capability, allocation is the policy's call — a seat only on the owner's word |
|
|
67
|
+
| **Junior** | Gemini **Flash** tiers, Codex Spark | mass reading, codebase sweeps, mechanical diffs, tight fix→check loops |
|
|
68
|
+
|
|
69
|
+
Two traps this table exists to prevent — both have bitten in practice:
|
|
70
|
+
|
|
71
|
+
1. **A lane serving "Claude Opus" is not necessarily the current Opus.** Antigravity's Claude
|
|
72
|
+
models trail the frontier by a full generation. Cheap, capable, worth using — but *middle*,
|
|
73
|
+
not senior. Their verdicts are an input you weigh, not an authority you defer to.
|
|
74
|
+
Conversely, do **not** file GLM as cheap because it is cheap: it is a senior lane.
|
|
75
|
+
2. **`codex` is the Codex CLI, not a Google model.** The lane name is not evidence of the engine.
|
|
76
|
+
|
|
77
|
+
**On irreversible work** — deletions, migrations, money paths, auth, public contracts — a middle
|
|
78
|
+
lane may never be the only senior-grade eye. Pair it with `codex`, or adjudicate that hunk yourself.
|
|
79
|
+
|
|
80
|
+
Match the WP to the tier, then reserve the most-depleted senior model for the lightest role —
|
|
81
|
+
usually the coach's own adjudication — and push volume down to junior lanes.
|
|
82
|
+
|
|
83
|
+
## Junior fan-out — where it is safe
|
|
84
|
+
|
|
85
|
+
Route down by default when **all** of these hold:
|
|
86
|
+
|
|
87
|
+
1. **Self-verifiable outcome** — a green command the worker runs itself. An objective gate replaces
|
|
88
|
+
senior judgment.
|
|
89
|
+
2. **Tight file boundary plus a pattern to mirror** — CRUD by existing convention, a test scaffolded
|
|
90
|
+
from a neighbouring spec, a story or doc from a template.
|
|
91
|
+
3. **Low blast radius** — one WP, dark or flagged code, no schema or contract change; a bad diff is
|
|
92
|
+
cheap to throw away.
|
|
93
|
+
4. **Read-only by nature** — recon, sweeps, summarization, fact-check tables. Zero risk beyond
|
|
94
|
+
wasted tokens.
|
|
95
|
+
|
|
96
|
+
## Where juniors are never safe
|
|
97
|
+
|
|
98
|
+
Regardless of how well specified — these stay with a senior lane or the coach:
|
|
99
|
+
|
|
100
|
+
- **Money paths**: billing, entitlement ledgers, IAP/Stripe webhooks, refunds, idempotency.
|
|
101
|
+
- **Schema migrations, backfills, destructive data operations.**
|
|
102
|
+
- **Auth, session, and security-sensitive code.**
|
|
103
|
+
- **Frozen public API contracts** and whatever the repo's policy marks human-review-required.
|
|
104
|
+
- **Anything without a one-sentence done-condition.** Juniors do not resolve spec ambiguity, they
|
|
105
|
+
amplify it.
|
|
106
|
+
|
|
107
|
+
A junior diff that touches this list is a routing bug: pull the WP back, do not patch it in review.
|
|
108
|
+
|
|
109
|
+
## Escalation
|
|
110
|
+
|
|
111
|
+
A worker that fails its own gate **twice** escalates one tier — never a third attempt at the same
|
|
112
|
+
tier. Two failures at the top tier means the WP is wrong, not the worker: re-scope it.
|
|
113
|
+
|
|
114
|
+
## Two-stage senior review
|
|
115
|
+
|
|
116
|
+
Juniors produce the fact-check and consistency reports; seniors render verdicts *on top of those
|
|
117
|
+
reports* rather than re-reading the world. This is the cheapest way to buy senior judgment.
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# The ledger and its hook
|
|
2
|
+
|
|
3
|
+
`~/.playmaker/ledger.jsonl` holds one JSON object per work package — the cross-repo evidence the
|
|
4
|
+
routing step reads. `scripts/ledger.py` maintains it (`add` / `fix` / `escape` / `stats` / `tail`);
|
|
5
|
+
it honours `PM_LEDGER` for tests. Rows whose note starts with "backfill" are historical — never
|
|
6
|
+
rewrite them.
|
|
7
|
+
|
|
8
|
+
## Fields
|
|
9
|
+
|
|
10
|
+
`date`, `repo`, `wp`, `class`, `risk`, `impl_lane`, `impl_model`, `gate_first_pass`, `reviewers`
|
|
11
|
+
(list), `blocking_r1`, `blocking_accepted`, `blocking_rejected`, `cycles`, `rounds`, `wall_min`,
|
|
12
|
+
`quota_note`, `outcome`, `commit`, `note`, `escaped_from` (list). Unknown fields are `-`; ints are
|
|
13
|
+
ints or `-`.
|
|
14
|
+
|
|
15
|
+
## How the hook maps a review dir to a row
|
|
16
|
+
|
|
17
|
+
On every `git commit` the PostToolUse hook (`scripts/ledger-hook.py`) matches the commit to the WP
|
|
18
|
+
under `.playmaker/reviews/` whose `diff-r*.patch` file list overlaps the commit's files the most
|
|
19
|
+
(ties → newest patch; dirs untouched for 14+ days are ignored), then fills:
|
|
20
|
+
|
|
21
|
+
| Field | Source |
|
|
22
|
+
|---|---|
|
|
23
|
+
| `date`, `repo`, `wp`, `commit`, `outcome` | today, toplevel basename, review-dir name, short sha, `landed` |
|
|
24
|
+
| `class`, `impl_lane`, `impl_model`, `risk` | spec.md header `class:` / `impl: <lane> <model>` / `risk:` (`routine|normal|high|seams`; see review-board.md); risk may also come from a pre-header spec title ("… risk high") or `board.env`; else `-` |
|
|
25
|
+
| `reviewers` | union of lanes over `sessions.txt` and every archived `sessions.txt`, short form (`kimi-k3`, `glm-5.3`, `agy-gemini-pro`, `codex`, `muse`, `claude-opus`) |
|
|
26
|
+
| `rounds` | number of `diff-r*.patch` |
|
|
27
|
+
| `cycles` | number of `fix-r<N>.md` with N ≥ 1 |
|
|
28
|
+
| `gate_first_pass` | `n` if `fix-r0.md` exists (the first gate did not pass), else `-` |
|
|
29
|
+
| `blocking_r1` | blocking findings across round-1 verdicts (oldest archive when rounds > 1), deduped by `file:line` |
|
|
30
|
+
| `blocking_accepted`, `blocking_rejected` | `-` (soft — the junior or the coach fills them) |
|
|
31
|
+
| `wall_min` | minutes from spec.md mtime to the commit |
|
|
32
|
+
| `note` | `auto-hook` plus the junior outcome |
|
|
33
|
+
|
|
34
|
+
A commit that matches nothing gets a one-line hint (`ledger.py add wp=… class=… impl_lane=…`); a
|
|
35
|
+
commit whose `(wp, commit)` row already exists is skipped. Every run is logged to
|
|
36
|
+
`~/.playmaker/logs/ledger-hook.log`. The hook never blocks or fails the commit.
|
|
37
|
+
|
|
38
|
+
## The junior
|
|
39
|
+
|
|
40
|
+
A junior model fills the fields the mechanics left as `-`; it never blanks or overrides a set value.
|
|
41
|
+
It runs as `codex exec -m gpt-5.3-codex-spark -s read-only -C <review dir> --skip-git-repo-check
|
|
42
|
+
--output-last-message <tmp>` with a 60 s timeout, reading spec.md, all verdicts, the fix lists,
|
|
43
|
+
board.env and the board.md line, and answering one JSON object. Configure it in
|
|
44
|
+
`~/.playmaker/config.toml` (or turn it off with `"none"`):
|
|
45
|
+
|
|
46
|
+
```toml
|
|
47
|
+
[ledger]
|
|
48
|
+
junior = "codex:gpt-5.3-codex-spark"
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Any failure, timeout or non-JSON keeps the mechanical values; the reason lands in `note`.
|
|
52
|
+
|
|
53
|
+
## Fixing a row
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
python3 ~/.playmaker/scripts/ledger.py fix wp=<wp> [repo=<r>] [commit=<sha>] k=v …
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
updates the LAST matching row in place, validated like `add`. When the hook matched nothing, append
|
|
60
|
+
the row by hand with `ledger.py add`.
|
|
61
|
+
|
|
62
|
+
## Registering the hook (the coach edits ~/.claude/settings.json)
|
|
63
|
+
|
|
64
|
+
```json
|
|
65
|
+
{
|
|
66
|
+
"hooks": {
|
|
67
|
+
"PostToolUse": [
|
|
68
|
+
{
|
|
69
|
+
"matcher": "Bash",
|
|
70
|
+
"hooks": [
|
|
71
|
+
{"type": "command", "command": "python3 <installed skill dir>/scripts/ledger-hook.py", "timeout": 120}
|
|
72
|
+
]
|
|
73
|
+
}
|
|
74
|
+
]
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
```
|
{playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/prompt-templates.md
RENAMED
|
@@ -8,20 +8,25 @@ scope, gate, done-condition, context.
|
|
|
8
8
|
```
|
|
9
9
|
<one sentence: what to build and why it exists>
|
|
10
10
|
|
|
11
|
+
Working directory is <cwd>. Every path below is relative to it — never absolute, never a bare filename.
|
|
12
|
+
|
|
11
13
|
SCOPE — edit ONLY these files:
|
|
12
14
|
<path/a.ts>
|
|
13
15
|
<path/b.spec.ts>
|
|
14
16
|
Do not touch anything else. If the change appears to require another file, stop and say so in your
|
|
15
|
-
final answer instead of editing it.
|
|
17
|
+
final answer instead of editing it. Never commit. Never install or update dependencies — if the task
|
|
18
|
+
seems to need one, stop and report.
|
|
16
19
|
|
|
17
20
|
CONTEXT you need (do not go looking for more):
|
|
18
21
|
- Mirror the pattern in <path/neighbour.ts>.
|
|
19
22
|
- Convention: <naming / error handling / logging rule>.
|
|
20
23
|
- Spec: <the two or three sentences that actually constrain this>.
|
|
21
24
|
|
|
22
|
-
ACCEPTANCE — run this yourself and
|
|
25
|
+
ACCEPTANCE — run this yourself and iterate until it exits 0, at most two failed attempts; after the
|
|
26
|
+
second failure stop and report both attempts with their last output:
|
|
23
27
|
<cmd, e.g. npx tsc --noEmit -p <tsconfig> && npx eslint <paths>>
|
|
24
|
-
Paste the
|
|
28
|
+
Paste the last ~30 lines of that command's output, its exit code, and the test and file counts it
|
|
29
|
+
printed.
|
|
25
30
|
|
|
26
31
|
DONE when: <one sentence the coach can confirm in seconds>.
|
|
27
32
|
|
|
@@ -31,6 +36,11 @@ prompt did not specify.
|
|
|
31
36
|
|
|
32
37
|
## Reviewer
|
|
33
38
|
|
|
39
|
+
Generated by `scripts/review-board.sh` — do not hand-write it; the two versions used to drift
|
|
40
|
+
(patch path, git rules, round wording). To seat a reviewer by hand — a substitute for a dead lane —
|
|
41
|
+
copy `.playmaker/reviews/<wp>/prompt-<lane>-<lens>.md`, dispatch it `--read-only` on the new lane,
|
|
42
|
+
and append `<id>|<lane>|<model>|<lens>` to `sessions.txt`. The shape, for reference:
|
|
43
|
+
|
|
34
44
|
```
|
|
35
45
|
You are reviewing one work package. You did NOT write it. Your job is to REFUTE it, not to
|
|
36
46
|
summarize it. Change nothing — this is a read-only review.
|
|
@@ -78,7 +88,8 @@ Reply with a one-line note per item saying what you changed.
|
|
|
78
88
|
## Recon (read-only)
|
|
79
89
|
|
|
80
90
|
```
|
|
81
|
-
Recon only — change no files.
|
|
91
|
+
Recon only — change no files. Working directory is <cwd>. One session: no sub-agents of your own.
|
|
92
|
+
Report as your final message, write nothing to disk. Budget: 15 minutes.
|
|
82
93
|
|
|
83
94
|
Find, in <dir>:
|
|
84
95
|
(a) <thing>
|