playmaker-cli 0.13.0__tar.gz → 0.14.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/CHANGELOG.md +60 -0
  2. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/PKG-INFO +16 -2
  3. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/README.md +15 -1
  4. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/pyproject.toml +1 -1
  5. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/SKILL.md +83 -26
  6. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/agent-gotchas.md +0 -33
  7. playmaker_cli-0.14.0/skills/playmaker-coach/references/lanes.md +117 -0
  8. playmaker_cli-0.14.0/skills/playmaker-coach/references/ledger.md +77 -0
  9. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/prompt-templates.md +15 -4
  10. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/quotas.md +16 -18
  11. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/review-board.md +45 -8
  12. playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger-hook.py +571 -0
  13. playmaker_cli-0.14.0/skills/playmaker-coach/scripts/ledger.py +264 -0
  14. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/scripts/review-board.sh +89 -27
  15. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/claude.py +40 -0
  16. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/cli.py +46 -0
  17. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/quotas.py +98 -58
  18. playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_inventory.json +7 -0
  19. playmaker_cli-0.14.0/tests/fixtures/codex_banked_resets_usage.json +19 -0
  20. playmaker_cli-0.14.0/tests/test_claude_effort.py +294 -0
  21. playmaker_cli-0.14.0/tests/test_ledger_concurrency.py +112 -0
  22. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_claude.py +120 -0
  23. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_codex.py +84 -0
  24. playmaker_cli-0.13.0/skills/playmaker-coach/references/lanes.md +0 -104
  25. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/.gitignore +0 -0
  26. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/LICENSE +0 -0
  27. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/skills/playmaker-coach/references/commands.md +0 -0
  28. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/__init__.py +0 -0
  29. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/__main__.py +0 -0
  30. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/__init__.py +0 -0
  31. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/agy.py +0 -0
  32. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/base.py +0 -0
  33. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/codex.py +0 -0
  34. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/gemini.py +0 -0
  35. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/kimi.py +0 -0
  36. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/muse.py +0 -0
  37. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/agents/opencode.py +0 -0
  38. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/config.py +0 -0
  39. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/notify.py +0 -0
  40. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/registry.py +0 -0
  41. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/state.py +0 -0
  42. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/src/playmaker/watcher.py +0 -0
  43. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/__init__.py +0 -0
  44. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/kimi_wire.jsonl +0 -0
  45. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/muse_exec.jsonl +0 -0
  46. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/fixtures/muse_session.jsonl +0 -0
  47. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_agy.py +0 -0
  48. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_batch.py +0 -0
  49. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_binary.py +0 -0
  50. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_claude.py +0 -0
  51. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_codex.py +0 -0
  52. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_kimi.py +0 -0
  53. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_muse.py +0 -0
  54. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_no_changes.py +0 -0
  55. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_opencode.py +0 -0
  56. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_permissions.py +0 -0
  57. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_antigravity.py +0 -0
  58. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_freshness.py +0 -0
  59. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_kimi.py +0 -0
  60. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_quotas_zai.py +0 -0
  61. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_registry.py +0 -0
  62. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_skill.py +0 -0
  63. {playmaker_cli-0.13.0 → playmaker_cli-0.14.0}/tests/test_state.py +0 -0
@@ -5,6 +5,66 @@ All notable changes to this project will be documented in this file.
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [0.14.0] - 2026-09-30
9
+
10
+ ### Added
11
+
12
+ - **`--effort` for the `claude` lane.** `[agents.claude] effort = "xhigh"` or
13
+ `playmaker dispatch|continue … --effort <low|medium|high|xhigh|max>` is forwarded to
14
+ Claude Code's own `--effort`, right after `--model`, on dispatch and resume. The flag
15
+ overrides the config for one run (carried through `PLAYMAKER_CLAUDE_EFFORT`, never
16
+ persisted); an invalid value fails before any process starts; other lanes accept and
17
+ ignore it with a note.
18
+ - **Banked limit resets in `playmaker quotas`.** The Codex block shows
19
+ `Banked resets N (M usable now)` and each reset's expiry (from
20
+ `wham/rate-limit-reset-credits`, same bearer token). Claude's reset is not reachable
21
+ with the CLI token (CodexBar reads it from the web session) and stays out.
22
+ - **The coach's ledger.** `scripts/ledger.py` keeps one JSON line per landed work package
23
+ in `~/.playmaker/ledger.jsonl` (`add`, `fix`, `escape`, `stats`, `tail`), and
24
+ `scripts/ledger-hook.py` — a Claude Code `PostToolUse` hook on `Bash` — writes the row
25
+ on every `git commit` from the WP's review directory, with an optional junior model
26
+ (`[ledger] junior = "codex:<model>"`) filling the soft fields. Fields, mapping and the
27
+ settings snippet: `references/ledger.md`.
28
+ - **`--risk seams` in `review-board.sh`.** A one-seat board over the joints between the
29
+ work packages of one fan-out, after each passed its own board and the global gate.
30
+
31
+ ### Changed
32
+
33
+ - **The coach skill, 2026-09-30 revision** (`skills/playmaker-coach`): routing is
34
+ «fit before headroom» (class the WP, then the ledger, then the pool floors, then
35
+ headroom); map vs. semantic recon; an integration step with a seam review before
36
+ landing; fixed board columns; a first-minute check on every detached dispatch and no
37
+ waiting on closed lanes; policy load order is personal, then repo. Reviewer roster
38
+ lines are an ordered preference list — the script seats exactly 1/2/3/1 for
39
+ routine/normal/high/seams after skipping the implementer and holds the rest in
40
+ reserve; fewer than required stops it (`PM_REVIEW_ALLOW_SHORT=1` overrides).
41
+ - **`review-board.sh` protects its artefacts.** Previous verdicts are archived on a new
42
+ dispatch and ignored by `--collect` when older than the current patch; `--dry-run`
43
+ no longer truncates `sessions.txt`; untracked files under the review paths are listed
44
+ with the `git add -N` hint; `board.env` records risk, round and required count;
45
+ `--collect` marks `failed`/`killed` seats DEAD, parses the last object carrying the
46
+ contract, and prints `verdicts: k/N` (`BOARD INCOMPLETE` while short). The prompt
47
+ interpolates the base ref, forbids edits and scratch files inside the tree, tells a
48
+ seat that cannot run the gate to report it under `unverifiable`, and asks the `risk`
49
+ lens to falsify the acceptance criteria.
50
+
51
+ ### Fixed
52
+
53
+ - **`playmaker quotas` no longer refreshes the Claude OAuth token.** The probe shared
54
+ the keychain entry and the single-use refresh token with the Claude Code CLI, and the
55
+ race logged the CLI out. The probe now only reads; when the token is expired it lets
56
+ the CLI refresh it (`[quotas] claude_refresh_via_cli`, default on) or shows
57
+ `login expired — run: claude auth login`. An outage after a successful refresh is
58
+ reported as an outage, not as an expired login.
59
+ - **The reset-credits inventory is best-effort**: a failing secondary request no
60
+ longer sinks the whole Codex block.
61
+ - **The ledger's writers take a lock and rewrites are atomic.** `ledger.py add` (and so
62
+ the hook) appends under an `flock`; `fix` and `escape` hold the same lock across their
63
+ read-modify-write and publish through a temp file and `os.replace`, so two coach sessions
64
+ committing at once cannot lose a row or tear the file; a trailing partial line is skipped
65
+ by the readers instead of stopping them. `--effort` is validated in the CLI before a
66
+ detached run spawns.
67
+
8
68
  ## [0.13.0] - 2026-09-30
9
69
 
10
70
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: playmaker-cli
3
- Version: 0.13.0
3
+ Version: 0.14.0
4
4
  Summary: Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel.
5
5
  Project-URL: Homepage, https://github.com/vladsafedev/playmaker
6
6
  Project-URL: Repository, https://github.com/vladsafedev/playmaker
@@ -375,6 +375,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
375
375
  `.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
376
376
  otherwise, and playmaker would report the agent as unavailable.
377
377
 
378
+ **claude** also accepts an effort level — Claude Code's `--effort
379
+ low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
380
+ motivating case: running Opus review boards at `xhigh`). Set the lane default
381
+ once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
382
+ claude --effort <level>` and `playmaker continue <id> --effort <level>`
383
+ override it for one run without persisting anything. The flag is accepted and
384
+ ignored on every other lane so scripts can pass it uniformly, and when it is
385
+ unset everywhere no `--effort` flag is sent at all — Claude Code's own default
386
+ applies.
387
+
378
388
  ## Notifications
379
389
 
380
390
  Every detached dispatch pings when it finishes. With
@@ -436,9 +446,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
436
446
  **separate buckets**. So is the `Codex — Spark` block, every agy row, and the
437
447
  whole Z.ai block. Routing a subtask is choosing which of them to spend.
438
448
 
439
- - **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
449
+ - **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
450
+ it expires, Claude Code itself refreshes it; the quota probe never rotates the
451
+ shared refresh token.
440
452
  Model-scoped weekly buckets come from the usage API's `limits[]` array and
441
453
  print as `Weekly · <model>`.
454
+ `[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
455
+ report `login expired — run: claude auth login` without spawning `claude`.
442
456
  - **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
443
457
  Spark model's own 5-hour and weekly windows come from
444
458
  `additional_rate_limits[]` and print as their own `Codex — Spark` block.
@@ -348,6 +348,16 @@ A bare name is resolved on `PATH`; a path is used as-is. opencode installs to
348
348
  `.zshrc` — so a dispatch from cron, an editor, or the coach can't find it
349
349
  otherwise, and playmaker would report the agent as unavailable.
350
350
 
351
+ **claude** also accepts an effort level — Claude Code's `--effort
352
+ low|medium|high|xhigh|max`, which trades speed for reasoning depth (the
353
+ motivating case: running Opus review boards at `xhigh`). Set the lane default
354
+ once with `effort = "xhigh"` under `[agents.claude]`; `playmaker dispatch
355
+ claude --effort <level>` and `playmaker continue <id> --effort <level>`
356
+ override it for one run without persisting anything. The flag is accepted and
357
+ ignored on every other lane so scripts can pass it uniformly, and when it is
358
+ unset everywhere no `--effort` flag is sent at all — Claude Code's own default
359
+ applies.
360
+
351
361
  ## Notifications
352
362
 
353
363
  Every detached dispatch pings when it finishes. With
@@ -409,9 +419,13 @@ The `Weekly`, `Weekly · Fable` and `Sonnet` rows above are the point: they are
409
419
  **separate buckets**. So is the `Codex — Spark` block, every agy row, and the
410
420
  whole Z.ai block. Routing a subtask is choosing which of them to spend.
411
421
 
412
- - **Claude** — OAuth usage API; token from the Claude Code Keychain entry.
422
+ - **Claude** — OAuth usage API; token from the Claude Code Keychain entry. When
423
+ it expires, Claude Code itself refreshes it; the quota probe never rotates the
424
+ shared refresh token.
413
425
  Model-scoped weekly buckets come from the usage API's `limits[]` array and
414
426
  print as `Weekly · <model>`.
427
+ `[quotas] claude_refresh_via_cli` defaults to `true`; set it to `false` to
428
+ report `login expired — run: claude auth login` without spawning `claude`.
415
429
  - **Codex** — ChatGPT `wham/usage` API; token from `~/.codex/auth.json`. The
416
430
  Spark model's own 5-hour and weekly windows come from
417
431
  `additional_rate_limits[]` and print as their own `Codex — Spark` block.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "playmaker-cli"
3
- version = "0.13.0"
3
+ version = "0.14.0"
4
4
  description = "Playing-coach CLI for orchestrating Claude Code, Codex, Antigravity, Kimi Code, Muse Code and opencode sub-agents in parallel."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -1,14 +1,13 @@
1
1
  ---
2
2
  name: playmaker-coach
3
- description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode/Kimi Code(kimi)/Muse Code(muse) workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
3
+ description: Team-lead mode for coding work. Decompose the task into work packages, dispatch them to Claude/Codex/Antigravity(agy)/opencode workers through the `playmaker` CLI, then run an automatic multi-agent review board over every diff before it lands, and drive the fix cycles. Use for ANY request that will change code in more than one place, needs an independent review pass, or has two or more parts that can run at once — "implement", "add", "fix", "refactor", "wire up", "сделай", "почини", "добавь", "реализуй", "собери". NOT for answering a question, reading or explaining code, a single-line edit, or a git/ops command.
4
4
  ---
5
5
 
6
6
  # playmaker-coach — you are the tech lead, not the typist
7
7
 
8
- `playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / Kimi Code (`kimi`) /
9
- Muse Code (`muse`) / a sibling Claude, tracks them, and returns their threads. This skill is the
10
- judgment on top: what to split, who gets which slice, how to size it so verifying is cheap, and
11
- **how the review board runs**.
8
+ `playmaker` dispatches sub-tasks to Codex / Antigravity (`agy`) / opencode / a sibling Claude,
9
+ tracks them, and returns their threads. This skill is the judgment on top: what to split,
10
+ who gets which slice, how to size it so verifying is cheap, and **how the review board runs**.
12
11
 
13
12
  **The coach produces plans, prompts, verdicts and integration — not feature diffs.** Your context
14
13
  window is the most expensive resource on the table. Every file you read yourself and every line you
@@ -18,9 +17,9 @@ summarization, and even the drafting of worker prompts when the task is big enou
18
17
  Two loops run under your hand, always both:
19
18
 
20
19
  ```
21
- decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue → land
22
- ↑______________________|
23
- max 2 cycles
20
+ decompose → dispatch → prove on disk → REVIEW BOARD → adjudicate → continue
21
+ ↑______________________| max 2 cycles
22
+ all WPs green → integrate → GLOBAL GATE → SEAM REVIEW (≥2 interacting WPs) → land → ledger row
24
23
  ```
25
24
 
26
25
  ## 1. Activation gate
@@ -48,21 +47,29 @@ outside the skill so it survives `playmaker skill install --force`. Read, in thi
48
47
  later files override earlier ones:
49
48
 
50
49
  ```bash
51
- cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions)
52
- cat ~/.playmaker/policy.md 2>/dev/null # personal policy (quota economics, lane defaults)
50
+ cat ~/.playmaker/policy.md 2>/dev/null # personal policy (tiers, quota economics, lane defaults)
51
+ cat ./.playmaker/policy.md 2>/dev/null # repo policy (gates, protected paths, conventions) — wins
53
52
  ls ./.playmaker/agents/*.md 2>/dev/null || ls ~/.playmaker/agents/*.md 2>/dev/null
54
53
  ```
55
54
 
56
- Agent profiles describe each lane's strengths, ceiling, and quota position — trust a profile over
57
- the generic defaults here. If no policy file exists, say so once in the plan and use these defaults.
55
+ Agent profiles describe each lane's traps, mechanics and quota position — trust a profile over the
56
+ generic defaults here, never over the policy's tier table or a dated owner order: profiles say how a
57
+ lane behaves, policy says what it may be given. A newer dated order beats an older line anywhere. If no policy file exists, say so once in the plan and use these defaults.
58
58
 
59
59
  ## 3. Protocol
60
60
 
61
61
  ### 3.1 Recon — delegate it
62
62
 
63
- Codebase exploration is the highest-leverage thing to delegate: raw reading is exactly what the
64
- cheapest model does as well as you do. Before your own `grep`/`Read` sweep, dispatch a read-only
65
- recon with an explicit deliverable and `--sync`, and read a 200-word report instead of ten files:
63
+ Codebase exploration is the highest-leverage thing to delegate. Two kinds, two tiers:
64
+
65
+ - **Map recon** — locate files, symbols, call sites, line ranges, inventories. Raw reading: the
66
+ cheapest lane (Gemini Flash) does it as well as you do.
67
+ - **Semantic recon** — invariants, ownership, contracts, side effects, impact radius, «why is it like
68
+ this». Judgment: a senior lane (Kimi K3 for many-file sweeps, Gemini Pro, Codex). A cheap model
69
+ returns a confident wrong map, and you plan on it.
70
+
71
+ Before your own `grep`/`Read` sweep, dispatch the matching kind with an explicit deliverable and
72
+ `--sync`, and read a 200-word report instead of ten files:
66
73
 
67
74
  ```bash
68
75
  playmaker dispatch agy --model <cheap-tier> --cwd "$(pwd)" --sync --read-only \
@@ -78,14 +85,17 @@ playmaker quotas # capacity per MODEL, not per agent; re-probes it
78
85
  ```
79
86
 
80
87
  Read at model granularity: separate buckets inside one provider are separate capacity. The aim is
81
- **level-loading** — finish the week with every pool drawn down evenly, except the one reserved for
82
- the coach — not hoarding the pools other people also use. Details and per-provider quirks:
88
+ **accepted work per point of scarce quota**, with prepaid capacity spent before it resets.
89
+ Level-loading — every pool drawn down evenly by week's end, except the coach's own — is the
90
+ tie-breaker between lanes that fit a WP equally, never a reason to queue a WP behind a closed lane. Details and per-provider quirks:
83
91
  `references/quotas.md`.
84
92
 
85
93
  ### 3.3 Decompose into work packages
86
94
 
87
- 2–5 WPs on the first pass. A WP is dispatchable only when all five hold — this is what makes review
88
- cheap and re-prompting rare:
95
+ 2–5 WPs on the first pass. Name each WP's **class** first — `mechanical | feature | terminal-heavy |
96
+ repo-recon | architecture | high-risk` — it decides which senior lanes are eligible (policy: fit before
97
+ headroom), which lenses the board gets, and it is the first field of the ledger row. Then a WP is
98
+ dispatchable only when all five hold — this is what makes review cheap and re-prompting rare:
89
99
 
90
100
  1. **Hard file boundary.** "Edit only `x.ts` and its spec" — never "do the backend part".
91
101
  2. **A gate the worker runs itself** — `tsc --noEmit`, a named spec file, a lint pass. It must exit 0
@@ -105,7 +115,7 @@ policy first, then `references/lanes.md`.
105
115
 
106
116
  ### 3.4 Propose, then wait
107
117
 
108
- Post the plan as: WP → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
118
+ Post the plan as: WP → class → lane+model → why → gate → done-condition, plus the reviewer pair per WP and
109
119
  the current per-model capacity. **Dispatch nothing until the user approves.** Approval may be
110
120
  partial ("go but reroute tests"); restate the modified plan in one line, then dispatch.
111
121
 
@@ -119,19 +129,27 @@ playmaker dispatch <agent> --model <name> --batch "$B" --cwd "$(pwd)" --prompt "
119
129
  Detached by default — that is the point; never `--sync` a whole fan-out. Always pass `--cwd`,
120
130
  `--batch` (one summary ping for the batch), and `--model` unless the profile says otherwise.
121
131
  Prompt shape: `references/prompt-templates.md`. Per-agent traps (agy scratch dir, opencode relative
122
- paths, codex model roster, kimi's K2.7 default and exit codes, muse's sandbox): `references/agent-gotchas.md`.
132
+ paths, codex model roster): `references/agent-gotchas.md`.
123
133
 
124
134
  Parallel WPs that touch the same files go in **git worktrees**, one per WP, or they will collide.
125
135
 
136
+ Within a minute of every detached dispatch, `playmaker get <id> --json`: a quota or auth refusal
137
+ finishes in seconds as `failed`, pings nobody, and looks exactly like a slow worker. Give every
138
+ dispatch a deadline (recon 15 minutes; a WP by its size) and on expiry check the process is alive
139
+ (`playmaker logs <id>`, its RSS) before waiting longer. Never wait for a closed lane.
140
+
126
141
  ### 3.6 Prove it on disk before you believe it
127
142
 
128
143
  `done` means "the process exited cleanly with text", not "code changed". playmaker ≥0.9 marks a
129
144
  zero-change write task `no_changes` — treat it exactly as a failure. On older builds, check yourself:
130
145
 
131
146
  ```bash
132
- git -C "<cwd>" status --short # empty tree on a write WP = NOT done
147
+ git -C "<cwd>" status --short --untracked-files=all # empty tree on a write WP = NOT done
133
148
  ```
134
149
 
150
+ New files are invisible to `git diff`: `git add -N <new files>` before the board, and compare
151
+ `git diff --stat` with what the worker claims it changed.
152
+
135
153
  Then run the WP's own gate yourself, once, cheaply. A WP that fails its gate never reaches the
136
154
  review board — it goes straight back to the worker.
137
155
 
@@ -157,10 +175,18 @@ Non-negotiables:
157
175
  - Every finding carries `file:line` + a concrete failure scenario. **No evidence → dropped.** You
158
176
  arbitrate, and a reviewer's confidence is an input, not a verdict.
159
177
  - Fixes go back via `playmaker continue <impl-id>` as a numbered list — the worker still has its
160
- context. Re-review runs on the **delta only**.
178
+ context. Re-review runs on the **delta only** — which means passing the round-1 state as the
179
+ base-ref (a wip commit in the worktree, squashed before landing); with the original base the
180
+ patch is cumulative and reviewers re-open settled code.
181
+ - A ruling that changes the scope goes into `spec.md` before the next round — reviewers refute
182
+ against the spec, and a stale spec produces false blockers. Before charging a finding to the
183
+ worker, check it is not already present on the base ref.
161
184
  - **Two cycles maximum.** Still blocking after two → stop and escalate to the user with both
162
185
  verdicts. A third round at the same lane is the most expensive way to use a cheap model.
163
186
  - Land only at **zero blocking findings**.
187
+ - On `high`, the `risk` seat is told to **falsify** the acceptance criteria — a command, input or repro
188
+ per criterion it attacked, not only an argument; its `pass` lists what it tried. A seat that cannot
189
+ execute (the `claude` lane runs with binaries forbidden) never holds `correctness` — give it `contracts`.
164
190
 
165
191
  ### 3.8 Keep a board file
166
192
 
@@ -173,11 +199,39 @@ Long fan-outs outlive your context. Maintain `./.playmaker/board.md` — one row
173
199
  Update it at dispatch, at gate, after each review round. On resume, read the board before anything
174
200
  else. It is also what you paste back to the user as the status report.
175
201
 
202
+ The columns are fixed — one layout, not one per board: a resumed session and the ledger both parse
203
+ them. At landing, the ledger hook writes the WP's row to the cross-repo ledger on `git commit`; read its
204
+ `[ledger]` line and correct the soft fields with `python3 ~/.playmaker/scripts/ledger.py fix wp=… k=v`.
205
+ When the hook matched nothing, append the row by hand: `python3 ~/.playmaker/scripts/ledger.py add
206
+ wp=… class=… impl_lane=… …` (fields: `~/.playmaker/policy.md`, «Ledger»). It keeps what board prose loses: class, first-gate pass, blocking findings that survived
207
+ adjudication, cycles, wall time — the evidence the routing step reads.
208
+
176
209
  ### 3.9 Failures
177
210
 
178
- Surface them; never silently retry. A failed dispatch (missing binary, bad auth, rejected model)
179
- gets a re-routed plan proposed to the user, not a second attempt at the same string. Diagnosis per
180
- agent: `references/agent-gotchas.md`.
211
+ Surface them; never silently retry — and never wait. A failed dispatch (missing binary, bad auth,
212
+ rejected model, quota) is neither a fix cycle nor a worker failure: re-route the WP to the next live
213
+ senior lane now, with the full context in a fresh prompt, and record the substitution in the board.
214
+ Ask the user only when no eligible lane is left. Diagnosis per agent: `references/agent-gotchas.md`.
215
+
216
+ ### 3.10 Integrate before landing — when the fan-out had two or more WPs
217
+
218
+ Two green WPs can be wrong together. Per-WP gates and boards prove the parts; nothing above proves the
219
+ whole. Once every WP is green:
220
+
221
+ ```
222
+ integrate on one tree (worktrees merged in dependency order)
223
+ → GLOBAL GATE — the repo's minimum bar plus the union of the WPs' own gates, run by you
224
+ → if ≥2 WPs touched interacting modules:
225
+ SEAM REVIEW — pm-review <batch>-seams <base-before-first-WP> --risk seams --impl-agent <main lane>
226
+ → commit, one WP per commit, in integration order → ledger rows
227
+ ```
228
+
229
+ "Interacting" = shared types or packages, the same service or screen, a migration and its consumers,
230
+ event or config names, shared state. The seam reviewer gets a spec that lists the seams and reads only
231
+ those: changed interfaces, duplicated or conflicting assumptions, incompatible types, migration order,
232
+ naming, config, shared state. The packages' internals already passed their own boards — re-reviewing
233
+ them is the third round the protocol forbids. Roster: the `seams` block in `reviewers.conf`;
234
+ `--impl-agent` is the lane that implemented most of the fan-out.
181
235
 
182
236
  ## 4. What the coach may still type by hand
183
237
 
@@ -210,3 +264,6 @@ editor on product code, ask whether that is a WP you failed to write.
210
264
  - **Dispatching a WP you cannot verify in one command and one paragraph.**
211
265
  - **Omitting `--model`** and letting a CLI default drain a top-tier bucket.
212
266
  - **Reading whole agent threads** when `summary` answers the question.
267
+ - **Landing two green WPs without running the tree they make together.** Per-WP gates prove the
268
+ parts; the global gate and the seam review prove the whole.
269
+ - **Routing by headroom alone.** Class first, then the ledger's floor, then the pool floors, then headroom.
@@ -65,39 +65,6 @@ line from `agy models` / `opencode models` rather than typing it.
65
65
  - Neither trap applies to **review** dispatches, which write nothing — which makes opencode a
66
66
  perfectly good reviewer even where it is a shaky implementer.
67
67
 
68
- ## kimi (Kimi Code CLI)
69
-
70
- - There is **no read-only mode below the prompt**: `-p` refuses `--auto`, `--yolo` and `--plan`
71
- (exit 1) and already runs with auto-approval. Only dispatch work you would run unattended anyway.
72
- - The session id arrives **only in the trailing `session.resume_hint` line**, so `playmaker list`
73
- shows the agent session late — do not conclude a dispatch failed just because the id has not
74
- appeared yet.
75
- - Sessions are **per-cwd**: `kimi session list` from another directory shows nothing. Track the
76
- session through playmaker, not through the CLI.
77
- - The stream carries **no token/cost fields** — there is nothing to budget against mid-run.
78
- - Exit codes: **exit 1 is non-retryable** (auth, quota, unknown model — "is not configured in
79
- config.toml"); **exit 75 is retryable**. Fix the cause on 1, re-dispatch on 75.
80
- - K3 is **slow on real tickets** — never `--sync` a big WP; dispatch detached and poll.
81
- - Needs **Node ≥ 22.19** — hence the wrapper binary; point `[agents.kimi] binary` at it.
82
- - **Always pass `-m kimi-code/k3-256k`** — the CLI default is the weaker K2.7 `kimi-for-coding`.
83
-
84
- ## muse (Muse Code CLI)
85
-
86
- - The default run keeps Muse's **sandbox**: approval off, workspace trusted, writes inside the
87
- repo fine — but writes **outside** it are denied, so `uv run pytest` fails initializing
88
- `~/.cache/uv` (npm/pnpm caches alike). A WP whose gate needs those caches needs
89
- `[agents.muse] yolo = true`.
90
- - **Without workspace trust Muse ignores the repo's AGENTS.md.** playmaker passes
91
- `--trust-workspace` by default; `trust_workspace = false` drops it, and repo rules stop
92
- applying.
93
- - Exit codes: **0** turn completed; **1** run failed (bad model, auth) — the reason is in the
94
- terminal event and stderr's last line; **2** usage error; **130/143** SIGINT/SIGTERM.
95
- - **Resume only from the same cwd** — a `--session-id` elsewhere would need
96
- `--allow-workspace-switch`, which playmaker never passes.
97
- - stderr **always** carries informational lines such as `muse: workspace root: …`, even on
98
- success — they are not errors.
99
- - The stream has **no cost fields** — token usage exists only in the session log on disk.
100
-
101
68
  ## Worktrees
102
69
 
103
70
  Parallel WPs that touch the same files collide. Give each its own git worktree and dispatch with
@@ -0,0 +1,117 @@
1
+ > **Opus 5 lane — OUT of the review board (owner order 2026-09-08):** `playmaker dispatch claude -m opus` draws the Anthropic all-models weekly — the bucket the coach session falls back to (as Opus) when the Fable weekly runs out — and was eating it fast. It no longer holds the `high` risk seat by default (that was the 2026-09-02 order); `codex` does. Opus is weighted by the Fable weekly instead: while `Weekly · Fable` ≥ 50% it fills the risk seat on `high` boards that Codex implemented, ¼ of that at 25–50%, nothing below 25% or when the all-models weekly is under 50%. Never for implementation or recon, never on `normal`. Table and flip: `~/.playmaker/policy.md`, `~/.playmaker/reviewers.conf`.
2
+
3
+ > **GEMINI ONLY on agy (owner order 2026-09-02):** never dispatch `claude-*` or `gpt-oss-*` models on the `agy` lane. Want a Claude? Don't dispatch one: the coach session is the Claude, and the Anthropic bucket is out of the board (2026-09-08). See `~/.playmaker/policy.md`.
4
+
5
+ # Execution lanes — where work runs
6
+
7
+ All Claude work — coach, in-session sub-agents, external `claude -p` — draws from the **same Claude
8
+ subscription**. So routing is about *which bucket* you spend and *where the result lands*, not who
9
+ pays. Four lanes:
10
+
11
+ 1. **Coach (this thread).** Interactive Claude Code on the top-tier weekly bucket — the scarcest,
12
+ hardest to replenish. Serial, and the most expensive in *context*. Reserve for orchestration,
13
+ architecture judgment, adjudication, final integration.
14
+
15
+ 2. **In-session sub-agents (the Task/Agent tool).** Run inside this session, write files (they
16
+ inherit the coach's permission mode), return straight into the coach's context, run in parallel.
17
+ Same subscription. Use when the coach folds the result in directly.
18
+
19
+ 3. **External dispatch — `playmaker dispatch claude`.** A tracked, detached stream you can monitor
20
+ and `continue` independently. Same subscription, and since September 2026 the same bucket: Sonnet has no
21
+ separate weekly (`seven_day_sonnet: null`), so this lane spends the coach's own all-models weekly.
22
+ Default `--model sonnet` (`haiku` for trivial mechanical work), and only when policy allows it. playmaker runs it in `acceptEdits`: it writes freely
23
+ inside `--cwd` and is refused outside it, so keep every path in the prompt inside `--cwd`.
24
+
25
+ 4. **External dispatch — `codex` / `agy` / `opencode`.** Each on its own subscription or plan — the
26
+ home for write-heavy parallel implementation that can leave the Anthropic subscription.
27
+ - **`agy` (Antigravity)** carries more than Google models: alongside Gemini Flash and Pro tiers it
28
+ serves **Claude Sonnet/Opus (Thinking)** and a GPT-OSS tier. Its Claude runs on *Google's* pool —
29
+ capable judgment that spends none of the Anthropic bucket, but a generation behind the
30
+ frontier: treat the whole lane as **middle** (see Model tiers below). Note the internal
31
+ split: all Gemini models share one bucket, Claude and GPT-OSS share another.
32
+ - **`opencode`** is the widest lane: one CLI over ~75 providers addressed as `provider/model` — a
33
+ GLM coding plan, or a model running locally on this machine, which spends no subscription quota
34
+ at all.
35
+
36
+ **Never write an agy or opencode model name from memory** — run `agy models` / `opencode models` and
37
+ copy a line. Both rosters and their spelling move with releases, and playmaker validates `--model`
38
+ against the live roster, failing the dispatch on a stale name.
39
+
40
+ ## Routing cheat-sheet
41
+
42
+ | The work… | Lane | Why |
43
+ |---|---|---|
44
+ | writes files, coach integrates the result directly | in-session sub-agent | write-capable, returns into context |
45
+ | is an independent stream to monitor separately | `dispatch claude --model sonnet` | tracked, detached — but on the coach's own subscription (Sonnet has no separate bucket): policy says when, and it is rarely |
46
+ | is heavy reasoning only the coach can do | coach | top tier, serial |
47
+ | is write-heavy and can leave Claude | codex / agy / opencode | their own quotas |
48
+ | wants Claude-flavoured judgment without spending the Anthropic bucket | `dispatch agy --model <claude-opus-thinking>` | middle tier — a generation behind; never the only senior eye |
49
+ | is bulk work with every subscription low | `dispatch opencode --model <plan>/<model>` | a separate plan, untouched by the others |
50
+ | is mechanical and privacy-sensitive, or all quotas spent | `dispatch opencode --model <local>/<model>` | runs on this machine, costs wall-clock only |
51
+
52
+ ## Model tiers — senior / middle / junior
53
+
54
+ > **Superseded on this machine by the tier table in `~/.playmaker/policy.md` (owner order
55
+ > 2026-09-04): there is no middle tier there, and `agy` Claude models are junior. The rows below are
56
+ > the skill's generic defaults for a machine without a policy file.
57
+
58
+ **Judge a lane by the model actually serving it, never by the lane's name.** Model rosters move
59
+ faster than this file; re-derive the table whenever a lane surprises you, and correct it here.
60
+
61
+ | Tier | What is actually there | Give it |
62
+ |---|---|---|
63
+ | **Senior** | the coach's own session (Fable 5.1, falling back to Opus 5.5); **`codex`** — the real Codex CLI on the ChatGPT plan (`--model` omitted; verify with `codex --version`); **`opencode` / GLM** (`zai-coding-plan/glm-*`) | architecture, spec interpretation, cross-module integration, adjudication, anything irreversible |
64
+ | **Near-senior** | **`agy gemini-3.1-pro-high`** — not quite the three above, but close enough to carry a review lens or a hard WP on its own | the always-on review seat, demanding implementation, deep recon |
65
+ | **Middle** | the rest of `agy` / Antigravity — its **Claude 4.6 / Sonnet 4.6** are a generation behind the frontier despite the name — plus GPT-OSS | well-specified implementation against a ready plan, refactors, CRUD by convention, a second opinion |
66
+ | **Reserve** | `dispatch claude --model sonnet` / `--model opus` — Sonnet 5.5 and Opus 5.5, senior capability on the coach's own subscription | nothing by default: tier is capability, allocation is the policy's call — a seat only on the owner's word |
67
+ | **Junior** | Gemini **Flash** tiers, Codex Spark | mass reading, codebase sweeps, mechanical diffs, tight fix→check loops |
68
+
69
+ Two traps this table exists to prevent — both have bitten in practice:
70
+
71
+ 1. **A lane serving "Claude Opus" is not necessarily the current Opus.** Antigravity's Claude
72
+ models trail the frontier by a full generation. Cheap, capable, worth using — but *middle*,
73
+ not senior. Their verdicts are an input you weigh, not an authority you defer to.
74
+ Conversely, do **not** file GLM as cheap because it is cheap: it is a senior lane.
75
+ 2. **`codex` is the Codex CLI, not a Google model.** The lane name is not evidence of the engine.
76
+
77
+ **On irreversible work** — deletions, migrations, money paths, auth, public contracts — a middle
78
+ lane may never be the only senior-grade eye. Pair it with `codex`, or adjudicate that hunk yourself.
79
+
80
+ Match the WP to the tier, then reserve the most-depleted senior model for the lightest role —
81
+ usually the coach's own adjudication — and push volume down to junior lanes.
82
+
83
+ ## Junior fan-out — where it is safe
84
+
85
+ Route down by default when **all** of these hold:
86
+
87
+ 1. **Self-verifiable outcome** — a green command the worker runs itself. An objective gate replaces
88
+ senior judgment.
89
+ 2. **Tight file boundary plus a pattern to mirror** — CRUD by existing convention, a test scaffolded
90
+ from a neighbouring spec, a story or doc from a template.
91
+ 3. **Low blast radius** — one WP, dark or flagged code, no schema or contract change; a bad diff is
92
+ cheap to throw away.
93
+ 4. **Read-only by nature** — recon, sweeps, summarization, fact-check tables. Zero risk beyond
94
+ wasted tokens.
95
+
96
+ ## Where juniors are never safe
97
+
98
+ Regardless of how well specified — these stay with a senior lane or the coach:
99
+
100
+ - **Money paths**: billing, entitlement ledgers, IAP/Stripe webhooks, refunds, idempotency.
101
+ - **Schema migrations, backfills, destructive data operations.**
102
+ - **Auth, session, and security-sensitive code.**
103
+ - **Frozen public API contracts** and whatever the repo's policy marks human-review-required.
104
+ - **Anything without a one-sentence done-condition.** Juniors do not resolve spec ambiguity, they
105
+ amplify it.
106
+
107
+ A junior diff that touches this list is a routing bug: pull the WP back, do not patch it in review.
108
+
109
+ ## Escalation
110
+
111
+ A worker that fails its own gate **twice** escalates one tier — never a third attempt at the same
112
+ tier. Two failures at the top tier means the WP is wrong, not the worker: re-scope it.
113
+
114
+ ## Two-stage senior review
115
+
116
+ Juniors produce the fact-check and consistency reports; seniors render verdicts *on top of those
117
+ reports* rather than re-reading the world. This is the cheapest way to buy senior judgment.
@@ -0,0 +1,77 @@
1
+ # The ledger and its hook
2
+
3
+ `~/.playmaker/ledger.jsonl` holds one JSON object per work package — the cross-repo evidence the
4
+ routing step reads. `scripts/ledger.py` maintains it (`add` / `fix` / `escape` / `stats` / `tail`);
5
+ it honours `PM_LEDGER` for tests. Rows whose note starts with "backfill" are historical — never
6
+ rewrite them.
7
+
8
+ ## Fields
9
+
10
+ `date`, `repo`, `wp`, `class`, `risk`, `impl_lane`, `impl_model`, `gate_first_pass`, `reviewers`
11
+ (list), `blocking_r1`, `blocking_accepted`, `blocking_rejected`, `cycles`, `rounds`, `wall_min`,
12
+ `quota_note`, `outcome`, `commit`, `note`, `escaped_from` (list). Unknown fields are `-`; ints are
13
+ ints or `-`.
14
+
15
+ ## How the hook maps a review dir to a row
16
+
17
+ On every `git commit` the PostToolUse hook (`scripts/ledger-hook.py`) matches the commit to the WP
18
+ under `.playmaker/reviews/` whose `diff-r*.patch` file list overlaps the commit's files the most
19
+ (ties → newest patch; dirs untouched for 14+ days are ignored), then fills:
20
+
21
+ | Field | Source |
22
+ |---|---|
23
+ | `date`, `repo`, `wp`, `commit`, `outcome` | today, toplevel basename, review-dir name, short sha, `landed` |
24
+ | `class`, `impl_lane`, `impl_model`, `risk` | spec.md header `class:` / `impl: <lane> <model>` / `risk:` (`routine|normal|high|seams`; see review-board.md); risk may also come from a pre-header spec title ("… risk high") or `board.env`; else `-` |
25
+ | `reviewers` | union of lanes over `sessions.txt` and every archived `sessions.txt`, short form (`kimi-k3`, `glm-5.3`, `agy-gemini-pro`, `codex`, `muse`, `claude-opus`) |
26
+ | `rounds` | number of `diff-r*.patch` |
27
+ | `cycles` | number of `fix-r<N>.md` with N ≥ 1 |
28
+ | `gate_first_pass` | `n` if `fix-r0.md` exists (the first gate did not pass), else `-` |
29
+ | `blocking_r1` | blocking findings across round-1 verdicts (oldest archive when rounds > 1), deduped by `file:line` |
30
+ | `blocking_accepted`, `blocking_rejected` | `-` (soft — the junior or the coach fills them) |
31
+ | `wall_min` | minutes from spec.md mtime to the commit |
32
+ | `note` | `auto-hook` plus the junior outcome |
33
+
34
+ A commit that matches nothing gets a one-line hint (`ledger.py add wp=… class=… impl_lane=…`); a
35
+ commit whose `(wp, commit)` row already exists is skipped. Every run is logged to
36
+ `~/.playmaker/logs/ledger-hook.log`. The hook never blocks or fails the commit.
37
+
38
+ ## The junior
39
+
40
+ A junior model fills the fields the mechanics left as `-`; it never blanks or overrides a set value.
41
+ It runs as `codex exec -m gpt-5.3-codex-spark -s read-only -C <review dir> --skip-git-repo-check
42
+ --output-last-message <tmp>` with a 60 s timeout, reading spec.md, all verdicts, the fix lists,
43
+ board.env and the board.md line, and answering one JSON object. Configure it in
44
+ `~/.playmaker/config.toml` (or turn it off with `"none"`):
45
+
46
+ ```toml
47
+ [ledger]
48
+ junior = "codex:gpt-5.3-codex-spark"
49
+ ```
50
+
51
+ Any failure, timeout or non-JSON keeps the mechanical values; the reason lands in `note`.
52
+
53
+ ## Fixing a row
54
+
55
+ ```bash
56
+ python3 ~/.playmaker/scripts/ledger.py fix wp=<wp> [repo=<r>] [commit=<sha>] k=v …
57
+ ```
58
+
59
+ updates the LAST matching row in place, validated like `add`. When the hook matched nothing, append
60
+ the row by hand with `ledger.py add`.
61
+
62
+ ## Registering the hook (the coach edits ~/.claude/settings.json)
63
+
64
+ ```json
65
+ {
66
+ "hooks": {
67
+ "PostToolUse": [
68
+ {
69
+ "matcher": "Bash",
70
+ "hooks": [
71
+ {"type": "command", "command": "python3 <installed skill dir>/scripts/ledger-hook.py", "timeout": 120}
72
+ ]
73
+ }
74
+ ]
75
+ }
76
+ }
77
+ ```
@@ -8,20 +8,25 @@ scope, gate, done-condition, context.
8
8
  ```
9
9
  <one sentence: what to build and why it exists>
10
10
 
11
+ Working directory is <cwd>. Every path below is relative to it — never absolute, never a bare filename.
12
+
11
13
  SCOPE — edit ONLY these files:
12
14
  <path/a.ts>
13
15
  <path/b.spec.ts>
14
16
  Do not touch anything else. If the change appears to require another file, stop and say so in your
15
- final answer instead of editing it.
17
+ final answer instead of editing it. Never commit. Never install or update dependencies — if the task
18
+ seems to need one, stop and report.
16
19
 
17
20
  CONTEXT you need (do not go looking for more):
18
21
  - Mirror the pattern in <path/neighbour.ts>.
19
22
  - Convention: <naming / error handling / logging rule>.
20
23
  - Spec: <the two or three sentences that actually constrain this>.
21
24
 
22
- ACCEPTANCE — run this yourself and keep iterating until it exits 0:
25
+ ACCEPTANCE — run this yourself and iterate until it exits 0, at most two failed attempts; after the
26
+ second failure stop and report both attempts with their last output:
23
27
  <cmd, e.g. npx tsc --noEmit -p <tsconfig> && npx eslint <paths>>
24
- Paste the final output of that command into your answer.
28
+ Paste the last ~30 lines of that command's output, its exit code, and the test and file counts it
29
+ printed.
25
30
 
26
31
  DONE when: <one sentence the coach can confirm in seconds>.
27
32
 
@@ -31,6 +36,11 @@ prompt did not specify.
31
36
 
32
37
  ## Reviewer
33
38
 
39
+ Generated by `scripts/review-board.sh` — do not hand-write it; the two versions used to drift
40
+ (patch path, git rules, round wording). To seat a reviewer by hand — a substitute for a dead lane —
41
+ copy `.playmaker/reviews/<wp>/prompt-<lane>-<lens>.md`, dispatch it `--read-only` on the new lane,
42
+ and append `<id>|<lane>|<model>|<lens>` to `sessions.txt`. The shape, for reference:
43
+
34
44
  ```
35
45
  You are reviewing one work package. You did NOT write it. Your job is to REFUTE it, not to
36
46
  summarize it. Change nothing — this is a read-only review.
@@ -78,7 +88,8 @@ Reply with a one-line note per item saying what you changed.
78
88
  ## Recon (read-only)
79
89
 
80
90
  ```
81
- Recon only — change no files.
91
+ Recon only — change no files. Working directory is <cwd>. One session: no sub-agents of your own.
92
+ Report as your final message, write nothing to disk. Budget: 15 minutes.
82
93
 
83
94
  Find, in <dir>:
84
95
  (a) <thing>