@hecer/yoke 1.14.0 → 1.15.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +36 -0
  4. package/README.md +15 -9
  5. package/canon/AGENTS.md +7 -0
  6. package/canon/manifest.yaml +1 -1
  7. package/canon/skills/executing-plans/SKILL.md +1 -1
  8. package/canon/skills/requesting-code-review/SKILL.md +2 -2
  9. package/canon/skills/requesting-code-review/code-reviewer.md +12 -0
  10. package/canon/skills/subagent-driven-development/SKILL.md +7 -3
  11. package/canon/skills/subagent-driven-development/code-quality-reviewer-prompt.md +7 -0
  12. package/canon/skills/subagent-driven-development/implementer-prompt.md +7 -0
  13. package/canon/skills/subagent-driven-development/spec-reviewer-prompt.md +7 -0
  14. package/canon/skills/systematic-debugging/SKILL.md +7 -11
  15. package/canon/skills/systematic-debugging/condition-based-waiting.md +7 -0
  16. package/canon/skills/systematic-debugging/defense-in-depth.md +7 -0
  17. package/canon/skills/systematic-debugging/root-cause-tracing.md +9 -0
  18. package/canon/skills/tdd/SKILL.md +1 -1
  19. package/canon/skills/tdd/testing-anti-patterns.md +9 -0
  20. package/dist/agents/pi-telemetry.js +53 -0
  21. package/dist/agents/process-streams.js +13 -0
  22. package/dist/agents/providers.js +4 -1
  23. package/dist/agents/supervision.js +2 -0
  24. package/dist/agents/telemetry.js +22 -25
  25. package/dist/cli.js +16 -2
  26. package/dist/code-intelligence/adapters/graft.js +11 -0
  27. package/dist/code-intelligence/adapters/graphify.js +8 -0
  28. package/dist/code-intelligence/adapters/index.js +5 -0
  29. package/dist/code-intelligence/adapters/mcp.js +37 -0
  30. package/dist/code-intelligence/adapters/serena.js +11 -0
  31. package/dist/code-intelligence/adapters/types.js +1 -0
  32. package/dist/code-intelligence/contracts.js +137 -0
  33. package/dist/code-intelligence/coordinator.js +370 -0
  34. package/dist/code-intelligence/edit-plans.js +53 -0
  35. package/dist/code-intelligence/evidence.js +77 -0
  36. package/dist/code-intelligence/index.js +5 -0
  37. package/dist/code-intelligence/internal-types.js +1 -0
  38. package/dist/code-intelligence/mcp-client.js +139 -0
  39. package/dist/code-intelligence/mcp-server.js +93 -0
  40. package/dist/code-intelligence/snapshots.js +117 -0
  41. package/dist/code-intelligence/transactions.js +60 -0
  42. package/dist/estimation/schedule.js +40 -26
  43. package/dist/retrofit/command.js +3 -1
  44. package/dist/retrofit/config.js +10 -0
  45. package/dist/retrofit/gitignore.js +1 -0
  46. package/dist/retrofit/plan.js +3 -3
  47. package/dist/retrofit/planners/claude.js +3 -3
  48. package/dist/retrofit/planners/codex.js +5 -5
  49. package/dist/retrofit/planners/gemini.js +2 -2
  50. package/dist/retrofit/planners/kilo.js +2 -2
  51. package/dist/retrofit/planners/opencode.js +2 -2
  52. package/dist/retrofit/planners/pi.js +2 -2
  53. package/dist/retrofit/planners/qwen.js +2 -2
  54. package/dist/retrofit/skill-actions.js +1 -1
  55. package/dist/retrofit/tools.js +9 -7
  56. package/dist/setup/command.js +3 -1
  57. package/docs/AGENT-HARDENING-2026-09-16.md +32 -0
  58. package/docs/CODE-INTELLIGENCE.md +36 -0
  59. package/docs/HARNESSES.md +6 -0
  60. package/gemini-extension.json +1 -1
  61. package/package.json +1 -1
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.13.0",
5
+ "version": "1.15.1",
6
6
  "description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo and Pi: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.13.0",
3
+ "version": "1.15.1",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows for seven supported harnesses",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -1,5 +1,41 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.15.1 — 2026-09-16
4
+
5
+ ### Fixed
6
+ - Sum finalized Pi assistant-turn usage without counting streamed snapshots or replayed transcripts twice; preserve missing measurements and multiple model identities.
7
+ - Recognize successful Pi/OpenCode/Kilo tool events as watchdog progress without relaxing execution budgets.
8
+ - Reject structured verdicts followed by terminal provider errors, reject errored/aborted Pi verdicts, and exclude Claude child-agent verdicts/usage from parent results.
9
+ - Validate OpenCode/Kilo effort aliases consistently; emit a single variant flag.
10
+ - Correct Pi settings-relative skill discovery and enforce manual-only skill invocation using Pi's native frontmatter.
11
+ - Estimate schedule ranges from per-story scenarios and retain low confidence for sparse task-specific evidence.
12
+ - Ship eight missing delegation/review/debugging/testing resources and clarify host capability, worker-budget and verification rules in the shared skills; remove unsupported quality/speed claims.
13
+ - Keep coordinator unit tests offline by stubbing the separate preview backend; verify actual isolated edits and preservation of the source project.
14
+ - Synchronize previously stale Claude/Codex plugin, Gemini extension and Canon versions with the npm package.
15
+
16
+ ### Migration and validation limits
17
+ - Refresh generated skills with a reviewed `yoke retrofit . --agent=all` (or the selected agent). Current Pi requires explicit project trust to load project-local resources; Yoke does not grant it automatically. See [harness setup](docs/HARNESSES.md).
18
+ - Existing custom configuration remains authoritative. No timeout or acceptance-gate defaults were weakened. Time ranges remain empirical, not guaranteed deadlines.
19
+ - See the [audit and validation limits](docs/AGENT-HARDENING-2026-09-16.md). No authenticated seven-agent benchmark or guaranteed speed/quality improvement is claimed.
20
+ - Local verification: 1,281 tests passed, two platform-specific tests skipped; TypeScript lint/build, Canon validation, documentation metadata, package dry run and dependency audit passed.
21
+
22
+ ## 1.15.0 — 2026-09-09
23
+
24
+ ### Added
25
+ - Add opt-in federated Code Intelligence that composes Graft, Graphify and Serena behind one Yoke-controlled MCP facade with six stable tools for context, symbols, traces, impact and edit workflows.
26
+ - Add pinned backend adapters, explicit coverage/provenance/freshness evidence, content-addressed workspace snapshots and bounded local policy checks.
27
+ - Add isolated edit previews and guarded apply transactions with approval, snapshot freshness, exclusive locking and idempotency checks; Yoke's existing review, verify and commit gates remain authoritative.
28
+
29
+ ### Changed
30
+ - Add `off`, `shadow` and `active` Code Intelligence modes to setup and retrofit. `off` preserves the legacy `codeGraph` path unchanged; `shadow` is read-only; `active` enables preview and approved edits.
31
+ - Keep backend runtimes external and configurable instead of vendoring or silently installing them. Pin the validated integration targets to Graft `0.17.0`, Graphify `0.9.56` and Serena `1.7.1-dev`.
32
+ - Add the [Code Intelligence guide](docs/CODE-INTELLIGENCE.md), including setup, backend requirements, safety boundaries, limitations and the Pi integration note.
33
+
34
+ ### Migration and validation limits
35
+ - No migration is required. Existing projects remain on the legacy path until `yoke setup` or `yoke retrofit` is run with `--code-intelligence=shadow` or `--code-intelligence=active`.
36
+ - Validated with the Code Intelligence contract/coordinator/snapshot/MCP tests, TypeScript lint/build, documentation metadata, package dry run and the facade MCP handshake. The full suite retains one pre-existing provider-process timing failure; it is reproduced independently and is not caused by this release.
37
+ - This release does not claim that every language or backend is available in every environment. Backend failures are surfaced as partial coverage or an explicit unavailable capability, never silently treated as complete evidence.
38
+
3
39
  ## 1.14.0 — 2026-09-09
4
40
 
5
41
  ### Added
package/README.md CHANGED
@@ -1,9 +1,9 @@
1
1
  <div align="center">
2
2
 
3
- <h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.14.0/docs/assets/yoke-logo.png" alt="Yoke" width="100" height="63"></h1>
3
+ <h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.15.1/docs/assets/yoke-logo.png" alt="Yoke" width="100" height="63"></h1>
4
4
 
5
- <!-- yoke:version:start -->1.14.0<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->1258<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.15.1<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->1283<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->34<!-- yoke:skills:end -->
8
8
  <!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi<!-- yoke:agents:end -->
9
9
 
@@ -17,7 +17,7 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-1258%20defined-blue.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-1283%20defined-blue.svg)
21
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini%20%7C%20Qwen%20%7C%20OpenCode%20%7C%20Kilo%20%7C%20Pi-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
@@ -27,10 +27,12 @@
27
27
 
28
28
  > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
29
 
30
- **New in 1.13.0:** first-class [OpenCode, Kilo and Pi integrations](docs/HARNESSES.md), including native headless invocation, provider/model/variant routing, retrofit artifacts, role configuration and provider telemetry. [Qwen Code hardening and explicit DeepSeek/Kimi API model profiles](docs/QWEN-MODEL-SUPPORT.md) remain available. Since 1.10.0, Yoke also includes [dashboard search, filters and period comparisons](docs/DASHBOARD-EVOLUTION.md), [batch task assessments with separate planning models](docs/CAPABILITY-ROUTING.md), and [Windows sandbox preflight and process supervision](docs/WINDOWS-RUNNER-VALIDATION.md). Existing routing settings remain authoritative. See the [changelog](CHANGELOG.md) and the [harness integration guide](docs/HARNESSES.md) for limitations and setup.
30
+ **New in 1.15.0:** federated [Code Intelligence](docs/CODE-INTELLIGENCE.md) composes Graft, Graphify and Serena behind one Yoke-controlled MCP surface, with structural and semantic evidence, content-addressed snapshots, partial-coverage reporting and isolated edit previews. It is opt-in: use `off` for the unchanged legacy path, `shadow` for read-only canaries, or `active` for previews and approved edits. The [1.14.0 dashboard overhaul](docs/DASHBOARD-OVERHAUL.md) and first-class [OpenCode, Kilo and Pi integrations](docs/HARNESSES.md) remain available. See the [changelog](CHANGELOG.md) and [Code Intelligence guide](docs/CODE-INTELLIGENCE.md) for setup and limitations.
31
31
 
32
32
  OpenCode, Kilo and Pi are real CLI integrations, not bundled runtimes or credentials. OpenCode/Kilo use their JSON headless modes and local MCP configuration; Pi uses JSONL and explicit tool allowlists, but has no native MCP, sub-agent or plan layer. Read the [integration guide](docs/HARNESSES.md) before selecting a permission profile.
33
33
 
34
+ **Fixed in 1.15.1:** multi-turn Pi usage accounting, provider error/progress handling, task-specific time ranges and missing skill resources. Pi has been supported since **1.13.0**; current Pi versions require explicit project trust for project-local skills in headless runs. The [agent/skill hardening audit](docs/AGENT-HARDENING-2026-09-16.md) documents the fixes, migration notes and validation limits.
35
+
34
36
  ### One dashboard, multiple projects
35
37
 
36
38
  ```sh
@@ -208,10 +210,11 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
208
210
  | `yoke projects add\|list\|remove` | Register a project, list registrations or remove a reference by ID | `0` · `2` invalid/unavailable |
209
211
  | `yoke check [dir] [--json] [--requirement=] [--protect [--refresh]]` | Execute acceptance checks or explicitly pin their infrastructure | `0` passed/pinned · `1` failed · `2` unverified/unavailable |
210
212
  | `yoke goal set\|run\|resume\|pause\|status\|handoff\|budget [dir]` | Durable objectives, provider handoff, protected checks and checkpoint budgets | run/resume: `0` complete · `1` unfinished · `2` unavailable |
211
- | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for all seven harnesses; optional DeepSeek/Kimi API profiles run through Qwen | `0` · `1` invalid setup |
213
+ | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--code-intelligence=off\|shadow\|active] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for all seven harnesses; optional federated code intelligence and DeepSeek/Kimi API profiles | `0` · `1` invalid setup |
214
+ | `yoke code-intelligence-server [--workspace=] [--mode=off\|shadow\|active]` | Serve the single Yoke-controlled MCP facade for federated code intelligence | `0` · `1` invalid/unavailable |
212
215
  | `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
213
216
  | `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
214
- | `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen,opencode,kilo,pi\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
217
+ | `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen,opencode,kilo,pi\|all] [--code-graph=graphify\|serena] [--code-intelligence=off\|shadow\|active] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
215
218
  | `yoke prd draft [dir] --idea= [--runner=] [--force]` | Idea → 5–12 stories with testable acceptance criteria | `0` · `1` invalid/guarded · `2` agent unavailable |
216
219
  | `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
217
220
  | `yoke change add\|status [dir] [--idea=]` | Queue a change at any time; the loop turns it into append-only stories at the next safe boundary | `0` · `1` invalid inbox/request |
@@ -852,7 +855,7 @@ Yoke's guardrails are **mechanical, not advisory** — the loop blocks on a dirt
852
855
 
853
856
  ## 🧠 Choose your code-graph
854
857
 
855
- `yoke retrofit --code-graph=graphify|serena` (default `graphify`, remembered per project). The `yoke-retrofit` skill asks and recommends based on the project.
858
+ `yoke retrofit --code-graph=graphify|serena` (default `graphify`, remembered per project) selects the legacy single graph. For complete code intelligence, enable the federated facade with `yoke retrofit --code-intelligence=active` (or `shadow` for a read-only canary). See the [Code Intelligence guide](docs/CODE-INTELLIGENCE.md).
856
859
 
857
860
  | | **graphify** | **Serena** |
858
861
  |---|---|---|
@@ -862,6 +865,8 @@ Yoke's guardrails are **mechanical, not advisory** — the loop blocks on a dirt
862
865
  | Best for | rapid exploration / migration / onboarding | systematic refactoring in typed codebases |
863
866
  | Caveat | heuristic edges; static index can go stale | one language server per language |
864
867
 
868
+ The federated mode composes both structural and semantic evidence and adds Graphify's architecture/document graph. It exposes one Yoke-controlled MCP surface, content-addressed snapshots, partial-coverage reporting and isolated edit previews; the legacy `codeGraph` setting remains valid and unchanged when code intelligence is `off`.
869
+
865
870
  ## 🪙 Token efficiency
866
871
 
867
872
  Yoke attacks tokens on two complementary surfaces:
@@ -903,6 +908,7 @@ canon/ # the source of truth — harness-agnostic
903
908
  AGENTS.md skills/ policy/ loop/ tools/ manifest.yaml
904
909
  src/
905
910
  canon/ # manifest schema + validator (yoke validate)
911
+ code-intelligence/ # federated MCP facade, adapters, snapshots and guarded edits
906
912
  change/ # append-only change inbox · planning · independent coverage review
907
913
  retrofit/ # detect · plan · apply · planners (all seven harnesses) · tools
908
914
  loop/ # prd · gates · runner · verify · git/worktree · loop · run-command · lock · cleanup
@@ -925,7 +931,7 @@ release provenance.
925
931
  ## 🧪 Development
926
932
 
927
933
  ```bash
928
- npm test # vitest (1258 tests)
934
+ npm test # vitest (1283 tests)
929
935
  npm run build # tsc, no emit errors
930
936
  npm run yoke -- validate canon
931
937
  ```
package/canon/AGENTS.md CHANGED
@@ -9,6 +9,13 @@ You are operating in a project retrofitted by Yoke. Follow these always:
9
9
 
10
10
  This file is the portable baseline. Agent-specific instructions are generated alongside it (CLAUDE.md, GEMINI.md).
11
11
 
12
+ ## Host capabilities and execution budgets
13
+
14
+ - Use only tools and skills actually installed in the host. Legacy `superpowers:` references in adapted skills refer to the corresponding local skill; `test-driven-development` maps to `tdd`. Do not install another plugin just to resolve a namespace.
15
+ - When Yoke owns the loop, worker concurrency, verification, and integration remain under its control. Do not start nested loops or native subagents from a worker. Without native delegation, work serially and expose missing independent-review evidence.
16
+ - Run focused checks for quick feedback, but never skip required acceptance, protected, integration, or release gates. Cache evidence only when code, configuration, and environment still match.
17
+ - Distinguish observed durations from future estimates. Report the sample count, empirical range, and unknown waiting time; a timeout is a budget, not a promised completion time. Do not reduce reasoning or test scope merely to meet an estimate.
18
+
12
19
  ## Skill routing & precedence
13
20
 
14
21
  When several skills could match the same task, resolve deterministically:
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.14.0
2
+ version: 1.15.1
3
3
  agents: [claude, codex, gemini, qwen, opencode, kilo, pi]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
@@ -11,7 +11,7 @@ Load plan, review critically, execute all tasks, report when complete.
11
11
 
12
12
  **Announce at start:** "I'm using the executing-plans skill to implement this plan."
13
13
 
14
- **Note:** Tell your human partner that Superpowers works much better with access to subagents. The quality of its work will be significantly higher if run on a platform with subagent support (such as Claude Code or Codex). If subagents are available, use superpowers:subagent-driven-development instead of this skill.
14
+ Use `subagent-driven-development` only when the host exposes delegation and the current run permits it. Inside a Yoke worker, execute the assigned task without launching nested workers; Yoke owns concurrency and independent review. Without delegation, execute serially and report any unavailable independent review explicitly. Subagents alone do not guarantee better quality or faster completion.
15
15
 
16
16
  ## The Process
17
17
 
@@ -31,7 +31,7 @@ HEAD_SHA=$(git rev-parse HEAD)
31
31
 
32
32
  **2. Dispatch code-reviewer subagent:**
33
33
 
34
- Use Task tool with superpowers:code-reviewer type, fill template at `code-reviewer.md`
34
+ Use the host's available read-only reviewer mechanism and fill the [review template](code-reviewer.md). Do not assume a `Task` tool exists. Inside a Yoke worker, hand evidence to Yoke's configured reviewer instead of launching nested agents. If independent review is unavailable, report that limitation; self-review is not an equivalent substitute.
35
35
 
36
36
  **Placeholders:**
37
37
  - `{WHAT_WAS_IMPLEMENTED}` - What you just built
@@ -102,4 +102,4 @@ You: [Fix progress indicators]
102
102
  - Show code/tests that prove it works
103
103
  - Request clarification
104
104
 
105
- See template at: requesting-code-review/code-reviewer.md
105
+ See the [review template](code-reviewer.md).
@@ -0,0 +1,12 @@
1
+ # Code review request
2
+
3
+ - Implementation: {WHAT_WAS_IMPLEMENTED}
4
+ - Acceptance criteria and requirements: {PLAN_OR_REQUIREMENTS}
5
+ - Base commit: {BASE_SHA}
6
+ - Candidate commit: {HEAD_SHA}
7
+ - Context and constraints: {DESCRIPTION}
8
+ - Permitted read scope, available checks, artifacts and known validation gaps: fill explicitly.
9
+
10
+ Independently inspect the actual candidate against requirements. Check correctness, failure handling, security, concurrency, maintainability and meaningful regression tests. Do not edit the candidate, weaken tests, merge or publish.
11
+
12
+ Return evidence-backed critical/important findings separately from minor suggestions. Each finding needs a location, failure scenario and validation approach. Distinguish verified, failed and unverified criteria. State limitations and whether unresolved blockers prevent acceptance; a test summary alone is insufficient.
@@ -5,6 +5,10 @@ description: Use when executing implementation plans with independent tasks in t
5
5
 
6
6
  # Subagent-Driven Development
7
7
 
8
+ ## Host capability and budget check
9
+
10
+ Use this workflow only if native delegation is available and authorized. Do not invent a `Task` or `TodoWrite` tool on hosts that lack it. Pi has no built-in subagents; use Yoke's scheduler or `executing-plans` instead. When running inside an existing Yoke worker, do not start nested agents or another loop: return evidence to the controlling runner. Independent review must not be replaced by claiming a self-review was independent.
11
+
8
12
  Execute plan by dispatching fresh subagent per task, with two-stage review after each: spec compliance review first, then code quality review.
9
13
 
10
14
  **Why subagents:** You delegate tasks to specialized agents with isolated context. By precisely crafting their instructions and context, you ensure they stay focused and succeed at their task. They should never inherit your session's context or history — you construct exactly what they need. This also preserves your own context for coordination work.
@@ -119,9 +123,9 @@ Implementer subagents report one of four statuses. Handle each appropriately:
119
123
 
120
124
  ## Prompt Templates
121
125
 
122
- - `./implementer-prompt.md` - Dispatch implementer subagent
123
- - `./spec-reviewer-prompt.md` - Dispatch spec compliance reviewer subagent
124
- - `./code-quality-reviewer-prompt.md` - Dispatch code quality reviewer subagent
126
+ - [Implementer prompt](./implementer-prompt.md) - Dispatch implementer subagent
127
+ - [Spec reviewer prompt](./spec-reviewer-prompt.md) - Dispatch spec compliance reviewer subagent
128
+ - [Code quality reviewer prompt](./code-quality-reviewer-prompt.md) - Dispatch code quality reviewer subagent
125
129
 
126
130
  ## Example Workflow
127
131
 
@@ -0,0 +1,7 @@
1
+ # Code quality reviewer handoff
2
+
3
+ Supply the task, Acceptance criteria, base and candidate commits, architecture constraints, test evidence, and known risks. Review with read-only permissions after the specification review.
4
+
5
+ Inspect correctness, error paths, concurrency, resource cleanup, security boundaries, compatibility, and regression coverage. Prioritize reproducible defects over stylistic preferences. Check that optimizations preserve required gates and that unknown usage or timing evidence is not reported as zero or certainty.
6
+
7
+ Return severity-ranked findings with precise locations, failure scenarios, and suggested regression tests. Separate required fixes from optional improvements. State validation limits even when no defects are found. Never edit the candidate, self-merge, or infer release readiness solely from a green test summary.
@@ -0,0 +1,7 @@
1
+ # Implementer handoff
2
+
3
+ Supply the task, repository/worktree path, base commit, allowed write scope, relevant interfaces, and Acceptance criteria with their required checks. Include only relevant context and previous failure evidence.
4
+
5
+ Implement only the assigned task. Preserve unrelated changes. Reproduce defects with a failing test before fixing them; do not weaken protected tests or acceptance checks. Run focused checks during iteration and all required gates before handoff. Do not spawn nested workers, publish, merge, or expand permissions. Commit only if the controller explicitly delegates that responsibility.
6
+
7
+ Return: outcome, changed files, exact checks and exit results, remaining risks, blockers, and artifact paths. Report incomplete work as incomplete. If blocked by a product decision or missing credentials, retain the work and explain what is needed; do not repeat the same failing action indefinitely.
@@ -0,0 +1,7 @@
1
+ # Specification reviewer handoff
2
+
3
+ Supply the original task, Acceptance criteria, base and candidate commits, allowed read scope, and validation commands. Review independently from the implementer, with read-only permissions.
4
+
5
+ Compare the actual diff and observable behavior to each criterion. Check omissions, unauthorized additions, weakened tests, and whether evidence applies to the candidate being reviewed. Do not rely on the implementer's completion claim. Run permitted non-mutating checks where feasible; mark unavailable checks as unverified.
6
+
7
+ Return one disposition per criterion: verified, failed, or unverified, with file/line or command evidence. Give actionable defects and distinguish blockers from optional suggestions. Do not modify files, approve an unverified requirement, merge, or publish.
@@ -111,7 +111,7 @@ You MUST complete each phase before proceeding to the next.
111
111
 
112
112
  **WHEN error is deep in call stack:**
113
113
 
114
- See `root-cause-tracing.md` in this directory for the complete backward tracing technique.
114
+ See [root-cause tracing](root-cause-tracing.md) for the backward tracing technique.
115
115
 
116
116
  **Quick version:**
117
117
  - Where does bad value originate?
@@ -273,24 +273,20 @@ If systematic investigation reveals issue is truly environmental, timing-depende
273
273
  3. Implement appropriate handling (retry, timeout, error message)
274
274
  4. Add monitoring/logging for future investigation
275
275
 
276
- **But:** 95% of "no root cause" cases are incomplete investigation.
276
+ Before declaring the cause unknowable, record what was ruled out and which observations are still missing.
277
277
 
278
278
  ## Supporting Techniques
279
279
 
280
280
  These techniques are part of systematic debugging and available in this directory:
281
281
 
282
- - **`root-cause-tracing.md`** - Trace bugs backward through call stack to find original trigger
283
- - **`defense-in-depth.md`** - Add validation at multiple layers after finding root cause
284
- - **`condition-based-waiting.md`** - Replace arbitrary timeouts with condition polling
282
+ - [Root-cause tracing](root-cause-tracing.md) - Trace bugs backward to the original trigger
283
+ - [Defense in depth](defense-in-depth.md) - Validate relevant trust boundaries after finding the cause
284
+ - [Condition-based waiting](condition-based-waiting.md) - Replace arbitrary sleeps with bounded condition waits
285
285
 
286
286
  **Related skills:**
287
287
  - **superpowers:test-driven-development** - For creating failing test case (Phase 4, Step 1)
288
288
  - **superpowers:verification-before-completion** - Verify fix worked before claiming success
289
289
 
290
- ## Real-World Impact
290
+ ## Measuring impact
291
291
 
292
- From debugging sessions:
293
- - Systematic approach: 15-30 minutes to fix
294
- - Random fixes approach: 2-3 hours of thrashing
295
- - First-time fix rate: 95% vs 40%
296
- - New bugs introduced: Near zero vs common
292
+ Record time to reproduce, time to a verified fix, retries and escaped regressions for the actual project. No fixed speedup, success rate or absence of new bugs is guaranteed by this workflow.
@@ -0,0 +1,7 @@
1
+ # Condition-based waiting
2
+
3
+ Prefer a completion event or promise over sleeping for an assumed duration. Subscribe before starting work so a fast completion is not missed. Handle success, failure, cancellation and timeout, and remove listeners/timers in every terminal path.
4
+
5
+ If polling is unavoidable, choose a measurable condition, bounded deadline and modest interval. Preserve the last observed state in the timeout error. Do not treat repeated log output as successful progress.
6
+
7
+ Tests should control time or use an explicit readiness signal. Keep one real process integration test where needed. A larger timeout is justified only by measured runtime and does not repair a race or an unintended network call.
@@ -0,0 +1,7 @@
1
+ # Defense in depth
2
+
3
+ Fix the originating contract violation first. Then identify independent trust boundaries where malformed or stale data could enter again: configuration, external process output, persisted state and public interfaces.
4
+
5
+ Validate at those boundaries with explicit errors and preserve unknown values. Avoid scattering duplicate checks through code that already has a validated type. Do not catch every exception and return success or zero.
6
+
7
+ Test malformed input, missing evidence, stale state, cancellation and cleanup. For destructive operations verify target scope and authority independently of the requesting model. Protection layers must not weaken acceptance tests or convert a blocked action into an automatic permission bypass.
@@ -0,0 +1,9 @@
1
+ # Root-cause tracing
2
+
3
+ Capture the smallest failing command, inputs, expected/actual behavior, environment and code revision. Reproduce before changing code.
4
+
5
+ Trace backward from the first observed invalid value or state transition to its producer. At each boundary, record the input, output and contract without logging credentials. Distinguish a downstream symptom from the earliest violated invariant.
6
+
7
+ Form one falsifiable hypothesis and choose an observation that separates it from alternatives. Change one factor at a time. If reproduction depends on concurrency, preserve scheduling and cancellation evidence rather than adding arbitrary delays.
8
+
9
+ Write a regression test at the violated boundary, demonstrate its failure, fix the cause and run dependent checks. Document remaining uncertainty instead of attributing an intermittent pass to a fix without evidence.
@@ -356,7 +356,7 @@ Never fix bugs without a test.
356
356
 
357
357
  ## Testing Anti-Patterns
358
358
 
359
- When adding mocks or test utilities, read @testing-anti-patterns.md to avoid common pitfalls:
359
+ When adding mocks or test utilities, read [testing anti-patterns](testing-anti-patterns.md) to avoid common pitfalls:
360
360
  - Testing mock behavior instead of real behavior
361
361
  - Adding test-only methods to production classes
362
362
  - Mocking without understanding dependencies
@@ -0,0 +1,9 @@
1
+ # Testing anti-patterns
2
+
3
+ - Test observable behavior, not that a mock returned the value you supplied. Assert resulting state and failure behavior as well as calls.
4
+ - Mock external boundaries deliberately. If production constructs a second client for an isolated workspace, replace that constructor too; a unit test must not silently invoke a real network backend.
5
+ - Do not mock the implementation under test or delete assertions to make a change pass. Keep at least one contract/integration check for important boundaries.
6
+ - Demonstrate that the regression test fails for the original defect and passes after the fix. A new green test alone does not establish that it detects the bug.
7
+ - Avoid arbitrary sleeps and broad timeout increases. Use readiness events, controlled clocks and bounded cancellation.
8
+ - Preserve protected acceptance tests and required gates. Focused tests speed feedback but do not replace release validation.
9
+ - Report skipped tests, unavailable platforms and missing credentials separately from passing checks. Never convert absent measurements into measured zero.
@@ -0,0 +1,53 @@
1
+ const record = (value) => typeof value === 'object' && value !== null && !Array.isArray(value);
2
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
3
+ /** Pi usage is per assistant message, not cumulative across a run. */
4
+ export function createPiTelemetry() {
5
+ const totals = {};
6
+ const counts = {};
7
+ const models = new Set();
8
+ let turns = 0;
9
+ let legacyUsage;
10
+ const add = (usage) => {
11
+ turns++;
12
+ const fields = {
13
+ inputTokens: usage?.input, outputTokens: usage?.output,
14
+ cachedInputTokens: usage?.cacheRead, cacheWriteInputTokens: usage?.cacheWrite,
15
+ reasoningOutputTokens: usage?.reasoning,
16
+ totalCostUsd: record(usage?.cost) ? usage.cost.total : undefined,
17
+ };
18
+ for (const [key, value] of Object.entries(fields))
19
+ if (finite(value)) {
20
+ totals[key] = (totals[key] ?? 0) + value;
21
+ counts[key] = (counts[key] ?? 0) + 1;
22
+ }
23
+ };
24
+ return {
25
+ consume(event) {
26
+ // Compatibility with earlier top-level snapshots; never override a final usage.
27
+ if (event.type === 'message_update' && record(event.usage))
28
+ legacyUsage = event.usage;
29
+ if (event.type !== 'message_end' || !record(event.message) || event.message.role !== 'assistant')
30
+ return;
31
+ if (typeof event.message.model === 'string' && event.message.model)
32
+ models.add(event.message.model);
33
+ add(record(event.message.usage) ? event.message.usage : legacyUsage);
34
+ legacyUsage = undefined;
35
+ },
36
+ finish() {
37
+ const reportedModels = [...models];
38
+ const complete = turns > 0 && counts.inputTokens === turns && counts.outputTokens === turns;
39
+ if (complete) {
40
+ // Optional totals must also cover every turn; omitted is not measured zero.
41
+ const measured = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] === turns));
42
+ return { usageAvailable: true, tokens: {
43
+ ...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens,
44
+ ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}),
45
+ }, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
46
+ }
47
+ return { usageAvailable: false,
48
+ ...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}),
49
+ ...(reportedModels.length ? { reportedModels } : {}),
50
+ };
51
+ },
52
+ };
53
+ }
@@ -1,4 +1,5 @@
1
1
  import { parseProviderTelemetry } from './telemetry.js';
2
+ import { createPiTelemetry } from './pi-telemetry.js';
2
3
  export function createBoundedOutput(limitBytes) {
3
4
  let text = '';
4
5
  let truncated = false;
@@ -17,6 +18,7 @@ export function createBoundedOutput(limitBytes) {
17
18
  };
18
19
  }
19
20
  export function createTelemetryAccumulator(agent) {
21
+ const pi = agent === 'pi' ? createPiTelemetry() : undefined;
20
22
  let trailing = '';
21
23
  let telemetry = { usageAvailable: false };
22
24
  let reportedModels = [];
@@ -25,6 +27,15 @@ export function createTelemetryAccumulator(agent) {
25
27
  : undefined;
26
28
  const update = (lines) => {
27
29
  for (const line of lines) {
30
+ if (pi) {
31
+ try {
32
+ const event = JSON.parse(line);
33
+ if (event && typeof event === 'object' && !Array.isArray(event))
34
+ pi.consume(event);
35
+ }
36
+ catch { /* non-JSON diagnostics carry no usage */ }
37
+ continue;
38
+ }
28
39
  const next = parseProviderTelemetry(agent, [line]);
29
40
  if (next.reportedModels)
30
41
  reportedModels = next.reportedModels;
@@ -75,6 +86,8 @@ export function createTelemetryAccumulator(agent) {
75
86
  if (trailing)
76
87
  update([trailing]);
77
88
  trailing = '';
89
+ if (pi)
90
+ return pi.finish();
78
91
  if (stepTotals && (stepTotals.hasInput || stepTotals.hasOutput)) {
79
92
  const latest = telemetry.tokens;
80
93
  const inputTokens = stepTotals.hasInput ? stepTotals.input : latest?.inputTokens;
@@ -45,6 +45,9 @@ const argsFor = (agent, permissions) => {
45
45
  };
46
46
  export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}, output = {}) {
47
47
  const parsedSelection = ModelSelectionSchema.parse(selection);
48
+ if (['opencode', 'kilo', 'pi'].includes(agent) && parsedSelection.reasoningEffort && parsedSelection.variant && parsedSelection.reasoningEffort !== parsedSelection.variant) {
49
+ throw new Error(`${agent} reasoningEffort and variant selections must match`);
50
+ }
48
51
  if (agent === 'gemini' && parsedSelection.bare)
49
52
  throw new Error('Gemini does not support the bare startup selection');
50
53
  if (agent === 'gemini' && parsedSelection.reasoningEffort)
@@ -106,7 +109,7 @@ export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe'
106
109
  args.push('--thinking', parsedSelection.reasoningEffort);
107
110
  }
108
111
  if (parsedSelection.variant) {
109
- if (agent === 'opencode' || agent === 'kilo')
112
+ if ((agent === 'opencode' || agent === 'kilo') && !parsedSelection.reasoningEffort)
110
113
  args.push('--variant', parsedSelection.variant);
111
114
  else if (agent === 'pi') {
112
115
  if (parsedSelection.reasoningEffort && parsedSelection.reasoningEffort !== parsedSelection.variant)
@@ -36,6 +36,8 @@ export function inspectProviderEvent(line) {
36
36
  return { failure: 'provider-terminal-error' };
37
37
  }
38
38
  return { progress: (event.type === 'item.completed' && ((item?.type === 'command_execution' && item.exit_code === 0) || (item?.type === 'file_change' && item.status === 'completed')))
39
+ || (event.type === 'tool_execution_end' && event.isError === false)
40
+ || (event.type === 'tool_use' && event.part?.type === 'tool' && event.part.state?.status === 'completed')
39
41
  || (event.type === 'tool_result' && event.status === 'success')
40
42
  || (event.type === 'user' && Array.isArray(event.message?.content) && event.message.content.some((part) => part.type === 'tool_result' && part.is_error === false)) };
41
43
  }
@@ -1,3 +1,4 @@
1
+ import { createPiTelemetry } from './pi-telemetry.js';
1
2
  const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
2
3
  function parseJson(value) {
3
4
  try {
@@ -32,15 +33,21 @@ export function parseProviderResult(agent, output) {
32
33
  if (agent === 'qwen')
33
34
  return parseQwenResult(output);
34
35
  const fragments = [];
36
+ let structuredResult;
35
37
  for (const line of output.split(/\r?\n/u)) {
36
38
  const parsed = parseJson(line);
37
39
  if (!parsed.ok || !isRecord(parsed.value))
38
40
  continue;
39
41
  const event = parsed.value;
42
+ if (agent === 'claude' && event.parent_tool_use_id != null)
43
+ continue;
44
+ if (event.type === 'error' || event.type === 'turn.failed' ||
45
+ (event.type === 'result' && (event.is_error === true || event.status === 'error')))
46
+ return null;
40
47
  switch (agent) {
41
48
  case 'claude':
42
49
  if (event.type === 'result' && directMachineResult(event.structured_output) !== undefined)
43
- return event.structured_output;
50
+ structuredResult = event.structured_output;
44
51
  if (event.type === 'result' && typeof event.result === 'string')
45
52
  fragments.push(event.result);
46
53
  break;
@@ -63,6 +70,8 @@ export function parseProviderResult(agent, output) {
63
70
  }
64
71
  case 'pi':
65
72
  if (event.type === 'message_end' && isRecord(event.message) && event.message.role === 'assistant') {
73
+ if (event.message.stopReason === 'error' || event.message.stopReason === 'aborted')
74
+ return null;
66
75
  const text = textContent(event.message.content);
67
76
  if (text)
68
77
  fragments.push(text);
@@ -70,6 +79,8 @@ export function parseProviderResult(agent, output) {
70
79
  break;
71
80
  }
72
81
  }
82
+ if (structuredResult !== undefined)
83
+ return structuredResult;
73
84
  const joined = parseJson(fragments.join(''));
74
85
  if (joined.ok) {
75
86
  const direct = directMachineResult(joined.value);
@@ -115,6 +126,15 @@ function parseQwenResult(output) {
115
126
  return candidate;
116
127
  }
117
128
  export function parseProviderTelemetry(agent, lines) {
129
+ if (agent === 'pi') {
130
+ const accumulator = createPiTelemetry();
131
+ for (const line of lines) {
132
+ const parsed = parseJson(line);
133
+ if (parsed.ok && isRecord(parsed.value))
134
+ accumulator.consume(parsed.value);
135
+ }
136
+ return accumulator.finish();
137
+ }
118
138
  let inputTokens;
119
139
  let cachedInputTokens;
120
140
  let cacheWriteInputTokens;
@@ -126,7 +146,6 @@ export function parseProviderTelemetry(agent, lines) {
126
146
  const harnessTotals = agent === 'opencode' || agent === 'kilo'
127
147
  ? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
128
148
  : undefined;
129
- let piUsage;
130
149
  for (const line of lines) {
131
150
  let parsed;
132
151
  try {
@@ -138,7 +157,7 @@ export function parseProviderTelemetry(agent, lines) {
138
157
  if (!parsed || typeof parsed !== 'object' || Array.isArray(parsed))
139
158
  continue;
140
159
  const event = parsed;
141
- if (agent === 'qwen' && event.parent_tool_use_id != null)
160
+ if ((agent === 'qwen' || agent === 'claude') && event.parent_tool_use_id != null)
142
161
  continue;
143
162
  const message = event.message && typeof event.message === 'object' ? event.message : undefined;
144
163
  const stats = event.stats && typeof event.stats === 'object' ? event.stats : undefined;
@@ -177,8 +196,6 @@ export function parseProviderTelemetry(agent, lines) {
177
196
  harnessTotals.hasCost = true;
178
197
  }
179
198
  }
180
- if (agent === 'pi' && event.type === 'message_update' && event.usage && typeof event.usage === 'object')
181
- piUsage = event.usage;
182
199
  const usage = (event.usage && typeof event.usage === 'object'
183
200
  ? event.usage
184
201
  : message?.usage && typeof message.usage === 'object'
@@ -247,26 +264,6 @@ export function parseProviderTelemetry(agent, lines) {
247
264
  if (typeof eventModel === 'string' && eventModel && reportedModels.length <= 1)
248
265
  model = eventModel;
249
266
  }
250
- if (piUsage) {
251
- const piInput = finite(piUsage.input);
252
- const piOutput = finite(piUsage.output);
253
- const piCached = finite(piUsage.cacheRead);
254
- const piCacheWrite = finite(piUsage.cacheWrite);
255
- const piReasoning = finite(piUsage.reasoning);
256
- const piCost = isRecord(piUsage.cost) ? finite(piUsage.cost.total) : undefined;
257
- if (piInput !== undefined)
258
- inputTokens = piInput;
259
- if (piOutput !== undefined)
260
- outputTokens = piOutput;
261
- if (piCached !== undefined)
262
- cachedInputTokens = piCached;
263
- if (piCacheWrite !== undefined)
264
- cacheWriteInputTokens = piCacheWrite;
265
- if (piReasoning !== undefined)
266
- reasoningOutputTokens = piReasoning;
267
- if (piCost !== undefined)
268
- totalCostUsd = piCost;
269
- }
270
267
  if (harnessTotals) {
271
268
  if (harnessTotals.hasInput)
272
269
  inputTokens = harnessTotals.input;