@hecer/yoke 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/.claude-plugin/plugin.json +3 -3
  2. package/.codex-plugin/plugin.json +2 -2
  3. package/CHANGELOG.md +41 -0
  4. package/README.md +42 -30
  5. package/canon/loop/prd.schema.md +2 -2
  6. package/canon/manifest.yaml +2 -2
  7. package/canon/skills/authoring-prd/SKILL.md +1 -1
  8. package/canon/skills/yoke-retrofit/SKILL.md +2 -2
  9. package/canon/skills/yoke-workflow/SKILL.md +1 -1
  10. package/canon/tools/graphify.md +1 -1
  11. package/canon/tools/playwright-mcp.md +1 -1
  12. package/canon/tools/serena.md +1 -1
  13. package/dist/agents/catalog.js +7 -0
  14. package/dist/agents/contracts.js +3 -1
  15. package/dist/agents/host.js +4 -0
  16. package/dist/agents/process-streams.js +62 -0
  17. package/dist/agents/process.js +43 -3
  18. package/dist/agents/providers.js +45 -3
  19. package/dist/agents/telemetry.js +99 -2
  20. package/dist/canon/manifest.js +2 -1
  21. package/dist/change/inbox.js +1 -1
  22. package/dist/cli.js +22 -24
  23. package/dist/dashboard/analytics.js +193 -29
  24. package/dist/dashboard/contracts.js +23 -0
  25. package/dist/dashboard/page.js +39 -94
  26. package/dist/dashboard/panels.js +87 -33
  27. package/dist/dashboard/server.js +190 -15
  28. package/dist/goals/command.js +3 -2
  29. package/dist/loop/claims.js +2 -1
  30. package/dist/loop/decision.js +3 -2
  31. package/dist/loop/parallel-command.js +4 -2
  32. package/dist/loop/prd.js +2 -1
  33. package/dist/loop/reporter.js +1 -0
  34. package/dist/loop/run-command.js +31 -10
  35. package/dist/observability/events.js +1 -1
  36. package/dist/observability/history.js +1 -1
  37. package/dist/prd/command.js +3 -3
  38. package/dist/quality/candidate-comparison.js +6 -1
  39. package/dist/quality/command.js +17 -2
  40. package/dist/quality/types.js +6 -1
  41. package/dist/retrofit/apply.js +87 -1
  42. package/dist/retrofit/config.js +8 -0
  43. package/dist/retrofit/detect.js +6 -0
  44. package/dist/retrofit/plan.js +6 -0
  45. package/dist/retrofit/planners/kilo.js +44 -0
  46. package/dist/retrofit/planners/opencode.js +44 -0
  47. package/dist/retrofit/planners/pi.js +24 -0
  48. package/dist/retrofit/skill-actions.js +3 -0
  49. package/dist/retrofit/tools.js +8 -0
  50. package/dist/review/command.js +3 -2
  51. package/dist/review/verdict.js +1 -1
  52. package/dist/routing/capability.js +2 -2
  53. package/dist/routing/planning.js +2 -0
  54. package/dist/routing/registry.js +3 -1
  55. package/dist/routing/router.js +7 -3
  56. package/dist/setup/command.js +13 -3
  57. package/docs/DASHBOARD-EVOLUTION.md +16 -2
  58. package/docs/DASHBOARD-OVERHAUL.md +146 -0
  59. package/docs/HARNESSES.md +81 -0
  60. package/gemini-extension.json +2 -2
  61. package/package.json +6 -2
@@ -2,12 +2,12 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.12.0",
6
- "description": "Cross-agent coding harness: one curated skill canon (TDD, brainstorming, plans, reviews, shipping, design verification) plus mechanical safety gates and an autonomous loop via the yoke CLI.",
5
+ "version": "1.13.0",
6
+ "description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo and Pi: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
9
9
  "repository": "https://github.com/HECer/yoke",
10
10
  "license": "MIT",
11
- "keywords": ["harness", "cross-agent", "tdd", "code-review", "autonomous-loop", "codex", "gemini-cli", "qwen-code"],
11
+ "keywords": ["harness", "cross-agent", "tdd", "code-review", "autonomous-loop", "codex", "gemini-cli", "qwen-code", "opencode", "kilo", "pi"],
12
12
  "skills": "./canon/skills/"
13
13
  }
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.12.0",
4
- "description": "Cross-agent coding discipline, mechanical gates, and release workflows",
3
+ "version": "1.13.0",
4
+ "description": "Cross-agent coding discipline, mechanical gates, and release workflows for seven supported harnesses",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
7
7
  }
package/CHANGELOG.md CHANGED
@@ -1,5 +1,46 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.14.0 — 2026-09-09
4
+
5
+ ### Added
6
+ - Add a local-first workspace control room that ranks registered projects by attention, activity, recorded tokens, reported cost, acceptance, or name, with composable search, status filters, UTC scopes, and restorable view links.
7
+ - Add workspace and project analytics with time buckets, token/call/duration/outcome summaries, provider/model/agent/role/phase/run rankings, usage comparisons, and visible measurement coverage.
8
+ - Add project live operations for task and phase state, worker metadata, bounded event timelines, safe-boundary pause/resume, append-only operator notes, and queued change requests.
9
+ - Add a responsive dashboard shell with dark/light themes, keyboard and reduced-motion support, explicit loading/empty/error/stale states, and readable narrow-screen navigation.
10
+
11
+ ### Changed
12
+ - Base dashboard views on durable local history and versioned events while keeping unknown, partial, corrupt, unavailable, and stale telemetry explicit instead of treating it as zero.
13
+ - Keep dashboard controls on the existing loop/goal runner and lock boundaries; browser requests remain typed, same-origin, loopback-only, and unable to execute arbitrary shell commands.
14
+
15
+ ### Fixed
16
+ - Scope dashboard loop verification to the intended local project and retain focused coverage for dashboard authorization, path safety, partial history, concurrency, navigation, and control behavior.
17
+ - Improve light-theme contrast for the active navigation state and keep long project names inside desktop and mobile navigation areas.
18
+
19
+ ### Migration and validation limits
20
+ - No migration is required. Start the local view with `yoke dashboard --no-register`, then register projects with `yoke projects add <path>` as needed. Existing dashboard settings remain local and authoritative.
21
+ - Validated with the dashboard tests, TypeScript lint/build, canonical manifest validation, release metadata checks, package dry run, and real local browser screenshots at desktop and mobile sizes.
22
+ - Dashboard data is limited to explicitly registered local projects and retained local telemetry. It does not provide remote multi-user access, reconstruct missing history, prove requested models were used, or execute arbitrary browser-supplied commands. Pause/resume operates only at existing safe loop/goal boundaries.
23
+ - This release targets npm package 1.14.0; npm publication is triggered by the matching published GitHub release and verified separately.
24
+
25
+ ## 1.13.0 — 2026-09-08
26
+
27
+ ### Added
28
+ - Add first-class OpenCode, Kilo and Pi coding-agent adapters across setup, retrofit, loop execution, reviews, quality critics/repairs, goals, PRD affinity and adaptive routing.
29
+ - Generate idiomatic OpenCode/Kilo skill packages, `AGENTS.md` instructions, merged MCP config and read-only reviewer agents; generate Pi skill packages and project settings without inventing unsupported MCP or sub-agent features.
30
+ - Support provider/model/variant selection for OpenCode and Kilo, and provider/model/thinking selection for Pi. Preserve provider and variant identity in routing evidence and dashboard status.
31
+ - Parse OpenCode/Kilo JSON text and per-step usage/cost events and Pi JSONL assistant/usage events, retaining partial or unknown measurements instead of treating them as free calls.
32
+
33
+ ### Changed
34
+ - Extend the shared agent contract and all CLI validation/help text from four to seven supported harnesses: Claude, Codex, Gemini, Qwen, OpenCode, Kilo and Pi.
35
+ - Add JSONC-aware merging for existing Kilo configuration files so comments, trailing commas and user-owned settings survive retrofit.
36
+ - Add OpenCode/Kilo/Pi capability-tier routing defaults and carry provider/variant choices into parallel workers and quality candidate comparison.
37
+
38
+ ### Migration and validation limits
39
+ - Run `yoke retrofit . --agent=opencode,kilo,pi` to add the new native artifacts, or use `--agent=all` for all seven harnesses. Install and authenticate each external CLI separately; Yoke does not bundle runtimes or credentials.
40
+ - OpenCode and Kilo safe execution uses their headless approval mode, while Pi uses explicit tool allowlists. None provides the same OS-level sandbox boundary as Codex/Gemini; Pi has no native MCP, sub-agent or plan layer. See [OpenCode, Kilo and Pi](docs/HARNESSES.md) before using `unsafe`.
41
+ - Regression fixtures cover invocation, routing, quality configuration, retrofit, host detection, result parsing and telemetry. No authenticated provider matrix, production sandbox equivalence, or model-quality/cost benchmark is claimed for these harnesses. Provider streams that omit final usage events remain partial or unknown.
42
+ - This release targets npm package 1.13.0; publication is triggered by the matching published GitHub release and verified separately.
43
+
3
44
  ## 1.12.0 — 2026-09-08
4
45
 
5
46
  ### Fixed
package/README.md CHANGED
@@ -1,15 +1,15 @@
1
1
  <div align="center">
2
2
 
3
- <h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.12.0/docs/assets/yoke-logo.png" alt="Yoke" width="100" height="63"></h1>
3
+ <h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.14.0/docs/assets/yoke-logo.png" alt="Yoke" width="100" height="63"></h1>
4
4
 
5
- <!-- yoke:version:start -->1.12.0<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->1213<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.14.0<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->1258<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->34<!-- yoke:skills:end -->
8
- <!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen<!-- yoke:agents:end -->
8
+ <!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi<!-- yoke:agents:end -->
9
9
 
10
- ### One harness, four agents — and zero trust in "done."
10
+ ### One harness, seven agents — and zero trust in "done."
11
11
 
12
- **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, Gemini CLI, and Qwen Code**. Its opt-in loop implements and verifies stories before committing. Independent review and browser proofs run when configured; screenshots and videos require the browser smoke gate.
12
+ **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, and Pi coding agent**. Its opt-in loop implements and verifies stories before committing. Independent review and browser proofs run when configured; screenshots and videos require the browser smoke gate.
13
13
 
14
14
  [![npm](https://img.shields.io/npm/v/%40hecer%2Fyoke?logo=npm&color=CB3837)](https://www.npmjs.com/package/@hecer/yoke)
15
15
  [![npm downloads](https://img.shields.io/npm/dm/%40hecer%2Fyoke?logo=npm)](https://www.npmjs.com/package/@hecer/yoke)
@@ -17,8 +17,8 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-1213%20defined-blue.svg)
21
- ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini%20%7C%20Qwen-8A2BE2)
20
+ ![Tests](https://img.shields.io/badge/tests-1258%20defined-blue.svg)
21
+ ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini%20%7C%20Qwen%20%7C%20OpenCode%20%7C%20Kilo%20%7C%20Pi-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
24
24
  **Install:** [`npm i -g @hecer/yoke`](https://www.npmjs.com/package/@hecer/yoke)
@@ -27,7 +27,9 @@
27
27
 
28
28
  > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
29
 
30
- **New in 1.12.0:** [Qwen Code hardening and explicit DeepSeek/Kimi API model profiles](docs/QWEN-MODEL-SUPPORT.md). Since 1.10.0, Yoke also includes [dashboard search, filters and period comparisons](docs/DASHBOARD-EVOLUTION.md), [batch task assessments with separate planning models](docs/CAPABILITY-ROUTING.md), and [Windows sandbox preflight and process supervision](docs/WINDOWS-RUNNER-VALIDATION.md). Existing routing settings remain authoritative. See the [changelog](CHANGELOG.md) for migration and validation limits.
30
+ **New in 1.13.0:** first-class [OpenCode, Kilo and Pi integrations](docs/HARNESSES.md), including native headless invocation, provider/model/variant routing, retrofit artifacts, role configuration and provider telemetry. [Qwen Code hardening and explicit DeepSeek/Kimi API model profiles](docs/QWEN-MODEL-SUPPORT.md) remain available. Since 1.10.0, Yoke also includes [dashboard search, filters and period comparisons](docs/DASHBOARD-EVOLUTION.md), [batch task assessments with separate planning models](docs/CAPABILITY-ROUTING.md), and [Windows sandbox preflight and process supervision](docs/WINDOWS-RUNNER-VALIDATION.md). Existing routing settings remain authoritative. See the [changelog](CHANGELOG.md) and the [harness integration guide](docs/HARNESSES.md) for limitations and setup.
31
+
32
+ OpenCode, Kilo and Pi are real CLI integrations, not bundled runtimes or credentials. OpenCode/Kilo use their JSON headless modes and local MCP configuration; Pi uses JSONL and explicit tool allowlists, but has no native MCP, sub-agent or plan layer. Read the [integration guide](docs/HARNESSES.md) before selecting a permission profile.
31
33
 
32
34
  ### One dashboard, multiple projects
33
35
 
@@ -38,11 +40,11 @@ yoke projects add /path/to/backend
38
40
  yoke dashboard --no-register
39
41
  ```
40
42
 
41
- Open the printed `http://127.0.0.1:...` URL. Each registered project has its own goals, tasks and evidence. The dashboard shows available worker state, per-task duration estimates, planned start offsets, input/output tokens, costs and unknown measurements. You can request a goal pause at a safe boundary.
43
+ Open the printed `http://127.0.0.1:...` URL. Each registered project has its own goals, tasks and evidence. The dashboard is a local control room with an overview/ranking screen, project live view, bounded **History explorer**, and Workspace analytics by UTC time bucket. It shows worker state, agent/provider/model/variant/role/phase metadata when recorded, per-task duration estimates, planned start offsets, input/output tokens, reported costs, outcomes, and explicit unknown or partial measurements. Dark and light themes, keyboard navigation, responsive layouts, and reduced-motion handling are included.
42
44
 
43
- The dashboard includes attention-first project search and filters, restorable view links, and usage comparisons against the preceding period. See [dashboard behavior and measurement limits](docs/DASHBOARD-EVOLUTION.md).
45
+ The dashboard includes attention-first project search and filters, project ranking by attention, last activity, token usage, reported cost, acceptance, or name, restorable view links, and usage comparisons against the preceding period. From a project’s live view you can request a safe-boundary pause or resume, add an operator note, and use **Queue a change** for the next planning boundary. See [dashboard behavior and measurement limits](docs/DASHBOARD-EVOLUTION.md) and the [dashboard overhaul contract](docs/DASHBOARD-OVERHAUL.md).
44
46
 
45
- Projects are registered explicitly; this version does not automatically discover every process or aggregate other computers. Start/resume and budget changes use the CLI. Missing history appears as unknown; time ranges are empirical estimates, not exact deadlines.
47
+ Projects are registered explicitly; this version does not automatically discover every process or aggregate other computers. Dashboard controls call the existing goal/loop pause and resume boundaries and never execute arbitrary shell commands. Change requests are append-only pending inbox entries, not immediate code changes. The server stays loopback-only and POST actions require same-origin session authorization; the local Yoke process remains the authority for execution. Missing history appears as unknown; time ranges are empirical estimates, not exact deadlines. Read the [overhaul contract](docs/DASHBOARD-OVERHAUL.md) for data limits and non-goals.
46
48
 
47
49
  ### Verified goals and efficient execution
48
50
 
@@ -100,7 +102,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
100
102
  | 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
101
103
  | 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
102
104
 
103
- **Who it's for:** anyone driving Claude Code, Codex CLI, Gemini CLI, or Qwen Code on real projects — especially if you use more than one, want autonomous runs you can trust, or are tired of "done" meaning "probably". Greenfield (`yoke new`) and brownfield (`yoke retrofit`) both work.
105
+ **Who it's for:** anyone driving Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo or Pi on real projects — especially if you use more than one, want autonomous runs you can trust, or are tired of "done" meaning "probably". Greenfield (`yoke new`) and brownfield (`yoke retrofit`) both work.
104
106
 
105
107
  **Who it's not for:** if you want a chat pair-programmer with no process, you don't need a harness. Yoke is for shipping with discipline.
106
108
 
@@ -155,7 +157,7 @@ The canon is also packaged as a Claude Code plugin — the repo is its own marke
155
157
  /plugin install yoke@yoke
156
158
  ```
157
159
 
158
- That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini/Qwen) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
160
+ That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini/Qwen/OpenCode/Kilo/Pi) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
159
161
 
160
162
  For Codex, no preinstalled skill is required: run `npx @hecer/yoke setup .` in a terminal, or
161
163
  ask Codex to run the six-question Yoke setup flow. The retrofit writes native skills to
@@ -182,7 +184,7 @@ Auto-upgrade is deliberately **not** the default: a gate harness shouldn't chang
182
184
 
183
185
  ## 🤖 Driving it through an agent
184
186
 
185
- Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, Gemini CLI, and Qwen Code):
187
+ Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, and Pi):
186
188
 
187
189
  > **Set it up** — *"Set up Yoke in this project. Ask me the Yoke setup questions one at a time with your recommendation, then run `yoke setup . --yes` with the selected host, agents, code graph, loop, runner, and decision policy. Commit in my configured identity."*
188
190
 
@@ -206,10 +208,10 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
206
208
  | `yoke projects add\|list\|remove` | Register a project, list registrations or remove a reference by ID | `0` · `2` invalid/unavailable |
207
209
  | `yoke check [dir] [--json] [--requirement=] [--protect [--refresh]]` | Execute acceptance checks or explicitly pin their infrastructure | `0` passed/pinned · `1` failed · `2` unverified/unavailable |
208
210
  | `yoke goal set\|run\|resume\|pause\|status\|handoff\|budget [dir]` | Durable objectives, provider handoff, protected checks and checkpoint budgets | run/resume: `0` complete · `1` unfinished · `2` unavailable |
209
- | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for Claude, Codex, Gemini and Qwen; optional DeepSeek/Kimi API profiles run through Qwen | `0` · `1` invalid setup |
211
+ | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for all seven harnesses; optional DeepSeek/Kimi API profiles run through Qwen | `0` · `1` invalid setup |
210
212
  | `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
211
213
  | `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
212
- | `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
214
+ | `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen,opencode,kilo,pi\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
213
215
  | `yoke prd draft [dir] --idea= [--runner=] [--force]` | Idea → 5–12 stories with testable acceptance criteria | `0` · `1` invalid/guarded · `2` agent unavailable |
214
216
  | `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
215
217
  | `yoke change add\|status [dir] [--idea=]` | Queue a change at any time; the loop turns it into append-only stories at the next safe boundary | `0` · `1` invalid inbox/request |
@@ -229,7 +231,7 @@ Three excellent projects, three different jobs. Honest version:
229
231
  | | [superpowers](https://github.com/obra/superpowers) (obra) | [gstack](https://github.com/garrytan/gstack) (Garry Tan) | **Yoke** |
230
232
  |---|---|---|---|
231
233
  | **What it is** | The canonical *skills methodology*: brainstorm → plan → TDD → review as composable skills | A *software factory* for Claude Code: ~40 role skills (QA, CSO, ship…) + a real Chromium browser layer | A *cross-agent harness*: one canon → native installs, plus a gated autonomous loop |
232
- | **Agents** | Claude Code first | Claude Code + hosts like Codex/Cursor/Kiro — **no Gemini CLI** | **Claude Code, Codex CLI, Gemini CLI** from one source of truth |
234
+ | **Agents** | Claude Code first | Claude Code + hosts like Codex/Cursor/Kiro — **no Gemini CLI** | **Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi** from one source of truth |
233
235
  | **Enforcement** | Advisory — skills *describe* the discipline; following them is up to the agent | Skill-driven; browser QA is genuinely real | **Mechanical** — gates live in code: clean tree, acceptance criteria, green tests, review verdict, commit integrity |
234
236
  | **Autonomy** | Interactive sessions | Interactive slash-commands (`/qa`, `/ship`, …) | Opt-in **Ralph loop** with watchdog, worktree isolation, single-flight lock, per-story proofs |
235
237
  | **Visual QA** | — | **Best-in-class**: live browser daemon (Chromium/CDP) with deep interactive QA | Built-in `flow-smoke` gate: screenshots always, video on failure, labelled per story — lighter, but *enforced* and cross-agent |
@@ -237,7 +239,7 @@ Three excellent projects, three different jobs. Honest version:
237
239
  | **Footprint** | Markdown skills (plugin) | ~230 MB with browser runtime; hourly auto-update | Node CLI + markdown canon; Playwright only if you use flow-smoke, resolved **from your project** |
238
240
  | **License** | MIT | MIT | MIT |
239
241
 
240
- **They compose — use all three where they're strongest.** Yoke's canon *ships* the superpowers methodology natively for all four agents (13 skills, [attributed](canon/skills/ATTRIBUTION.md)). And if gstack is installed, `yoke retrofit` detects it and adds a routing note to `CLAUDE.md` telling Claude to prefer gstack's live-browser `/qa`, `/cso`, and ship pipeline for what Yoke deliberately doesn't bundle — no dependency, no conflict, and Codex/Gemini artifacts stay uniform.
242
+ **They compose — use all three where they're strongest.** Yoke's canon *ships* the superpowers methodology natively for all seven agents (13 skills, [attributed](canon/skills/ATTRIBUTION.md)). And if gstack is installed, `yoke retrofit` detects it and adds a routing note to `CLAUDE.md` telling Claude to prefer gstack's live-browser `/qa`, `/cso`, and ship pipeline for what Yoke deliberately doesn't bundle — no dependency, no conflict, and non-Claude artifacts stay uniform.
241
243
 
242
244
  **Choose Yoke when** you run more than one agent, want autonomy you can audit (gates + proofs + logs), or want one place to maintain your team's methodology. **Choose gstack when** you live 100% in Claude Code and want the deepest interactive browser QA. **Choose superpowers when** you want the methodology alone, interactively, in Claude Code — or just use it *through* Yoke.
243
245
 
@@ -254,11 +256,17 @@ flowchart TD
254
256
  Skill --> Codex["Codex CLI<br/>AGENTS.md · config.toml · RTK.md"]
255
257
  Skill --> Gemini["Gemini CLI<br/>GEMINI.md · commands · settings.json"]
256
258
  Skill --> Qwen["Qwen Code<br/>QWEN.md · skills · settings.json"]
259
+ Skill --> OpenCode["OpenCode<br/>AGENTS.md · opencode.json · .opencode/skills"]
260
+ Skill --> Kilo["Kilo<br/>AGENTS.md · kilo.jsonc · .kilo/skills"]
261
+ Skill --> Pi["Pi<br/>AGENTS.md · .pi/settings.json · .pi/skills"]
257
262
  Loop["🤖 yoke loop — autonomous Ralph loop<br/>gates · verify · review · isolation · proofs"]
258
263
  Claude -. drives .-> Loop
259
264
  Codex -. drives .-> Loop
260
265
  Gemini -. drives .-> Loop
261
266
  Qwen -. drives .-> Loop
267
+ OpenCode -. drives .-> Loop
268
+ Kilo -. drives .-> Loop
269
+ Pi -. drives .-> Loop
262
270
  ```
263
271
 
264
272
  Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`) → **Loop** (`yoke loop`) — on top of a durable **Context layer** (`yoke context`).
@@ -271,6 +279,9 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
271
279
  | **Codex** | Complete skill packages under `.agents/skills/`, per-skill implicit-invocation policy, `AGENTS.md`, `RTK.md`, `.codex/config.toml`, native hooks, reusable `.codex/agents/*.toml`, and package plugin metadata |
272
280
  | **Gemini** | Complete skill packages under `.gemini/skills/`, an auto-invocation index, `GEMINI.md`, `.gemini/commands/*.toml`, and `.gemini/settings.json` (MCP + `AGENTS.md` context) |
273
281
  | **Qwen** | Complete skill packages under `.qwen/skills/`, `QWEN.md`, `.qwen/settings.json`, native invocation restrictions, and an RTK PreToolUse retry guard |
282
+ | **OpenCode** | Complete skill packages under `.opencode/skills/`, shared `AGENTS.md`, merged `opencode.json` (instructions + local MCP), and `.opencode/agents/yoke-reviewer.md` |
283
+ | **Kilo** | Complete skill packages under `.kilo/skills/`, shared `AGENTS.md`, merged `kilo.jsonc` (instructions + local MCP), and `.kilo/agents/yoke-reviewer.md` |
284
+ | **Pi** | Complete skill packages under `.pi/skills/`, shared `AGENTS.md`, and merged `.pi/settings.json`; Pi's tool allowlists are applied at invocation time |
274
285
 
275
286
  > **rtk integration:** Claude receives its PreToolUse hook; Codex receives a native hook adapter around `rtk hook check`; Gemini retains instruction-mode fallback where its CLI has no equivalent command-rewrite lifecycle.
276
287
 
@@ -285,13 +296,14 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
285
296
 
286
297
  ## 🧰 What's in the canon — 34 skills
287
298
 
288
- `yoke retrofit` installs all of these into each agent natively. Provenance is credited in [`canon/skills/ATTRIBUTION.md`](canon/skills/ATTRIBUTION.md).
299
+ `yoke retrofit` installs all of these into each selected agent natively. Provenance is credited in [`canon/skills/ATTRIBUTION.md`](canon/skills/ATTRIBUTION.md).
289
300
 
290
- To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.md` carries a **skill routing & precedence** block (methodology before role; one canonical entrypoint per concern — e.g. pre-merge code review is always `review`), emitted into all four agents.
301
+ To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.md` carries a **skill routing & precedence** block (methodology before role; one canonical entrypoint per concern — e.g. pre-merge code review is always `review`), emitted into all seven agents.
291
302
 
292
303
  Each manifest entry also declares `invocation: auto|manual`. Retrofit translates that intent into
293
304
  the provider's native controls: Claude and Qwen disable model invocation for manual skills, Codex writes
294
- `agents/openai.yaml`, and Gemini lists only automatic skills in its generated index. Validation
305
+ `agents/openai.yaml`, Gemini lists only automatic skills in its generated index, and OpenCode/Kilo/Pi
306
+ receive complete project-local skill packages. Validation
295
307
  rejects conflicting package metadata and broken local Markdown links before anything is installed.
296
308
 
297
309
  **Process / methodology** — *superpowers-derived discipline (13)*
@@ -587,16 +599,16 @@ new projects should use `decisionPolicy: auto|critical`.
587
599
  worker profiles, automatic execution keeps the selected parent. When enabled, the selected
588
600
  parent remains the strong planner/controller. Before each bounded story it receives only the
589
601
  story, acceptance criteria, and at most three eligible worker profiles, then returns one
590
- machine-readable choice. The worker can be a cheaper/faster Claude, Codex, or Gemini profile;
602
+ machine-readable choice. The worker can be a cheaper/faster profile for any configured harness;
591
603
  `SELF` keeps difficult work on the parent. Explicit project rules skip the controller.
592
- Loop runners disable native delegation in Codex, Claude, Gemini and Qwen so it cannot multiply
604
+ Loop runners disable native delegation in Codex, Claude, Gemini, Qwen, OpenCode and Kilo so it cannot multiply
593
605
  the Yoke worker budget. Integration retains its execution slot until the candidate lands.
594
606
 
595
607
  **Provider support:** adaptive routing uses Yoke's shared provider adapter and works with Claude
596
- Code, Codex CLI, Gemini CLI, and Qwen Code, including mixed-provider worker lists. Internal contract tests
597
- cover invocation and routing behavior for all four providers. The measured performance evidence
598
- below is intentionally **Codex-only**; it does not claim equivalent Claude, Gemini or Qwen savings
599
- until authenticated, repeated in-the-wild runs exist for those providers.
608
+ Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo and Pi, including mixed-provider worker lists.
609
+ Internal contract tests cover invocation and routing behavior for all seven providers. The measured
610
+ performance evidence below is intentionally **Codex-only**; it does not claim equivalent savings
611
+ until authenticated, repeated in-the-wild runs exist for each provider.
600
612
 
601
613
  ```yaml
602
614
  runner:
@@ -892,7 +904,7 @@ canon/ # the source of truth — harness-agnostic
892
904
  src/
893
905
  canon/ # manifest schema + validator (yoke validate)
894
906
  change/ # append-only change inbox · planning · independent coverage review
895
- retrofit/ # detect · plan · apply · planners (claude/codex/gemini) · tools
907
+ retrofit/ # detect · plan · apply · planners (all seven harnesses) · tools
896
908
  loop/ # prd · gates · runner · verify · git/worktree · loop · run-command · lock · cleanup
897
909
  quality/ # reference collection · blind critic · bounded repair · candidate comparison
898
910
  new/ # yoke new — greenfield bootstrap
@@ -913,7 +925,7 @@ release provenance.
913
925
  ## 🧪 Development
914
926
 
915
927
  ```bash
916
- npm test # vitest (1213 tests)
928
+ npm test # vitest (1258 tests)
917
929
  npm run build # tsc, no emit errors
918
930
  npm run yoke -- validate canon
919
931
  ```
@@ -8,7 +8,7 @@ The loop is driven by a continuous PRD backlog. Each story:
8
8
  priority: 1 # lower = higher priority
9
9
  needs: [] # optional dependency IDs; no unknown IDs, self-links, or cycles
10
10
  area: api # optional collision domain for parallel scheduling
11
- agent: codex # optional claude|codex|gemini affinity
11
+ agent: codex # optional supported harness affinity
12
12
  acceptance:
13
13
  - id: valid-request-returns-200
14
14
  text: The endpoint returns 200 for a valid request.
@@ -42,7 +42,7 @@ dispatch blocks missing or stale assessments. See [capability routing](../../doc
42
42
 
43
43
  Stories without `needs`, `area`, or `agent` retain serial behavior. A story is ready only when
44
44
  every ID in `needs` passes. The scheduler orders ready work by priority, avoids simultaneously
45
- active areas, and uses `agent` as an affinity hint.
45
+ active areas, and uses `agent` as a supported harness affinity hint.
46
46
 
47
47
  The backlog is continuous, not a release object. A momentary stop condition is every story
48
48
  having `passes: true`; if configured, `completion.command` must then prove the integrated
@@ -1,6 +1,6 @@
1
1
  name: yoke-canon
2
- version: 1.12.0
3
- agents: [claude, codex, gemini, qwen]
2
+ version: 1.14.0
3
+ agents: [claude, codex, gemini, qwen, opencode, kilo, pi]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
6
6
  - { id: yoke-retrofit, path: skills/yoke-retrofit, kind: methodology, invocation: auto }
@@ -21,7 +21,7 @@ new stories; they do not require a release object.
21
21
  7. Resolve planning questions before unattended execution. `yoke prd check` rejects unresolved
22
22
  placeholders; critical irreversible choices use the structured decision channel.
23
23
  8. Use `needs` only for hard prerequisites, `area` for collision domains, and `agent` only as
24
- a Claude/Codex/Gemini affinity hint.
24
+ a supported harness affinity hint.
25
25
  9. Keep planning on the start model. Add an `assessment` to each story: `taskClass`
26
26
  (`mechanical`, `implementation`, `debugging`, `architecture`), `difficulty`, `uncertainty`,
27
27
  `risk`, `scope`, `testability` (each `low`, `medium`, `high`), a concise `reason`, and
@@ -1,13 +1,13 @@
1
1
  ---
2
2
  name: yoke-retrofit
3
- description: Use when asked to "retrofit", "yoke this project", or set up the Yoke harness in a project — runs the shared setup wizard and configures the same behavior for Claude, Codex, and Gemini.
3
+ description: Use when asked to "retrofit", "yoke this project", or set up the Yoke harness in a project — runs the shared setup wizard and configures the same behavior for the supported Yoke harnesses.
4
4
  ---
5
5
 
6
6
  # Yoke Retrofit
7
7
 
8
8
  Set up or update Yoke through the shared `yoke setup` contract.
9
9
 
10
- 1. Inspect the project and identify the current host (`claude`, `codex`, or `gemini`).
10
+ 1. Inspect the project and identify the current host (`claude`, `codex`, `gemini`, `qwen`, `opencode`, `kilo`, or `pi`).
11
11
  2. Ask these setup questions one at a time and give a direct recommendation:
12
12
  - target agents (recommend the current host; use `all` for deliberately cross-agent projects),
13
13
  - code-graph tool,
@@ -5,7 +5,7 @@ description: Use when the user asks Yoke to plan and build a feature, run storie
5
5
 
6
6
  # Yoke Workflow
7
7
 
8
- Provide the same interaction contract in Claude, Codex, and Gemini.
8
+ Provide the same interaction contract in every supported Yoke harness.
9
9
 
10
10
  1. Read `.yoke/config.yaml`. If it is missing, offer `yoke setup . --host=<current-agent>` and run the setup flow before planning.
11
11
  2. Plan before starting the loop. Inspect the project, then ask one focused question at a time only where the answer changes product behavior, scope, architecture, security, data ownership, external cost, or an irreversible choice. Include a recommended answer. Resolve routine implementation details yourself.
@@ -1,3 +1,3 @@
1
1
  # Tool: graphify (code-graph)
2
2
 
3
- MIT, multimodal code/doc graph. Wired as an MCP server for all three agents (stdio). Prefer symbol/graph lookups over reading whole files. Caveat: heuristic edges (INFERRED/AMBIGUOUS) and a static index that can go stale — rebuild on significant changes.
3
+ MIT, multimodal code/doc graph. Wired as an MCP server for Claude, Codex, Gemini, Qwen, OpenCode and Kilo (stdio). Pi has no native MCP layer. Prefer symbol/graph lookups over reading whole files. Caveat: heuristic edges (INFERRED/AMBIGUOUS) and a static index that can go stale — rebuild on significant changes.
@@ -1,3 +1,3 @@
1
1
  # Tool: Playwright MCP (browser / dogfooding)
2
2
 
3
- Microsoft Playwright MCP, Apache-2.0. Wired as an MCP server for all three agents — the only browser tool with native MCP parity across Claude/Codex/Gemini. Used for QA, dogfooding user flows, screenshots, and deploy verification.
3
+ Microsoft Playwright MCP, Apache-2.0. Wired as an MCP server for the harnesses that support native MCP configuration (Claude, Codex, Gemini, Qwen, OpenCode and Kilo). Pi has no native MCP layer, so Pi projects use the portable skills and explicit Yoke gates instead. Used for QA, dogfooding user flows, screenshots, and deploy verification.
@@ -2,7 +2,7 @@
2
2
 
3
3
  MIT, MCP-first. The alternative to graphify, selected via `yoke retrofit --code-graph=serena`. Serena uses real language servers (LSP) for symbol-accurate, cross-file retrieval and refactoring (`find_symbol`, `find_referencing_symbols`, rename/move) — no static index that goes stale, so it will not miss a reference.
4
4
 
5
- Wired as an MCP server for all three agents. Best for large, strongly-typed codebases (TypeScript, Python, Go) doing systematic refactoring, where missing a caller is costly.
5
+ Wired as an MCP server for Claude, Codex, Gemini, Qwen, OpenCode and Kilo. Pi has no native MCP layer. Best for large, strongly-typed codebases (TypeScript, Python, Go) doing systematic refactoring, where missing a caller is costly.
6
6
 
7
7
  Caveat: needs one language server per language (can be fiddly on Windows for exotic languages) and requires `uv`. The launch command is a best-effort template — adjust to your install, e.g. `uvx --from git+https://github.com/oraios/serena serena-mcp-server`.
8
8
 
@@ -0,0 +1,7 @@
1
+ import { AgentSchema } from './contracts.js';
2
+ /** Single source of truth for CLI help, setup prompts, and default resolution. */
3
+ export const SUPPORTED_AGENTS = AgentSchema.options;
4
+ export const AGENT_LIST = SUPPORTED_AGENTS.join(',');
5
+ export function isSupportedAgent(value) {
6
+ return SUPPORTED_AGENTS.includes(value);
7
+ }
@@ -1,9 +1,11 @@
1
1
  import { z } from 'zod';
2
- export const AgentSchema = z.enum(['claude', 'codex', 'gemini', 'qwen']);
2
+ export const AgentSchema = z.enum(['claude', 'codex', 'gemini', 'qwen', 'opencode', 'kilo', 'pi']);
3
3
  export const PermissionProfileSchema = z.enum(['safe', 'unsafe', 'read-only']);
4
4
  export const ModelSelectionSchema = z.object({
5
+ provider: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/).optional(),
5
6
  model: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,127}$/).optional(),
6
7
  reasoningEffort: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$/).optional(),
8
+ variant: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$/).optional(),
7
9
  nativeMultiAgent: z.boolean().optional(),
8
10
  bare: z.boolean().optional(),
9
11
  });
@@ -9,6 +9,10 @@ export function detectHostAgent(env = process.env) {
9
9
  return 'gemini';
10
10
  if (env.QWEN_CODE || env.QWEN_CODE_SESSION_ID || env.QWEN_CLI)
11
11
  return 'qwen';
12
+ if (env.OPENCODE_CLIENT || env.OPENCODE_CONFIG || env.OPENCODE_CONFIG_DIR)
13
+ return 'opencode';
14
+ if (env.KILO_CLIENT || env.KILO_CONFIG || env.KILO_CONFIG_DIR)
15
+ return 'kilo';
12
16
  if (env.CODEX_HOME)
13
17
  return 'codex';
14
18
  if (env.CLAUDE_CONFIG_DIR)
@@ -20,6 +20,9 @@ export function createTelemetryAccumulator(agent) {
20
20
  let trailing = '';
21
21
  let telemetry = { usageAvailable: false };
22
22
  let reportedModels = [];
23
+ const stepTotals = agent === 'opencode' || agent === 'kilo'
24
+ ? { input: 0, output: 0, cached: 0, cacheWrite: 0, reasoning: 0, cost: 0, hasInput: false, hasOutput: false, hasCached: false, hasCacheWrite: false, hasReasoning: false, hasCost: false }
25
+ : undefined;
23
26
  const update = (lines) => {
24
27
  for (const line of lines) {
25
28
  const next = parseProviderTelemetry(agent, [line]);
@@ -31,6 +34,35 @@ export function createTelemetryAccumulator(agent) {
31
34
  // never add it to earlier results or to assistant-message snapshots.
32
35
  if (next.tokens || next.partialUsage)
33
36
  telemetry = next;
37
+ if (stepTotals && isStepFinish(line)) {
38
+ const usage = next.tokens ?? next.partialUsage;
39
+ if (usage) {
40
+ if (usage.inputTokens !== undefined) {
41
+ stepTotals.input += usage.inputTokens;
42
+ stepTotals.hasInput = true;
43
+ }
44
+ if (usage.outputTokens !== undefined) {
45
+ stepTotals.output += usage.outputTokens;
46
+ stepTotals.hasOutput = true;
47
+ }
48
+ if (usage.cachedInputTokens !== undefined) {
49
+ stepTotals.cached += usage.cachedInputTokens;
50
+ stepTotals.hasCached = true;
51
+ }
52
+ if (usage.cacheWriteInputTokens !== undefined) {
53
+ stepTotals.cacheWrite += usage.cacheWriteInputTokens;
54
+ stepTotals.hasCacheWrite = true;
55
+ }
56
+ if (usage.reasoningOutputTokens !== undefined) {
57
+ stepTotals.reasoning += usage.reasoningOutputTokens;
58
+ stepTotals.hasReasoning = true;
59
+ }
60
+ if (usage.totalCostUsd !== undefined) {
61
+ stepTotals.cost += usage.totalCostUsd;
62
+ stepTotals.hasCost = true;
63
+ }
64
+ }
65
+ }
34
66
  }
35
67
  };
36
68
  return {
@@ -43,6 +75,26 @@ export function createTelemetryAccumulator(agent) {
43
75
  if (trailing)
44
76
  update([trailing]);
45
77
  trailing = '';
78
+ if (stepTotals && (stepTotals.hasInput || stepTotals.hasOutput)) {
79
+ const latest = telemetry.tokens;
80
+ const inputTokens = stepTotals.hasInput ? stepTotals.input : latest?.inputTokens;
81
+ const outputTokens = stepTotals.hasOutput ? stepTotals.output : latest?.outputTokens;
82
+ const partialUsage = {
83
+ ...(inputTokens !== undefined ? { inputTokens } : {}),
84
+ ...(outputTokens !== undefined ? { outputTokens } : {}),
85
+ ...(stepTotals.hasCached ? { cachedInputTokens: stepTotals.cached } : latest?.cachedInputTokens !== undefined ? { cachedInputTokens: latest.cachedInputTokens } : {}),
86
+ ...(stepTotals.hasCacheWrite ? { cacheWriteInputTokens: stepTotals.cacheWrite } : latest?.cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens: latest.cacheWriteInputTokens } : {}),
87
+ ...(stepTotals.hasReasoning ? { reasoningOutputTokens: stepTotals.reasoning } : latest?.reasoningOutputTokens !== undefined ? { reasoningOutputTokens: latest.reasoningOutputTokens } : {}),
88
+ ...(stepTotals.hasCost ? { totalCostUsd: stepTotals.cost } : latest?.totalCostUsd !== undefined ? { totalCostUsd: latest.totalCostUsd } : {}),
89
+ ...(latest?.model ? { model: latest.model } : {}),
90
+ };
91
+ if (inputTokens !== undefined && outputTokens !== undefined) {
92
+ telemetry = { usageAvailable: true, tokens: { ...partialUsage, inputTokens, outputTokens } };
93
+ }
94
+ else {
95
+ telemetry = { usageAvailable: false, partialUsage };
96
+ }
97
+ }
46
98
  if (telemetry.tokens) {
47
99
  const { model: _model, ...tokens } = telemetry.tokens;
48
100
  return { usageAvailable: telemetry.usageAvailable, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
@@ -52,3 +104,13 @@ export function createTelemetryAccumulator(agent) {
52
104
  },
53
105
  };
54
106
  }
107
+ function isStepFinish(line) {
108
+ try {
109
+ const value = JSON.parse(line);
110
+ const part = value.part && typeof value.part === 'object' ? value.part : undefined;
111
+ return value.type === 'step_finish' || part?.type === 'step-finish';
112
+ }
113
+ catch {
114
+ return false;
115
+ }
116
+ }
@@ -58,7 +58,32 @@ export function startProviderProcess(agent, invocation, options = {}) {
58
58
  catch {
59
59
  return true;
60
60
  }
61
- if (startedAt.startsWith('unverified:') || processIncarnation(processPid) !== startedAt)
61
+ // A slow or unavailable Windows identity query must not strand the exact
62
+ // child process Yoke just spawned. The live ChildProcess handle proves
63
+ // ownership more strongly than a late PID lookup. Terminate that exact
64
+ // handle immediately and ask taskkill asynchronously to catch descendants;
65
+ // waiting synchronously for taskkill can itself exceed the provider timeout
66
+ // on a heavily loaded Windows host. If the identity was verified, retain the
67
+ // PID-reuse guard for cleanup records.
68
+ if (startedAt.startsWith('unverified:')) {
69
+ if (child.exitCode !== null || child.signalCode !== null || child.killed)
70
+ return true;
71
+ try {
72
+ const killed = child.kill('SIGKILL');
73
+ if (killed && process.platform === 'win32') {
74
+ try {
75
+ const tree = spawn('taskkill', ['/PID', String(processPid), '/T', '/F'], { stdio: 'ignore', windowsHide: true });
76
+ tree.unref();
77
+ }
78
+ catch { /* direct child termination already succeeded */ }
79
+ }
80
+ return killed;
81
+ }
82
+ catch {
83
+ return false;
84
+ }
85
+ }
86
+ if (processIncarnation(processPid) !== startedAt)
62
87
  return false;
63
88
  return killProcessTreeForCleanup(processPid);
64
89
  });
@@ -76,6 +101,7 @@ export function startProviderProcess(agent, invocation, options = {}) {
76
101
  let completionTimer;
77
102
  let recordFailure;
78
103
  let terminationConfirmed = false;
104
+ let forcedTerminationAttempted = false;
79
105
  let settled = false;
80
106
  let resolveCompletion = () => { };
81
107
  const completion = new Promise(resolveCompletionValue => {
@@ -119,14 +145,24 @@ export function startProviderProcess(agent, invocation, options = {}) {
119
145
  stderrTruncated: stderr.truncated,
120
146
  telemetry: telemetry.finish(),
121
147
  });
148
+ const forceTerminateExactChild = () => {
149
+ if (child.exitCode !== null || child.signalCode !== null || child.killed)
150
+ return;
151
+ try {
152
+ child.kill('SIGKILL');
153
+ }
154
+ catch { /* the tree killer may have won the race */ }
155
+ };
122
156
  const finalize = (exitCode) => {
123
157
  if (settled)
124
158
  return;
125
159
  supervision.flush();
126
160
  // Windows can emit close before taskkill's process-tree state is observable.
127
161
  // Reconfirm here so successful termination does not leave a stale ownership record.
128
- if (termination && pid !== undefined && !terminationConfirmed) {
162
+ if (termination && pid !== undefined && !terminationConfirmed && !forcedTerminationAttempted) {
129
163
  terminationConfirmed = terminateProcessTree(pid, true);
164
+ if (terminationConfirmed)
165
+ forceTerminateExactChild();
130
166
  }
131
167
  const details = evidence();
132
168
  if (recordFailure) {
@@ -154,8 +190,12 @@ export function startProviderProcess(agent, invocation, options = {}) {
154
190
  if (pid !== undefined)
155
191
  terminationConfirmed = terminateProcessTree(pid, false);
156
192
  forceTimer = setTimeout(() => {
157
- if (pid !== undefined && !settled)
193
+ if (pid !== undefined && !settled) {
194
+ forcedTerminationAttempted = true;
158
195
  terminationConfirmed = terminateProcessTree(pid, true);
196
+ if (terminationConfirmed)
197
+ forceTerminateExactChild();
198
+ }
159
199
  // Allow close/pipe draining to confirm termination before the bounded fallback.
160
200
  if (!settled)
161
201
  completionTimer = setTimeout(() => {