@hecer/yoke 1.15.0 → 1.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +3 -3
- package/.codex-plugin/plugin.json +2 -2
- package/CHANGELOG.md +41 -0
- package/README.md +26 -21
- package/canon/AGENTS.md +7 -0
- package/canon/manifest.yaml +2 -2
- package/canon/skills/executing-plans/SKILL.md +1 -1
- package/canon/skills/requesting-code-review/SKILL.md +2 -2
- package/canon/skills/requesting-code-review/code-reviewer.md +12 -0
- package/canon/skills/subagent-driven-development/SKILL.md +7 -3
- package/canon/skills/subagent-driven-development/code-quality-reviewer-prompt.md +7 -0
- package/canon/skills/subagent-driven-development/implementer-prompt.md +7 -0
- package/canon/skills/subagent-driven-development/spec-reviewer-prompt.md +7 -0
- package/canon/skills/systematic-debugging/SKILL.md +7 -11
- package/canon/skills/systematic-debugging/condition-based-waiting.md +7 -0
- package/canon/skills/systematic-debugging/defense-in-depth.md +7 -0
- package/canon/skills/systematic-debugging/root-cause-tracing.md +9 -0
- package/canon/skills/tdd/SKILL.md +1 -1
- package/canon/skills/tdd/testing-anti-patterns.md +9 -0
- package/canon/skills/yoke-retrofit/SKILL.md +1 -1
- package/dist/agents/contracts.js +1 -1
- package/dist/agents/host.js +4 -0
- package/dist/agents/pi-telemetry.js +53 -0
- package/dist/agents/process-streams.js +13 -0
- package/dist/agents/providers.js +29 -5
- package/dist/agents/supervision.js +3 -1
- package/dist/agents/telemetry.js +38 -28
- package/dist/change/inbox.js +1 -1
- package/dist/estimation/schedule.js +40 -26
- package/dist/prd/command.js +2 -2
- package/dist/retrofit/detect.js +2 -0
- package/dist/retrofit/plan.js +2 -0
- package/dist/retrofit/planners/hermes.js +32 -0
- package/dist/retrofit/planners/pi.js +1 -1
- package/dist/retrofit/skill-actions.js +2 -1
- package/dist/review/command.js +1 -1
- package/dist/review/verdict.js +1 -1
- package/dist/routing/capability.js +1 -1
- package/dist/routing/router.js +1 -1
- package/dist/setup/command.js +3 -0
- package/docs/AGENT-HARDENING-2026-09-16.md +32 -0
- package/docs/HARNESSES.md +35 -14
- package/gemini-extension.json +2 -2
- package/package.json +3 -2
|
@@ -2,12 +2,12 @@
|
|
|
2
2
|
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
|
|
3
3
|
"name": "yoke",
|
|
4
4
|
"displayName": "Yoke",
|
|
5
|
-
"version": "1.
|
|
6
|
-
"description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo and
|
|
5
|
+
"version": "1.16.0",
|
|
6
|
+
"description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi and Hermes: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
|
|
7
7
|
"author": { "name": "HECer", "url": "https://github.com/HECer" },
|
|
8
8
|
"homepage": "https://github.com/HECer/yoke#readme",
|
|
9
9
|
"repository": "https://github.com/HECer/yoke",
|
|
10
10
|
"license": "MIT",
|
|
11
|
-
"keywords": ["harness", "cross-agent", "tdd", "code-review", "autonomous-loop", "codex", "gemini-cli", "qwen-code", "opencode", "kilo", "pi"],
|
|
11
|
+
"keywords": ["harness", "cross-agent", "tdd", "code-review", "autonomous-loop", "codex", "gemini-cli", "qwen-code", "opencode", "kilo", "pi", "hermes"],
|
|
12
12
|
"skills": "./canon/skills/"
|
|
13
13
|
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "Cross-agent coding discipline, mechanical gates, and release workflows for
|
|
3
|
+
"version": "1.16.0",
|
|
4
|
+
"description": "Cross-agent coding discipline, mechanical gates, and release workflows for eight supported harnesses",
|
|
5
5
|
"skills": "./canon/skills/",
|
|
6
6
|
"hooks": "./hooks/hooks.json"
|
|
7
7
|
}
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,46 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.16.0 — 2026-09-17
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
- Add first-class adapter support for Nous Research's Hermes Agent (`hermes` CLI) as the eighth supported harness.
|
|
7
|
+
- Add non-interactive execution for Hermes via `hermes chat --format stream-json --query-file -`.
|
|
8
|
+
- Map permission profiles to Hermes toolset flags: `safe` to `--toolsets file,terminal`, `read-only` to `--toolsets file`, and `unsafe` to `--yolo`.
|
|
9
|
+
- Add telemetry extraction for Hermes `stream-json` streams: assistant text deltas, token usage (`input`, `output`, `cache_read`, `cache_write`), total cost USD, and active model identities.
|
|
10
|
+
- Recognize successful Hermes `tool_result` events as watchdog progress activity.
|
|
11
|
+
- Add Hermes project marker detection (`.hermes`, `HERMES.md`, `hermes.yaml`, `hermes.json`) and host environment marker detection (`HERMES_SESSION_ID`, `HERMES_CONFIG`, `HERMES_HOME`).
|
|
12
|
+
- Add Hermes retrofit planner generating complete skill packages under `.hermes/skills/`, shared `AGENTS.md` instructions, and a read-only reviewer agent (`.hermes/agents/yoke-reviewer.md`).
|
|
13
|
+
- Add `hermes` to setup CLI options, routing worker presets (`hermes-standard`), fallback review providers, and runner affinity validation.
|
|
14
|
+
- Add Hermes integration documentation in `docs/HARNESSES.md`.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
- Update README, package metadata, and manifests to reflect eight supported harnesses across invocation, routing, and retrofit architecture.
|
|
18
|
+
- Synchronize Claude plugin, Codex plugin, Gemini extension, Canon, and npm package versions to 1.16.0.
|
|
19
|
+
|
|
20
|
+
### Migration and validation limits
|
|
21
|
+
- Run `yoke retrofit . --agent=hermes` or `yoke retrofit . --agent=all` to install native Hermes artifacts. External `hermes` CLI must be installed and configured separately; Yoke does not bundle runtimes or credentials.
|
|
22
|
+
- Hermes runner does not support bare startup mode or native multi-agent delegation; Yoke enforces managed worker and single-flight execution boundaries.
|
|
23
|
+
- Local verification: 1,285 tests passed, two platform-specific tests skipped, 141 test files passed; TypeScript lint/build, Canon validation, documentation metadata check, package dry run, and dependency audit passed.
|
|
24
|
+
|
|
25
|
+
## 1.15.1 — 2026-09-16
|
|
26
|
+
|
|
27
|
+
### Fixed
|
|
28
|
+
- Sum finalized Pi assistant-turn usage without counting streamed snapshots or replayed transcripts twice; preserve missing measurements and multiple model identities.
|
|
29
|
+
- Recognize successful Pi/OpenCode/Kilo tool events as watchdog progress without relaxing execution budgets.
|
|
30
|
+
- Reject structured verdicts followed by terminal provider errors, reject errored/aborted Pi verdicts, and exclude Claude child-agent verdicts/usage from parent results.
|
|
31
|
+
- Validate OpenCode/Kilo effort aliases consistently; emit a single variant flag.
|
|
32
|
+
- Correct Pi settings-relative skill discovery and enforce manual-only skill invocation using Pi's native frontmatter.
|
|
33
|
+
- Estimate schedule ranges from per-story scenarios and retain low confidence for sparse task-specific evidence.
|
|
34
|
+
- Ship eight missing delegation/review/debugging/testing resources and clarify host capability, worker-budget and verification rules in the shared skills; remove unsupported quality/speed claims.
|
|
35
|
+
- Keep coordinator unit tests offline by stubbing the separate preview backend; verify actual isolated edits and preservation of the source project.
|
|
36
|
+
- Synchronize previously stale Claude/Codex plugin, Gemini extension and Canon versions with the npm package.
|
|
37
|
+
|
|
38
|
+
### Migration and validation limits
|
|
39
|
+
- Refresh generated skills with a reviewed `yoke retrofit . --agent=all` (or the selected agent). Current Pi requires explicit project trust to load project-local resources; Yoke does not grant it automatically. See [harness setup](docs/HARNESSES.md).
|
|
40
|
+
- Existing custom configuration remains authoritative. No timeout or acceptance-gate defaults were weakened. Time ranges remain empirical, not guaranteed deadlines.
|
|
41
|
+
- See the [audit and validation limits](docs/AGENT-HARDENING-2026-09-16.md). No authenticated seven-agent benchmark or guaranteed speed/quality improvement is claimed.
|
|
42
|
+
- Local verification: 1,281 tests passed, two platform-specific tests skipped; TypeScript lint/build, Canon validation, documentation metadata, package dry run and dependency audit passed.
|
|
43
|
+
|
|
3
44
|
## 1.15.0 — 2026-09-09
|
|
4
45
|
|
|
5
46
|
### Added
|
package/README.md
CHANGED
|
@@ -1,15 +1,15 @@
|
|
|
1
1
|
<div align="center">
|
|
2
2
|
|
|
3
|
-
<h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.
|
|
3
|
+
<h1><img src="https://raw.githubusercontent.com/HECer/yoke/v1.16.0/docs/assets/yoke-logo.png" alt="Yoke" width="100" height="63"></h1>
|
|
4
4
|
|
|
5
|
-
<!-- yoke:version:start -->1.
|
|
6
|
-
<!-- yoke:tests:start -->
|
|
5
|
+
<!-- yoke:version:start -->1.16.0<!-- yoke:version:end -->
|
|
6
|
+
<!-- yoke:tests:start -->1287<!-- yoke:tests:end -->
|
|
7
7
|
<!-- yoke:skills:start -->34<!-- yoke:skills:end -->
|
|
8
|
-
<!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi<!-- yoke:agents:end -->
|
|
8
|
+
<!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi | Hermes<!-- yoke:agents:end -->
|
|
9
9
|
|
|
10
|
-
### One harness,
|
|
10
|
+
### One harness, eight agents — and zero trust in "done."
|
|
11
11
|
|
|
12
|
-
**Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo,
|
|
12
|
+
**Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi coding agent, and Hermes Agent**. Its opt-in loop implements and verifies stories before committing. Independent review and browser proofs run when configured; screenshots and videos require the browser smoke gate.
|
|
13
13
|
|
|
14
14
|
[](https://www.npmjs.com/package/@hecer/yoke)
|
|
15
15
|
[](https://www.npmjs.com/package/@hecer/yoke)
|
|
@@ -17,8 +17,8 @@
|
|
|
17
17
|
[](#-license)
|
|
18
18
|

|
|
19
19
|

|
|
20
|
-

|
|
20
|
+

|
|
21
|
+

|
|
22
22
|

|
|
23
23
|
|
|
24
24
|
**Install:** [`npm i -g @hecer/yoke`](https://www.npmjs.com/package/@hecer/yoke)
|
|
@@ -31,6 +31,8 @@
|
|
|
31
31
|
|
|
32
32
|
OpenCode, Kilo and Pi are real CLI integrations, not bundled runtimes or credentials. OpenCode/Kilo use their JSON headless modes and local MCP configuration; Pi uses JSONL and explicit tool allowlists, but has no native MCP, sub-agent or plan layer. Read the [integration guide](docs/HARNESSES.md) before selecting a permission profile.
|
|
33
33
|
|
|
34
|
+
**Fixed in 1.15.1:** multi-turn Pi usage accounting, provider error/progress handling, task-specific time ranges and missing skill resources. Pi has been supported since **1.13.0**; current Pi versions require explicit project trust for project-local skills in headless runs. The [agent/skill hardening audit](docs/AGENT-HARDENING-2026-09-16.md) documents the fixes, migration notes and validation limits.
|
|
35
|
+
|
|
34
36
|
### One dashboard, multiple projects
|
|
35
37
|
|
|
36
38
|
```sh
|
|
@@ -102,7 +104,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
|
|
|
102
104
|
| 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
|
|
103
105
|
| 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
|
|
104
106
|
|
|
105
|
-
**Who it's for:** anyone driving Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo or
|
|
107
|
+
**Who it's for:** anyone driving Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi or Hermes on real projects — especially if you use more than one, want autonomous runs you can trust, or are tired of "done" meaning "probably". Greenfield (`yoke new`) and brownfield (`yoke retrofit`) both work.
|
|
106
108
|
|
|
107
109
|
**Who it's not for:** if you want a chat pair-programmer with no process, you don't need a harness. Yoke is for shipping with discipline.
|
|
108
110
|
|
|
@@ -157,7 +159,7 @@ The canon is also packaged as a Claude Code plugin — the repo is its own marke
|
|
|
157
159
|
/plugin install yoke@yoke
|
|
158
160
|
```
|
|
159
161
|
|
|
160
|
-
That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini/Qwen/OpenCode/Kilo/Pi) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
|
|
162
|
+
That gives you all canon skills under the `yoke:` namespace (e.g. `yoke:tdd`, `yoke:review`) inside Claude Code — no retrofit needed. The `yoke` CLI (loop, gates, retrofit for Codex/Gemini/Qwen/OpenCode/Kilo/Pi/Hermes) still comes from `npm i -g @hecer/yoke`. Gemini CLI users can likewise `gemini extensions install https://github.com/HECer/yoke`.
|
|
161
163
|
|
|
162
164
|
For Codex, no preinstalled skill is required: run `npx @hecer/yoke setup .` in a terminal, or
|
|
163
165
|
ask Codex to run the six-question Yoke setup flow. The retrofit writes native skills to
|
|
@@ -184,7 +186,7 @@ Auto-upgrade is deliberately **not** the default: a gate harness shouldn't chang
|
|
|
184
186
|
|
|
185
187
|
## 🤖 Driving it through an agent
|
|
186
188
|
|
|
187
|
-
Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, and
|
|
189
|
+
Yoke is meant to be operated *by* your coding agent — after a retrofit, the agent has the skills, the safety policy, and the routing, so it knows the methodology. Copy-paste prompts (identical wording works for Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi, and Hermes):
|
|
188
190
|
|
|
189
191
|
> **Set it up** — *"Set up Yoke in this project. Ask me the Yoke setup questions one at a time with your recommendation, then run `yoke setup . --yes` with the selected host, agents, code graph, loop, runner, and decision policy. Commit in my configured identity."*
|
|
190
192
|
|
|
@@ -208,11 +210,11 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
|
|
|
208
210
|
| `yoke projects add\|list\|remove` | Register a project, list registrations or remove a reference by ID | `0` · `2` invalid/unavailable |
|
|
209
211
|
| `yoke check [dir] [--json] [--requirement=] [--protect [--refresh]]` | Execute acceptance checks or explicitly pin their infrastructure | `0` passed/pinned · `1` failed · `2` unverified/unavailable |
|
|
210
212
|
| `yoke goal set\|run\|resume\|pause\|status\|handoff\|budget [dir]` | Durable objectives, provider handoff, protected checks and checkpoint budgets | run/resume: `0` complete · `1` unfinished · `2` unavailable |
|
|
211
|
-
| `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--code-intelligence=off\|shadow\|active] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for all
|
|
213
|
+
| `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--code-intelligence=off\|shadow\|active] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing] [--model-provider=deepseek,kimi]` | Shared setup for all eight harnesses; optional federated code intelligence and DeepSeek/Kimi API profiles | `0` · `1` invalid setup |
|
|
212
214
|
| `yoke code-intelligence-server [--workspace=] [--mode=off\|shadow\|active]` | Serve the single Yoke-controlled MCP facade for federated code intelligence | `0` · `1` invalid/unavailable |
|
|
213
215
|
| `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
|
|
214
216
|
| `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
|
|
215
|
-
| `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen,opencode,kilo,pi\|all] [--code-graph=graphify\|serena] [--code-intelligence=off\|shadow\|active] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
|
|
217
|
+
| `yoke retrofit [dir] [--agent=claude,codex,gemini,qwen,opencode,kilo,pi,hermes\|all] [--code-graph=graphify\|serena] [--code-intelligence=off\|shadow\|active] [--loop]` | Install/update the harness for the selected agents, non-destructively | `0` |
|
|
216
218
|
| `yoke prd draft [dir] --idea= [--runner=] [--force]` | Idea → 5–12 stories with testable acceptance criteria | `0` · `1` invalid/guarded · `2` agent unavailable |
|
|
217
219
|
| `yoke prd check [dir]` | PRD lint gate (schema, dependencies, cycles, duplicate ids, acceptance) | `0` valid · `1` violations |
|
|
218
220
|
| `yoke change add\|status [dir] [--idea=]` | Queue a change at any time; the loop turns it into append-only stories at the next safe boundary | `0` · `1` invalid inbox/request |
|
|
@@ -232,7 +234,7 @@ Three excellent projects, three different jobs. Honest version:
|
|
|
232
234
|
| | [superpowers](https://github.com/obra/superpowers) (obra) | [gstack](https://github.com/garrytan/gstack) (Garry Tan) | **Yoke** |
|
|
233
235
|
|---|---|---|---|
|
|
234
236
|
| **What it is** | The canonical *skills methodology*: brainstorm → plan → TDD → review as composable skills | A *software factory* for Claude Code: ~40 role skills (QA, CSO, ship…) + a real Chromium browser layer | A *cross-agent harness*: one canon → native installs, plus a gated autonomous loop |
|
|
235
|
-
| **Agents** | Claude Code first | Claude Code + hosts like Codex/Cursor/Kiro — **no Gemini CLI** | **Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi** from one source of truth |
|
|
237
|
+
| **Agents** | Claude Code first | Claude Code + hosts like Codex/Cursor/Kiro — **no Gemini CLI** | **Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi, Hermes** from one source of truth |
|
|
236
238
|
| **Enforcement** | Advisory — skills *describe* the discipline; following them is up to the agent | Skill-driven; browser QA is genuinely real | **Mechanical** — gates live in code: clean tree, acceptance criteria, green tests, review verdict, commit integrity |
|
|
237
239
|
| **Autonomy** | Interactive sessions | Interactive slash-commands (`/qa`, `/ship`, …) | Opt-in **Ralph loop** with watchdog, worktree isolation, single-flight lock, per-story proofs |
|
|
238
240
|
| **Visual QA** | — | **Best-in-class**: live browser daemon (Chromium/CDP) with deep interactive QA | Built-in `flow-smoke` gate: screenshots always, video on failure, labelled per story — lighter, but *enforced* and cross-agent |
|
|
@@ -240,7 +242,7 @@ Three excellent projects, three different jobs. Honest version:
|
|
|
240
242
|
| **Footprint** | Markdown skills (plugin) | ~230 MB with browser runtime; hourly auto-update | Node CLI + markdown canon; Playwright only if you use flow-smoke, resolved **from your project** |
|
|
241
243
|
| **License** | MIT | MIT | MIT |
|
|
242
244
|
|
|
243
|
-
**They compose — use all three where they're strongest.** Yoke's canon *ships* the superpowers methodology natively for all
|
|
245
|
+
**They compose — use all three where they're strongest.** Yoke's canon *ships* the superpowers methodology natively for all eight agents (13 skills, [attributed](canon/skills/ATTRIBUTION.md)). And if gstack is installed, `yoke retrofit` detects it and adds a routing note to `CLAUDE.md` telling Claude to prefer gstack's live-browser `/qa`, `/cso`, and ship pipeline for what Yoke deliberately doesn't bundle — no dependency, no conflict, and non-Claude artifacts stay uniform.
|
|
244
246
|
|
|
245
247
|
**Choose Yoke when** you run more than one agent, want autonomy you can audit (gates + proofs + logs), or want one place to maintain your team's methodology. **Choose gstack when** you live 100% in Claude Code and want the deepest interactive browser QA. **Choose superpowers when** you want the methodology alone, interactively, in Claude Code — or just use it *through* Yoke.
|
|
246
248
|
|
|
@@ -260,6 +262,7 @@ flowchart TD
|
|
|
260
262
|
Skill --> OpenCode["OpenCode<br/>AGENTS.md · opencode.json · .opencode/skills"]
|
|
261
263
|
Skill --> Kilo["Kilo<br/>AGENTS.md · kilo.jsonc · .kilo/skills"]
|
|
262
264
|
Skill --> Pi["Pi<br/>AGENTS.md · .pi/settings.json · .pi/skills"]
|
|
265
|
+
Skill --> Hermes["Hermes<br/>AGENTS.md · .hermes/skills · yoke-reviewer.md"]
|
|
263
266
|
Loop["🤖 yoke loop — autonomous Ralph loop<br/>gates · verify · review · isolation · proofs"]
|
|
264
267
|
Claude -. drives .-> Loop
|
|
265
268
|
Codex -. drives .-> Loop
|
|
@@ -268,6 +271,7 @@ flowchart TD
|
|
|
268
271
|
OpenCode -. drives .-> Loop
|
|
269
272
|
Kilo -. drives .-> Loop
|
|
270
273
|
Pi -. drives .-> Loop
|
|
274
|
+
Hermes -. drives .-> Loop
|
|
271
275
|
```
|
|
272
276
|
|
|
273
277
|
Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`) → **Loop** (`yoke loop`) — on top of a durable **Context layer** (`yoke context`).
|
|
@@ -283,6 +287,7 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
|
|
|
283
287
|
| **OpenCode** | Complete skill packages under `.opencode/skills/`, shared `AGENTS.md`, merged `opencode.json` (instructions + local MCP), and `.opencode/agents/yoke-reviewer.md` |
|
|
284
288
|
| **Kilo** | Complete skill packages under `.kilo/skills/`, shared `AGENTS.md`, merged `kilo.jsonc` (instructions + local MCP), and `.kilo/agents/yoke-reviewer.md` |
|
|
285
289
|
| **Pi** | Complete skill packages under `.pi/skills/`, shared `AGENTS.md`, and merged `.pi/settings.json`; Pi's tool allowlists are applied at invocation time |
|
|
290
|
+
| **Hermes** | Complete skill packages under `.hermes/skills/`, shared `AGENTS.md`, and `.hermes/agents/yoke-reviewer.md` |
|
|
286
291
|
|
|
287
292
|
> **rtk integration:** Claude receives its PreToolUse hook; Codex receives a native hook adapter around `rtk hook check`; Gemini retains instruction-mode fallback where its CLI has no equivalent command-rewrite lifecycle.
|
|
288
293
|
|
|
@@ -299,11 +304,11 @@ Three layers — **Canon** (`yoke validate`) → **Retrofit** (`yoke retrofit`)
|
|
|
299
304
|
|
|
300
305
|
`yoke retrofit` installs all of these into each selected agent natively. Provenance is credited in [`canon/skills/ATTRIBUTION.md`](canon/skills/ATTRIBUTION.md).
|
|
301
306
|
|
|
302
|
-
To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.md` carries a **skill routing & precedence** block (methodology before role; one canonical entrypoint per concern — e.g. pre-merge code review is always `review`), emitted into all
|
|
307
|
+
To stop overlapping skills from auto-invoking against each other, `canon/AGENTS.md` carries a **skill routing & precedence** block (methodology before role; one canonical entrypoint per concern — e.g. pre-merge code review is always `review`), emitted into all eight agents.
|
|
303
308
|
|
|
304
309
|
Each manifest entry also declares `invocation: auto|manual`. Retrofit translates that intent into
|
|
305
310
|
the provider's native controls: Claude and Qwen disable model invocation for manual skills, Codex writes
|
|
306
|
-
`agents/openai.yaml`, Gemini lists only automatic skills in its generated index, and OpenCode/Kilo/Pi
|
|
311
|
+
`agents/openai.yaml`, Gemini lists only automatic skills in its generated index, and OpenCode/Kilo/Pi/Hermes
|
|
307
312
|
receive complete project-local skill packages. Validation
|
|
308
313
|
rejects conflicting package metadata and broken local Markdown links before anything is installed.
|
|
309
314
|
|
|
@@ -606,8 +611,8 @@ Loop runners disable native delegation in Codex, Claude, Gemini, Qwen, OpenCode
|
|
|
606
611
|
the Yoke worker budget. Integration retains its execution slot until the candidate lands.
|
|
607
612
|
|
|
608
613
|
**Provider support:** adaptive routing uses Yoke's shared provider adapter and works with Claude
|
|
609
|
-
Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo and
|
|
610
|
-
Internal contract tests cover invocation and routing behavior for all
|
|
614
|
+
Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo, Pi, and Hermes, including mixed-provider worker lists.
|
|
615
|
+
Internal contract tests cover invocation and routing behavior for all eight providers. The measured
|
|
611
616
|
performance evidence below is intentionally **Codex-only**; it does not claim equivalent savings
|
|
612
617
|
until authenticated, repeated in-the-wild runs exist for each provider.
|
|
613
618
|
|
|
@@ -908,7 +913,7 @@ src/
|
|
|
908
913
|
canon/ # manifest schema + validator (yoke validate)
|
|
909
914
|
code-intelligence/ # federated MCP facade, adapters, snapshots and guarded edits
|
|
910
915
|
change/ # append-only change inbox · planning · independent coverage review
|
|
911
|
-
retrofit/ # detect · plan · apply · planners (all
|
|
916
|
+
retrofit/ # detect · plan · apply · planners (all eight harnesses) · tools
|
|
912
917
|
loop/ # prd · gates · runner · verify · git/worktree · loop · run-command · lock · cleanup
|
|
913
918
|
quality/ # reference collection · blind critic · bounded repair · candidate comparison
|
|
914
919
|
new/ # yoke new — greenfield bootstrap
|
|
@@ -929,7 +934,7 @@ release provenance.
|
|
|
929
934
|
## 🧪 Development
|
|
930
935
|
|
|
931
936
|
```bash
|
|
932
|
-
npm test # vitest (
|
|
937
|
+
npm test # vitest (1287 tests)
|
|
933
938
|
npm run build # tsc, no emit errors
|
|
934
939
|
npm run yoke -- validate canon
|
|
935
940
|
```
|
package/canon/AGENTS.md
CHANGED
|
@@ -9,6 +9,13 @@ You are operating in a project retrofitted by Yoke. Follow these always:
|
|
|
9
9
|
|
|
10
10
|
This file is the portable baseline. Agent-specific instructions are generated alongside it (CLAUDE.md, GEMINI.md).
|
|
11
11
|
|
|
12
|
+
## Host capabilities and execution budgets
|
|
13
|
+
|
|
14
|
+
- Use only tools and skills actually installed in the host. Legacy `superpowers:` references in adapted skills refer to the corresponding local skill; `test-driven-development` maps to `tdd`. Do not install another plugin just to resolve a namespace.
|
|
15
|
+
- When Yoke owns the loop, worker concurrency, verification, and integration remain under its control. Do not start nested loops or native subagents from a worker. Without native delegation, work serially and expose missing independent-review evidence.
|
|
16
|
+
- Run focused checks for quick feedback, but never skip required acceptance, protected, integration, or release gates. Cache evidence only when code, configuration, and environment still match.
|
|
17
|
+
- Distinguish observed durations from future estimates. Report the sample count, empirical range, and unknown waiting time; a timeout is a budget, not a promised completion time. Do not reduce reasoning or test scope merely to meet an estimate.
|
|
18
|
+
|
|
12
19
|
## Skill routing & precedence
|
|
13
20
|
|
|
14
21
|
When several skills could match the same task, resolve deterministically:
|
package/canon/manifest.yaml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
name: yoke-canon
|
|
2
|
-
version: 1.
|
|
3
|
-
agents: [claude, codex, gemini, qwen, opencode, kilo, pi]
|
|
2
|
+
version: 1.16.0
|
|
3
|
+
agents: [claude, codex, gemini, qwen, opencode, kilo, pi, hermes]
|
|
4
4
|
skills:
|
|
5
5
|
- { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
|
|
6
6
|
- { id: yoke-retrofit, path: skills/yoke-retrofit, kind: methodology, invocation: auto }
|
|
@@ -11,7 +11,7 @@ Load plan, review critically, execute all tasks, report when complete.
|
|
|
11
11
|
|
|
12
12
|
**Announce at start:** "I'm using the executing-plans skill to implement this plan."
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
Use `subagent-driven-development` only when the host exposes delegation and the current run permits it. Inside a Yoke worker, execute the assigned task without launching nested workers; Yoke owns concurrency and independent review. Without delegation, execute serially and report any unavailable independent review explicitly. Subagents alone do not guarantee better quality or faster completion.
|
|
15
15
|
|
|
16
16
|
## The Process
|
|
17
17
|
|
|
@@ -31,7 +31,7 @@ HEAD_SHA=$(git rev-parse HEAD)
|
|
|
31
31
|
|
|
32
32
|
**2. Dispatch code-reviewer subagent:**
|
|
33
33
|
|
|
34
|
-
Use
|
|
34
|
+
Use the host's available read-only reviewer mechanism and fill the [review template](code-reviewer.md). Do not assume a `Task` tool exists. Inside a Yoke worker, hand evidence to Yoke's configured reviewer instead of launching nested agents. If independent review is unavailable, report that limitation; self-review is not an equivalent substitute.
|
|
35
35
|
|
|
36
36
|
**Placeholders:**
|
|
37
37
|
- `{WHAT_WAS_IMPLEMENTED}` - What you just built
|
|
@@ -102,4 +102,4 @@ You: [Fix progress indicators]
|
|
|
102
102
|
- Show code/tests that prove it works
|
|
103
103
|
- Request clarification
|
|
104
104
|
|
|
105
|
-
See
|
|
105
|
+
See the [review template](code-reviewer.md).
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
# Code review request
|
|
2
|
+
|
|
3
|
+
- Implementation: {WHAT_WAS_IMPLEMENTED}
|
|
4
|
+
- Acceptance criteria and requirements: {PLAN_OR_REQUIREMENTS}
|
|
5
|
+
- Base commit: {BASE_SHA}
|
|
6
|
+
- Candidate commit: {HEAD_SHA}
|
|
7
|
+
- Context and constraints: {DESCRIPTION}
|
|
8
|
+
- Permitted read scope, available checks, artifacts and known validation gaps: fill explicitly.
|
|
9
|
+
|
|
10
|
+
Independently inspect the actual candidate against requirements. Check correctness, failure handling, security, concurrency, maintainability and meaningful regression tests. Do not edit the candidate, weaken tests, merge or publish.
|
|
11
|
+
|
|
12
|
+
Return evidence-backed critical/important findings separately from minor suggestions. Each finding needs a location, failure scenario and validation approach. Distinguish verified, failed and unverified criteria. State limitations and whether unresolved blockers prevent acceptance; a test summary alone is insufficient.
|
|
@@ -5,6 +5,10 @@ description: Use when executing implementation plans with independent tasks in t
|
|
|
5
5
|
|
|
6
6
|
# Subagent-Driven Development
|
|
7
7
|
|
|
8
|
+
## Host capability and budget check
|
|
9
|
+
|
|
10
|
+
Use this workflow only if native delegation is available and authorized. Do not invent a `Task` or `TodoWrite` tool on hosts that lack it. Pi has no built-in subagents; use Yoke's scheduler or `executing-plans` instead. When running inside an existing Yoke worker, do not start nested agents or another loop: return evidence to the controlling runner. Independent review must not be replaced by claiming a self-review was independent.
|
|
11
|
+
|
|
8
12
|
Execute plan by dispatching fresh subagent per task, with two-stage review after each: spec compliance review first, then code quality review.
|
|
9
13
|
|
|
10
14
|
**Why subagents:** You delegate tasks to specialized agents with isolated context. By precisely crafting their instructions and context, you ensure they stay focused and succeed at their task. They should never inherit your session's context or history — you construct exactly what they need. This also preserves your own context for coordination work.
|
|
@@ -119,9 +123,9 @@ Implementer subagents report one of four statuses. Handle each appropriately:
|
|
|
119
123
|
|
|
120
124
|
## Prompt Templates
|
|
121
125
|
|
|
122
|
-
-
|
|
123
|
-
-
|
|
124
|
-
-
|
|
126
|
+
- [Implementer prompt](./implementer-prompt.md) - Dispatch implementer subagent
|
|
127
|
+
- [Spec reviewer prompt](./spec-reviewer-prompt.md) - Dispatch spec compliance reviewer subagent
|
|
128
|
+
- [Code quality reviewer prompt](./code-quality-reviewer-prompt.md) - Dispatch code quality reviewer subagent
|
|
125
129
|
|
|
126
130
|
## Example Workflow
|
|
127
131
|
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Code quality reviewer handoff
|
|
2
|
+
|
|
3
|
+
Supply the task, Acceptance criteria, base and candidate commits, architecture constraints, test evidence, and known risks. Review with read-only permissions after the specification review.
|
|
4
|
+
|
|
5
|
+
Inspect correctness, error paths, concurrency, resource cleanup, security boundaries, compatibility, and regression coverage. Prioritize reproducible defects over stylistic preferences. Check that optimizations preserve required gates and that unknown usage or timing evidence is not reported as zero or certainty.
|
|
6
|
+
|
|
7
|
+
Return severity-ranked findings with precise locations, failure scenarios, and suggested regression tests. Separate required fixes from optional improvements. State validation limits even when no defects are found. Never edit the candidate, self-merge, or infer release readiness solely from a green test summary.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Implementer handoff
|
|
2
|
+
|
|
3
|
+
Supply the task, repository/worktree path, base commit, allowed write scope, relevant interfaces, and Acceptance criteria with their required checks. Include only relevant context and previous failure evidence.
|
|
4
|
+
|
|
5
|
+
Implement only the assigned task. Preserve unrelated changes. Reproduce defects with a failing test before fixing them; do not weaken protected tests or acceptance checks. Run focused checks during iteration and all required gates before handoff. Do not spawn nested workers, publish, merge, or expand permissions. Commit only if the controller explicitly delegates that responsibility.
|
|
6
|
+
|
|
7
|
+
Return: outcome, changed files, exact checks and exit results, remaining risks, blockers, and artifact paths. Report incomplete work as incomplete. If blocked by a product decision or missing credentials, retain the work and explain what is needed; do not repeat the same failing action indefinitely.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Specification reviewer handoff
|
|
2
|
+
|
|
3
|
+
Supply the original task, Acceptance criteria, base and candidate commits, allowed read scope, and validation commands. Review independently from the implementer, with read-only permissions.
|
|
4
|
+
|
|
5
|
+
Compare the actual diff and observable behavior to each criterion. Check omissions, unauthorized additions, weakened tests, and whether evidence applies to the candidate being reviewed. Do not rely on the implementer's completion claim. Run permitted non-mutating checks where feasible; mark unavailable checks as unverified.
|
|
6
|
+
|
|
7
|
+
Return one disposition per criterion: verified, failed, or unverified, with file/line or command evidence. Give actionable defects and distinguish blockers from optional suggestions. Do not modify files, approve an unverified requirement, merge, or publish.
|
|
@@ -111,7 +111,7 @@ You MUST complete each phase before proceeding to the next.
|
|
|
111
111
|
|
|
112
112
|
**WHEN error is deep in call stack:**
|
|
113
113
|
|
|
114
|
-
See
|
|
114
|
+
See [root-cause tracing](root-cause-tracing.md) for the backward tracing technique.
|
|
115
115
|
|
|
116
116
|
**Quick version:**
|
|
117
117
|
- Where does bad value originate?
|
|
@@ -273,24 +273,20 @@ If systematic investigation reveals issue is truly environmental, timing-depende
|
|
|
273
273
|
3. Implement appropriate handling (retry, timeout, error message)
|
|
274
274
|
4. Add monitoring/logging for future investigation
|
|
275
275
|
|
|
276
|
-
|
|
276
|
+
Before declaring the cause unknowable, record what was ruled out and which observations are still missing.
|
|
277
277
|
|
|
278
278
|
## Supporting Techniques
|
|
279
279
|
|
|
280
280
|
These techniques are part of systematic debugging and available in this directory:
|
|
281
281
|
|
|
282
|
-
-
|
|
283
|
-
-
|
|
284
|
-
-
|
|
282
|
+
- [Root-cause tracing](root-cause-tracing.md) - Trace bugs backward to the original trigger
|
|
283
|
+
- [Defense in depth](defense-in-depth.md) - Validate relevant trust boundaries after finding the cause
|
|
284
|
+
- [Condition-based waiting](condition-based-waiting.md) - Replace arbitrary sleeps with bounded condition waits
|
|
285
285
|
|
|
286
286
|
**Related skills:**
|
|
287
287
|
- **superpowers:test-driven-development** - For creating failing test case (Phase 4, Step 1)
|
|
288
288
|
- **superpowers:verification-before-completion** - Verify fix worked before claiming success
|
|
289
289
|
|
|
290
|
-
##
|
|
290
|
+
## Measuring impact
|
|
291
291
|
|
|
292
|
-
|
|
293
|
-
- Systematic approach: 15-30 minutes to fix
|
|
294
|
-
- Random fixes approach: 2-3 hours of thrashing
|
|
295
|
-
- First-time fix rate: 95% vs 40%
|
|
296
|
-
- New bugs introduced: Near zero vs common
|
|
292
|
+
Record time to reproduce, time to a verified fix, retries and escaped regressions for the actual project. No fixed speedup, success rate or absence of new bugs is guaranteed by this workflow.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Condition-based waiting
|
|
2
|
+
|
|
3
|
+
Prefer a completion event or promise over sleeping for an assumed duration. Subscribe before starting work so a fast completion is not missed. Handle success, failure, cancellation and timeout, and remove listeners/timers in every terminal path.
|
|
4
|
+
|
|
5
|
+
If polling is unavoidable, choose a measurable condition, bounded deadline and modest interval. Preserve the last observed state in the timeout error. Do not treat repeated log output as successful progress.
|
|
6
|
+
|
|
7
|
+
Tests should control time or use an explicit readiness signal. Keep one real process integration test where needed. A larger timeout is justified only by measured runtime and does not repair a race or an unintended network call.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Defense in depth
|
|
2
|
+
|
|
3
|
+
Fix the originating contract violation first. Then identify independent trust boundaries where malformed or stale data could enter again: configuration, external process output, persisted state and public interfaces.
|
|
4
|
+
|
|
5
|
+
Validate at those boundaries with explicit errors and preserve unknown values. Avoid scattering duplicate checks through code that already has a validated type. Do not catch every exception and return success or zero.
|
|
6
|
+
|
|
7
|
+
Test malformed input, missing evidence, stale state, cancellation and cleanup. For destructive operations verify target scope and authority independently of the requesting model. Protection layers must not weaken acceptance tests or convert a blocked action into an automatic permission bypass.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Root-cause tracing
|
|
2
|
+
|
|
3
|
+
Capture the smallest failing command, inputs, expected/actual behavior, environment and code revision. Reproduce before changing code.
|
|
4
|
+
|
|
5
|
+
Trace backward from the first observed invalid value or state transition to its producer. At each boundary, record the input, output and contract without logging credentials. Distinguish a downstream symptom from the earliest violated invariant.
|
|
6
|
+
|
|
7
|
+
Form one falsifiable hypothesis and choose an observation that separates it from alternatives. Change one factor at a time. If reproduction depends on concurrency, preserve scheduling and cancellation evidence rather than adding arbitrary delays.
|
|
8
|
+
|
|
9
|
+
Write a regression test at the violated boundary, demonstrate its failure, fix the cause and run dependent checks. Document remaining uncertainty instead of attributing an intermittent pass to a fix without evidence.
|
|
@@ -356,7 +356,7 @@ Never fix bugs without a test.
|
|
|
356
356
|
|
|
357
357
|
## Testing Anti-Patterns
|
|
358
358
|
|
|
359
|
-
When adding mocks or test utilities, read
|
|
359
|
+
When adding mocks or test utilities, read [testing anti-patterns](testing-anti-patterns.md) to avoid common pitfalls:
|
|
360
360
|
- Testing mock behavior instead of real behavior
|
|
361
361
|
- Adding test-only methods to production classes
|
|
362
362
|
- Mocking without understanding dependencies
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Testing anti-patterns
|
|
2
|
+
|
|
3
|
+
- Test observable behavior, not that a mock returned the value you supplied. Assert resulting state and failure behavior as well as calls.
|
|
4
|
+
- Mock external boundaries deliberately. If production constructs a second client for an isolated workspace, replace that constructor too; a unit test must not silently invoke a real network backend.
|
|
5
|
+
- Do not mock the implementation under test or delete assertions to make a change pass. Keep at least one contract/integration check for important boundaries.
|
|
6
|
+
- Demonstrate that the regression test fails for the original defect and passes after the fix. A new green test alone does not establish that it detects the bug.
|
|
7
|
+
- Avoid arbitrary sleeps and broad timeout increases. Use readiness events, controlled clocks and bounded cancellation.
|
|
8
|
+
- Preserve protected acceptance tests and required gates. Focused tests speed feedback but do not replace release validation.
|
|
9
|
+
- Report skipped tests, unavailable platforms and missing credentials separately from passing checks. Never convert absent measurements into measured zero.
|
|
@@ -7,7 +7,7 @@ description: Use when asked to "retrofit", "yoke this project", or set up the Yo
|
|
|
7
7
|
|
|
8
8
|
Set up or update Yoke through the shared `yoke setup` contract.
|
|
9
9
|
|
|
10
|
-
1. Inspect the project and identify the current host (`claude`, `codex`, `gemini`, `qwen`, `opencode`, `kilo`, or `
|
|
10
|
+
1. Inspect the project and identify the current host (`claude`, `codex`, `gemini`, `qwen`, `opencode`, `kilo`, `pi`, or `hermes`).
|
|
11
11
|
2. Ask these setup questions one at a time and give a direct recommendation:
|
|
12
12
|
- target agents (recommend the current host; use `all` for deliberately cross-agent projects),
|
|
13
13
|
- code-graph tool,
|
package/dist/agents/contracts.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
|
-
export const AgentSchema = z.enum(['claude', 'codex', 'gemini', 'qwen', 'opencode', 'kilo', 'pi']);
|
|
2
|
+
export const AgentSchema = z.enum(['claude', 'codex', 'gemini', 'qwen', 'opencode', 'kilo', 'pi', 'hermes']);
|
|
3
3
|
export const PermissionProfileSchema = z.enum(['safe', 'unsafe', 'read-only']);
|
|
4
4
|
export const ModelSelectionSchema = z.object({
|
|
5
5
|
provider: z.string().regex(/^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$/).optional(),
|
package/dist/agents/host.js
CHANGED
|
@@ -13,6 +13,8 @@ export function detectHostAgent(env = process.env) {
|
|
|
13
13
|
return 'opencode';
|
|
14
14
|
if (env.KILO_CLIENT || env.KILO_CONFIG || env.KILO_CONFIG_DIR)
|
|
15
15
|
return 'kilo';
|
|
16
|
+
if (env.HERMES_SESSION_ID || env.HERMES_CONFIG)
|
|
17
|
+
return 'hermes';
|
|
16
18
|
if (env.CODEX_HOME)
|
|
17
19
|
return 'codex';
|
|
18
20
|
if (env.CLAUDE_CONFIG_DIR)
|
|
@@ -21,6 +23,8 @@ export function detectHostAgent(env = process.env) {
|
|
|
21
23
|
return 'gemini';
|
|
22
24
|
if (env.QWEN_CLI_HOME)
|
|
23
25
|
return 'qwen';
|
|
26
|
+
if (env.HERMES_HOME)
|
|
27
|
+
return 'hermes';
|
|
24
28
|
return undefined;
|
|
25
29
|
}
|
|
26
30
|
export function resolveRunnerAgent(config, explicit, host) {
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
const record = (value) => typeof value === 'object' && value !== null && !Array.isArray(value);
|
|
2
|
+
const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0;
|
|
3
|
+
/** Pi usage is per assistant message, not cumulative across a run. */
|
|
4
|
+
export function createPiTelemetry() {
|
|
5
|
+
const totals = {};
|
|
6
|
+
const counts = {};
|
|
7
|
+
const models = new Set();
|
|
8
|
+
let turns = 0;
|
|
9
|
+
let legacyUsage;
|
|
10
|
+
const add = (usage) => {
|
|
11
|
+
turns++;
|
|
12
|
+
const fields = {
|
|
13
|
+
inputTokens: usage?.input, outputTokens: usage?.output,
|
|
14
|
+
cachedInputTokens: usage?.cacheRead, cacheWriteInputTokens: usage?.cacheWrite,
|
|
15
|
+
reasoningOutputTokens: usage?.reasoning,
|
|
16
|
+
totalCostUsd: record(usage?.cost) ? usage.cost.total : undefined,
|
|
17
|
+
};
|
|
18
|
+
for (const [key, value] of Object.entries(fields))
|
|
19
|
+
if (finite(value)) {
|
|
20
|
+
totals[key] = (totals[key] ?? 0) + value;
|
|
21
|
+
counts[key] = (counts[key] ?? 0) + 1;
|
|
22
|
+
}
|
|
23
|
+
};
|
|
24
|
+
return {
|
|
25
|
+
consume(event) {
|
|
26
|
+
// Compatibility with earlier top-level snapshots; never override a final usage.
|
|
27
|
+
if (event.type === 'message_update' && record(event.usage))
|
|
28
|
+
legacyUsage = event.usage;
|
|
29
|
+
if (event.type !== 'message_end' || !record(event.message) || event.message.role !== 'assistant')
|
|
30
|
+
return;
|
|
31
|
+
if (typeof event.message.model === 'string' && event.message.model)
|
|
32
|
+
models.add(event.message.model);
|
|
33
|
+
add(record(event.message.usage) ? event.message.usage : legacyUsage);
|
|
34
|
+
legacyUsage = undefined;
|
|
35
|
+
},
|
|
36
|
+
finish() {
|
|
37
|
+
const reportedModels = [...models];
|
|
38
|
+
const complete = turns > 0 && counts.inputTokens === turns && counts.outputTokens === turns;
|
|
39
|
+
if (complete) {
|
|
40
|
+
// Optional totals must also cover every turn; omitted is not measured zero.
|
|
41
|
+
const measured = Object.fromEntries(Object.entries(totals).filter(([key]) => counts[key] === turns));
|
|
42
|
+
return { usageAvailable: true, tokens: {
|
|
43
|
+
...measured, inputTokens: totals.inputTokens, outputTokens: totals.outputTokens,
|
|
44
|
+
...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}),
|
|
45
|
+
}, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
|
|
46
|
+
}
|
|
47
|
+
return { usageAvailable: false,
|
|
48
|
+
...(Object.keys(totals).length ? { partialUsage: { ...totals } } : {}),
|
|
49
|
+
...(reportedModels.length ? { reportedModels } : {}),
|
|
50
|
+
};
|
|
51
|
+
},
|
|
52
|
+
};
|
|
53
|
+
}
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { parseProviderTelemetry } from './telemetry.js';
|
|
2
|
+
import { createPiTelemetry } from './pi-telemetry.js';
|
|
2
3
|
export function createBoundedOutput(limitBytes) {
|
|
3
4
|
let text = '';
|
|
4
5
|
let truncated = false;
|
|
@@ -17,6 +18,7 @@ export function createBoundedOutput(limitBytes) {
|
|
|
17
18
|
};
|
|
18
19
|
}
|
|
19
20
|
export function createTelemetryAccumulator(agent) {
|
|
21
|
+
const pi = agent === 'pi' ? createPiTelemetry() : undefined;
|
|
20
22
|
let trailing = '';
|
|
21
23
|
let telemetry = { usageAvailable: false };
|
|
22
24
|
let reportedModels = [];
|
|
@@ -25,6 +27,15 @@ export function createTelemetryAccumulator(agent) {
|
|
|
25
27
|
: undefined;
|
|
26
28
|
const update = (lines) => {
|
|
27
29
|
for (const line of lines) {
|
|
30
|
+
if (pi) {
|
|
31
|
+
try {
|
|
32
|
+
const event = JSON.parse(line);
|
|
33
|
+
if (event && typeof event === 'object' && !Array.isArray(event))
|
|
34
|
+
pi.consume(event);
|
|
35
|
+
}
|
|
36
|
+
catch { /* non-JSON diagnostics carry no usage */ }
|
|
37
|
+
continue;
|
|
38
|
+
}
|
|
28
39
|
const next = parseProviderTelemetry(agent, [line]);
|
|
29
40
|
if (next.reportedModels)
|
|
30
41
|
reportedModels = next.reportedModels;
|
|
@@ -75,6 +86,8 @@ export function createTelemetryAccumulator(agent) {
|
|
|
75
86
|
if (trailing)
|
|
76
87
|
update([trailing]);
|
|
77
88
|
trailing = '';
|
|
89
|
+
if (pi)
|
|
90
|
+
return pi.finish();
|
|
78
91
|
if (stepTotals && (stepTotals.hasInput || stepTotals.hasOutput)) {
|
|
79
92
|
const latest = telemetry.tokens;
|
|
80
93
|
const inputTokens = stepTotals.hasInput ? stepTotals.input : latest?.inputTokens;
|