@hecer/yoke 1.6.2 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +42 -0
  4. package/README.md +61 -22
  5. package/canon/manifest.yaml +1 -1
  6. package/canon/tools/gemini-rtk-hook.mjs +25 -0
  7. package/dist/agents/contracts.js +2 -0
  8. package/dist/agents/process-streams.js +18 -4
  9. package/dist/agents/providers.js +32 -4
  10. package/dist/agents/telemetry.js +58 -10
  11. package/dist/check/command.js +114 -0
  12. package/dist/cli.js +124 -5
  13. package/dist/context/packet.js +32 -0
  14. package/dist/dashboard/analytics.js +123 -0
  15. package/dist/dashboard/page.js +34 -0
  16. package/dist/dashboard/panels.js +19 -0
  17. package/dist/dashboard/registry.js +68 -0
  18. package/dist/dashboard/server.js +177 -0
  19. package/dist/estimation/durations.js +20 -0
  20. package/dist/estimation/schedule.js +41 -0
  21. package/dist/execution/actions.js +24 -0
  22. package/dist/goals/command.js +191 -0
  23. package/dist/loop/dispatcher.js +19 -5
  24. package/dist/loop/git.js +5 -4
  25. package/dist/loop/loop.js +23 -3
  26. package/dist/loop/parallel-adapters.js +3 -2
  27. package/dist/loop/parallel-command.js +52 -4
  28. package/dist/loop/prd.js +3 -0
  29. package/dist/loop/recovery.js +51 -0
  30. package/dist/loop/reporter.js +105 -8
  31. package/dist/loop/run-command.js +62 -13
  32. package/dist/loop/runner.js +20 -16
  33. package/dist/loop/scheduler.js +38 -1
  34. package/dist/observability/events.js +72 -0
  35. package/dist/observability/history.js +80 -0
  36. package/dist/quality/candidate-comparison.js +1 -1
  37. package/dist/quality/command.js +16 -3
  38. package/dist/retrofit/config.js +11 -0
  39. package/dist/retrofit/gitignore.js +6 -0
  40. package/dist/retrofit/planners/gemini.js +11 -3
  41. package/dist/routing/router.js +50 -10
  42. package/dist/setup/command.js +3 -2
  43. package/dist/workspace/fingerprint.js +66 -0
  44. package/dist/workspace/state.js +20 -0
  45. package/docs/PRODUCT-DIRECTION-2026-09-05.md +199 -0
  46. package/docs/VERIFIED-PROJECTS-VALIDATION.md +29 -0
  47. package/docs/VERIFIED-PROJECTS.md +167 -0
  48. package/docs/superpowers/plans/2026-09-05-verified-projects.md +83 -0
  49. package/gemini-extension.json +1 -1
  50. package/hooks/bounded-gemini.mjs +70 -0
  51. package/package.json +1 -1
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.6.2",
5
+ "version": "1.8.0",
6
6
  "description": "Cross-agent coding harness: one curated skill canon (TDD, brainstorming, plans, reviews, shipping, design verification) plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.6.2",
3
+ "version": "1.8.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -2,6 +2,48 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ ## 1.8.0 — 2026-09-06
6
+
7
+ ### Added
8
+ - Add persistent local measurement history and dashboard views for current work, usage/time and results, including UTC day/week/month filters, model history, project comparisons and consumption charts.
9
+ - Display current tasks, worker phases, integration progress and status age, with automatic refresh of the current-work view.
10
+ - Record explicit acceptances and show measured tokens and time per acceptance. Attribute available reviewer, critic and repair usage; distinguish actual reported models, unknown calls and partial costs.
11
+
12
+ ### Changed
13
+ - Enable routing in new setups and automatically select up to three parallel workers when all pending tasks declare write scopes. Dependencies and overlapping scopes still constrain dispatch. Preserve explicit opt-outs and use isolated worktrees by default.
14
+ - Share routing decisions between synchronous and asynchronous runners; support routed parallel workers, stable recovery history and explicit provider affinity.
15
+ - Reserve execution capacity through integration and disable native delegation for loop providers; preserve Gemini system policy in a temporary bounded-execution configuration.
16
+ - Require a dated changelog entry, synchronized version metadata and verified release checks for every new version in the project instructions.
17
+
18
+ ### Fixed
19
+ - Avoid conflicting Codex sandbox arguments and prevent Codex-only options from leaking into Gemini workers.
20
+ - Keep compact measurement history after recent activity expires, deduplicate archived events, and report incomplete history instead of treating missing usage as zero.
21
+ - Preserve reviewer telemetry and worker/model attribution across parallel execution and recovery.
22
+
23
+ ### Migration and validation limits
24
+ - Use `--parallel=N` to choose a worker limit, `--parallel=auto` for automatic selection, `--no-routing` to opt out of routing, and `--no-isolate` to opt out of default isolation. Explicit existing configuration remains authoritative. Unknown write scopes, tool actions and worktree recovery select serial execution in auto mode.
25
+ - Automatic routing needs configured profiles; otherwise it keeps the selected parent provider. Explicit `--routing` without profiles reports a configuration error.
26
+ - Missing historical usage cannot be reconstructed. Tokens per minute describe interval or summed call consumption, not measured generation speed. Live authenticated provider benchmarks, resource-adaptive concurrency and calibrated time/cost predictions are not established by this release.
27
+
28
+ ## 1.7.0 — 2026-09-05
29
+
30
+ ### Added
31
+ - Independent `yoke check` with executable acceptance mapping, protected test infrastructure and content-bound evidence.
32
+ - Durable project goals, provider handoff, checkpoint budgets, interruption accounting and project-scoped recovery.
33
+ - Local project registry and loopback dashboard with goals, task estimates, evidence, consumption and pause controls.
34
+ - Explicit routing rules with persisted gate-driven escalation; bounded tool actions without model calls.
35
+ - Task-aware context packets, advisory write scopes, dependency-depth scheduling and empirical time ranges with prediction error records.
36
+
37
+ ### Fixed
38
+ - Failed serial isolated worktrees are retained and can be explicitly resumed against their original target and PRD.
39
+ - Reviewer fingerprints include untracked contents and acceptance inputs; unsupported nested repository identity fails closed.
40
+ - Gemini always emits streaming telemetry, preserves model identity, honors aggregate token aliases and rejects unsupported selections. Its RTK hook merges native settings without a shell dependency.
41
+ - Runtime evidence is excluded from Git gates and commits in existing projects. Partial usage and costs remain visibly incomplete.
42
+ - Protected acceptance is checked after verification and repair, and linked goal state cannot overwrite unrelated files through pause.
43
+
44
+ ### Validation limits
45
+ - Live authenticated provider comparisons and calibrated development-time/cost estimates are not established by the automated tests. Browser proof and independent model review still require explicit configuration.
46
+
5
47
  ## 1.6.2 — 2026-09-02
6
48
 
7
49
  ### Added
package/README.md CHANGED
@@ -2,14 +2,14 @@
2
2
 
3
3
  # 🐂 Yoke
4
4
 
5
- <!-- yoke:version:start -->1.6.2<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->1021<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.8.0<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->1117<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->34<!-- yoke:skills:end -->
8
8
  <!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
9
9
 
10
10
  ### One harness, three agents — and zero trust in "done."
11
11
 
12
- **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Then, when you want it, an opt-in autonomous loop ships your spec story-by-story: tested, cross-model-reviewed, committed — **with a screenshot to prove every story and a video for every failure**.
12
+ **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Its opt-in loop implements and verifies stories before committing. Independent review and browser proofs run when configured; screenshots and videos require the browser smoke gate.
13
13
 
14
14
  [![npm](https://img.shields.io/npm/v/%40hecer%2Fyoke?logo=npm&color=CB3837)](https://www.npmjs.com/package/@hecer/yoke)
15
15
  [![npm downloads](https://img.shields.io/npm/dm/%40hecer%2Fyoke?logo=npm)](https://www.npmjs.com/package/@hecer/yoke)
@@ -17,7 +17,7 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-1021%20passing-brightgreen.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-1117%20defined-blue.svg)
21
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
@@ -27,13 +27,43 @@
27
27
 
28
28
  > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
29
 
30
+ **New in 1.8.0:** [automatic routing, up to three parallel workers and expanded project dashboards](docs/VERIFIED-PROJECTS.md). Inspect current work, compare recorded consumption across projects and models by day, week or month, and track measured effort per accepted change. New setups enable routing and isolated execution by default; explicit overrides remain available.
31
+
32
+ ### One dashboard, multiple projects
33
+
34
+ ```sh
35
+ npm install -g @hecer/yoke@latest
36
+ yoke projects add /path/to/frontend
37
+ yoke projects add /path/to/backend
38
+ yoke dashboard --no-register
39
+ ```
40
+
41
+ Open the printed `http://127.0.0.1:...` URL. Each registered project has its own goals, tasks and evidence. The dashboard shows available worker state, per-task duration estimates, planned start offsets, input/output tokens, costs and unknown measurements. You can request a goal pause at a safe boundary.
42
+
43
+ Projects are registered explicitly; this version does not automatically discover every process or aggregate other computers. Start/resume and budget changes use the CLI. Missing history appears as unknown; time ranges are empirical estimates, not exact deadlines.
44
+
45
+ ### Verified goals and efficient execution
46
+
47
+ ```sh
48
+ yoke check /path/to/project --json
49
+ # First define executable criteria and protected tests in .yoke/acceptance.yaml.
50
+ yoke goal set /path/to/project --objective="Complete guest checkout" --attempts=3 --minutes=30
51
+ yoke goal run /path/to/project --runner=codex
52
+ yoke goal resume /path/to/project --runner=claude
53
+ yoke goal handoff /path/to/project
54
+ ```
55
+
56
+ Goals persist their objective, attempts and check evidence across runs. Changed protected tests block acceptance; failed work is retained. Explicit routing rules bypass controller calls, configured tool actions use no model, and failed rule-based attempts can escalate to a stronger worker. Context selection stays within a character budget; declared write scopes and dependencies guide parallel scheduling.
57
+
58
+ See the [1.7 workflow guide](docs/VERIFIED-PROJECTS.md) for setup, Gemini selection, recovery and budgets. Provider contracts do not establish equal model quality; live comparative savings and calibrated time predictions remain unmeasured. Token budgets apply between provider calls, and browser proofs still require a configured smoke gate.
59
+
30
60
  Yoke 1.5 keeps failed gate output compact without throwing evidence away: deterministic previews
31
61
  retain actionable failures and final summaries, while large complete stdout/stderr remains available
32
62
  in private, content-addressed local artifacts. Existing projects keep their serial behavior and use
33
63
  safe 2 KiB preview / 8 KiB artifact defaults unless configured otherwise.
34
64
 
35
- Yoke 1.4 adds opt-in parallel workers and a bounded, reference-driven quality gauntlet without
36
- changing existing serial loop defaults. See [the 1.4 migration guide](docs/MIGRATING-TO-1.4.md)
65
+ Yoke 1.4 introduced opt-in parallel workers and a bounded, reference-driven quality gauntlet.
66
+ Yoke 1.8.0 uses automatic parallelism for tasks with declared write scopes. See [the 1.4 migration guide](docs/MIGRATING-TO-1.4.md)
37
67
  for the new flags, configuration, cleanup behavior, and review-verdict contract.
38
68
 
39
69
  Yoke 1.1 is safe-by-default: provider CLIs use autonomous sandbox profiles unless `--unsafe`
@@ -50,7 +80,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
50
80
 
51
81
  | The pain | What actually happens | What Yoke does about it |
52
82
  |---|---|---|
53
- | 🎭 **The verification gap** — *"agent says done, but it isn't"* | Agents submit confidently on 100% of runs while resolving far fewer; "all tests pass" when they were never run ([silent-failures research](https://arxiv.org/pdf/2603.25764)) | The loop trusts **your verify command's exit code**, never the agent's word. A story is `passes: true` only after tests are green, the reviewer approved, and the commit landed — atomically. Plus: **screenshot proofs** per story. |
83
+ | 🎭 **The verification gap** — *"agent says done, but it isn't"* | A success message can omit untested acceptance criteria. | The loop executes acceptance and project checks. Enabled review and browser gates must pass before a story lands. `yoke check` exposes unmapped outcomes as unverified. |
54
84
  | 🔀 **Three agents, three configs** | Teams hand-maintain `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, skills, and MCP wiring separately — copy-paste drift everywhere | **One canon → `yoke retrofit`** generates the idiomatic native artifacts for each agent. Change the canon once, re-retrofit everywhere. |
55
85
  | 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
56
86
  | 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
@@ -59,7 +89,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
59
89
 
60
90
  **Who it's not for:** if you want a chat pair-programmer with no process, you don't need a harness. Yoke is for shipping with discipline.
61
91
 
62
- ## ⏱️ 60 seconds: idea → tested, photographed software
92
+ ## Example: idea → verified implementation
63
93
 
64
94
  ```console
65
95
  $ yoke new reading-app --idea="a web app that tracks my reading list"
@@ -78,10 +108,10 @@ $ yoke loop run reading-app --isolate --review --max=10
78
108
  # nothing was committed. fix, then re-run.
79
109
 
80
110
  $ ls reading-app/.yoke/proof/STORY-2/
81
- home.png list.png # photographic evidence, labelled per story
111
+ home.png list.png # example when browser smoke is configured
82
112
  ```
83
113
 
84
- Every claim in that transcript is enforced by code paths with tests behind them — 1021 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
114
+ This is an illustrative transcript. Actual durations depend on the project, provider, retries and enabled gates; browser proof requires a configured smoke flow. See [how it was built](#-why--how-it-was-built).
85
115
 
86
116
  ## 🚀 Quickstart
87
117
 
@@ -157,7 +187,11 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
157
187
 
158
188
  | Command | What it does | Exit codes |
159
189
  |---|---|---|
160
- | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing]` | Shared six-question setup for Claude, Codex, and Gemini; adaptive routing is always an explicit opt-in | `0` · `1` invalid setup |
190
+ | `yoke dashboard [dir] [--no-register] [--port=N]` | Local overview for every registered project; optionally register `dir` first | `0` stopped normally · `1` shutdown failure · `2` unavailable |
191
+ | `yoke projects add\|list\|remove` | Register a project, list registrations or remove a reference by ID | `0` · `2` invalid/unavailable |
192
+ | `yoke check [dir] [--json] [--requirement=] [--protect [--refresh]]` | Execute acceptance checks or explicitly pin their infrastructure | `0` passed/pinned · `1` failed · `2` unverified/unavailable |
193
+ | `yoke goal set\|run\|resume\|pause\|status\|handoff\|budget [dir]` | Durable objectives, provider handoff, protected checks and checkpoint budgets | run/resume: `0` complete · `1` unfinished · `2` unavailable |
194
+ | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing]` | Shared six-question setup for Claude, Codex, and Gemini; routing defaults on for new setups and preserves explicit opt-outs | `0` · `1` invalid setup |
161
195
  | `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
162
196
  | `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
163
197
  | `yoke retrofit [dir] [--agent=claude,codex,gemini\|all] [--code-graph=graphify\|serena] [--loop]` | Install/update the harness, non-destructively | `0` |
@@ -529,14 +563,16 @@ use `yoke loop resume . --discard`; pending decisions are never deleted by that
529
563
  `loop.onAmbiguity: resolve|abort` and `--on-ambiguity=` remain supported as compatibility aliases;
530
564
  new projects should use `decisionPolicy: auto|critical`.
531
565
 
532
- ### Adaptive model routing (explicit opt-in)
566
+ ### Adaptive model routing
533
567
 
534
- `yoke setup` asks before enabling routing; the default is **off**. When enabled, the selected
568
+ `yoke setup` enables routing by default and preserves explicit opt-outs. Without configured
569
+ worker profiles, automatic execution keeps the selected parent. When enabled, the selected
535
570
  parent remains the strong planner/controller. Before each bounded story it receives only the
536
571
  story, acceptance criteria, and at most three eligible worker profiles, then returns one
537
572
  machine-readable choice. The worker can be a cheaper/faster Claude, Codex, or Gemini profile;
538
- `SELF` keeps difficult work on the parent. Provider-native subagents are disabled for these
539
- runs so Yoke does not pay for two orchestration layers.
573
+ `SELF` keeps difficult work on the parent. Explicit project rules skip the controller.
574
+ Loop runners disable native delegation in Codex, Claude and Gemini so it cannot multiply
575
+ the Yoke worker budget. Integration retains its execution slot until the candidate lands.
540
576
 
541
577
  **Provider support:** adaptive routing uses Yoke's shared provider adapter and works with Claude
542
578
  Code, Codex CLI, and Gemini CLI, including mixed-provider worker lists. Internal contract tests
@@ -550,7 +586,7 @@ runner:
550
586
  model: gpt-5.6-sol # optional; provider model strings stay opaque to Yoke
551
587
  reasoningEffort: high
552
588
  routing:
553
- enabled: true # setup defaults false; setup --routing opts in
589
+ enabled: true # new setup default; false preserves an explicit opt-out
554
590
  strategy: balanced # balanced | cost | speed | quality
555
591
  maxCandidates: 3
556
592
  workers:
@@ -571,7 +607,7 @@ routing:
571
607
  capabilities: [large-context, implementation]
572
608
  ```
573
609
 
574
- Use `yoke loop run . --routing` for a one-run opt-in or `--no-routing` for a controlled
610
+ Use `yoke loop run . --routing` to explicitly require configured routing or `--no-routing` for a controlled
575
611
  baseline. Routing control calls are read-only and deliberately tiny; malformed output or no
576
612
  eligible worker falls back to `SELF`. Yoke does not ship a universal, fast-aging
577
613
  "intelligence score". Candidate model IDs come from project configuration while setup defaults
@@ -581,9 +617,12 @@ after 30 days. It stores no prompts, source, or project paths—only a project h
581
617
  time/token/outcome evidence. Writes are immutable one-event files, so concurrent Yoke instances
582
618
  cannot overwrite a shared registry file.
583
619
 
584
- Routing is not free: it adds one controller call per story. It is most promising when a bounded
585
- worker saves more than that call costs; tiny stories may be slower. Keep it opt-in and measure it
586
- on your own backlog rather than assuming a win.
620
+ Routing is not free: stories without a matching rule can add a controller call. Measure it
621
+ on your own backlog rather than assuming a win. Routing now also runs within asynchronous
622
+ parallel workers. Automatic parallelism starts at up to three workers when pending tasks declare
623
+ write scopes; unknown scopes and configured tool actions keep execution serial. Isolation is
624
+ on by default. Explicit `--parallel=N`, `--no-routing` and `--no-isolate` remain available.
625
+ See [execution defaults and dashboard measurement details](docs/VERIFIED-PROJECTS.md#execution-defaults-in-180).
587
626
 
588
627
  ### Performance budgets: efficiency as a gate, not a style
589
628
 
@@ -807,7 +846,7 @@ the routed median used **33.8% less wall time, 11.0% less fresh input, 49.5% few
807
846
  and 78.2% fewer reasoning tokens**. All three pairs improved wall time and fresh input.
808
847
 
809
848
  The boundary matters: an earlier architecture/privacy task correctly stayed on `SELF` and paid
810
- controller overhead, so routing is an explicit opt-in rather than a universal win. Codex did not
849
+ controller overhead, so the routing default is not evidence of universal savings. Codex did not
811
850
  emit dollar cost for these plan-backed runs; Yoke reports the measured token breakdown instead of
812
851
  inventing a price. Method, ranges, controller cost, caveats, analyzer, and six raw JSON rows are in
813
852
  [`bench/RESULTS.md`](bench/RESULTS.md#codex-only-full-repository-routing-study-2026-08-02).
@@ -856,7 +895,7 @@ release provenance.
856
895
  ## 🧪 Development
857
896
 
858
897
  ```bash
859
- npm test # vitest (1021 tests)
898
+ npm test # vitest (1117 tests)
860
899
  npm run build # tsc, no emit errors
861
900
  npm run yoke -- validate canon
862
901
  ```
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.6.2
2
+ version: 1.7.0
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
@@ -0,0 +1,25 @@
1
+ #!/usr/bin/env node
2
+ // Pure BeforeTool argument adapter. Never evaluates or executes tool commands.
3
+ import { readFileSync } from 'node:fs'
4
+
5
+ const record = value => value !== null && typeof value === 'object' && !Array.isArray(value)
6
+ try {
7
+ const event = JSON.parse(readFileSync(0, 'utf8'))
8
+ if (!record(event) || typeof event.tool_name !== 'string' ||
9
+ (event.hook_event_name !== undefined && event.hook_event_name !== 'BeforeTool')) throw new Error('event')
10
+ let response = {}
11
+ if (event.tool_name === 'run_shell_command') {
12
+ if (!record(event.tool_input) || typeof event.tool_input.command !== 'string' || !event.tool_input.command.trim() || event.tool_input.command.includes('\0')) throw new Error('command')
13
+ const command = event.tool_input.command
14
+ // Restrict rewriting to simple supported invocations. Leave shell syntax,
15
+ // quoted executables, assignments and existing RTK wrappers untouched.
16
+ if (!/[;&|<>`$\r\n()]/u.test(command) && /^\s*(?:git|rg|npm|npx|cargo|pytest|go|docker|kubectl)\s/u.test(command)) {
17
+ const rewritten = command.trimStart().replace(/^rg\s/u, 'grep ')
18
+ response = { hookSpecificOutput: { tool_input: { ...event.tool_input, command: `rtk ${rewritten}` } } }
19
+ }
20
+ }
21
+ process.stdout.write(JSON.stringify(response) + '\n')
22
+ } catch {
23
+ process.stderr.write('Invalid Gemini BeforeTool hook input\n')
24
+ process.exitCode = 2
25
+ }
@@ -25,6 +25,8 @@ const ProviderTokenUsageSchema = z.object({
25
25
  export const ProviderTelemetrySchema = z.object({
26
26
  usageAvailable: z.boolean(),
27
27
  tokens: ProviderTokenUsageSchema.optional(),
28
+ partialUsage: ProviderTokenUsageSchema.partial().optional(),
29
+ reportedModels: z.array(z.string().min(1)).optional(),
28
30
  }).superRefine((telemetry, ctx) => {
29
31
  if (telemetry.usageAvailable && !telemetry.tokens) {
30
32
  ctx.addIssue({ code: 'custom', path: ['tokens'], message: 'usageAvailable telemetry requires token totals' });
@@ -19,10 +19,19 @@ export function createBoundedOutput(limitBytes) {
19
19
  export function createTelemetryAccumulator(agent) {
20
20
  let trailing = '';
21
21
  let telemetry = { usageAvailable: false };
22
+ let reportedModels = [];
22
23
  const update = (lines) => {
23
- const next = parseProviderTelemetry(agent, [...lines]);
24
- if (next.usageAvailable)
25
- telemetry = next;
24
+ for (const line of lines) {
25
+ const next = parseProviderTelemetry(agent, [line]);
26
+ if (next.reportedModels)
27
+ reportedModels = next.reportedModels;
28
+ else if (next.tokens?.model)
29
+ reportedModels = [next.tokens.model];
30
+ // Provider result usage is cumulative: replace the latest measurement,
31
+ // never add it to earlier results or to assistant-message snapshots.
32
+ if (next.tokens || next.partialUsage)
33
+ telemetry = next;
34
+ }
26
35
  };
27
36
  return {
28
37
  append(text) {
@@ -34,7 +43,12 @@ export function createTelemetryAccumulator(agent) {
34
43
  if (trailing)
35
44
  update([trailing]);
36
45
  trailing = '';
37
- return telemetry;
46
+ if (telemetry.tokens) {
47
+ const { model: _model, ...tokens } = telemetry.tokens;
48
+ return { usageAvailable: telemetry.usageAvailable, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
49
+ ...(reportedModels.length > 1 ? { reportedModels } : {}) };
50
+ }
51
+ return { ...telemetry, ...(reportedModels.length ? { reportedModels } : {}) };
38
52
  },
39
53
  };
40
54
  }
@@ -1,4 +1,5 @@
1
1
  import { ModelSelectionSchema } from './contracts.js';
2
+ import { fileURLToPath } from 'node:url';
2
3
  export { providerSpawnOptions, startProviderProcess, } from './process.js';
3
4
  const argsFor = (agent, permissions) => {
4
5
  if (agent === 'claude') {
@@ -13,16 +14,38 @@ const argsFor = (agent, permissions) => {
13
14
  return ['exec', '--dangerously-bypass-approvals-and-sandbox', '--json'];
14
15
  if (permissions === 'read-only')
15
16
  return ['exec', '--sandbox', 'read-only', '--json'];
16
- return ['exec', '--sandbox', 'workspace-write', '--approve-for-me', '--json'];
17
+ // Automatic review already selects workspace-write and conflicts with --sandbox.
18
+ return ['exec', '--approve-for-me', '--json'];
17
19
  }
18
20
  if (permissions === 'unsafe')
19
- return ['--yolo'];
21
+ return ['--yolo', '--output-format', 'stream-json'];
20
22
  const approval = permissions === 'read-only' ? 'plan' : 'auto_edit';
21
- return ['--approval-mode', approval, '--sandbox'];
23
+ return ['--approval-mode', approval, '--sandbox', '--output-format', 'stream-json'];
22
24
  };
23
- export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}) {
25
+ export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}, output = {}) {
24
26
  const parsedSelection = ModelSelectionSchema.parse(selection);
27
+ if (agent === 'gemini' && parsedSelection.bare)
28
+ throw new Error('Gemini does not support the bare startup selection');
29
+ if (agent === 'gemini' && parsedSelection.reasoningEffort)
30
+ throw new Error('Gemini does not support the reasoningEffort selection');
31
+ if (agent === 'gemini' && parsedSelection.nativeMultiAgent === true)
32
+ throw new Error('Gemini does not support enabling the nativeMultiAgent selection');
25
33
  const args = argsFor(agent, permissions);
34
+ if (output.schemaFile !== undefined || output.jsonSchema !== undefined) {
35
+ if (agent === 'codex' && output.schemaFile && output.jsonSchema === undefined) {
36
+ if (/[\0\r\n]/u.test(output.schemaFile) || (process.platform === 'win32' && !/^[A-Za-z0-9_./:\\-]+$/u.test(output.schemaFile)))
37
+ throw new Error('Invalid output schema file path');
38
+ args.push('--output-schema', output.schemaFile);
39
+ }
40
+ else if (agent === 'claude' && output.jsonSchema && output.schemaFile === undefined) {
41
+ const schema = JSON.stringify(output.jsonSchema);
42
+ if (process.platform === 'win32')
43
+ throw new Error('Inline structured output is unsupported by the Windows provider shell shim');
44
+ args.push('--json-schema', schema);
45
+ }
46
+ else
47
+ throw new Error(`${agent} structured output schema requires ${agent === 'codex' ? 'schemaFile' : agent === 'claude' ? 'jsonSchema' : 'a supported native schema option (unavailable)'}`);
48
+ }
26
49
  if (parsedSelection.model)
27
50
  args.push('--model', parsedSelection.model);
28
51
  if (parsedSelection.reasoningEffort) {
@@ -33,11 +56,16 @@ export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe'
33
56
  }
34
57
  if (agent === 'codex' && parsedSelection.nativeMultiAgent === false)
35
58
  args.push('--disable', 'multi_agent');
59
+ if (agent === 'claude' && parsedSelection.nativeMultiAgent === false)
60
+ args.push('--disallowedTools', 'Agent', 'Task', 'TeamCreate', 'SendMessage');
36
61
  if (parsedSelection.bare) {
37
62
  if (agent === 'codex')
38
63
  args.push('--ignore-user-config');
39
64
  else if (agent === 'claude')
40
65
  args.push('--bare');
41
66
  }
67
+ if (agent === 'gemini' && parsedSelection.nativeMultiAgent === false) {
68
+ return { command: process.execPath, args: [fileURLToPath(new URL('../../hooks/bounded-gemini.mjs', import.meta.url)), ...args], input: prompt, cwd };
69
+ }
42
70
  return { command: agent, args, input: prompt, cwd };
43
71
  }
@@ -1,4 +1,4 @@
1
- const finite = (value) => typeof value === 'number' && Number.isFinite(value) ? value : undefined;
1
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
2
2
  function parseJson(value) {
3
3
  try {
4
4
  return { ok: true, value: JSON.parse(value) };
@@ -30,6 +30,8 @@ export function parseProviderResult(agent, output) {
30
30
  const event = parsed.value;
31
31
  switch (agent) {
32
32
  case 'claude':
33
+ if (event.type === 'result' && directMachineResult(event.structured_output) !== undefined)
34
+ return event.structured_output;
33
35
  if (event.type === 'result' && typeof event.result === 'string')
34
36
  fragments.push(event.result);
35
37
  break;
@@ -69,6 +71,7 @@ export function parseProviderTelemetry(agent, lines) {
69
71
  let reasoningOutputTokens;
70
72
  let totalCostUsd;
71
73
  let model;
74
+ let reportedModels = [];
72
75
  for (const line of lines) {
73
76
  let parsed;
74
77
  try {
@@ -89,11 +92,44 @@ export function parseProviderTelemetry(agent, lines) {
89
92
  : stats?.usage && typeof stats.usage === 'object'
90
93
  ? stats.usage
91
94
  : stats);
92
- const models = stats?.models && typeof stats.models === 'object' ? stats.models : undefined;
93
- const firstModel = models ? Object.entries(models)[0] : undefined;
95
+ const models = isRecord(stats?.models) ? stats.models : undefined;
96
+ const modelEntries = models ? Object.entries(models) : [];
97
+ const firstModel = modelEntries.length === 1 ? modelEntries[0] : undefined;
94
98
  const modelUsage = firstModel?.[1] && typeof firstModel[1] === 'object' ? firstModel[1] : undefined;
95
99
  const nestedModelTokens = modelUsage?.tokens && typeof modelUsage.tokens === 'object' ? modelUsage.tokens : undefined;
96
- const source = nestedModelTokens ?? modelUsage ?? usage;
100
+ // Streaming stats report aggregate input_tokens (including cached input).
101
+ // Older JSON stats only provide model-local token objects. Sum a field
102
+ // only when every model measured it; a missing measurement is not zero.
103
+ let source = usage ?? nestedModelTokens ?? modelUsage;
104
+ if (agent === 'gemini' && modelEntries.length > 0) {
105
+ reportedModels = modelEntries.map(([name]) => name);
106
+ model = firstModel?.[0];
107
+ const fields = {
108
+ input_tokens: ['input_tokens', 'inputTokens', 'promptTokenCount', 'input'],
109
+ output_tokens: ['output_tokens', 'outputTokens', 'candidatesTokenCount', 'output'],
110
+ cached_input_tokens: ['cached_input_tokens', 'cachedInputTokens', 'cachedContentTokenCount', 'cached'],
111
+ reasoning_output_tokens: ['reasoning_output_tokens', 'reasoningOutputTokens', 'thoughtsTokenCount', 'thoughts'],
112
+ };
113
+ const totals = {};
114
+ for (const [field, aliases] of Object.entries(fields)) {
115
+ const aggregate = aliases.map(key => finite(usage?.[key])).find(value => value !== undefined);
116
+ if (aggregate !== undefined) {
117
+ totals[field] = aggregate;
118
+ continue;
119
+ }
120
+ const values = modelEntries.map(([, value]) => {
121
+ const entry = isRecord(value) ? value : {};
122
+ const tokens = isRecord(entry.tokens) ? entry.tokens : entry;
123
+ return aliases.map(key => finite(tokens[key])).find(value => value !== undefined);
124
+ });
125
+ if (values.every(value => value !== undefined))
126
+ totals[field] = values.reduce((sum, value) => sum + value, 0);
127
+ }
128
+ source = { ...totals, ...usage };
129
+ const aggregateCached = finite(usage?.cached_input_tokens ?? usage?.cached);
130
+ if (aggregateCached !== undefined)
131
+ source.cached_input_tokens = aggregateCached;
132
+ }
97
133
  const inValue = finite(source?.input_tokens ?? source?.inputTokens ?? source?.prompt_tokens ?? source?.promptTokenCount ?? source?.input);
98
134
  const cachedValue = finite(source?.cached_input_tokens ?? source?.cache_read_input_tokens ?? source?.cachedInputTokens ?? source?.cachedContentTokenCount ?? source?.cached);
99
135
  const cacheWriteValue = finite(source?.cache_write_input_tokens ?? source?.cache_creation_input_tokens ?? source?.cacheWriteInputTokens ?? source?.cacheWrite);
@@ -113,19 +149,31 @@ export function parseProviderTelemetry(agent, lines) {
113
149
  if (costValue !== undefined)
114
150
  totalCostUsd = costValue;
115
151
  const eventModel = event.model ?? message?.model ?? firstModel?.[0];
116
- if (typeof eventModel === 'string' && eventModel)
152
+ if (typeof eventModel === 'string' && eventModel && reportedModels.length <= 1)
117
153
  model = eventModel;
118
154
  }
119
- if (inputTokens === undefined && outputTokens === undefined)
120
- return { usageAvailable: false };
155
+ if (inputTokens === undefined || outputTokens === undefined) {
156
+ const partialUsage = {
157
+ ...(inputTokens !== undefined ? { inputTokens } : {}),
158
+ ...(outputTokens !== undefined ? { outputTokens } : {}),
159
+ ...(cachedInputTokens !== undefined ? { cachedInputTokens } : {}),
160
+ ...(cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens } : {}),
161
+ ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}),
162
+ ...(totalCostUsd !== undefined ? { totalCostUsd } : {}),
163
+ };
164
+ return { usageAvailable: false,
165
+ ...(Object.keys(partialUsage).length ? { partialUsage } : {}),
166
+ ...(reportedModels.length ? { reportedModels } : model ? { reportedModels: [model] } : {}),
167
+ };
168
+ }
121
169
  const tokens = {
122
- inputTokens: inputTokens ?? 0,
170
+ inputTokens,
123
171
  ...(cachedInputTokens !== undefined ? { cachedInputTokens } : {}),
124
172
  ...(cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens } : {}),
125
- outputTokens: outputTokens ?? 0,
173
+ outputTokens,
126
174
  ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}),
127
175
  ...(totalCostUsd !== undefined ? { totalCostUsd } : {}),
128
176
  ...(model ? { model } : {}),
129
177
  };
130
- return { usageAvailable: true, tokens };
178
+ return { usageAvailable: true, tokens, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
131
179
  }