@hecer/yoke 1.6.1 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.codex-plugin/plugin.json +1 -1
  3. package/CHANGELOG.md +31 -2
  4. package/README.md +43 -9
  5. package/canon/manifest.yaml +1 -1
  6. package/canon/tools/gemini-rtk-hook.mjs +25 -0
  7. package/dist/agents/contracts.js +2 -0
  8. package/dist/agents/process-streams.js +18 -4
  9. package/dist/agents/process.js +2 -0
  10. package/dist/agents/providers.js +25 -4
  11. package/dist/agents/telemetry.js +58 -10
  12. package/dist/check/command.js +114 -0
  13. package/dist/cli.js +120 -1
  14. package/dist/context/packet.js +32 -0
  15. package/dist/dashboard/page.js +32 -0
  16. package/dist/dashboard/registry.js +68 -0
  17. package/dist/dashboard/server.js +160 -0
  18. package/dist/estimation/durations.js +20 -0
  19. package/dist/estimation/schedule.js +41 -0
  20. package/dist/execution/actions.js +24 -0
  21. package/dist/goals/command.js +187 -0
  22. package/dist/loop/cleanup.js +3 -0
  23. package/dist/loop/dispatcher.js +14 -3
  24. package/dist/loop/git.js +5 -4
  25. package/dist/loop/loop.js +23 -3
  26. package/dist/loop/parallel-adapters.js +3 -2
  27. package/dist/loop/parallel-command.js +10 -1
  28. package/dist/loop/prd.js +3 -0
  29. package/dist/loop/recovery.js +51 -0
  30. package/dist/loop/reporter.js +92 -8
  31. package/dist/loop/run-command.js +20 -3
  32. package/dist/loop/runner.js +9 -12
  33. package/dist/loop/scheduler.js +38 -1
  34. package/dist/loop/watchdog.js +2 -0
  35. package/dist/observability/events.js +67 -0
  36. package/dist/retrofit/config.js +9 -0
  37. package/dist/retrofit/gitignore.js +5 -0
  38. package/dist/retrofit/planners/gemini.js +11 -3
  39. package/dist/routing/router.js +26 -8
  40. package/dist/workspace/fingerprint.js +66 -0
  41. package/dist/workspace/state.js +20 -0
  42. package/docs/PRODUCT-DIRECTION-2026-09-05.md +183 -0
  43. package/docs/PUBLISHING.md +25 -2
  44. package/docs/VERIFIED-PROJECTS-VALIDATION.md +29 -0
  45. package/docs/VERIFIED-PROJECTS.md +147 -0
  46. package/docs/superpowers/plans/2026-09-05-verified-projects.md +83 -0
  47. package/gemini-extension.json +1 -1
  48. package/package.json +1 -1
@@ -2,7 +2,7 @@
2
2
  "$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
3
3
  "name": "yoke",
4
4
  "displayName": "Yoke",
5
- "version": "1.6.1",
5
+ "version": "1.7.0",
6
6
  "description": "Cross-agent coding harness: one curated skill canon (TDD, brainstorming, plans, reviews, shipping, design verification) plus mechanical safety gates and an autonomous loop via the yoke CLI.",
7
7
  "author": { "name": "HECer", "url": "https://github.com/HECer" },
8
8
  "homepage": "https://github.com/HECer/yoke#readme",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.6.1",
3
+ "version": "1.7.0",
4
4
  "description": "Cross-agent coding discipline, mechanical gates, and release workflows",
5
5
  "skills": "./canon/skills/",
6
6
  "hooks": "./hooks/hooks.json"
package/CHANGELOG.md CHANGED
@@ -2,11 +2,40 @@
2
2
 
3
3
  ## Unreleased
4
4
 
5
+ ## 1.7.0 — 2026-09-05
6
+
7
+ ### Added
8
+ - Independent `yoke check` with executable acceptance mapping, protected test infrastructure and content-bound evidence.
9
+ - Durable project goals, provider handoff, checkpoint budgets, interruption accounting and project-scoped recovery.
10
+ - Local project registry and loopback dashboard with goals, task estimates, evidence, consumption and pause controls.
11
+ - Explicit routing rules with persisted gate-driven escalation; bounded tool actions without model calls.
12
+ - Task-aware context packets, advisory write scopes, dependency-depth scheduling and empirical time ranges with prediction error records.
13
+
14
+ ### Fixed
15
+ - Failed serial isolated worktrees are retained and can be explicitly resumed against their original target and PRD.
16
+ - Reviewer fingerprints include untracked contents and acceptance inputs; unsupported nested repository identity fails closed.
17
+ - Gemini always emits streaming telemetry, preserves model identity, honors aggregate token aliases and rejects unsupported selections. Its RTK hook merges native settings without a shell dependency.
18
+ - Runtime evidence is excluded from Git gates and commits in existing projects. Partial usage and costs remain visibly incomplete.
19
+ - Protected acceptance is checked after verification and repair, and linked goal state cannot overwrite unrelated files through pause.
20
+
21
+ ### Validation limits
22
+ - Live authenticated provider comparisons and calibrated development-time/cost estimates are not established by the automated tests. Browser proof and independent model review still require explicit configuration.
23
+
24
+ ## 1.6.2 — 2026-09-02
25
+
26
+ ### Added
27
+ - GitHub Releases can now publish `@hecer/yoke` through npm trusted publishing with short-lived OIDC credentials and automatic provenance, without a long-lived npm token.
28
+
29
+ ### Fixed
30
+ - Codex safe-mode invocations now use the supported `workspace-write` sandbox with `--approve-for-me`; the removed `--full-auto` flag no longer blocks current Codex CLI releases.
31
+ - Windows provider cleanup rechecks termination after process close and accepts an already-absent process as successfully cleaned up, avoiding stale ownership records and unnecessary watchdog waits.
32
+ - Successful stale-loop cleanup removes obsolete runtime status, so `yoke loop status` no longer reports a dead run as `RUNNING`.
33
+
5
34
  ## 1.6.1 — 2026-08-21
6
35
 
7
36
  ### Fixed
8
- - Release metadata now counts platform-conditional tests consistently on Windows and Ubuntu, so `docs:check` no longer fails after an otherwise green cross-platform test matrix.
9
- - Provider cleanup now rechecks termination after the child closes, preventing stale ownership records when Windows reports process exit asynchronously.
37
+ - Release metadata counts platform-conditional tests consistently on Windows and Ubuntu.
38
+ - Provider cleanup confirms termination after child close, preventing stale ownership records when Windows reports process exit asynchronously.
10
39
 
11
40
  ## 1.6.0 — 2026-08-20
12
41
 
package/README.md CHANGED
@@ -2,14 +2,14 @@
2
2
 
3
3
  # 🐂 Yoke
4
4
 
5
- <!-- yoke:version:start -->1.6.1<!-- yoke:version:end -->
6
- <!-- yoke:tests:start -->1019<!-- yoke:tests:end -->
5
+ <!-- yoke:version:start -->1.7.0<!-- yoke:version:end -->
6
+ <!-- yoke:tests:start -->1100<!-- yoke:tests:end -->
7
7
  <!-- yoke:skills:start -->34<!-- yoke:skills:end -->
8
8
  <!-- yoke:agents:start -->Claude | Codex | Gemini<!-- yoke:agents:end -->
9
9
 
10
10
  ### One harness, three agents — and zero trust in "done."
11
11
 
12
- **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Then, when you want it, an opt-in autonomous loop ships your spec story-by-story: tested, cross-model-reviewed, committed — **with a screenshot to prove every story and a video for every failure**.
12
+ **Yoke** installs one curated canon of skills, **mechanical safety gates**, and tool wiring into any project — natively for **Claude Code, OpenAI Codex CLI, and Gemini CLI**. Its opt-in loop implements and verifies stories before committing. Independent review and browser proofs run when configured; screenshots and videos require the browser smoke gate.
13
13
 
14
14
  [![npm](https://img.shields.io/npm/v/%40hecer%2Fyoke?logo=npm&color=CB3837)](https://www.npmjs.com/package/@hecer/yoke)
15
15
  [![npm downloads](https://img.shields.io/npm/dm/%40hecer%2Fyoke?logo=npm)](https://www.npmjs.com/package/@hecer/yoke)
@@ -17,7 +17,7 @@
17
17
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](#-license)
18
18
  ![Node](https://img.shields.io/badge/node-%E2%89%A520-339933?logo=node.js&logoColor=white)
19
19
  ![TypeScript](https://img.shields.io/badge/TypeScript-3178C6?logo=typescript&logoColor=white)
20
- ![Tests](https://img.shields.io/badge/tests-1019%20passing-brightgreen.svg)
20
+ ![Tests](https://img.shields.io/badge/tests-1100%20defined-blue.svg)
21
21
  ![Agents](https://img.shields.io/badge/agents-Claude%20%7C%20Codex%20%7C%20Gemini-8A2BE2)
22
22
  ![Built with TDD](https://img.shields.io/badge/built%20with-TDD%20%2B%20review-ff69b4.svg)
23
23
 
@@ -27,6 +27,36 @@
27
27
 
28
28
  > **TL;DR** — `yoke setup .` asks six questions and installs the native harness for your agent. `yoke new my-app --idea="..."` bootstraps a project and drafts its story backlog. `yoke loop run my-app --isolate --review` then implements it behind hard gates: **clean tree → acceptance criteria → your real tests green → an independent model approves → commit**. Add `--parallel=N` for dependency-aware workers, or declare a reference and add `--quality` for a bounded critic/repair gauntlet. If any blocking gate is red, nothing is committed. Proof lives in `.yoke/proof/<story>/`.
29
29
 
30
+ **New in 1.7.0:** [verified project goals, recovery and one local dashboard for all your registered projects](docs/VERIFIED-PROJECTS.md). Check an existing project, continue a bounded goal with Codex, Claude or Gemini, and inspect tasks, acceptance evidence, recorded consumption and estimated timing in one place.
31
+
32
+ ### One dashboard, multiple projects
33
+
34
+ ```sh
35
+ npm install -g @hecer/yoke@latest
36
+ yoke projects add /path/to/frontend
37
+ yoke projects add /path/to/backend
38
+ yoke dashboard --no-register
39
+ ```
40
+
41
+ Open the printed `http://127.0.0.1:...` URL. Each registered project has its own goals, tasks and evidence. The dashboard shows available worker state, per-task duration estimates, planned start offsets, input/output tokens, costs and unknown measurements. You can request a goal pause at a safe boundary.
42
+
43
+ Projects are registered explicitly; this version does not automatically discover every process or aggregate other computers. Start/resume and budget changes use the CLI. Missing history appears as unknown; time ranges are empirical estimates, not exact deadlines.
44
+
45
+ ### Verified goals and efficient execution
46
+
47
+ ```sh
48
+ yoke check /path/to/project --json
49
+ # First define executable criteria and protected tests in .yoke/acceptance.yaml.
50
+ yoke goal set /path/to/project --objective="Complete guest checkout" --attempts=3 --minutes=30
51
+ yoke goal run /path/to/project --runner=codex
52
+ yoke goal resume /path/to/project --runner=claude
53
+ yoke goal handoff /path/to/project
54
+ ```
55
+
56
+ Goals persist their objective, attempts and check evidence across runs. Changed protected tests block acceptance; failed work is retained. Explicit routing rules bypass controller calls, configured tool actions use no model, and failed rule-based attempts can escalate to a stronger worker. Context selection stays within a character budget; declared write scopes and dependencies guide parallel scheduling.
57
+
58
+ See the [1.7 workflow guide](docs/VERIFIED-PROJECTS.md) for setup, Gemini selection, recovery and budgets. Provider contracts do not establish equal model quality; live comparative savings and calibrated time predictions remain unmeasured. Token budgets apply between provider calls, and browser proofs still require a configured smoke gate.
59
+
30
60
  Yoke 1.5 keeps failed gate output compact without throwing evidence away: deterministic previews
31
61
  retain actionable failures and final summaries, while large complete stdout/stderr remains available
32
62
  in private, content-addressed local artifacts. Existing projects keep their serial behavior and use
@@ -50,7 +80,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
50
80
 
51
81
  | The pain | What actually happens | What Yoke does about it |
52
82
  |---|---|---|
53
- | 🎭 **The verification gap** — *"agent says done, but it isn't"* | Agents submit confidently on 100% of runs while resolving far fewer; "all tests pass" when they were never run ([silent-failures research](https://arxiv.org/pdf/2603.25764)) | The loop trusts **your verify command's exit code**, never the agent's word. A story is `passes: true` only after tests are green, the reviewer approved, and the commit landed — atomically. Plus: **screenshot proofs** per story. |
83
+ | 🎭 **The verification gap** — *"agent says done, but it isn't"* | A success message can omit untested acceptance criteria. | The loop executes acceptance and project checks. Enabled review and browser gates must pass before a story lands. `yoke check` exposes unmapped outcomes as unverified. |
54
84
  | 🔀 **Three agents, three configs** | Teams hand-maintain `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, skills, and MCP wiring separately — copy-paste drift everywhere | **One canon → `yoke retrofit`** generates the idiomatic native artifacts for each agent. Change the canon once, re-retrofit everywhere. |
55
85
  | 🌀 **Overnight loops going off the rails** | Raw Ralph-loop users "wake up to broken codebases that don't compile" | Yoke is **"Ralph, but with gates"**: clean-worktree gate, acceptance-criteria gate, green-tests gate, review gate, per-story worktree isolation, idle-timeout watchdog, single-flight lock, commit integrity. |
56
86
  | 😵 **Review fatigue** | AI adoption nearly doubles PR volume and review time; humans start skimming | **`yoke review`**: a second model writes a schema-validated pass/fail verdict — chainable into verify, pre-push, or CI. Cross-model review catches what self-review misses. |
@@ -59,7 +89,7 @@ Agentic coding in 2026 fails in four well-documented ways. Yoke answers each one
59
89
 
60
90
  **Who it's not for:** if you want a chat pair-programmer with no process, you don't need a harness. Yoke is for shipping with discipline.
61
91
 
62
- ## ⏱️ 60 seconds: idea → tested, photographed software
92
+ ## Example: idea → verified implementation
63
93
 
64
94
  ```console
65
95
  $ yoke new reading-app --idea="a web app that tracks my reading list"
@@ -78,10 +108,10 @@ $ yoke loop run reading-app --isolate --review --max=10
78
108
  # nothing was committed. fix, then re-run.
79
109
 
80
110
  $ ls reading-app/.yoke/proof/STORY-2/
81
- home.png list.png # photographic evidence, labelled per story
111
+ home.png list.png # example when browser smoke is configured
82
112
  ```
83
113
 
84
- Every claim in that transcript is enforced by code paths with tests behind them — 1019 of them, and this repo was built by its own loop and gates ([how it was built](#-why--how-it-was-built)).
114
+ This is an illustrative transcript. Actual durations depend on the project, provider, retries and enabled gates; browser proof requires a configured smoke flow. See [how it was built](#-why--how-it-was-built).
85
115
 
86
116
  ## 🚀 Quickstart
87
117
 
@@ -157,6 +187,10 @@ Yoke's CLI is deterministic and chainable by design: an agent (or a shell `&&`)
157
187
 
158
188
  | Command | What it does | Exit codes |
159
189
  |---|---|---|
190
+ | `yoke dashboard [dir] [--no-register] [--port=N]` | Local overview for every registered project; optionally register `dir` first | `0` stopped normally · `1` shutdown failure · `2` unavailable |
191
+ | `yoke projects add\|list\|remove` | Register a project, list registrations or remove a reference by ID | `0` · `2` invalid/unavailable |
192
+ | `yoke check [dir] [--json] [--requirement=] [--protect [--refresh]]` | Execute acceptance checks or explicitly pin their infrastructure | `0` passed/pinned · `1` failed · `2` unverified/unavailable |
193
+ | `yoke goal set\|run\|resume\|pause\|status\|handoff\|budget [dir]` | Durable objectives, provider handoff, protected checks and checkpoint budgets | run/resume: `0` complete · `1` unfinished · `2` unavailable |
160
194
  | `yoke setup [dir] [--yes] [--host=] [--agent=] [--runner=] [--code-graph=] [--decision-policy=] [--loop\|--no-loop] [--routing\|--no-routing]` | Shared six-question setup for Claude, Codex, and Gemini; adaptive routing is always an explicit opt-in | `0` · `1` invalid setup |
161
195
  | `yoke validate [canonDir]` | Validate the canon (schema, frontmatter, templates) | `0` valid · `1` errors |
162
196
  | `yoke new <dir> [--idea=] [--agent=] [--runner=] [--loop]` | Greenfield bootstrap: git init → scaffold → retrofit → context → PRD (drafted from `--idea`) → committed | `0` · `1` usage / non-empty dir / draft failed (scaffold survives) · `2` draft agent unavailable |
@@ -856,7 +890,7 @@ release provenance.
856
890
  ## 🧪 Development
857
891
 
858
892
  ```bash
859
- npm test # vitest (1019 tests)
893
+ npm test # vitest (1100 tests)
860
894
  npm run build # tsc, no emit errors
861
895
  npm run yoke -- validate canon
862
896
  ```
@@ -1,5 +1,5 @@
1
1
  name: yoke-canon
2
- version: 1.6.1
2
+ version: 1.7.0
3
3
  agents: [claude, codex, gemini]
4
4
  skills:
5
5
  - { id: tdd, path: skills/tdd, kind: methodology, invocation: auto }
@@ -0,0 +1,25 @@
1
+ #!/usr/bin/env node
2
+ // Pure BeforeTool argument adapter. Never evaluates or executes tool commands.
3
+ import { readFileSync } from 'node:fs'
4
+
5
+ const record = value => value !== null && typeof value === 'object' && !Array.isArray(value)
6
+ try {
7
+ const event = JSON.parse(readFileSync(0, 'utf8'))
8
+ if (!record(event) || typeof event.tool_name !== 'string' ||
9
+ (event.hook_event_name !== undefined && event.hook_event_name !== 'BeforeTool')) throw new Error('event')
10
+ let response = {}
11
+ if (event.tool_name === 'run_shell_command') {
12
+ if (!record(event.tool_input) || typeof event.tool_input.command !== 'string' || !event.tool_input.command.trim() || event.tool_input.command.includes('\0')) throw new Error('command')
13
+ const command = event.tool_input.command
14
+ // Restrict rewriting to simple supported invocations. Leave shell syntax,
15
+ // quoted executables, assignments and existing RTK wrappers untouched.
16
+ if (!/[;&|<>`$\r\n()]/u.test(command) && /^\s*(?:git|rg|npm|npx|cargo|pytest|go|docker|kubectl)\s/u.test(command)) {
17
+ const rewritten = command.trimStart().replace(/^rg\s/u, 'grep ')
18
+ response = { hookSpecificOutput: { tool_input: { ...event.tool_input, command: `rtk ${rewritten}` } } }
19
+ }
20
+ }
21
+ process.stdout.write(JSON.stringify(response) + '\n')
22
+ } catch {
23
+ process.stderr.write('Invalid Gemini BeforeTool hook input\n')
24
+ process.exitCode = 2
25
+ }
@@ -25,6 +25,8 @@ const ProviderTokenUsageSchema = z.object({
25
25
  export const ProviderTelemetrySchema = z.object({
26
26
  usageAvailable: z.boolean(),
27
27
  tokens: ProviderTokenUsageSchema.optional(),
28
+ partialUsage: ProviderTokenUsageSchema.partial().optional(),
29
+ reportedModels: z.array(z.string().min(1)).optional(),
28
30
  }).superRefine((telemetry, ctx) => {
29
31
  if (telemetry.usageAvailable && !telemetry.tokens) {
30
32
  ctx.addIssue({ code: 'custom', path: ['tokens'], message: 'usageAvailable telemetry requires token totals' });
@@ -19,10 +19,19 @@ export function createBoundedOutput(limitBytes) {
19
19
  export function createTelemetryAccumulator(agent) {
20
20
  let trailing = '';
21
21
  let telemetry = { usageAvailable: false };
22
+ let reportedModels = [];
22
23
  const update = (lines) => {
23
- const next = parseProviderTelemetry(agent, [...lines]);
24
- if (next.usageAvailable)
25
- telemetry = next;
24
+ for (const line of lines) {
25
+ const next = parseProviderTelemetry(agent, [line]);
26
+ if (next.reportedModels)
27
+ reportedModels = next.reportedModels;
28
+ else if (next.tokens?.model)
29
+ reportedModels = [next.tokens.model];
30
+ // Provider result usage is cumulative: replace the latest measurement,
31
+ // never add it to earlier results or to assistant-message snapshots.
32
+ if (next.tokens || next.partialUsage)
33
+ telemetry = next;
34
+ }
26
35
  };
27
36
  return {
28
37
  append(text) {
@@ -34,7 +43,12 @@ export function createTelemetryAccumulator(agent) {
34
43
  if (trailing)
35
44
  update([trailing]);
36
45
  trailing = '';
37
- return telemetry;
46
+ if (telemetry.tokens) {
47
+ const { model: _model, ...tokens } = telemetry.tokens;
48
+ return { usageAvailable: telemetry.usageAvailable, tokens: { ...tokens, ...(reportedModels.length === 1 ? { model: reportedModels[0] } : {}) },
49
+ ...(reportedModels.length > 1 ? { reportedModels } : {}) };
50
+ }
51
+ return { ...telemetry, ...(reportedModels.length ? { reportedModels } : {}) };
38
52
  },
39
53
  };
40
54
  }
@@ -81,6 +81,8 @@ export function startProviderProcess(agent, invocation, options = {}) {
81
81
  telemetry: telemetry.finish(),
82
82
  });
83
83
  const finalize = (exitCode) => {
84
+ // Windows can emit close before taskkill's process-tree state is observable.
85
+ // Reconfirm here so successful termination does not leave a stale ownership record.
84
86
  if (termination && pid !== undefined && !terminationConfirmed) {
85
87
  terminationConfirmed = terminateProcessTree(pid, true);
86
88
  }
@@ -13,16 +13,37 @@ const argsFor = (agent, permissions) => {
13
13
  return ['exec', '--dangerously-bypass-approvals-and-sandbox', '--json'];
14
14
  if (permissions === 'read-only')
15
15
  return ['exec', '--sandbox', 'read-only', '--json'];
16
- return ['exec', '--full-auto', '--json'];
16
+ return ['exec', '--sandbox', 'workspace-write', '--approve-for-me', '--json'];
17
17
  }
18
18
  if (permissions === 'unsafe')
19
- return ['--yolo'];
19
+ return ['--yolo', '--output-format', 'stream-json'];
20
20
  const approval = permissions === 'read-only' ? 'plan' : 'auto_edit';
21
- return ['--approval-mode', approval, '--sandbox'];
21
+ return ['--approval-mode', approval, '--sandbox', '--output-format', 'stream-json'];
22
22
  };
23
- export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}) {
23
+ export function buildProviderInvocation(agent, prompt, cwd, permissions = 'safe', selection = {}, output = {}) {
24
24
  const parsedSelection = ModelSelectionSchema.parse(selection);
25
+ if (agent === 'gemini' && parsedSelection.bare)
26
+ throw new Error('Gemini does not support the bare startup selection');
27
+ if (agent === 'gemini' && parsedSelection.reasoningEffort)
28
+ throw new Error('Gemini does not support the reasoningEffort selection');
29
+ if (agent === 'gemini' && parsedSelection.nativeMultiAgent !== undefined)
30
+ throw new Error('Gemini does not support the nativeMultiAgent selection');
25
31
  const args = argsFor(agent, permissions);
32
+ if (output.schemaFile !== undefined || output.jsonSchema !== undefined) {
33
+ if (agent === 'codex' && output.schemaFile && output.jsonSchema === undefined) {
34
+ if (/[\0\r\n]/u.test(output.schemaFile) || (process.platform === 'win32' && !/^[A-Za-z0-9_./:\\-]+$/u.test(output.schemaFile)))
35
+ throw new Error('Invalid output schema file path');
36
+ args.push('--output-schema', output.schemaFile);
37
+ }
38
+ else if (agent === 'claude' && output.jsonSchema && output.schemaFile === undefined) {
39
+ const schema = JSON.stringify(output.jsonSchema);
40
+ if (process.platform === 'win32')
41
+ throw new Error('Inline structured output is unsupported by the Windows provider shell shim');
42
+ args.push('--json-schema', schema);
43
+ }
44
+ else
45
+ throw new Error(`${agent} structured output schema requires ${agent === 'codex' ? 'schemaFile' : agent === 'claude' ? 'jsonSchema' : 'a supported native schema option (unavailable)'}`);
46
+ }
26
47
  if (parsedSelection.model)
27
48
  args.push('--model', parsedSelection.model);
28
49
  if (parsedSelection.reasoningEffort) {
@@ -1,4 +1,4 @@
1
- const finite = (value) => typeof value === 'number' && Number.isFinite(value) ? value : undefined;
1
+ const finite = (value) => typeof value === 'number' && Number.isFinite(value) && value >= 0 ? value : undefined;
2
2
  function parseJson(value) {
3
3
  try {
4
4
  return { ok: true, value: JSON.parse(value) };
@@ -30,6 +30,8 @@ export function parseProviderResult(agent, output) {
30
30
  const event = parsed.value;
31
31
  switch (agent) {
32
32
  case 'claude':
33
+ if (event.type === 'result' && directMachineResult(event.structured_output) !== undefined)
34
+ return event.structured_output;
33
35
  if (event.type === 'result' && typeof event.result === 'string')
34
36
  fragments.push(event.result);
35
37
  break;
@@ -69,6 +71,7 @@ export function parseProviderTelemetry(agent, lines) {
69
71
  let reasoningOutputTokens;
70
72
  let totalCostUsd;
71
73
  let model;
74
+ let reportedModels = [];
72
75
  for (const line of lines) {
73
76
  let parsed;
74
77
  try {
@@ -89,11 +92,44 @@ export function parseProviderTelemetry(agent, lines) {
89
92
  : stats?.usage && typeof stats.usage === 'object'
90
93
  ? stats.usage
91
94
  : stats);
92
- const models = stats?.models && typeof stats.models === 'object' ? stats.models : undefined;
93
- const firstModel = models ? Object.entries(models)[0] : undefined;
95
+ const models = isRecord(stats?.models) ? stats.models : undefined;
96
+ const modelEntries = models ? Object.entries(models) : [];
97
+ const firstModel = modelEntries.length === 1 ? modelEntries[0] : undefined;
94
98
  const modelUsage = firstModel?.[1] && typeof firstModel[1] === 'object' ? firstModel[1] : undefined;
95
99
  const nestedModelTokens = modelUsage?.tokens && typeof modelUsage.tokens === 'object' ? modelUsage.tokens : undefined;
96
- const source = nestedModelTokens ?? modelUsage ?? usage;
100
+ // Streaming stats report aggregate input_tokens (including cached input).
101
+ // Older JSON stats only provide model-local token objects. Sum a field
102
+ // only when every model measured it; a missing measurement is not zero.
103
+ let source = usage ?? nestedModelTokens ?? modelUsage;
104
+ if (agent === 'gemini' && modelEntries.length > 0) {
105
+ reportedModels = modelEntries.map(([name]) => name);
106
+ model = firstModel?.[0];
107
+ const fields = {
108
+ input_tokens: ['input_tokens', 'inputTokens', 'promptTokenCount', 'input'],
109
+ output_tokens: ['output_tokens', 'outputTokens', 'candidatesTokenCount', 'output'],
110
+ cached_input_tokens: ['cached_input_tokens', 'cachedInputTokens', 'cachedContentTokenCount', 'cached'],
111
+ reasoning_output_tokens: ['reasoning_output_tokens', 'reasoningOutputTokens', 'thoughtsTokenCount', 'thoughts'],
112
+ };
113
+ const totals = {};
114
+ for (const [field, aliases] of Object.entries(fields)) {
115
+ const aggregate = aliases.map(key => finite(usage?.[key])).find(value => value !== undefined);
116
+ if (aggregate !== undefined) {
117
+ totals[field] = aggregate;
118
+ continue;
119
+ }
120
+ const values = modelEntries.map(([, value]) => {
121
+ const entry = isRecord(value) ? value : {};
122
+ const tokens = isRecord(entry.tokens) ? entry.tokens : entry;
123
+ return aliases.map(key => finite(tokens[key])).find(value => value !== undefined);
124
+ });
125
+ if (values.every(value => value !== undefined))
126
+ totals[field] = values.reduce((sum, value) => sum + value, 0);
127
+ }
128
+ source = { ...totals, ...usage };
129
+ const aggregateCached = finite(usage?.cached_input_tokens ?? usage?.cached);
130
+ if (aggregateCached !== undefined)
131
+ source.cached_input_tokens = aggregateCached;
132
+ }
97
133
  const inValue = finite(source?.input_tokens ?? source?.inputTokens ?? source?.prompt_tokens ?? source?.promptTokenCount ?? source?.input);
98
134
  const cachedValue = finite(source?.cached_input_tokens ?? source?.cache_read_input_tokens ?? source?.cachedInputTokens ?? source?.cachedContentTokenCount ?? source?.cached);
99
135
  const cacheWriteValue = finite(source?.cache_write_input_tokens ?? source?.cache_creation_input_tokens ?? source?.cacheWriteInputTokens ?? source?.cacheWrite);
@@ -113,19 +149,31 @@ export function parseProviderTelemetry(agent, lines) {
113
149
  if (costValue !== undefined)
114
150
  totalCostUsd = costValue;
115
151
  const eventModel = event.model ?? message?.model ?? firstModel?.[0];
116
- if (typeof eventModel === 'string' && eventModel)
152
+ if (typeof eventModel === 'string' && eventModel && reportedModels.length <= 1)
117
153
  model = eventModel;
118
154
  }
119
- if (inputTokens === undefined && outputTokens === undefined)
120
- return { usageAvailable: false };
155
+ if (inputTokens === undefined || outputTokens === undefined) {
156
+ const partialUsage = {
157
+ ...(inputTokens !== undefined ? { inputTokens } : {}),
158
+ ...(outputTokens !== undefined ? { outputTokens } : {}),
159
+ ...(cachedInputTokens !== undefined ? { cachedInputTokens } : {}),
160
+ ...(cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens } : {}),
161
+ ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}),
162
+ ...(totalCostUsd !== undefined ? { totalCostUsd } : {}),
163
+ };
164
+ return { usageAvailable: false,
165
+ ...(Object.keys(partialUsage).length ? { partialUsage } : {}),
166
+ ...(reportedModels.length ? { reportedModels } : model ? { reportedModels: [model] } : {}),
167
+ };
168
+ }
121
169
  const tokens = {
122
- inputTokens: inputTokens ?? 0,
170
+ inputTokens,
123
171
  ...(cachedInputTokens !== undefined ? { cachedInputTokens } : {}),
124
172
  ...(cacheWriteInputTokens !== undefined ? { cacheWriteInputTokens } : {}),
125
- outputTokens: outputTokens ?? 0,
173
+ outputTokens,
126
174
  ...(reasoningOutputTokens !== undefined ? { reasoningOutputTokens } : {}),
127
175
  ...(totalCostUsd !== undefined ? { totalCostUsd } : {}),
128
176
  ...(model ? { model } : {}),
129
177
  };
130
- return { usageAvailable: true, tokens };
178
+ return { usageAvailable: true, tokens, ...(reportedModels.length > 1 ? { reportedModels } : {}) };
131
179
  }
@@ -0,0 +1,114 @@
1
+ import { createHash, randomUUID } from 'node:crypto';
2
+ import { existsSync, mkdirSync, readFileSync, realpathSync, writeFileSync } from 'node:fs';
3
+ import { homedir } from 'node:os';
4
+ import { dirname, isAbsolute, join, relative, resolve } from 'node:path';
5
+ import { parse } from 'yaml';
6
+ import { z } from 'zod';
7
+ import { defaultConfig, loadConfig, resolveVerifyCommand } from '../retrofit/config.js';
8
+ import { commandVerifier } from '../loop/verify.js';
9
+ import { workspaceFingerprint } from '../workspace/fingerprint.js';
10
+ import { statePath } from '../workspace/state.js';
11
+ const Criterion = z.object({ id: z.string().min(1).max(120), text: z.string().min(1).max(8000), commands: z.array(z.string().min(1).max(8000)).max(30) }).strict();
12
+ const Acceptance = z.object({ version: z.literal(1), criteria: z.array(Criterion).max(200), protected: z.array(z.string().min(1)).max(500).default([]) }).strict().superRefine((value, ctx) => {
13
+ if (new Set(value.criteria.map(c => c.id)).size !== value.criteria.length)
14
+ ctx.addIssue({ code: 'custom', message: 'Duplicate acceptance criterion id' });
15
+ });
16
+ export function loadAcceptance(root) {
17
+ const file = statePath(root, 'acceptance.yaml');
18
+ return existsSync(file) ? Acceptance.parse(parse(readFileSync(file, 'utf8'))) : null;
19
+ }
20
+ function protectedPath(root, path) {
21
+ if (isAbsolute(path))
22
+ throw new Error('Protected path must be relative');
23
+ const full = realpathSync(resolve(root, path));
24
+ const rel = relative(realpathSync(root), full);
25
+ if (rel === '..' || rel.startsWith('..\\') || rel.startsWith('../') || isAbsolute(rel))
26
+ throw new Error('Protected path escapes project');
27
+ return full;
28
+ }
29
+ function baselinePath(root) {
30
+ const id = createHash('sha256').update(realpathSync(root)).digest('hex');
31
+ return join(process.env.YOKE_STATE_DIR ?? join(homedir(), '.yoke', 'state'), 'acceptance', `${id}.json`);
32
+ }
33
+ function protectedHashes(root, paths) {
34
+ return Object.fromEntries(paths.map(path => [path, createHash('sha256').update(readFileSync(protectedPath(root, path))).digest('hex')]));
35
+ }
36
+ /** Explicitly pin acceptance outside the worker workspace. Never refreshed by check. */
37
+ export function protectAcceptance(root, refresh = false) {
38
+ const manifest = loadAcceptance(root);
39
+ if (!manifest)
40
+ throw new Error('Create .yoke/acceptance.yaml before protecting acceptance');
41
+ const paths = [...new Set(['.yoke/acceptance.yaml', ...manifest.protected, ...['package.json', 'package-lock.json', 'pnpm-lock.yaml', 'yarn.lock'].filter(p => existsSync(join(root, p)))])];
42
+ const file = baselinePath(root);
43
+ const hashes = protectedHashes(root, paths);
44
+ mkdirSync(dirname(file), { recursive: true });
45
+ writeFileSync(file, JSON.stringify({ version: 1, hashes }), { flag: refresh ? 'w' : 'wx', mode: 0o600 });
46
+ return file;
47
+ }
48
+ export function acceptanceProtectionProblem(root, baselineRoot = root) {
49
+ const file = baselinePath(baselineRoot);
50
+ if (!existsSync(file))
51
+ return null;
52
+ try {
53
+ const baseline = z.object({ version: z.literal(1), hashes: z.record(z.string().regex(/^[a-f0-9]{64}$/)) }).strict().parse(JSON.parse(readFileSync(file, 'utf8')));
54
+ if (!Object.keys(baseline.hashes).includes('.yoke/acceptance.yaml'))
55
+ return 'Invalid protected acceptance baseline';
56
+ const actual = protectedHashes(root, Object.keys(baseline.hashes));
57
+ const changed = Object.keys(actual).filter(path => actual[path] !== baseline.hashes[path]);
58
+ return changed.length ? `Protected acceptance changed: ${changed.join(', ')}` : null;
59
+ }
60
+ catch (error) {
61
+ return `Protected acceptance cannot be verified: ${error.message}`;
62
+ }
63
+ }
64
+ export function checkProject(directory, options = {}) {
65
+ const root = realpathSync(directory);
66
+ const started = Date.now();
67
+ const before = workspaceFingerprint(root);
68
+ const problem = acceptanceProtectionProblem(root);
69
+ const criteria = [];
70
+ const execute = options.execute ?? ((command, cwd) => commandVerifier(command, { phase: 'verify' })(cwd));
71
+ if (problem)
72
+ criteria.push({ id: 'protected-acceptance', text: 'Acceptance infrastructure unchanged', commands: [], status: 'failed', summary: problem });
73
+ else {
74
+ const manifest = loadAcceptance(root);
75
+ for (const criterion of manifest?.criteria ?? []) {
76
+ const results = criterion.commands.map(command => {
77
+ try {
78
+ return execute(command, root);
79
+ }
80
+ catch (error) {
81
+ return { passed: false, summary: error.message };
82
+ }
83
+ });
84
+ criteria.push({ ...criterion, status: results.length === 0 ? 'unverified' : results.every(r => r.passed) ? 'passed' : 'failed', summary: results.map(r => r.summary).join('\n') || 'No executable acceptance mapped' });
85
+ }
86
+ const command = resolveVerifyCommand(root, loadConfig(root) ?? defaultConfig('1.6.2'));
87
+ if (command) {
88
+ let result;
89
+ try {
90
+ result = execute(command, root);
91
+ }
92
+ catch (error) {
93
+ result = { passed: false, summary: error.message };
94
+ }
95
+ criteria.push({ id: 'project-suite', text: 'Configured project verification', commands: [command], status: result.passed ? 'passed' : 'failed', summary: result.summary });
96
+ }
97
+ if (options.requirement)
98
+ criteria.push({ id: 'requested-outcome', text: options.requirement, commands: [], status: 'unverified', summary: 'Map this outcome to executable criteria in .yoke/acceptance.yaml; a green suite alone is not proof of this requirement.' });
99
+ if (criteria.length === 0)
100
+ criteria.push({ id: 'acceptance', text: 'Project acceptance', commands: [], status: 'unverified', summary: 'No acceptance manifest or project verification command found' });
101
+ }
102
+ const changed = workspaceFingerprint(root) !== before;
103
+ const afterProblem = acceptanceProtectionProblem(root);
104
+ if (changed || afterProblem)
105
+ criteria.push({ id: 'source-integrity', text: 'Checked source remained stable', commands: [], status: 'failed', summary: afterProblem ?? 'Source changed during verification; run check again on a stable tree' });
106
+ const status = criteria.some(c => c.status === 'failed') ? 'failed' : criteria.some(c => c.status === 'unverified') ? 'unverified' : 'passed';
107
+ const id = randomUUID();
108
+ const evidencePath = statePath(root, 'checks', `${id}.json`);
109
+ const report = { version: 1, id, generatedAt: new Date().toISOString(), fingerprint: before, status, summary: changed ? 'Source changed during verification' : `${criteria.filter(c => c.status === 'passed').length}/${criteria.length} checks passed; ${status}`, criteria, durationMs: Date.now() - started, evidencePath };
110
+ mkdirSync(dirname(evidencePath), { recursive: true });
111
+ writeFileSync(evidencePath, JSON.stringify(report, null, 2) + '\n', { flag: 'wx', mode: 0o600 });
112
+ return report;
113
+ }
114
+ export function checkExitCode(report) { return report.status === 'passed' ? 0 : report.status === 'failed' ? 1 : 2; }