@hecer/yoke 1.11.0 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +13 -13
- package/.codex-plugin/plugin.json +7 -7
- package/CHANGELOG.md +416 -398
- package/README.md +931 -915
- package/TODOS.md +5 -5
- package/agents/docs.toml +6 -6
- package/agents/implementer.toml +6 -6
- package/agents/reviewer.toml +6 -6
- package/agents/security.toml +6 -6
- package/bench/README.md +86 -86
- package/bench/RESULTS.md +35 -35
- package/bench/output-compaction.mjs +65 -65
- package/bench/result-schema.mjs +12 -12
- package/bench/results/claude-2026-07-27T18-03-26.json +50 -50
- package/bench/results/codex-unavailable-1785175418318.json +15 -15
- package/bench/results/gemini-2026-07-27T18-03-44.json +46 -46
- package/bench/run-matrix.mjs +26 -26
- package/bench/run.mjs +106 -106
- package/canon/AGENTS.md +30 -30
- package/canon/context/DECISIONS.md +4 -4
- package/canon/context/GLOSSARY.md +11 -11
- package/canon/context/KNOWLEDGE.md +4 -4
- package/canon/context/PROJECT.md +15 -15
- package/canon/loop/loop-spec.md +65 -65
- package/canon/loop/prd.schema.md +41 -41
- package/canon/manifest.yaml +59 -59
- package/canon/policy/gates.md +7 -7
- package/canon/policy/roles.md +9 -9
- package/canon/skills/ATTRIBUTION.md +99 -99
- package/canon/skills/authoring-prd/SKILL.md +56 -56
- package/canon/skills/brainstorming/SKILL.md +164 -164
- package/canon/skills/codebase-design/DEEPENING.md +15 -15
- package/canon/skills/codebase-design/DESIGN-IT-TWICE.md +12 -12
- package/canon/skills/codebase-design/SKILL.md +39 -39
- package/canon/skills/dispatching-parallel-agents/SKILL.md +182 -182
- package/canon/skills/document-release/SKILL.md +302 -302
- package/canon/skills/domain-modeling/ADR-FORMAT.md +19 -19
- package/canon/skills/domain-modeling/CONTEXT-FORMAT.md +39 -39
- package/canon/skills/domain-modeling/SKILL.md +35 -35
- package/canon/skills/executing-plans/SKILL.md +70 -70
- package/canon/skills/finishing-a-development-branch/SKILL.md +200 -200
- package/canon/skills/health/SKILL.md +177 -177
- package/canon/skills/maintaining-context/SKILL.md +34 -34
- package/canon/skills/minimal-code/SKILL.md +21 -21
- package/canon/skills/no-ai-slop/SKILL.md +103 -103
- package/canon/skills/no-ai-slop/eval.md +43 -43
- package/canon/skills/plan-ceo-review/SKILL.md +541 -541
- package/canon/skills/plan-eng-review/SKILL.md +362 -362
- package/canon/skills/receiving-code-review/SKILL.md +213 -213
- package/canon/skills/requesting-code-review/SKILL.md +105 -105
- package/canon/skills/resolving-merge-conflicts/SKILL.md +18 -18
- package/canon/skills/retro/SKILL.md +397 -397
- package/canon/skills/review/SKILL.md +246 -246
- package/canon/skills/ship/SKILL.md +691 -691
- package/canon/skills/subagent-driven-development/SKILL.md +277 -277
- package/canon/skills/systematic-debugging/SKILL.md +296 -296
- package/canon/skills/tdd/SKILL.md +371 -371
- package/canon/skills/unslop-ui/SKILL.md +34 -34
- package/canon/skills/using-git-worktrees/SKILL.md +218 -218
- package/canon/skills/verification-before-completion/SKILL.md +139 -139
- package/canon/skills/visual-verification/SKILL.md +54 -54
- package/canon/skills/workflow/SKILL.md +22 -22
- package/canon/skills/writing-for-agents/SKILL-MECHANICS.md +27 -27
- package/canon/skills/writing-for-agents/SKILL.md +42 -42
- package/canon/skills/writing-plans/SKILL.md +152 -152
- package/canon/skills/writing-skills/SKILL.md +655 -655
- package/canon/skills/yoke-retrofit/SKILL.md +26 -26
- package/canon/skills/yoke-workflow/SKILL.md +20 -20
- package/canon/tools/codex-rtk-hook.mjs +35 -35
- package/canon/tools/gemini-rtk-hook.mjs +25 -25
- package/canon/tools/graphify.md +3 -3
- package/canon/tools/playwright-mcp.md +3 -3
- package/canon/tools/qwen-rtk-hook.mjs +25 -0
- package/canon/tools/rtk.md +7 -7
- package/canon/tools/serena.md +6 -6
- package/dist/agents/host.js +1 -1
- package/dist/agents/providers.js +18 -5
- package/dist/agents/telemetry.js +35 -36
- package/dist/cli.js +18 -10
- package/dist/dashboard/page.js +122 -122
- package/dist/dashboard/panels.js +91 -91
- package/dist/loop/run-command.js +3 -3
- package/dist/prd/command.js +17 -17
- package/dist/retrofit/apply.js +8 -1
- package/dist/retrofit/config.js +1 -1
- package/dist/retrofit/detect.js +2 -0
- package/dist/retrofit/planners/claude.js +14 -14
- package/dist/retrofit/planners/qwen.js +3 -3
- package/dist/retrofit/preserve.js +2 -2
- package/dist/retrofit/qwen-settings.js +17 -0
- package/dist/retrofit/skill-actions.js +1 -1
- package/dist/setup/command.js +22 -8
- package/dist/setup/model-presets.js +48 -0
- package/docs/CAPABILITY-ROUTING.md +51 -51
- package/docs/DASHBOARD-EVOLUTION.md +33 -33
- package/docs/MIGRATING-TO-1.0.md +33 -33
- package/docs/MIGRATING-TO-1.1.md +27 -27
- package/docs/MIGRATING-TO-1.4.md +70 -70
- package/docs/PRODUCT-DIRECTION-2026-09-05.md +210 -210
- package/docs/PUBLISHING.md +114 -114
- package/docs/QWEN-MODEL-SUPPORT.md +142 -0
- package/docs/VERIFIED-PROJECTS-VALIDATION.md +29 -29
- package/docs/VERIFIED-PROJECTS.md +167 -167
- package/docs/superpowers/plans/2026-06-28-baustein-e-context-layer.md +981 -981
- package/docs/superpowers/plans/2026-06-29-baustein-f-routing.md +258 -258
- package/docs/superpowers/plans/2026-06-29-baustein-g-loop-observability.md +1006 -1006
- package/docs/superpowers/plans/2026-06-29-baustein-h-loop-robustness.md +374 -374
- package/docs/superpowers/plans/2026-06-30-baustein-i-visual-design-verification.md +450 -450
- package/docs/superpowers/plans/2026-07-02-baustein-k-zero-to-100-bootstrap.md +1024 -1024
- package/docs/superpowers/plans/2026-07-02-baustein-m-flow-smoke-proofs.md +574 -574
- package/docs/superpowers/plans/2026-08-13-gauntlet-quality-loop.md +537 -537
- package/docs/superpowers/plans/2026-08-16-artifact-backed-output-compaction.md +329 -329
- package/docs/superpowers/plans/2026-09-05-verified-projects.md +83 -83
- package/docs/superpowers/specs/2026-06-28-baustein-e-context-layer-design.md +146 -146
- package/docs/superpowers/specs/2026-06-29-baustein-f-routing-design.md +106 -106
- package/docs/superpowers/specs/2026-06-29-baustein-g-loop-observability-design.md +186 -186
- package/docs/superpowers/specs/2026-06-29-baustein-h-loop-robustness-design.md +113 -113
- package/docs/superpowers/specs/2026-06-30-baustein-i-visual-design-verification-design.md +98 -98
- package/docs/superpowers/specs/2026-07-02-baustein-k-zero-to-100-bootstrap-design.md +200 -200
- package/docs/superpowers/specs/2026-07-02-baustein-m-flow-smoke-proofs-design.md +155 -155
- package/docs/superpowers/specs/2026-08-13-gauntlet-quality-loop-design.md +422 -422
- package/docs/superpowers/specs/2026-08-16-artifact-backed-output-compaction-design.md +166 -166
- package/gemini-extension.json +6 -6
- package/hooks/hooks.json +19 -19
- package/package.json +87 -87
- package/dist/dashboard/discovery.js +0 -73
- package/docs/community-outreach-2026-08-20.md +0 -85
- package/docs/launch-copy-2026-08-21.md +0 -193
package/TODOS.md
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
# Yoke follow-up work
|
|
2
|
-
|
|
3
|
-
- Add provider-native output schemas when all three CLIs expose compatible stable APIs.
|
|
4
|
-
- Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
|
|
5
|
-
- Add signed provenance and attestations to npm and GitHub releases.
|
|
1
|
+
# Yoke follow-up work
|
|
2
|
+
|
|
3
|
+
- Add provider-native output schemas when all three CLIs expose compatible stable APIs.
|
|
4
|
+
- Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
|
|
5
|
+
- Add signed provenance and attestations to npm and GitHub releases.
|
package/agents/docs.toml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
name = "docs"
|
|
2
|
-
description = "Documentation specialist for release and API consistency."
|
|
3
|
-
sandbox_mode = "workspace-write"
|
|
4
|
-
developer_instructions = """
|
|
5
|
-
Update only documentation required by the assigned change. Verify commands and version references against the repository.
|
|
6
|
-
"""
|
|
1
|
+
name = "docs"
|
|
2
|
+
description = "Documentation specialist for release and API consistency."
|
|
3
|
+
sandbox_mode = "workspace-write"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Update only documentation required by the assigned change. Verify commands and version references against the repository.
|
|
6
|
+
"""
|
package/agents/implementer.toml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
name = "implementer"
|
|
2
|
-
description = "Implementation specialist for one scoped story."
|
|
3
|
-
sandbox_mode = "workspace-write"
|
|
4
|
-
developer_instructions = """
|
|
5
|
-
Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
|
|
6
|
-
"""
|
|
1
|
+
name = "implementer"
|
|
2
|
+
description = "Implementation specialist for one scoped story."
|
|
3
|
+
sandbox_mode = "workspace-write"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Implement only the assigned scope. Use tests first, run verification, and do not review or commit your own work.
|
|
6
|
+
"""
|
package/agents/reviewer.toml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
name = "reviewer"
|
|
2
|
-
description = "Read-only reviewer for correctness and acceptance criteria."
|
|
3
|
-
sandbox_mode = "read-only"
|
|
4
|
-
developer_instructions = """
|
|
5
|
-
Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.
|
|
6
|
-
"""
|
|
1
|
+
name = "reviewer"
|
|
2
|
+
description = "Read-only reviewer for correctness and acceptance criteria."
|
|
3
|
+
sandbox_mode = "read-only"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Review observed diffs and test evidence. Do not modify files. Return only findings grounded in evidence.
|
|
6
|
+
"""
|
package/agents/security.toml
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
name = "security"
|
|
2
|
-
description = "Read-only security reviewer for changed code."
|
|
3
|
-
sandbox_mode = "read-only"
|
|
4
|
-
developer_instructions = """
|
|
5
|
-
Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.
|
|
6
|
-
"""
|
|
1
|
+
name = "security"
|
|
2
|
+
description = "Read-only security reviewer for changed code."
|
|
3
|
+
sandbox_mode = "read-only"
|
|
4
|
+
developer_instructions = """
|
|
5
|
+
Inspect changed code for exploitable security regressions. Do not modify files and avoid speculative findings.
|
|
6
|
+
"""
|
package/bench/README.md
CHANGED
|
@@ -1,86 +1,86 @@
|
|
|
1
|
-
# Yoke benchmark — tokens · speed · quality
|
|
2
|
-
|
|
3
|
-
Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
|
|
4
|
-
the same PRD within each comparison, and three measured dimensions:
|
|
5
|
-
|
|
6
|
-
Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
|
|
7
|
-
multi-run performance study currently measures Codex only; provider support tests are not treated
|
|
8
|
-
as performance evidence for Claude or Gemini.
|
|
9
|
-
|
|
10
|
-
| Dimension | How it is measured |
|
|
11
|
-
|---|---|
|
|
12
|
-
| **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
|
|
13
|
-
| **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
|
|
14
|
-
| **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
|
|
15
|
-
|
|
16
|
-
## The fixture (`fixtures/string-kit`)
|
|
17
|
-
|
|
18
|
-
A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
|
|
19
|
-
`node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
|
|
20
|
-
stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
|
|
21
|
-
final quality check runs everything. No npm installs, so results measure the agent — not the
|
|
22
|
-
network.
|
|
23
|
-
|
|
24
|
-
## The routing fixture (`fixtures/routing-queue`)
|
|
25
|
-
|
|
26
|
-
A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
|
|
27
|
-
`node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
|
|
28
|
-
letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
|
|
29
|
-
enough that a lower-cost worker can potentially repay one routing-controller call per story.
|
|
30
|
-
|
|
31
|
-
## Running it
|
|
32
|
-
|
|
33
|
-
```bash
|
|
34
|
-
npm run build
|
|
35
|
-
node bench/run.mjs --runner=claude # or gemini / codex
|
|
36
|
-
node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
|
|
37
|
-
node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
|
|
38
|
-
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
|
|
39
|
-
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
|
|
40
|
-
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
|
|
41
|
-
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
|
|
42
|
-
node bench/analyze-routing-study.mjs
|
|
43
|
-
node bench/run-matrix.mjs --label=release-1.0
|
|
44
|
-
node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
|
|
45
|
-
```
|
|
46
|
-
|
|
47
|
-
Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
|
|
48
|
-
explicit `--run-root`), git-inits it,
|
|
49
|
-
drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
|
|
50
|
-
`bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
|
|
51
|
-
The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
|
|
52
|
-
failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
|
|
53
|
-
|
|
54
|
-
### Gate-output compaction benchmark
|
|
55
|
-
|
|
56
|
-
`output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
|
|
57
|
-
It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
|
|
58
|
-
remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
|
|
59
|
-
bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
|
|
60
|
-
comparison to Aphrodite's corpus or published ratios.
|
|
61
|
-
|
|
62
|
-
`run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
|
|
63
|
-
routing registry, junctions this checkout's dependencies, and replays the seed's original
|
|
64
|
-
acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
|
|
65
|
-
visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
|
|
66
|
-
to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
|
|
67
|
-
aggregates the checked-in three-pair Codex-only study.
|
|
68
|
-
|
|
69
|
-
## Caveats (read before quoting numbers)
|
|
70
|
-
|
|
71
|
-
- Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
|
|
72
|
-
alternating-order pairs per arm. Re-run before setting broad policy defaults.
|
|
73
|
-
- Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
|
|
74
|
-
policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
|
|
75
|
-
to exclude personal MCP/plugin startup from both sides.
|
|
76
|
-
- Model identity matters more than CLI identity: `tokens.model` records what actually served
|
|
77
|
-
the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
|
|
78
|
-
- Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
|
|
79
|
-
treat cached input as equivalent to newly processed input. Dollar cost is reported only when
|
|
80
|
-
the provider emits it—Yoke does not guess prices from a model name.
|
|
81
|
-
- The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
|
|
82
|
-
hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
|
|
83
|
-
- Cumulative verify means a story's duration includes fixing any regressions it caused.
|
|
84
|
-
|
|
85
|
-
Results live in [`results/`](results/) — one JSON per run, summarized in
|
|
86
|
-
[`RESULTS.md`](RESULTS.md).
|
|
1
|
+
# Yoke benchmark — tokens · speed · quality
|
|
2
|
+
|
|
3
|
+
Reproducible cross-runner and routing A/B benchmarks for the Yoke loop. Fixed fixture projects,
|
|
4
|
+
the same PRD within each comparison, and three measured dimensions:
|
|
5
|
+
|
|
6
|
+
Adaptive routing is implemented for Claude Code, Codex CLI, and Gemini CLI. The checked-in
|
|
7
|
+
multi-run performance study currently measures Codex only; provider support tests are not treated
|
|
8
|
+
as performance evidence for Claude or Gemini.
|
|
9
|
+
|
|
10
|
+
| Dimension | How it is measured |
|
|
11
|
+
|---|---|
|
|
12
|
+
| **Tokens** | The loop's own provider telemetry in `.yoke/loop-status.json`, split into total input, cached input, fresh input (`input − cached`), output, and reasoning tokens when the provider reports them. Adaptive runs preserve one entry per orchestrator/worker call. Missing telemetry is `null`, never estimated. |
|
|
13
|
+
| **Speed** | Wall-clock, measured by the harness from outside: total run + per-story (from `--json` NDJSON event timestamps). The loop itself stores no durations. |
|
|
14
|
+
| **Quality** | Objective, not judged by any model: the fixture ships **pre-written tests** the agent never has to write (only satisfy). After the run, each story's test file is executed against the final tree. `srcLoc` (non-empty lines in `src/`) is a code-economy proxy. |
|
|
15
|
+
|
|
16
|
+
## The fixture (`fixtures/string-kit`)
|
|
17
|
+
|
|
18
|
+
A dependency-free ESM library with 3 stories (`slugify`, `truncate`, `titleCase`) and 16
|
|
19
|
+
`node:test` assertions total. `bench-verify.mjs` is cumulative: story N runs the tests of
|
|
20
|
+
stories 1…N (the loop exports `YOKE_STORY`), so later stories cannot break earlier work; the
|
|
21
|
+
final quality check runs everything. No npm installs, so results measure the agent — not the
|
|
22
|
+
network.
|
|
23
|
+
|
|
24
|
+
## The routing fixture (`fixtures/routing-queue`)
|
|
25
|
+
|
|
26
|
+
A dependency-free in-memory priority queue with two cumulative stories and 10 pre-written
|
|
27
|
+
`node:test` cases covering idempotency, priority/FIFO ordering, leases, retry/backoff, dead
|
|
28
|
+
letters, expired lease recovery, filters, and state statistics. It is intentionally substantial
|
|
29
|
+
enough that a lower-cost worker can potentially repay one routing-controller call per story.
|
|
30
|
+
|
|
31
|
+
## Running it
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
npm run build
|
|
35
|
+
node bench/run.mjs --runner=claude # or gemini / codex
|
|
36
|
+
node bench/run.mjs --runner=codex --fixture=routing-queue --routing=off --unsafe --run-root=G:\NN-Developed\Yoke-Testground
|
|
37
|
+
node bench/run.mjs --runner=codex --fixture=routing-queue --routing=on --unsafe --run-root=G:\NN-Developed\Yoke-Testground
|
|
38
|
+
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=off --run-root=G:\NN-Developed\Yoke-Testground
|
|
39
|
+
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-large-seed-2026-08-02 --routing=on --run-root=G:\NN-Developed\Yoke-Testground
|
|
40
|
+
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=on --label=codex-only-pair1-on --run-root=G:\NN-Developed\Yoke-Testground
|
|
41
|
+
node bench/run-large.mjs --seed=G:\NN-Developed\Yoke-Testground\yoke-codex-study-seed-2026-08-02 --routing=off --label=codex-only-pair1-off --run-root=G:\NN-Developed\Yoke-Testground
|
|
42
|
+
node bench/analyze-routing-study.mjs
|
|
43
|
+
node bench/run-matrix.mjs --label=release-1.0
|
|
44
|
+
node bench/output-compaction.mjs # deterministic local gate-output benchmark; no provider call
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Each run copies the fixture to `bench/.runs/<fixture>-<runner>-<routing>-<stamp>` (or the
|
|
48
|
+
explicit `--run-root`), git-inits it,
|
|
49
|
+
drives `yoke loop run --json --max=6 --timeout=10`, and writes a result JSON to
|
|
50
|
+
`bench/results/`. Runs are billed against your own accounts for the agent CLIs involved.
|
|
51
|
+
The matrix is sequential to avoid cross-provider load distortion. Missing CLIs and authentication
|
|
52
|
+
failures are stored as honest `unavailable`/`auth-failed` rows and never presented as quality measurements.
|
|
53
|
+
|
|
54
|
+
### Gate-output compaction benchmark
|
|
55
|
+
|
|
56
|
+
`output-compaction.mjs` exercises only Yoke's deterministic failure-preview and artifact path.
|
|
57
|
+
It emits a fixed noisy gate failure, verifies that the early compiler error and final test summary
|
|
58
|
+
remain visible, and checks the stored SHA-256 digest. Its byte/token approximation is not a provider
|
|
59
|
+
bill and says nothing about tool output generated inside Claude Code, Codex, or Gemini. It makes no
|
|
60
|
+
comparison to Aphrodite's corpus or published ratios.
|
|
61
|
+
|
|
62
|
+
`run-large.mjs` accepts an external full-repository seed, starts each arm with a separate empty
|
|
63
|
+
routing registry, junctions this checkout's dependencies, and replays the seed's original
|
|
64
|
+
acceptance tests under fresh filenames after the agent run. This prevents an agent editing a
|
|
65
|
+
visible test from turning into false benchmark evidence. A seed can provide `bench-acceptance.json`
|
|
66
|
+
to declare its fixture identity and hidden-test files. `analyze-routing-study.mjs` validates and
|
|
67
|
+
aggregates the checked-in three-pair Codex-only study.
|
|
68
|
+
|
|
69
|
+
## Caveats (read before quoting numbers)
|
|
70
|
+
|
|
71
|
+
- Agent runs are stochastic. Older rows are N=1; the 2026-08-02 Codex-only study uses three
|
|
72
|
+
alternating-order pairs per arm. Re-run before setting broad policy defaults.
|
|
73
|
+
- Compare routing on/off only when fixture, parent model/effort, permissions, native-subagent
|
|
74
|
+
policy, and host load are controlled. The checked-in routing fixture uses `runner.bare: true`
|
|
75
|
+
to exclude personal MCP/plugin startup from both sides.
|
|
76
|
+
- Model identity matters more than CLI identity: `tokens.model` records what actually served
|
|
77
|
+
the run. Different default models per CLI make "claude vs gemini" really "model X vs model Y".
|
|
78
|
+
- Codex input telemetry includes cache reads. Compare both total input and fresh input; do not
|
|
79
|
+
treat cached input as equivalent to newly processed input. Dollar cost is reported only when
|
|
80
|
+
the provider emits it—Yoke does not guess prices from a model name.
|
|
81
|
+
- The fixture is deliberately small (a loop-overhead + basic-competence probe, minutes not
|
|
82
|
+
hours). It does not measure large-context refactoring, UI work, or long-horizon planning.
|
|
83
|
+
- Cumulative verify means a story's duration includes fixing any regressions it caused.
|
|
84
|
+
|
|
85
|
+
Results live in [`results/`](results/) — one JSON per run, summarized in
|
|
86
|
+
[`RESULTS.md`](RESULTS.md).
|
package/bench/RESULTS.md
CHANGED
|
@@ -1,22 +1,22 @@
|
|
|
1
|
-
# Benchmark results
|
|
2
|
-
|
|
3
|
-
Result schema v1 records fixture version, sample label, permission profile, telemetry/model
|
|
4
|
-
availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
|
|
5
|
-
|
|
6
|
-
Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
|
|
7
|
-
[README.md](README.md). One row per run — raw JSON in [`results/`](results/).
|
|
8
|
-
|
|
9
|
-
## Runs
|
|
10
|
-
|
|
11
|
-
| Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
|
|
12
|
-
|---|---|---|---|---|---|---|---|---|
|
|
13
|
-
| 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
|
|
14
|
-
| 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
|
|
15
|
-
| 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
|
|
16
|
-
| 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
|
|
17
|
-
| 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
|
|
18
|
-
| — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
|
|
19
|
-
|
|
1
|
+
# Benchmark results
|
|
2
|
+
|
|
3
|
+
Result schema v1 records fixture version, sample label, permission profile, telemetry/model
|
|
4
|
+
availability, verdict/blocker, conflicts, wall time, iterations, and final fixture-test status.
|
|
5
|
+
|
|
6
|
+
Fixture `string-kit` (3 stories, 16 pre-written assertions). Methodology and caveats:
|
|
7
|
+
[README.md](README.md). One row per run — raw JSON in [`results/`](results/).
|
|
8
|
+
|
|
9
|
+
## Runs
|
|
10
|
+
|
|
11
|
+
| Date | Runner | Model (reported) | Result | Wall-clock | Input tok | Output tok | First-pass stories | src LOC |
|
|
12
|
+
|---|---|---|---|---|---|---|---|---|
|
|
13
|
+
| 2026-07-27 | claude | unavailable | ⛔ blocked before implementation; zero synthetic usage is not counted | 11 s | — | — | 0/3 | 3 |
|
|
14
|
+
| 2026-07-27 | codex | unavailable | ⛔ CLI probe failed on Windows with access denied | — | — | — | — | — |
|
|
15
|
+
| 2026-07-27 | gemini | unavailable | ⛔ runner exited before implementation | 14 s | — | — | 0/3 | 3 |
|
|
16
|
+
| 2026-07-10 | claude | claude-opus-4-8 | ✅ 3/3 complete | 4 m 28 s | 42 957 | 9 817 | 3/3 (1 iteration each) | 39 |
|
|
17
|
+
| 2026-07-10 | gemini | — | ⛔ blocked: CLI not authenticated on the bench machine (headless needs `GEMINI_API_KEY` or configured OAuth) | — | — | — | — | — |
|
|
18
|
+
| — | codex | — | ⛔ not installed on the bench machine | — | — | — | — | — |
|
|
19
|
+
|
|
20
20
|
Per-story wall-clock (claude run): STORY-1 72 s · STORY-2 114 s · STORY-3 81 s. Every story
|
|
21
21
|
passed verify on the first iteration; the final quality check (all 16 assertions on the final
|
|
22
22
|
tree, outside the loop) is green.
|
|
@@ -115,22 +115,22 @@ zero-token deterministic SELF fast path for clearly high-risk architecture/priva
|
|
|
115
115
|
This is again N=1. Raw valid rows:
|
|
116
116
|
[`routing off`](results/yoke-large-codex-routing-off-2026-08-01T22-58-01.json) and
|
|
117
117
|
[`routing on`](results/yoke-large-codex-routing-on-2026-08-01T23-09-39.json).
|
|
118
|
-
|
|
119
|
-
The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
|
|
120
|
-
before changing the fixture, and the Codex executable could not be probed in this Windows
|
|
121
|
-
environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
|
|
122
|
-
|
|
123
|
-
## What the harness caught before producing a single number
|
|
124
|
-
|
|
125
|
-
Building an honest benchmark is itself a verification pass. The first runs found two real
|
|
126
|
-
Yoke bugs, both fixed in 0.3.0:
|
|
127
|
-
|
|
128
|
-
1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
|
|
129
|
-
5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
|
|
130
|
-
2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
|
|
131
|
-
`-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
|
|
132
|
-
stdin (which selects headless mode) and passes only `--yolo`.
|
|
133
|
-
|
|
118
|
+
|
|
119
|
+
The 2026-07-27 release matrix produced no quality measurement: Claude and Gemini exited
|
|
120
|
+
before changing the fixture, and the Codex executable could not be probed in this Windows
|
|
121
|
+
environment. Raw rows preserve the exact runner diagnostics; no scores were inferred.
|
|
122
|
+
|
|
123
|
+
## What the harness caught before producing a single number
|
|
124
|
+
|
|
125
|
+
Building an honest benchmark is itself a verification pass. The first runs found two real
|
|
126
|
+
Yoke bugs, both fixed in 0.3.0:
|
|
127
|
+
|
|
128
|
+
1. **Availability-probe timeout** — `gemini --version` cold-starts in ~5.8 s on Windows; the
|
|
129
|
+
5 s probe timeout misreported an installed CLI as "not found on PATH". Now 20 s.
|
|
130
|
+
2. **Gemini invocation** — Gemini CLI 0.33+ requires a value after `-p`; the runner's bare
|
|
131
|
+
`-p --yolo` died with "Not enough arguments following: p". The runner now relies on piped
|
|
132
|
+
stdin (which selects headless mode) and passes only `--yolo`.
|
|
133
|
+
|
|
134
134
|
## Reading the numbers
|
|
135
135
|
|
|
136
136
|
- Tokens come from provider telemetry when available; missing telemetry is recorded as missing,
|
|
@@ -1,65 +1,65 @@
|
|
|
1
|
-
import { createHash } from 'node:crypto'
|
|
2
|
-
import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
|
|
3
|
-
import { tmpdir } from 'node:os'
|
|
4
|
-
import { join, resolve } from 'node:path'
|
|
5
|
-
import { fileURLToPath } from 'node:url'
|
|
6
|
-
|
|
7
|
-
async function runtimeModules() {
|
|
8
|
-
const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
|
|
9
|
-
const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
|
|
10
|
-
const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
|
|
11
|
-
return Promise.all([
|
|
12
|
-
import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
|
|
13
|
-
import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
|
|
14
|
-
])
|
|
15
|
-
}
|
|
16
|
-
|
|
17
|
-
function fixture() {
|
|
18
|
-
return [
|
|
19
|
-
'=== stdout ===',
|
|
20
|
-
'compiling application',
|
|
21
|
-
'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
|
|
22
|
-
'the failing call is part of the checkout flow',
|
|
23
|
-
...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
|
|
24
|
-
'=== stderr ===',
|
|
25
|
-
'Tests: 1 failed, 249 passed, 250 total',
|
|
26
|
-
].join('\n')
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
export async function runOutputCompactionBenchmark() {
|
|
30
|
-
const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
|
|
31
|
-
const raw = fixture()
|
|
32
|
-
const previewBudgetBytes = 512
|
|
33
|
-
const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
|
|
34
|
-
const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
|
|
35
|
-
try {
|
|
36
|
-
const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
|
|
37
|
-
const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
|
|
38
|
-
const storedDigest = createHash('sha256').update(stored).digest('hex')
|
|
39
|
-
const previewBytes = Buffer.byteLength(compacted.preview)
|
|
40
|
-
const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
|
|
41
|
-
return {
|
|
42
|
-
fixture: 'gate-output-v1',
|
|
43
|
-
rawBytes: Buffer.byteLength(raw),
|
|
44
|
-
rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
|
|
45
|
-
previewBudgetBytes,
|
|
46
|
-
previewBytes,
|
|
47
|
-
previewApproxTokens: Math.ceil(previewBytes / 4),
|
|
48
|
-
referencedBytes,
|
|
49
|
-
compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
|
|
50
|
-
earlyErrorRetained: compacted.preview.includes('error TS2304'),
|
|
51
|
-
finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
|
|
52
|
-
digestRoundTrip: storedDigest === artifact.sha256,
|
|
53
|
-
}
|
|
54
|
-
} finally {
|
|
55
|
-
rmSync(dir, { recursive: true, force: true })
|
|
56
|
-
}
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
60
|
-
const result = await runOutputCompactionBenchmark()
|
|
61
|
-
console.log(JSON.stringify(result, null, 2))
|
|
62
|
-
if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
|
|
63
|
-
process.exitCode = 1
|
|
64
|
-
}
|
|
65
|
-
}
|
|
1
|
+
import { createHash } from 'node:crypto'
|
|
2
|
+
import { existsSync, mkdtempSync, readFileSync, rmSync } from 'node:fs'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
4
|
+
import { join, resolve } from 'node:path'
|
|
5
|
+
import { fileURLToPath } from 'node:url'
|
|
6
|
+
|
|
7
|
+
async function runtimeModules() {
|
|
8
|
+
const builtCompact = new URL('../dist/output/compact.js', import.meta.url)
|
|
9
|
+
const builtArtifact = new URL('../dist/output/artifact.js', import.meta.url)
|
|
10
|
+
const useBuild = existsSync(fileURLToPath(builtCompact)) && existsSync(fileURLToPath(builtArtifact))
|
|
11
|
+
return Promise.all([
|
|
12
|
+
import(useBuild ? builtCompact.href : new URL('../src/output/compact.ts', import.meta.url).href),
|
|
13
|
+
import(useBuild ? builtArtifact.href : new URL('../src/output/artifact.ts', import.meta.url).href),
|
|
14
|
+
])
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function fixture() {
|
|
18
|
+
return [
|
|
19
|
+
'=== stdout ===',
|
|
20
|
+
'compiling application',
|
|
21
|
+
'src/core.ts:17: error TS2304: Cannot find name ImportantWidget',
|
|
22
|
+
'the failing call is part of the checkout flow',
|
|
23
|
+
...Array.from({ length: 600 }, (_, index) => `progress shard=${index % 12} status=unchanged cache=hit`),
|
|
24
|
+
'=== stderr ===',
|
|
25
|
+
'Tests: 1 failed, 249 passed, 250 total',
|
|
26
|
+
].join('\n')
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export async function runOutputCompactionBenchmark() {
|
|
30
|
+
const [{ compactCommandOutput }, { writeOutputArtifact }] = await runtimeModules()
|
|
31
|
+
const raw = fixture()
|
|
32
|
+
const previewBudgetBytes = 512
|
|
33
|
+
const compacted = compactCommandOutput(raw, { previewBytes: previewBudgetBytes })
|
|
34
|
+
const dir = mkdtempSync(join(tmpdir(), 'yoke-output-bench-'))
|
|
35
|
+
try {
|
|
36
|
+
const artifact = writeOutputArtifact(dir, raw, { phase: 'verify', storyId: 'BENCH' })
|
|
37
|
+
const stored = readFileSync(join(dir, ...artifact.relativePath.split('/')))
|
|
38
|
+
const storedDigest = createHash('sha256').update(stored).digest('hex')
|
|
39
|
+
const previewBytes = Buffer.byteLength(compacted.preview)
|
|
40
|
+
const referencedBytes = previewBytes + Buffer.byteLength(artifact.marker)
|
|
41
|
+
return {
|
|
42
|
+
fixture: 'gate-output-v1',
|
|
43
|
+
rawBytes: Buffer.byteLength(raw),
|
|
44
|
+
rawApproxTokens: Math.ceil(Buffer.byteLength(raw) / 4),
|
|
45
|
+
previewBudgetBytes,
|
|
46
|
+
previewBytes,
|
|
47
|
+
previewApproxTokens: Math.ceil(previewBytes / 4),
|
|
48
|
+
referencedBytes,
|
|
49
|
+
compressionRatio: Number((Buffer.byteLength(raw) / referencedBytes).toFixed(2)),
|
|
50
|
+
earlyErrorRetained: compacted.preview.includes('error TS2304'),
|
|
51
|
+
finalSummaryRetained: compacted.preview.includes('Tests: 1 failed, 249 passed'),
|
|
52
|
+
digestRoundTrip: storedDigest === artifact.sha256,
|
|
53
|
+
}
|
|
54
|
+
} finally {
|
|
55
|
+
rmSync(dir, { recursive: true, force: true })
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
|
|
60
|
+
const result = await runOutputCompactionBenchmark()
|
|
61
|
+
console.log(JSON.stringify(result, null, 2))
|
|
62
|
+
if (!result.earlyErrorRetained || !result.finalSummaryRetained || !result.digestRoundTrip || result.previewBytes > result.previewBudgetBytes) {
|
|
63
|
+
process.exitCode = 1
|
|
64
|
+
}
|
|
65
|
+
}
|
package/bench/result-schema.mjs
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
|
-
export const requiredResultFields = [
|
|
2
|
-
'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
|
|
3
|
-
'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
|
|
4
|
-
'iterations', 'finalTestsPass',
|
|
5
|
-
]
|
|
6
|
-
|
|
7
|
-
export function validateResult(result) {
|
|
8
|
-
const missing = requiredResultFields.filter(key => !(key in result))
|
|
9
|
-
if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
|
|
10
|
-
if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
|
|
11
|
-
return result
|
|
12
|
-
}
|
|
1
|
+
export const requiredResultFields = [
|
|
2
|
+
'schemaVersion', 'fixtureVersion', 'runner', 'sampleLabel', 'permissionProfile',
|
|
3
|
+
'usageAvailable', 'modelAvailable', 'verdict', 'conflicts', 'wallClockMs',
|
|
4
|
+
'iterations', 'finalTestsPass',
|
|
5
|
+
]
|
|
6
|
+
|
|
7
|
+
export function validateResult(result) {
|
|
8
|
+
const missing = requiredResultFields.filter(key => !(key in result))
|
|
9
|
+
if (missing.length) throw new Error(`benchmark result missing: ${missing.join(', ')}`)
|
|
10
|
+
if (!['completed', 'blocked', 'unavailable', 'auth-failed'].includes(result.verdict)) throw new Error(`invalid verdict: ${result.verdict}`)
|
|
11
|
+
return result
|
|
12
|
+
}
|
|
@@ -1,50 +1,50 @@
|
|
|
1
|
-
{
|
|
2
|
-
"schemaVersion": 1,
|
|
3
|
-
"fixtureVersion": "string-kit@1",
|
|
4
|
-
"runner": "claude",
|
|
5
|
-
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
-
"permissionProfile": "safe",
|
|
7
|
-
"yokeVersion": "1.0.0",
|
|
8
|
-
"fixture": "string-kit",
|
|
9
|
-
"startedAt": "2026-07-27T18:03:26.598Z",
|
|
10
|
-
"wallClockMs": 11345,
|
|
11
|
-
"exitCode": 1,
|
|
12
|
-
"finalState": "blocked",
|
|
13
|
-
"verdict": "blocked",
|
|
14
|
-
"blocker": "Claude runner exited before implementation; fixture verification remained red.",
|
|
15
|
-
"conflicts": 0,
|
|
16
|
-
"iterations": 1,
|
|
17
|
-
"finalTestsPass": false,
|
|
18
|
-
"progress": {
|
|
19
|
-
"passed": 0,
|
|
20
|
-
"total": 3
|
|
21
|
-
},
|
|
22
|
-
"usageAvailable": false,
|
|
23
|
-
"modelAvailable": false,
|
|
24
|
-
"tokens": {
|
|
25
|
-
"inputTokens": 0,
|
|
26
|
-
"outputTokens": 0,
|
|
27
|
-
"model": "<synthetic>"
|
|
28
|
-
},
|
|
29
|
-
"stories": [
|
|
30
|
-
{
|
|
31
|
-
"id": "STORY-1",
|
|
32
|
-
"durationMs": 10954,
|
|
33
|
-
"iterations": 1,
|
|
34
|
-
"finalTestsPass": false
|
|
35
|
-
},
|
|
36
|
-
{
|
|
37
|
-
"id": "STORY-2",
|
|
38
|
-
"durationMs": null,
|
|
39
|
-
"iterations": 0,
|
|
40
|
-
"finalTestsPass": false
|
|
41
|
-
},
|
|
42
|
-
{
|
|
43
|
-
"id": "STORY-3",
|
|
44
|
-
"durationMs": null,
|
|
45
|
-
"iterations": 0,
|
|
46
|
-
"finalTestsPass": false
|
|
47
|
-
}
|
|
48
|
-
],
|
|
49
|
-
"srcLoc": 3
|
|
50
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "claude",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"yokeVersion": "1.0.0",
|
|
8
|
+
"fixture": "string-kit",
|
|
9
|
+
"startedAt": "2026-07-27T18:03:26.598Z",
|
|
10
|
+
"wallClockMs": 11345,
|
|
11
|
+
"exitCode": 1,
|
|
12
|
+
"finalState": "blocked",
|
|
13
|
+
"verdict": "blocked",
|
|
14
|
+
"blocker": "Claude runner exited before implementation; fixture verification remained red.",
|
|
15
|
+
"conflicts": 0,
|
|
16
|
+
"iterations": 1,
|
|
17
|
+
"finalTestsPass": false,
|
|
18
|
+
"progress": {
|
|
19
|
+
"passed": 0,
|
|
20
|
+
"total": 3
|
|
21
|
+
},
|
|
22
|
+
"usageAvailable": false,
|
|
23
|
+
"modelAvailable": false,
|
|
24
|
+
"tokens": {
|
|
25
|
+
"inputTokens": 0,
|
|
26
|
+
"outputTokens": 0,
|
|
27
|
+
"model": "<synthetic>"
|
|
28
|
+
},
|
|
29
|
+
"stories": [
|
|
30
|
+
{
|
|
31
|
+
"id": "STORY-1",
|
|
32
|
+
"durationMs": 10954,
|
|
33
|
+
"iterations": 1,
|
|
34
|
+
"finalTestsPass": false
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"id": "STORY-2",
|
|
38
|
+
"durationMs": null,
|
|
39
|
+
"iterations": 0,
|
|
40
|
+
"finalTestsPass": false
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
"id": "STORY-3",
|
|
44
|
+
"durationMs": null,
|
|
45
|
+
"iterations": 0,
|
|
46
|
+
"finalTestsPass": false
|
|
47
|
+
}
|
|
48
|
+
],
|
|
49
|
+
"srcLoc": 3
|
|
50
|
+
}
|
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
{
|
|
2
|
-
"schemaVersion": 1,
|
|
3
|
-
"fixtureVersion": "string-kit@1",
|
|
4
|
-
"runner": "codex",
|
|
5
|
-
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
-
"permissionProfile": "safe",
|
|
7
|
-
"usageAvailable": false,
|
|
8
|
-
"modelAvailable": false,
|
|
9
|
-
"verdict": "unavailable",
|
|
10
|
-
"blocker": "Zugriff verweigert",
|
|
11
|
-
"conflicts": 0,
|
|
12
|
-
"wallClockMs": null,
|
|
13
|
-
"iterations": 0,
|
|
14
|
-
"finalTestsPass": false
|
|
15
|
-
}
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"fixtureVersion": "string-kit@1",
|
|
4
|
+
"runner": "codex",
|
|
5
|
+
"sampleLabel": "matrix-2026-07-27T18:03:26.202Z",
|
|
6
|
+
"permissionProfile": "safe",
|
|
7
|
+
"usageAvailable": false,
|
|
8
|
+
"modelAvailable": false,
|
|
9
|
+
"verdict": "unavailable",
|
|
10
|
+
"blocker": "Zugriff verweigert",
|
|
11
|
+
"conflicts": 0,
|
|
12
|
+
"wallClockMs": null,
|
|
13
|
+
"iterations": 0,
|
|
14
|
+
"finalTestsPass": false
|
|
15
|
+
}
|