@hecer/yoke 1.21.1 → 1.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +48 -0
- package/README.md +8 -1
- package/TODOS.md +6 -0
- package/bench/analyze-codex-comparison.mjs +90 -17
- package/bench/compare-codex.mjs +159 -36
- package/bench/result-schema.mjs +132 -0
- package/canon/manifest.yaml +1 -1
- package/canon/skills/visual-verification/SKILL.md +25 -2
- package/canon/tools/codex-rtk-hook.mjs +6 -16
- package/dist/agents/pi-telemetry.js +2 -1
- package/dist/agents/process-streams.js +12 -64
- package/dist/agents/provider-selection.js +12 -0
- package/dist/agents/telemetry.js +52 -52
- package/dist/change/inbox.js +8 -3
- package/dist/check/command.js +69 -17
- package/dist/check/delivery.js +121 -0
- package/dist/cli.js +91 -3
- package/dist/code-intelligence/adapters/mcp.js +1 -0
- package/dist/code-intelligence/budgets.js +138 -0
- package/dist/code-intelligence/contracts.js +2 -0
- package/dist/code-intelligence/coordinator.js +159 -85
- package/dist/code-intelligence/evidence.js +87 -34
- package/dist/code-intelligence/index.js +1 -0
- package/dist/code-intelligence/mcp-client.js +25 -6
- package/dist/code-intelligence/mcp-server.js +14 -11
- package/dist/code-intelligence/preflight.js +71 -0
- package/dist/dashboard/analytics.js +5 -3
- package/dist/goals/command.js +183 -53
- package/dist/goals/usage.js +87 -0
- package/dist/loop/cache-isolation.js +36 -0
- package/dist/loop/candidate-cleanup.js +47 -17
- package/dist/loop/candidates.js +17 -11
- package/dist/loop/dispatcher.js +89 -26
- package/dist/loop/failure.js +104 -0
- package/dist/loop/gate-snapshot.js +19 -0
- package/dist/loop/git.js +1 -1
- package/dist/loop/loop.js +124 -70
- package/dist/loop/parallel-adapters.js +57 -6
- package/dist/loop/parallel-command.js +49 -7
- package/dist/loop/proof-retention.js +70 -0
- package/dist/loop/recovery.js +23 -5
- package/dist/loop/reporter.js +22 -5
- package/dist/loop/run-command.js +101 -47
- package/dist/loop/runner.js +6 -5
- package/dist/loop/worker.js +152 -91
- package/dist/observability/history.js +2 -1
- package/dist/observability/invocation.js +42 -0
- package/dist/observability/local-report.js +120 -0
- package/dist/observability/usage.js +19 -0
- package/dist/prd/command.js +20 -7
- package/dist/prd/decompose.js +5 -2
- package/dist/retrofit/config.js +29 -2
- package/dist/retrofit/gitignore.js +12 -0
- package/dist/retrofit/planners/codex.js +20 -20
- package/dist/routing/attempts.js +241 -0
- package/dist/routing/capability.js +13 -9
- package/dist/routing/optimization.js +73 -0
- package/dist/routing/registry.js +7 -1
- package/dist/routing/router.js +282 -127
- package/dist/setup/command.js +8 -2
- package/dist/smoke/command.js +387 -85
- package/dist/update/check.js +1 -1
- package/docs/BENCHMARK-MANIFEST.md +131 -0
- package/docs/CODE-INTELLIGENCE.md +43 -1
- package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
- package/docs/DELIVERY-JOURNEYS.md +206 -0
- package/docs/ECONOMIC-ROUTING.md +180 -0
- package/docs/GOALS.md +61 -4
- package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
- package/docs/RELEASE-VALIDATION-1.23.0.md +39 -0
- package/docs/benchmarks/2026-10-04-efficiency/ANALYSE.md +182 -0
- package/docs/benchmarks/2026-10-04-efficiency/compare-help.py +55 -0
- package/docs/benchmarks/2026-10-04-efficiency/manifest.json +125 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-analysis.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-design.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/provenance-original-report.json +90 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/DEVELOPMENT_ANALYSIS.md +142 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/RESULT.md +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/commands.jsonl +26 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/environment.json +31 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/final-yoke-smoke.json +40 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/model-purpose-hints.csv +19 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observations.jsonl +21 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/observer-command-phases.csv +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/roles.csv +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/shell-categories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/stories.csv +8 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/summary.json +469 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-history.jsonl +104 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-1.log +58 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-2.log +29 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-3.log +5 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-loop-4.log +12 -0
- package/docs/benchmarks/2026-10-04-efficiency/raw/yoke-phases.csv +10 -0
- package/docs/benchmarks/2026-10-04-efficiency/regression-comparison.json +104 -0
- package/docs/parallel-execution.md +37 -9
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency-prd.json +11 -0
- package/docs/superpowers/plans/2026-10-04-yoke-1.23-efficiency.md +83 -0
- package/docs/superpowers/specs/2026-10-04-yoke-1.23-efficiency-design.md +120 -0
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
"$schema": "https://json.schemastore.org/claude-code-plugin-manifest.json",
|
|
3
3
|
"name": "yoke",
|
|
4
4
|
"displayName": "Yoke",
|
|
5
|
-
"version": "1.
|
|
5
|
+
"version": "1.23.0",
|
|
6
6
|
"description": "Cross-agent coding harness for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi and Hermes: one curated skill canon plus mechanical safety gates and an autonomous loop via the yoke CLI.",
|
|
7
7
|
"author": { "name": "HECer", "url": "https://github.com/HECer" },
|
|
8
8
|
"homepage": "https://github.com/HECer/yoke#readme",
|
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,53 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.23.0 — 2026-10-04
|
|
4
|
+
|
|
5
|
+
### Added
|
|
6
|
+
- Add read-only `yoke tools-preflight` and bounded local `yoke usage` reports. Separate configured tools from available capabilities, recorded usage from unknown coverage, and overlapping worker time from summed process durations.
|
|
7
|
+
- Retain immutable hashed runtime proofs before isolated candidate cleanup, with source, configuration and environment bindings. Preserve candidates when transfer or validation fails.
|
|
8
|
+
- Add optional `smoke.sourceIdentity: { path, sha256 }` configuration; production browser smoke now requires it and verifies served bytes before launch and after journeys.
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
- Make setup/retrofit help read-only and reject unknown or contradictory mutating flags before dispatch while preserving recovery options.
|
|
12
|
+
- Ignore supervision runtime files, diagnose already tracked copies, render actual parallel workers and integration states, and stop reporting success merely when an integrator becomes idle.
|
|
13
|
+
- Preserve redacted browser launch causes instead of reporting every failure as a missing package. Keep OAuth, authorization and configured form secrets out of diagnostics.
|
|
14
|
+
- Migrate Yoke-owned Codex RTK hooks to the native protocol without deleting foreign hooks; honor configured Code Intelligence workspace identifiers and await MCP shutdown before temporary directory cleanup.
|
|
15
|
+
- Validate actual candidate writes and sibling collisions before and after integration gates. Reject shared writable worktree runtime caches and completion commands that change source or final assets.
|
|
16
|
+
- Preserve partial/missing usage markers through routing and include recorded serial implementation durations without counting parent/child usage twice.
|
|
17
|
+
|
|
18
|
+
### Migration and validation limits
|
|
19
|
+
- Existing browser smoke projects must pin an app-specific served source path and its lowercase SHA-256. Update the pin intentionally after source changes; a static byte pin establishes response identity, not an independent deployment attestation.
|
|
20
|
+
- Shared package download stores remain allowed. Worktrees need their own `node_modules` root and writable `.vite-temp`, `.vite` and `.cache` directories. The guard covers these conventional roots, not arbitrary application-defined cache paths or concurrent changes after inspection.
|
|
21
|
+
- No dependency installer or retry policy was added: Yoke does not own one. Full-product model speed/cost and genuine npm/provider cold/warm cache comparisons remain unmeasured. The paired help experiment is a local regression check only.
|
|
22
|
+
- See the [analysis](docs/benchmarks/2026-10-04-efficiency/ANALYSE.md) and [validation record](docs/RELEASE-VALIDATION-1.23.0.md). Version metadata is prepared locally; publication requires independent review and release gates, followed by verified CI/npm evidence.
|
|
23
|
+
|
|
24
|
+
## 1.22.0 — 2026-10-03
|
|
25
|
+
|
|
26
|
+
### Added
|
|
27
|
+
- Add durable routing attempt reservations independent of optional analytics retention, per-call goal admission and persistent consumption accounting, including planning and interrupted calls.
|
|
28
|
+
- Add optional evidence-based economic ranking inside capability routing. Cost, speed and balanced objectives require comparable, complete execution evidence; configured capability floors and fallback limits remain authoritative.
|
|
29
|
+
- Add `yoke goal assess` to prepare a bound goal contract explicitly, and record versioned provenance for setup model profiles as configuration priors rather than measured price or capability claims.
|
|
30
|
+
- Add compact persistent failure signatures: repeated unchanged failures request diagnosis and then stop instead of exhausting repeated identical turns. Completion failures carry typed reasons into repair planning.
|
|
31
|
+
- Add optional multi-step browser journeys and structured proof reports. Project checks can bind declared artifacts and journeys to an acceptance digest, source fingerprint, runtime environment and individual requirement results.
|
|
32
|
+
- Add versioned direct-Codex benchmark manifests containing actual source/build, fixture, acceptance, model and startup-policy provenance. Only compatible verified pairs contribute to comparisons; legacy data remains diagnostic.
|
|
33
|
+
|
|
34
|
+
### Fixed
|
|
35
|
+
- Retain incomplete parallel work across failure, pause and decision boundaries, including unsuccessful candidate races, and resume implementation separately from already prepared integration candidates.
|
|
36
|
+
- Honor ambiguity aborts consistently in serial and parallel execution. Reuse successful gates only on unchanged source, task and gate inputs; rerun them after relevant changes and integration.
|
|
37
|
+
- Prevent exhausted routing budgets from resetting after statistics eviction or failed registry writes. Preserve the cost of planning even when routing blocks before implementation.
|
|
38
|
+
- Preserve known partial provider usage without converting unknown calls to zero-cost work or discarding measured worker usage. Record PRD drafting, decomposition and change-planning/coverage calls on success and failure.
|
|
39
|
+
- Apply goal assessment policy and routing rules, reset provider-specific defaults when changing runners, and separate historic native Codex bindings from the currently executing provider.
|
|
40
|
+
- Correct Code Intelligence reference edges, report unverified index freshness and traversal coverage honestly, and enforce shared request deadlines and bounded UTF-8 responses including evidence metadata.
|
|
41
|
+
- Synchronize package, lockfile, Canon, provider manifests and README release metadata.
|
|
42
|
+
|
|
43
|
+
### Migration and validation limits
|
|
44
|
+
- Existing configurations retain their routing ranking unless `routing.optimization` is added. New setup configurations select `balanced` with a minimum of 20 comparable samples per candidate; insufficient or incomplete evidence keeps the conservative order. Profile tiers and `costTier` are configuration priors, not live provider prices.
|
|
45
|
+
- Goal token ceilings are checked before each additional call. A provider reporting only at call completion can still exceed a ceiling within that call; the overrun is retained and prevents further budgeted dispatch. Unknown interrupted consumption requires an explicit budget decision.
|
|
46
|
+
- Retained work consumes disk space until successful integration or explicit reviewed cleanup. Candidate recovery resumes the first retained alternative; additional alternatives remain available for inspection and cleanup. Custom lifecycles without `retain` keep their existing cleanup policy. Interrupted attempts remain spent budget and are excluded from model-quality comparisons; recovery does not invent missing accounting history.
|
|
47
|
+
- Delivery declarations map project-authored commands to artifacts and journeys. Hashes compare pre-check and post-check artifact content; they do not detect changes reverted between snapshots or independently prove that a command exercised a particular device or deployment. Artifact snapshots are limited to 512 MiB per file, 1 GiB total and 10 seconds per snapshot. Build artifacts before `yoke check`; inspect and explicitly refresh protected acceptance after intentional manifest changes.
|
|
48
|
+
- Code Intelligence token budgets use a documented conservative UTF-8 estimate, not provider-measured token counts. A tiny budget can only return a bounded `BUDGET_EXCEEDED` error envelope; backend freshness stays unknown without independent index evidence.
|
|
49
|
+
- This release uses deterministic regression and packaging checks. No new paid model benchmark or general speed/cost improvement is claimed; see the [validation record](docs/RELEASE-VALIDATION-1.22.0.md).
|
|
50
|
+
|
|
3
51
|
## 1.21.1 — 2026-10-03
|
|
4
52
|
|
|
5
53
|
### Fixed
|
package/README.md
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
[](LICENSE)
|
|
12
12
|

|
|
13
13
|
|
|
14
|
-
<!-- yoke:version:start -->1.
|
|
14
|
+
<!-- yoke:version:start -->1.23.0<!-- yoke:version:end --> · <!-- yoke:tests:start -->1697<!-- yoke:tests:end --> test cases · <!-- yoke:skills:start -->34<!-- yoke:skills:end --> skills
|
|
15
15
|
|
|
16
16
|
<!-- yoke:agents:start -->Claude | Codex | Gemini | Qwen | OpenCode | Kilo | Pi | Hermes<!-- yoke:agents:end -->
|
|
17
17
|
|
|
@@ -21,6 +21,8 @@ Yoke turns a goal into acceptance-tested stories, coordinates one or more coding
|
|
|
21
21
|
|
|
22
22
|
## How it works
|
|
23
23
|
|
|
24
|
+
Version 1.23 adds local `yoke usage [dir] --json` and `yoke tools-preflight [dir] --json` diagnostics. Browser smoke now requires `smoke.sourceIdentity` with an app-specific same-origin path and the lowercase SHA-256 of its served bytes. Pin a stable source response and refresh it deliberately after source changes; see [browser delivery configuration](docs/DELIVERY-JOURNEYS.md). Local version preparation and validation limits are recorded in [1.23 validation](docs/RELEASE-VALIDATION-1.23.0.md).
|
|
25
|
+
|
|
24
26
|
```mermaid
|
|
25
27
|
flowchart LR
|
|
26
28
|
A[Goal and acceptance criteria] --> B[PRD stories and dependencies]
|
|
@@ -41,6 +43,8 @@ Each story carries observable acceptance criteria and targeted test commands. Pa
|
|
|
41
43
|
| **Verified completion** | Checks acceptance criteria, the project verify command, and any configured completion, review, quality, or browser gates before accepting work. |
|
|
42
44
|
| **Parallel execution** | Schedules independent stories in isolated worktrees, respects dependencies and write scopes, and coordinates a shared worker limit across projects. |
|
|
43
45
|
| **Durable goals** | Binds objectives to executable acceptance, shares worker capacity, records budgets, and optionally resumes native Codex goal threads. |
|
|
46
|
+
| **Measured routing** | Keeps hard attempt budgets independent of statistics, accounts for planning and reviews, and uses comparable complete evidence for optional economic model selection. |
|
|
47
|
+
| **Delivery evidence** | Records executable user journeys, acceptance results and stable declared build artifacts with their source and acceptance fingerprints. |
|
|
44
48
|
| **Long-running autonomy** | Recovers from provider failures and blocked work. Optional exploration discovers new, repository-evidenced tasks after the planned backlog drains. |
|
|
45
49
|
| **A choice of agents** | Uses the native CLI for Claude, Codex, Gemini, Qwen, OpenCode, Kilo, Pi, or Hermes. Install and authenticate the CLI you choose; Yoke does not bundle model runtimes or credentials. |
|
|
46
50
|
| **Project visibility** | A local dashboard shows project status, **Workspace analytics**, **History**, and controls such as **Queue a change**, safe-boundary pause/resume, and operator notes. It runs with the local Yoke process. Dark and light themes are available. |
|
|
@@ -119,9 +123,12 @@ The dashboard is a local control room. It does not discover every process or run
|
|
|
119
123
|
| [Continuous exploration](docs/CONTINUOUS-EXPLORATION.md) | Autonomous discovery, runtime limits, pause/resume, and stop detection |
|
|
120
124
|
| [Project workflows](docs/VERIFIED-PROJECTS.md) | Setup, verification, goals, and execution defaults |
|
|
121
125
|
| [Verified goals](docs/GOALS.md) | Acceptance binding, native Codex goals, budgets, and resource admission |
|
|
126
|
+
| [Economic routing](docs/ECONOMIC-ROUTING.md) | Durable attempt budgets, per-call costs, capability floors, and conservative model selection |
|
|
127
|
+
| [Delivery journeys](docs/DELIVERY-JOURNEYS.md) | Browser steps, project acceptance, artifact hashes, and proof limits |
|
|
122
128
|
| [Code Intelligence](docs/CODE-INTELLIGENCE.md) | Optional Graphify, Serena, and Graft evidence providers |
|
|
123
129
|
| [Dashboard](docs/DASHBOARD-EVOLUTION.md) | Project views, history, controls, and reporting boundaries |
|
|
124
130
|
| [Benchmarks](bench/RESULTS.md) | Direct Codex comparison, routing studies, sample limits, and integration findings |
|
|
131
|
+
| [Benchmark manifests](docs/BENCHMARK-MANIFEST.md) | Comparable inputs, model identity, build provenance, and legacy-data limitations |
|
|
125
132
|
| [Changelog](CHANGELOG.md) | Release features, behavior changes, and migration notes |
|
|
126
133
|
|
|
127
134
|
## Safety and limits
|
package/TODOS.md
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
# Yoke follow-up work
|
|
2
2
|
|
|
3
|
+
## Deferred from the user-authorized 1.23.0 release
|
|
4
|
+
|
|
5
|
+
- E7: establish dependency-installer ownership before enforcing lockfile-keyed install reuse and offline/network failure classification; measure genuine cold/warm setup conditions.
|
|
6
|
+
- E8: emit explicit failure categories at production boundaries where the cause is known; preserve unknown observer/guardian/approval coverage.
|
|
7
|
+
- E9: collect paired full-product benchmarks and rerun the populated NEXUS layout regression with a platform-correct dependency tree. The copied macOS dependencies could not start Vite on Windows. No general speed/cost improvement is established yet.
|
|
8
|
+
|
|
3
9
|
- Add provider-native output schemas when all three CLIs expose compatible stable APIs.
|
|
4
10
|
- Expand benchmark fixtures and collect multiple authenticated samples per provider/model.
|
|
5
11
|
- Add signed provenance and attestations to npm and GitHub releases.
|
|
@@ -1,22 +1,95 @@
|
|
|
1
1
|
// node bench/analyze-codex-comparison.mjs <results.json> ... > summary.json
|
|
2
2
|
import { readFileSync } from 'node:fs'
|
|
3
|
+
import { assessComparison, stableComparisonJson } from './result-schema.mjs'
|
|
4
|
+
|
|
3
5
|
const sources = process.argv.slice(2).map(path => JSON.parse(readFileSync(path, 'utf8')))
|
|
4
6
|
if (!sources.length) throw new Error('Supply comparison results.json files')
|
|
5
|
-
const
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
7
|
+
const runs = []
|
|
8
|
+
const cohorts = new Map()
|
|
9
|
+
const legacyReports = []
|
|
10
|
+
const median = values => {
|
|
11
|
+
const sorted = [...values].sort((a, b) => a - b)
|
|
12
|
+
const n = sorted.length
|
|
13
|
+
return n ? (sorted[Math.floor((n - 1) / 2)] + sorted[Math.floor(n / 2)]) / 2 : null
|
|
14
|
+
}
|
|
15
|
+
const measured = value => Number.isFinite(value) && value >= 0
|
|
16
|
+
const metric = (rows, read) => rows.every(row => measured(read(row))) ? median(rows.map(read)) : null
|
|
17
|
+
function summarize(rows) {
|
|
18
|
+
return {
|
|
19
|
+
fixture: rows[0].fixture, arm: rows[0].arm, runs: rows.length,
|
|
20
|
+
accepted: rows.filter(row => row.accepted === true).length,
|
|
21
|
+
medianWallMs: median(rows.map(row => row.wallMs)),
|
|
22
|
+
medianElapsedToAcceptanceMs: metric(rows, row => measured(row.acceptanceMs) ? row.wallMs + row.acceptanceMs : null),
|
|
23
|
+
medianFreshInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.freshInputTokens),
|
|
24
|
+
medianOutputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.outputTokens),
|
|
25
|
+
medianAllInputTokens: metric(rows, row => row.usage?.measurementComplete === false ? null : row.usage?.inputTokens),
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
function armSummaries(rows) {
|
|
29
|
+
return [...new Set(rows.map(row => row.arm))].map(arm => summarize(rows.filter(row => row.arm === arm)))
|
|
30
|
+
}
|
|
31
|
+
for (const [index, source] of sources.entries()) {
|
|
32
|
+
const input = source.results ?? source.runs
|
|
33
|
+
if (!Array.isArray(input) || !input.length) throw new Error('No completed runs in input ' + (index + 1))
|
|
34
|
+
const rows = input.map(({ dir, ...row }) => {
|
|
35
|
+
if (!measured(row.wallMs) || row.wallMs <= 0) throw new Error('Invalid measurement: wall time')
|
|
36
|
+
for (const field of ['inputTokens', 'cachedInputTokens', 'outputTokens']) {
|
|
37
|
+
if (row.usage?.[field] != null && !measured(row.usage[field])) throw new Error('Invalid measurement: ' + field)
|
|
38
|
+
}
|
|
39
|
+
if (row.acceptanceMs != null && !measured(row.acceptanceMs)) throw new Error('Invalid measurement: acceptance time')
|
|
40
|
+
const freshInputTokens = measured(row.usage?.inputTokens) && measured(row.usage?.cachedInputTokens)
|
|
41
|
+
? row.usage.inputTokens - row.usage.cachedInputTokens : null
|
|
42
|
+
if (freshInputTokens !== null && freshInputTokens < 0) throw new Error('Invalid measurement: cache exceeds input')
|
|
43
|
+
return { ...row, freshInputTokens }
|
|
44
|
+
})
|
|
45
|
+
if (!source.manifest) {
|
|
46
|
+
legacyReports.push({
|
|
47
|
+
status: 'legacy/unverified', reportedVersion: source.version ?? source.auditedVersion ?? null,
|
|
48
|
+
reportedCommit: source.commit ?? source.auditedCommit ?? null, runs: rows.length,
|
|
49
|
+
reason: 'No versioned manifest; provenance and cross-arm conditions cannot be verified',
|
|
50
|
+
diagnosticSummaries: armSummaries(rows),
|
|
51
|
+
})
|
|
52
|
+
runs.push(...rows.map(row => ({ ...row, evidenceStatus: 'legacy/unverified' })))
|
|
53
|
+
continue
|
|
54
|
+
}
|
|
55
|
+
// A malformed identity remains its own unverified cohort, never joins another file.
|
|
56
|
+
const id = typeof source.manifest.id === 'string' && source.manifest.id ? source.manifest.id : 'invalid-input-' + index
|
|
57
|
+
const cohort = cohorts.get(id) ?? { id, manifests: [], runs: [] }
|
|
58
|
+
cohort.manifests.push(source.manifest)
|
|
59
|
+
cohort.runs.push(...rows)
|
|
60
|
+
cohorts.set(id, cohort)
|
|
61
|
+
}
|
|
62
|
+
const comparisons = [...cohorts.values()].map(cohort => ({
|
|
63
|
+
id: cohort.id,
|
|
64
|
+
...assessComparison(cohort.manifests, cohort.runs),
|
|
65
|
+
manifest: cohort.manifests[0],
|
|
66
|
+
provenance: [...new Map(cohort.runs.map(row => [stableComparisonJson(row.context?.source), row.context?.source ?? null])).values()],
|
|
67
|
+
diagnosticSummaries: armSummaries(cohort.runs),
|
|
9
68
|
}))
|
|
10
|
-
|
|
11
|
-
for (const
|
|
12
|
-
const
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
console.log(JSON.stringify({
|
|
69
|
+
const groups = []
|
|
70
|
+
for (const comparison of comparisons) {
|
|
71
|
+
const rows = cohorts.get(comparison.id).runs
|
|
72
|
+
runs.push(...rows.map(row => ({ ...row, comparisonId: comparison.id, evidenceStatus: comparison.status })))
|
|
73
|
+
if (comparison.status === 'verified') {
|
|
74
|
+
groups.push(...armSummaries(rows).map(group => ({ comparisonId: comparison.id, ...group })))
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
const verifiedSources = comparisons.filter(comparison => comparison.status === 'verified').flatMap(comparison => comparison.provenance)
|
|
78
|
+
const unique = values => [...new Set(values)]
|
|
79
|
+
const versions = unique(verifiedSources.map(source => source.version))
|
|
80
|
+
const commits = unique(verifiedSources.map(source => source.commit))
|
|
81
|
+
console.log(JSON.stringify({
|
|
82
|
+
schemaVersion: 2,
|
|
83
|
+
auditedVersion: versions.length === 1 ? versions[0] : null,
|
|
84
|
+
auditedCommit: commits.length === 1 ? commits[0] : null,
|
|
85
|
+
requestedModels: unique(runs.map(row => row.context?.execution?.requestedModel ?? row.model).filter(Boolean)),
|
|
86
|
+
reportedModels: unique(runs.flatMap(row => row.context?.execution?.actualModels ?? [])),
|
|
87
|
+
comparisons, groups, legacyReports, runs,
|
|
88
|
+
limitations: unique([
|
|
89
|
+
...sources.flatMap(source => source.limits ?? source.limitations ?? []),
|
|
90
|
+
'Verified means the recorded manifest conditions match; it does not certify model behavior, host isolation, or statistical significance',
|
|
91
|
+
'Diagnostic summaries include unsuccessful or unverified runs and must not be used as accepted performance comparisons',
|
|
92
|
+
'Missing or partial token telemetry remains unknown; no USD price is inferred',
|
|
93
|
+
]),
|
|
94
|
+
pricesMeasured: false, cpuMeasured: false, memoryMeasured: false,
|
|
95
|
+
}, null, 2))
|
package/bench/compare-codex.mjs
CHANGED
|
@@ -1,71 +1,194 @@
|
|
|
1
|
-
// Controlled live comparison. Run
|
|
2
|
-
//
|
|
3
|
-
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync } from 'node:fs'
|
|
4
|
-
import { createHash } from 'node:crypto'
|
|
1
|
+
// Controlled live workflow comparison. Run only when authenticated runs are authorized.
|
|
2
|
+
// Run after npm run build. See docs/BENCHMARK-MANIFEST.md for scope and limits.
|
|
3
|
+
import { cpSync, mkdirSync, readFileSync, writeFileSync, readdirSync, lstatSync, existsSync } from 'node:fs'
|
|
4
|
+
import { createHash, randomUUID } from 'node:crypto'
|
|
5
5
|
import { spawn, spawnSync } from 'node:child_process'
|
|
6
|
+
import { arch, cpus, hostname, release, tmpdir, totalmem } from 'node:os'
|
|
6
7
|
import { dirname, join, resolve } from 'node:path'
|
|
7
8
|
import { fileURLToPath } from 'node:url'
|
|
8
9
|
import { parse, stringify } from 'yaml'
|
|
9
10
|
import { buildProviderInvocation, startProviderProcess } from '../dist/agents/providers.js'
|
|
11
|
+
import { readEvents } from '../dist/observability/events.js'
|
|
12
|
+
import { stableComparisonJson } from './result-schema.mjs'
|
|
13
|
+
|
|
10
14
|
const repo = dirname(dirname(fileURLToPath(import.meta.url)))
|
|
11
15
|
const version = JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version
|
|
12
|
-
const opts = Object.fromEntries(process.argv.slice(2).map(
|
|
16
|
+
const opts = Object.fromEntries(process.argv.slice(2).map(value => value.replace(/^--/, '').split('=')))
|
|
13
17
|
const fixture = opts.fixture ?? 'routing-queue'
|
|
14
18
|
if (!['routing-queue', 'string-kit', 'independent-utils'].includes(fixture)) throw new Error('Unsupported fixture')
|
|
15
19
|
const repeats = Number(opts.repeats ?? 2)
|
|
16
20
|
if (!Number.isInteger(repeats) || repeats < 1 || repeats > 10) throw new Error('repeats must be 1..10')
|
|
17
21
|
const model = opts.model ?? 'gpt-6.1-sol'
|
|
18
|
-
const
|
|
22
|
+
const effort = opts.effort ?? 'low'
|
|
23
|
+
const root = resolve(opts.root ?? join(tmpdir(), 'ykb-' + Date.now().toString(36)))
|
|
19
24
|
mkdirSync(root, { recursive: true })
|
|
20
25
|
const seed = join(repo, 'bench', 'fixtures', fixture)
|
|
21
|
-
const
|
|
22
|
-
const
|
|
23
|
-
const
|
|
24
|
-
const
|
|
25
|
-
const
|
|
26
|
+
const prdInput = readFileSync(join(seed, '.yoke/prd.yaml'), 'utf8')
|
|
27
|
+
const stories = parse(prdInput)
|
|
28
|
+
const originalTests = readdirSync(join(seed, 'tests')).filter(file => file.endsWith('.test.mjs')).sort()
|
|
29
|
+
const protectedFiles = new Map(['bench-verify.mjs', ...originalTests.map(file => join('tests', file))].map(file => [file, readFileSync(join(seed, file))]))
|
|
30
|
+
const digest = value => createHash('sha256').update(value).digest('hex')
|
|
31
|
+
const acceptanceDigest = digest(Buffer.concat([...protectedFiles].flatMap(([file, bytes]) => [Buffer.from(file.replaceAll('\\', '/') + '\0'), bytes])))
|
|
32
|
+
const prompt = 'Implement all requirements below. Run node bench-verify.mjs. Preserve tests and the verification script. Do not commit.\n' + stories.map(story => story.id + ': ' + story.title + '\n' + story.acceptance.join('\n')).join('\n\n')
|
|
26
33
|
const results = []
|
|
27
34
|
const requestedArms = opts.arms?.split(',')
|
|
28
35
|
if (requestedArms && (!requestedArms.length || requestedArms.some(arm => !['codex', 'yoke-serial', 'yoke-parallel'].includes(arm)))) throw new Error('arms must name codex,yoke-serial,yoke-parallel')
|
|
36
|
+
const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
|
|
37
|
+
const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
|
|
38
|
+
if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
|
|
39
|
+
const manifest = {
|
|
40
|
+
schemaVersion: 1, id: randomUUID(), kind: 'workflow', arms: selectedArms, repeats,
|
|
41
|
+
allowedDifferences: [
|
|
42
|
+
{ field: 'execution.workflow', reason: 'One direct provider session versus the normal Yoke story workflow' },
|
|
43
|
+
{ field: 'execution.parallel', reason: 'The explicitly selected parallel arm uses three Yoke workers' },
|
|
44
|
+
{ field: 'execution.promptDigest', reason: 'Workflow inputs differ while the fixture requirements and acceptance remain identical' },
|
|
45
|
+
{ field: 'execution.promptScope', reason: 'Direct provider prompt versus PRD/configuration inputs expanded by Yoke' },
|
|
46
|
+
{ field: 'startup.ignoreRules', reason: 'Preserve the documented direct-arm ignore-rules flag; Yoke retains normal project rule handling' },
|
|
47
|
+
{ field: 'startup.commitPolicy', reason: 'Direct arm is instructed not to commit; Yoke performs its normal story commits' },
|
|
48
|
+
{ field: 'startup.isolation', reason: 'Yoke uses its normal isolated worktrees inside a fresh fixture' },
|
|
49
|
+
{ field: 'startup.timeoutPolicy', reason: 'Direct session deadline versus bounded Yoke story dispatches' },
|
|
50
|
+
{ field: 'startup.userStatePolicy', reason: 'Yoke runtime registry is isolated using LOCALAPPDATA; direct CLI keeps inherited state' },
|
|
51
|
+
],
|
|
52
|
+
}
|
|
53
|
+
const limits = [
|
|
54
|
+
'Small fixture sample; not representative of production projects',
|
|
55
|
+
'Workflow comparison includes declared prompt, commit, isolation, timeout and rule-policy differences',
|
|
56
|
+
'CPU and RAM not measured; host background load is uncontrolled',
|
|
57
|
+
'Routing disabled; fixed requested model and effort; missing reported model identity prevents a verified comparison',
|
|
58
|
+
'Immutable tests replayed after execution; visible to agents, not hidden tests',
|
|
59
|
+
'User config disabled; installed plugins and discovered skills are not fully audited',
|
|
60
|
+
'Prompt digests bind submitted workflow inputs, not provider system prompts or every dynamically expanded Yoke prompt',
|
|
61
|
+
'Wall time starts after fixture setup and ends at runner exit; independent acceptance time is recorded separately',
|
|
62
|
+
]
|
|
63
|
+
function hashPaths(base, paths) {
|
|
64
|
+
const hash = createHash('sha256')
|
|
65
|
+
function visit(relative) {
|
|
66
|
+
const path = join(base, relative)
|
|
67
|
+
const label = relative.replaceAll('\\', '/')
|
|
68
|
+
if (!existsSync(path)) { hash.update('missing:' + label + '\0'); return }
|
|
69
|
+
const stat = lstatSync(path)
|
|
70
|
+
if (stat.isSymbolicLink()) throw new Error('Cannot certify linked benchmark input: ' + label)
|
|
71
|
+
if (stat.isDirectory()) {
|
|
72
|
+
hash.update('directory:' + label + '\0')
|
|
73
|
+
for (const name of readdirSync(path).sort()) visit(join(relative, name))
|
|
74
|
+
} else if (stat.isFile()) {
|
|
75
|
+
const bytes = readFileSync(path)
|
|
76
|
+
hash.update(Buffer.byteLength(label) + ':' + label + ':' + bytes.length + ':')
|
|
77
|
+
hash.update(bytes)
|
|
78
|
+
} else throw new Error('Unsupported benchmark input: ' + label)
|
|
79
|
+
}
|
|
80
|
+
for (const path of [...paths].sort()) visit(path)
|
|
81
|
+
return hash.digest('hex')
|
|
82
|
+
}
|
|
83
|
+
function git(args) {
|
|
84
|
+
const result = spawnSync('git', args, { cwd: repo, encoding: 'utf8' })
|
|
85
|
+
return result.status === 0 ? result.stdout.trim() : null
|
|
86
|
+
}
|
|
87
|
+
function sourceSnapshot() {
|
|
88
|
+
const gitRoot = git(['rev-parse', '--show-toplevel'])
|
|
89
|
+
const sameRoot = gitRoot && (process.platform === 'win32' ? resolve(gitRoot).toLowerCase() === repo.toLowerCase() : resolve(gitRoot) === repo)
|
|
90
|
+
const status = sameRoot ? git(['status', '--porcelain', '--untracked-files=normal']) : null
|
|
91
|
+
return {
|
|
92
|
+
version: JSON.parse(readFileSync(join(repo, 'package.json'), 'utf8')).version,
|
|
93
|
+
commit: sameRoot ? git(['rev-parse', 'HEAD']) : null,
|
|
94
|
+
dirty: status === null ? null : status.length > 0,
|
|
95
|
+
// Hash actual runtime artifacts even on a clean checkout: dist may be stale or untracked.
|
|
96
|
+
buildDigest: hashPaths(repo, ['dist', 'canon', 'agents', 'hooks', 'package.json', 'package-lock.json', 'bench/compare-codex.mjs', 'bench/result-schema.mjs']),
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
function environmentSnapshot() {
|
|
100
|
+
const provider = spawnSync('codex', ['--version'], { encoding: 'utf8', timeout: 20_000, shell: process.platform === 'win32' })
|
|
101
|
+
return {
|
|
102
|
+
platform: process.platform, arch: arch(), nodeVersion: process.version,
|
|
103
|
+
providerVersion: provider.status === 0 ? provider.stdout.trim() || null : null,
|
|
104
|
+
hostDigest: digest(stableComparisonJson({ hostname: hostname(), platform: process.platform, release: release(), arch: arch(), cpus: cpus().map(cpu => cpu.model), memory: totalmem() })),
|
|
105
|
+
hostLoad: 'uncontrolled', skillsPlugins: 'not-audited',
|
|
106
|
+
}
|
|
107
|
+
}
|
|
108
|
+
function reportedModelsFromEvents(dir) {
|
|
109
|
+
const records = readEvents(dir, 1000).filter(event => event.type === 'tokens')
|
|
110
|
+
const calls = records.flatMap(event => Array.isArray(event.data?.calls) && event.data.calls.length ? event.data.calls : [event.data ?? {}])
|
|
111
|
+
const models = calls.map(call => call.actualModel ?? call.model)
|
|
112
|
+
return models.length && models.every(value => typeof value === 'string' && value) ? [...new Set(models)].sort() : null
|
|
113
|
+
}
|
|
114
|
+
const fixtureContext = {
|
|
115
|
+
id: fixture, seedDigest: hashPaths(seed, ['.']),
|
|
116
|
+
acceptanceDigest, requirementsDigest: digest(stableComparisonJson(stories)),
|
|
117
|
+
}
|
|
29
118
|
for (let repeat = 0; repeat < repeats; repeat++) {
|
|
30
|
-
const baseArms = fixture === 'independent-utils' ? ['codex', 'yoke-serial', 'yoke-parallel'] : ['codex', 'yoke-serial']
|
|
31
|
-
const selectedArms = requestedArms ? baseArms.filter(arm => requestedArms.includes(arm)) : baseArms
|
|
32
|
-
if (!selectedArms.length) throw new Error('No requested arms apply to this fixture')
|
|
33
119
|
const arms = repeat % 2 ? [...selectedArms].reverse() : selectedArms
|
|
34
120
|
for (const arm of arms) {
|
|
35
|
-
const
|
|
36
|
-
|
|
121
|
+
const before = sourceSnapshot()
|
|
122
|
+
const environment = environmentSnapshot()
|
|
123
|
+
const dir = join(root, String(repeat + 1) + '-' + arm)
|
|
124
|
+
mkdirSync(dir)
|
|
125
|
+
cpSync(seed, dir, { recursive: true })
|
|
37
126
|
const parallel = arm === 'yoke-parallel' ? 3 : 1
|
|
38
|
-
|
|
127
|
+
const configInput = stringify({ canonVersion: '1.2.0', agents: ['codex'], loop: { enabled: true, parallel, isolate: true }, runner: { agent: 'codex', model, reasoningEffort: effort, bare: true }, routing: { enabled: false }, verify: { command: 'node bench-verify.mjs' } })
|
|
128
|
+
writeFileSync(join(dir, '.yoke/config.yaml'), configInput)
|
|
39
129
|
for (const args of [['init', '-q'], ['add', '-A'], ['-c', 'user.name=benchmark', '-c', 'user.email=benchmark@yoke.local', 'commit', '-qm', 'Immutable fixture seed']]) {
|
|
40
|
-
const
|
|
130
|
+
const result = spawnSync('git', args, { cwd: dir, encoding: 'utf8' })
|
|
131
|
+
if (result.status !== 0) throw new Error(result.stderr)
|
|
41
132
|
}
|
|
42
|
-
console.log(JSON.stringify({ type: 'start', fixture, repeat: repeat + 1, arm, model, dir }))
|
|
43
|
-
const start = Date.now()
|
|
133
|
+
console.log(JSON.stringify({ type: 'start', comparisonId: manifest.id, fixture, repeat: repeat + 1, arm, model, effort, dir }))
|
|
134
|
+
const start = Date.now()
|
|
135
|
+
let outcome, usage, actualModels = null
|
|
136
|
+
let events = []
|
|
44
137
|
if (arm === 'codex') {
|
|
45
|
-
const
|
|
46
|
-
|
|
47
|
-
|
|
138
|
+
const invocation = buildProviderInvocation('codex', prompt, dir, 'unsafe', { model, reasoningEffort: effort, bare: true, nativeMultiAgent: false })
|
|
139
|
+
// Historical workflow policy is retained and explicitly declared, not silently equalized.
|
|
140
|
+
invocation.args.push('--ignore-rules')
|
|
141
|
+
const controller = new AbortController()
|
|
142
|
+
const timer = setTimeout(() => controller.abort(), 600_000)
|
|
48
143
|
try {
|
|
49
|
-
const
|
|
50
|
-
outcome =
|
|
51
|
-
|
|
144
|
+
const result = await startProviderProcess('codex', invocation, { signal: controller.signal, idleTimeoutMs: 180_000 }).completion
|
|
145
|
+
outcome = result.kind
|
|
146
|
+
usage = result.telemetry.tokens
|
|
147
|
+
const reported = result.telemetry.reportedModels ?? (usage?.model ? [usage.model] : [])
|
|
148
|
+
actualModels = reported.length ? [...new Set(reported)].sort() : null
|
|
149
|
+
writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), JSON.stringify(result, null, 2))
|
|
52
150
|
} finally { clearTimeout(timer) }
|
|
53
151
|
} else {
|
|
54
|
-
const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing',
|
|
55
|
-
let output = '', errors = ''
|
|
56
|
-
|
|
152
|
+
const child = spawn(process.execPath, [join(repo, 'dist/cli.js'), 'loop', 'run', dir, '--json', '--runner=codex', '--no-routing', '--parallel=' + parallel, '--max=3', '--timeout=10', '--unsafe'], { cwd: dir, env: { ...process.env, LOCALAPPDATA: join(root, 'state') }, stdio: ['ignore', 'pipe', 'pipe'] })
|
|
153
|
+
let output = '', errors = ''
|
|
154
|
+
child.stdout.on('data', data => { output += data })
|
|
155
|
+
child.stderr.on('data', data => { errors += data })
|
|
156
|
+
outcome = await new Promise((resolveExit, reject) => { child.on('error', reject); child.on('close', resolveExit) })
|
|
57
157
|
events = output.split('\n').flatMap(line => { try { return [JSON.parse(line)] } catch { return [] } })
|
|
58
|
-
writeFileSync(join(root,
|
|
59
|
-
usage = events.filter(
|
|
158
|
+
writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.log'), output + '\nSTDERR\n' + errors)
|
|
159
|
+
usage = events.filter(event => event.type === 'status' && event.tokens).at(-1)?.tokens
|
|
160
|
+
actualModels = reportedModelsFromEvents(dir)
|
|
60
161
|
}
|
|
61
162
|
const wallMs = Date.now() - start
|
|
62
163
|
// Replay immutable originals after the agent exits. Agent-written tests never determine quality.
|
|
63
164
|
mkdirSync(join(dir, 'tests'), { recursive: true })
|
|
64
165
|
for (const [file, bytes] of protectedFiles) writeFileSync(join(dir, file), bytes)
|
|
65
|
-
const
|
|
66
|
-
const
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
166
|
+
const checkStart = Date.now()
|
|
167
|
+
const check = spawnSync(process.execPath, ['--test', ...originalTests.map(file => join('tests', file))], { cwd: dir, encoding: 'utf8', timeout: 60_000 })
|
|
168
|
+
const acceptanceMs = Date.now() - checkStart
|
|
169
|
+
writeFileSync(join(root, fixture + '-' + (repeat + 1) + '-' + arm + '.acceptance.log'), check.stdout + check.stderr)
|
|
170
|
+
const after = sourceSnapshot()
|
|
171
|
+
const context = {
|
|
172
|
+
source: { ...before, stable: stableComparisonJson(before) === stableComparisonJson(after) && hashPaths(seed, ['.']) === fixtureContext.seedDigest },
|
|
173
|
+
fixture: fixtureContext,
|
|
174
|
+
execution: {
|
|
175
|
+
provider: 'codex', requestedModel: model, actualModels, effort, routing: false, nativeMultiAgent: false, nativeGoal: false, parallel,
|
|
176
|
+
workflow: arm === 'codex' ? 'direct-provider-session' : 'yoke-story-loop',
|
|
177
|
+
promptDigest: digest(arm === 'codex' ? prompt : stableComparisonJson({ prdInput, configInput })),
|
|
178
|
+
promptScope: arm === 'codex' ? 'provider-prompt' : 'workflow-input-bundle',
|
|
179
|
+
},
|
|
180
|
+
startup: {
|
|
181
|
+
permissionProfile: 'unsafe', bare: true, ignoreRules: arm === 'codex',
|
|
182
|
+
commitPolicy: arm === 'codex' ? 'instructed-no-commit' : 'normal-story-commits',
|
|
183
|
+
isolation: arm === 'codex' ? 'fresh-fixture' : 'fresh-fixture-and-story-worktrees',
|
|
184
|
+
timeoutPolicy: arm === 'codex' ? '600000ms session; 180000ms idle' : 'three dispatches; 600000ms per runner',
|
|
185
|
+
userStatePolicy: arm === 'codex' ? 'inherited' : 'isolated-LOCALAPPDATA',
|
|
186
|
+
},
|
|
187
|
+
environment,
|
|
188
|
+
}
|
|
189
|
+
const result = { fixture, repeat: repeat + 1, arm, model, effort, routing: false, nativeMultiAgent: false, parallel, acceptanceDigest, context, wallMs, acceptanceMs, accepted: check.status === 0, outcome, usage: usage ?? null, events: events.length, dir }
|
|
190
|
+
results.push(result)
|
|
191
|
+
writeFileSync(join(root, 'results.json'), JSON.stringify({ schemaVersion: 2, version, manifest, root, results, limits }, null, 2))
|
|
192
|
+
console.log(JSON.stringify({ type: 'result', comparisonId: manifest.id, ...result }))
|
|
70
193
|
}
|
|
71
194
|
}
|