@hecer/yoke 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +35 -0
- package/README.md +23 -18
- package/canon/skills/authoring-prd/SKILL.md +7 -0
- package/dist/agents/providers.js +10 -3
- package/dist/change/inbox.js +2 -0
- package/dist/cli.js +12 -5
- package/dist/dashboard/analytics.js +123 -0
- package/dist/dashboard/page.js +6 -4
- package/dist/dashboard/panels.js +19 -0
- package/dist/dashboard/server.js +20 -3
- package/dist/goals/command.js +31 -4
- package/dist/loop/dispatcher.js +5 -2
- package/dist/loop/git.js +1 -1
- package/dist/loop/loop.js +44 -3
- package/dist/loop/parallel-command.js +49 -8
- package/dist/loop/prd.js +2 -0
- package/dist/loop/reporter.js +29 -9
- package/dist/loop/run-command.js +58 -15
- package/dist/loop/runner.js +17 -10
- package/dist/loop/worker.js +28 -1
- package/dist/observability/events.js +6 -1
- package/dist/observability/history.js +80 -0
- package/dist/quality/candidate-comparison.js +1 -1
- package/dist/quality/command.js +27 -8
- package/dist/retrofit/config.js +6 -1
- package/dist/retrofit/gitignore.js +1 -0
- package/dist/routing/assessment.js +66 -0
- package/dist/routing/capability.js +79 -0
- package/dist/routing/router.js +80 -10
- package/dist/setup/command.js +24 -11
- package/docs/CAPABILITY-ROUTING.md +54 -0
- package/docs/PRODUCT-DIRECTION-2026-09-05.md +26 -0
- package/docs/VERIFIED-PROJECTS.md +20 -0
- package/gemini-extension.json +1 -1
- package/hooks/bounded-gemini.mjs +70 -0
- package/package.json +1 -1
package/dist/routing/router.js
CHANGED
|
@@ -1,6 +1,8 @@
|
|
|
1
|
-
import { buildWatchdogInvocation, makeRunner, runCapturedAgent, runnerInvocation, } from '../loop/runner.js';
|
|
1
|
+
import { buildWatchdogInvocation, makeRunner, runCapturedAgent, runnerInvocation, contextBlockFor, } from '../loop/runner.js';
|
|
2
2
|
import { isAcceptanceCriterion } from '../loop/prd.js';
|
|
3
3
|
import { historyForWorkers, projectHash, readRoutingObservations, recordRoutingObservation, storyHash } from './registry.js';
|
|
4
|
+
import { assessmentInstructions, assessmentKey, parseAssessment } from './assessment.js';
|
|
5
|
+
import { chooseCapability, readAssessment, saveAssessment } from './capability.js';
|
|
4
6
|
const costRank = { low: 0, medium: 1, high: 2 };
|
|
5
7
|
export function rankWorkers(workers, strategy, maxCandidates) {
|
|
6
8
|
const history = historyForWorkers(workers);
|
|
@@ -112,6 +114,26 @@ function callUsage(role, provider, selection, tokens, durationMs, profile) {
|
|
|
112
114
|
};
|
|
113
115
|
}
|
|
114
116
|
export function makeAdaptiveRunner(options) {
|
|
117
|
+
const run = routingSteps(options);
|
|
118
|
+
return ctx => {
|
|
119
|
+
const steps = run(ctx);
|
|
120
|
+
let next = steps.next();
|
|
121
|
+
while (!next.done)
|
|
122
|
+
next = steps.next(next.value());
|
|
123
|
+
return next.value;
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
export function makeAsyncAdaptiveRunner(options) {
|
|
127
|
+
const run = routingSteps(options);
|
|
128
|
+
return async (ctx) => {
|
|
129
|
+
const steps = run(ctx);
|
|
130
|
+
let next = steps.next();
|
|
131
|
+
while (!next.done)
|
|
132
|
+
next = steps.next(await next.value());
|
|
133
|
+
return next.value;
|
|
134
|
+
};
|
|
135
|
+
}
|
|
136
|
+
function routingSteps(options) {
|
|
115
137
|
const now = options.now ?? Date.now;
|
|
116
138
|
const available = options.isAvailable ?? (() => true);
|
|
117
139
|
const eligibleWorkers = options.workers.filter(worker => available(worker.agent));
|
|
@@ -121,12 +143,52 @@ export function makeAdaptiveRunner(options) {
|
|
|
121
143
|
permissions: options.permissions ?? 'safe',
|
|
122
144
|
selection,
|
|
123
145
|
}));
|
|
124
|
-
return (ctx)
|
|
146
|
+
return function* (ctx) {
|
|
147
|
+
if (options.strategy === 'capability' && !options.rules?.some(rule => (!rule.area || rule.area === ctx.story.area) && (!rule.storyId || rule.storyId === ctx.story.id))) {
|
|
148
|
+
const root = options.projectRoot ?? ctx.targetDir;
|
|
149
|
+
let assessment = readAssessment(root, ctx.story);
|
|
150
|
+
let planning;
|
|
151
|
+
const calls = [];
|
|
152
|
+
if (!assessment) {
|
|
153
|
+
const prompt = [assessmentInstructions, 'Use the supplied task contract and project context to produce a bounded plan. Do not implement or change files.',
|
|
154
|
+
contextBlockFor(ctx.targetDir, ctx.story), JSON.stringify(ctx.story), 'Return exactly one line: YOKE_ASSESS {"taskClass":"implementation","difficulty":"medium","uncertainty":"low","risk":"low","scope":"low","testability":"high","reason":"evidence","approach":"steps and tests"}'].join('\n');
|
|
155
|
+
const selection = { ...options.parentSelection, nativeMultiAgent: false };
|
|
156
|
+
const started = now();
|
|
157
|
+
planning = yield () => options.captureRoute ? options.captureRoute(options.parent, ctx, prompt, selection)
|
|
158
|
+
: runCapturedAgent(options.parent, buildWatchdogInvocation(runnerInvocation(options.parent, prompt, ctx.targetDir, true, 'read-only', selection), options.idleTimeoutMs ?? 0));
|
|
159
|
+
calls.push(callUsage('orchestrator', options.parent, selection, planning.tokens, now() - started));
|
|
160
|
+
assessment = planning.success ? parseAssessment(planning.output) : undefined;
|
|
161
|
+
if (assessment)
|
|
162
|
+
saveAssessment(root, ctx.story, assessment, { provider: options.parent, model: planning.tokens?.model ?? selection.model });
|
|
163
|
+
}
|
|
164
|
+
if (!assessment)
|
|
165
|
+
return { success: false, summary: 'Routing assessment unavailable or invalid; implementation was not started', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
|
|
166
|
+
const choice = chooseCapability({ root, story: ctx.story, assessment, workers: eligibleWorkers, parent: options.parent, parentSelection: options.parentSelection, maxAttempts: options.maxAttempts });
|
|
167
|
+
options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, reason: choice.reason, next: choice.next, assessment });
|
|
168
|
+
if (choice.exhausted)
|
|
169
|
+
return { success: false, summary: 'Routing attempt budget exhausted; replan this task before retrying', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
|
|
170
|
+
const started = now();
|
|
171
|
+
const result = yield () => makeWorker(choice.provider, choice.selection)({ ...ctx, story: { ...ctx.story, assessment } });
|
|
172
|
+
calls.push(callUsage(choice.worker ? 'worker' : 'parent', choice.provider, choice.selection, result.tokens, now() - started, choice.worker?.id ?? 'SELF'));
|
|
173
|
+
let recorded = false;
|
|
174
|
+
return { ...result, summary: `route=${choice.worker?.id ?? 'SELF'} (${choice.reason}); ${result.summary}`,
|
|
175
|
+
tokens: { ...aggregateCalls(calls), storyId: ctx.story.id, escalated: choice.failures > 1 },
|
|
176
|
+
routing: { canRetry: !result.infrastructureFailure && choice.failures + 1 < (options.maxAttempts ?? 5), recordOutcome: (verified, failureKind) => {
|
|
177
|
+
if (recorded)
|
|
178
|
+
return;
|
|
179
|
+
recorded = true;
|
|
180
|
+
recordRoutingObservation({ projectHash: projectHash(root), storyHash: storyHash(projectHash(root), ctx.story.id), assessmentKey: assessmentKey(ctx.story), taskClass: assessment.taskClass, requiredTier: choice.requiredTier,
|
|
181
|
+
role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort,
|
|
182
|
+
actualModel: result.tokens?.model, orchestratorProvider: options.parent, orchestratorModel: options.parentSelection?.model, orchestratorDurationMs: calls.filter(c => c.role === 'orchestrator').reduce((s, c) => s + c.durationMs, 0), workerDurationMs: calls[calls.length - 1].durationMs,
|
|
183
|
+
processSuccess: result.success, verificationSuccess: verified, failureKind: failureKind ?? (result.infrastructureFailure ? 'infrastructure' : 'implementation'), usageAvailable: result.tokens !== undefined && result.tokens.measurementComplete !== false,
|
|
184
|
+
inputTokens: result.tokens?.inputTokens ?? 0, outputTokens: result.tokens?.outputTokens ?? 0, totalCostUsd: result.tokens?.totalCostUsd });
|
|
185
|
+
} } };
|
|
186
|
+
}
|
|
125
187
|
// Re-rank per story so a long-running loop can use gate outcomes learned by
|
|
126
188
|
// earlier stories without rebuilding the runner.
|
|
127
189
|
const rule = options.rules?.find(rule => (!rule.area || rule.area === ctx.story.area) && (!rule.storyId || rule.storyId === ctx.story.id) && (rule.area || rule.storyId));
|
|
128
190
|
if (rule) {
|
|
129
|
-
const project = projectHash(ctx.targetDir);
|
|
191
|
+
const project = projectHash(options.projectRoot ?? ctx.targetDir);
|
|
130
192
|
const prior = readRoutingObservations().reverse().find(event => event.projectHash === project && event.storyHash === storyHash(project, ctx.story.id) && typeof event.verificationSuccess === 'boolean');
|
|
131
193
|
if (prior?.verificationSuccess === false)
|
|
132
194
|
failedStories.add(ctx.story.id);
|
|
@@ -134,12 +196,12 @@ export function makeAdaptiveRunner(options) {
|
|
|
134
196
|
const ruleWorker = rule && failedStories.has(ctx.story.id) ? rule.escalateTo ?? 'SELF' : rule?.worker;
|
|
135
197
|
const candidates = rule ? eligibleWorkers : rankWorkers(eligibleWorkers, options.strategy, options.maxCandidates);
|
|
136
198
|
if (candidates.length === 0) {
|
|
137
|
-
return makeWorker(options.parent, options.parentSelection ?? {})(ctx);
|
|
199
|
+
return yield () => makeWorker(options.parent, options.parentSelection ?? {})(ctx);
|
|
138
200
|
}
|
|
139
201
|
const prompt = buildRoutingPrompt(ctx, candidates, options.strategy);
|
|
140
|
-
const orchestratorSelection = { ...(options.parentSelection ?? {}), ...(options.orchestratorSelection ?? {}),
|
|
202
|
+
const orchestratorSelection = { ...(options.parentSelection ?? {}), ...(options.orchestratorSelection ?? {}), nativeMultiAgent: false };
|
|
141
203
|
const orchestratorStarted = now();
|
|
142
|
-
const routeRun = rule ? { success: true, summary: 'Explicit rule', output: '', tokens: { inputTokens: 0, outputTokens: 0 } } : options.captureRoute
|
|
204
|
+
const routeRun = rule ? { success: true, summary: 'Explicit rule', output: '', tokens: { inputTokens: 0, outputTokens: 0 } } : yield () => options.captureRoute
|
|
143
205
|
? options.captureRoute(options.parent, ctx, prompt, orchestratorSelection)
|
|
144
206
|
: runCapturedAgent(options.parent, buildWatchdogInvocation(runnerInvocation(options.parent, prompt, ctx.targetDir, true, 'read-only', orchestratorSelection), options.idleTimeoutMs ?? 0));
|
|
145
207
|
const orchestratorDurationMs = Math.max(0, now() - orchestratorStarted);
|
|
@@ -150,16 +212,18 @@ export function makeAdaptiveRunner(options) {
|
|
|
150
212
|
const worker = selected === 'SELF' ? undefined : candidates.find(candidate => candidate.id === selected);
|
|
151
213
|
const provider = worker?.agent ?? options.parent;
|
|
152
214
|
const selection = worker
|
|
153
|
-
? { model: worker.model, reasoningEffort: worker.reasoningEffort,
|
|
154
|
-
: { ...(options.parentSelection ?? {}),
|
|
215
|
+
? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' ? { bare: options.parentSelection?.bare } : {}) }
|
|
216
|
+
: { ...(options.parentSelection ?? {}), nativeMultiAgent: false };
|
|
155
217
|
const workerStarted = now();
|
|
156
|
-
const result = makeWorker(provider, selection)(ctx);
|
|
218
|
+
const result = yield () => makeWorker(provider, selection)(ctx);
|
|
157
219
|
const workerDurationMs = Math.max(0, now() - workerStarted);
|
|
158
220
|
const calls = [
|
|
159
221
|
...(!rule ? [callUsage('orchestrator', options.parent, orchestratorSelection, routeRun.tokens, orchestratorDurationMs)] : []),
|
|
160
222
|
callUsage(worker ? 'worker' : 'parent', provider, selection, result.tokens, workerDurationMs, selected),
|
|
161
223
|
];
|
|
162
224
|
const tokens = {
|
|
225
|
+
storyId: ctx.story.id,
|
|
226
|
+
escalated: Boolean(rule && failedStories.has(ctx.story.id)),
|
|
163
227
|
inputTokens: (routeRun.tokens?.inputTokens ?? 0) + (result.tokens?.inputTokens ?? 0),
|
|
164
228
|
...((routeRun.tokens?.cachedInputTokens !== undefined || result.tokens?.cachedInputTokens !== undefined)
|
|
165
229
|
? { cachedInputTokens: (routeRun.tokens?.cachedInputTokens ?? 0) + (result.tokens?.cachedInputTokens ?? 0) }
|
|
@@ -188,7 +252,7 @@ export function makeAdaptiveRunner(options) {
|
|
|
188
252
|
failedStories.add(ctx.story.id);
|
|
189
253
|
else
|
|
190
254
|
failedStories.delete(ctx.story.id);
|
|
191
|
-
const project = projectHash(ctx.targetDir);
|
|
255
|
+
const project = projectHash(options.projectRoot ?? ctx.targetDir);
|
|
192
256
|
recordRoutingObservation({
|
|
193
257
|
projectHash: project,
|
|
194
258
|
storyHash: storyHash(project, ctx.story.id),
|
|
@@ -214,3 +278,9 @@ export function makeAdaptiveRunner(options) {
|
|
|
214
278
|
return { ...result, summary: `${routeSummary}; ${result.summary}`, tokens, routing: { recordOutcome } };
|
|
215
279
|
};
|
|
216
280
|
}
|
|
281
|
+
function aggregateCalls(calls) {
|
|
282
|
+
return { inputTokens: calls.reduce((n, c) => n + c.inputTokens, 0), outputTokens: calls.reduce((n, c) => n + c.outputTokens, 0),
|
|
283
|
+
cachedInputTokens: calls.reduce((n, c) => n + (c.cachedInputTokens ?? 0), 0), cacheWriteInputTokens: calls.reduce((n, c) => n + (c.cacheWriteInputTokens ?? 0), 0),
|
|
284
|
+
...(calls.some(c => c.totalCostUsd !== undefined) ? { totalCostUsd: calls.reduce((n, c) => n + (c.totalCostUsd ?? 0), 0) } : {}),
|
|
285
|
+
calls, measurementComplete: calls.every(c => c.usageAvailable), costMeasurementComplete: calls.every(c => c.totalCostUsd !== undefined) };
|
|
286
|
+
}
|
package/dist/setup/command.js
CHANGED
|
@@ -7,14 +7,26 @@ import { runRetrofit } from '../retrofit/command.js';
|
|
|
7
7
|
const ALL_AGENTS = ['claude', 'codex', 'gemini'];
|
|
8
8
|
export function defaultRoutingWorkers(agents) {
|
|
9
9
|
const workers = {
|
|
10
|
-
claude:
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
10
|
+
claude: [
|
|
11
|
+
{ id: 'claude-fast', agent: 'claude', model: 'haiku', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
12
|
+
{ id: 'claude-standard', agent: 'claude', model: 'sonnet', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
13
|
+
{ id: 'claude-strong', agent: 'claude', model: 'sonnet', reasoningEffort: 'high', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
14
|
+
{ id: 'claude-frontier', agent: 'claude', model: 'opus', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
15
|
+
],
|
|
16
|
+
codex: [
|
|
17
|
+
{ id: 'codex-light', agent: 'codex', model: 'gpt-5.6-luna', reasoningEffort: 'low', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
18
|
+
{ id: 'codex-standard', agent: 'codex', model: 'gpt-5.6-terra', reasoningEffort: 'medium', tier: 'standard', costTier: 'low', capabilities: ['implementation'] },
|
|
19
|
+
{ id: 'codex-strong', agent: 'codex', model: 'gpt-5.6-sol', reasoningEffort: 'high', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
20
|
+
{ id: 'codex-frontier', agent: 'codex', model: 'gpt-6-astra', reasoningEffort: 'high', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
21
|
+
],
|
|
22
|
+
gemini: [
|
|
23
|
+
{ id: 'gemini-light', agent: 'gemini', model: 'gemini-2.5-flash', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
24
|
+
{ id: 'gemini-standard', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
25
|
+
{ id: 'gemini-strong', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
26
|
+
{ id: 'gemini-frontier', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
27
|
+
],
|
|
16
28
|
};
|
|
17
|
-
return agents.
|
|
29
|
+
return agents.flatMap(agent => workers[agent]);
|
|
18
30
|
}
|
|
19
31
|
function parseAgents(value, fallback) {
|
|
20
32
|
if (value.trim().toLowerCase() === 'all')
|
|
@@ -46,7 +58,7 @@ export async function runSetup(targetDir, opts = {}) {
|
|
|
46
58
|
const defaultLoop = opts.loop ?? existing?.loop.enabled ?? true;
|
|
47
59
|
const defaultRunner = opts.runner ?? existing?.runner?.agent ?? (host && defaultAgents.includes(host) ? host : defaultAgents[0] ?? host ?? 'claude');
|
|
48
60
|
const defaultPolicy = opts.decisionPolicy ?? existing?.loop.decisionPolicy ?? (existing?.loop.onAmbiguity === 'abort' ? 'critical' : 'auto');
|
|
49
|
-
const defaultRouting = opts.routing ?? existing?.routing?.enabled ??
|
|
61
|
+
const defaultRouting = opts.routing ?? existing?.routing?.enabled ?? true;
|
|
50
62
|
const interactive = opts.interactive ?? (process.stdin.isTTY === true && process.stdout.isTTY === true);
|
|
51
63
|
let close;
|
|
52
64
|
let ask = opts.ask;
|
|
@@ -84,15 +96,16 @@ export async function runSetup(targetDir, opts = {}) {
|
|
|
84
96
|
const config = loadConfig(targetDir);
|
|
85
97
|
if (!config)
|
|
86
98
|
return 1;
|
|
87
|
-
config.loop = { ...config.loop, enabled: loop, decisionPolicy };
|
|
99
|
+
config.loop = { parallel: 'auto', isolate: true, ...config.loop, enabled: loop, decisionPolicy };
|
|
88
100
|
config.runner = { ...config.runner, agent: runner };
|
|
89
101
|
const existingWorkers = config.routing?.workers ?? [];
|
|
90
102
|
config.routing = {
|
|
103
|
+
...config.routing,
|
|
91
104
|
enabled: routing,
|
|
92
|
-
strategy: config.routing?.strategy ?? '
|
|
105
|
+
strategy: opts.routingStrategy ?? config.routing?.strategy ?? 'capability',
|
|
93
106
|
maxCandidates: config.routing?.maxCandidates ?? 3,
|
|
94
107
|
...(config.routing?.orchestrator ? { orchestrator: config.routing.orchestrator } : {}),
|
|
95
|
-
workers: existingWorkers.length > 0 ? existingWorkers : defaultRoutingWorkers(agents),
|
|
108
|
+
workers: existingWorkers.length > 0 && !opts.routingPreset ? existingWorkers : defaultRoutingWorkers(agents),
|
|
96
109
|
};
|
|
97
110
|
saveConfig(targetDir, config);
|
|
98
111
|
console.log(`Yoke setup complete: agents=${agents.join(',')} · runner=${runner} · loop=${loop ? 'on' : 'off'} · routing=${routing ? 'on' : 'off'} · decisions=${decisionPolicy}`);
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# Routing by task requirements
|
|
2
|
+
|
|
3
|
+
Available in Yoke 1.9.0.
|
|
4
|
+
|
|
5
|
+
New setups use `routing.strategy: capability`. Existing explicit strategies and profiles remain unchanged. To opt an existing project into capability routing with its current profiles:
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
yoke setup . --yes --routing --routing-strategy=capability
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Give each existing worker a `tier: light|standard|strong|frontier`. Profiles without a tier remain usable with legacy strategies but are not candidates for capability selection. To explicitly replace worker profiles with the supplied provider presets, add `--routing-preset`. This replaces customized worker profiles; omit it to retain them.
|
|
12
|
+
|
|
13
|
+
## Planning and selection
|
|
14
|
+
|
|
15
|
+
The configured start provider/model plans new change requests. The planner supplies an `assessment` with the task class, difficulty, uncertainty, risk, scope, testability, rationale and implementation approach. Existing tasks without an assessment receive one read-only assessment call using the start model. This call does not use a cheaper orchestration override. Its result is cached under `.yoke/routing/`, keyed by the task contract. Changing the contract invalidates the cached assessment; toggling `passes` does not.
|
|
16
|
+
|
|
17
|
+
An assessment is a planning judgment, not a measured success probability. High testability means executable checks can detect an incorrect implementation. High uncertainty, architecture work or high risk require the frontier tier; difficult or broadly coupled work requires strong; routine implementation requires standard. Light is reserved for clear, low-risk mechanical work with strong checks. Weak testability raises the minimum tier. Reviews and critics have a standard minimum even for light tasks.
|
|
18
|
+
|
|
19
|
+
```yaml
|
|
20
|
+
assessment:
|
|
21
|
+
taskClass: implementation
|
|
22
|
+
difficulty: medium
|
|
23
|
+
uncertainty: low
|
|
24
|
+
risk: low
|
|
25
|
+
scope: low
|
|
26
|
+
testability: high
|
|
27
|
+
reason: Existing handler pattern and executable contract tests
|
|
28
|
+
approach: Extend the handler, cover the boundary cases, run contract tests
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Yoke chooses an eligible profile at or above the required tier, then compares declared cost tiers. Optional `roles: [implementation, reviewer, critic, repair]` limits a profile's uses. Task `agent` affinity restricts implementation to that provider. Explicit routing rules and explicit quality role models retain precedence. A missing suitable profile falls back to the start model (or the explicitly bound provider's default) and labels the fallback; it does not prove that the fallback has sufficient capability. An invalid assessment blocks implementation.
|
|
32
|
+
|
|
33
|
+
## Initial profiles
|
|
34
|
+
|
|
35
|
+
| Tier | Codex | Claude | Gemini |
|
|
36
|
+
| --- | --- | --- | --- |
|
|
37
|
+
| light | gpt-5.6-luna, low | haiku | gemini-2.5-flash |
|
|
38
|
+
| standard | gpt-5.6-terra, medium | sonnet | gemini-2.5-pro |
|
|
39
|
+
| strong | gpt-5.6-sol, high | sonnet, high effort | gemini-2.5-pro |
|
|
40
|
+
| frontier | gpt-6-astra, high | opus | gemini-2.5-pro |
|
|
41
|
+
|
|
42
|
+
These are editable starting hypotheses, not measured equivalences or price claims. The Codex names follow the requested profile family. Account access is not established by finding an installed CLI. Gemini uses documented explicit model IDs and receives no unsupported reasoning-effort parameter. Several Gemini tiers deliberately share Pro; moving between those tiers alone is not a stronger-model transition. Adjust the presets to the models available to your account. Claude aliases can resolve to different concrete models over time. Provider-reported model identity remains separate from requested identity.
|
|
43
|
+
|
|
44
|
+
Provider references: [Claude model configuration](https://code.claude.com/docs/en/model-config), [Gemini model selection](https://geminicli.com/docs/cli/model/).
|
|
45
|
+
|
|
46
|
+
## Repair, escalation and evidence
|
|
47
|
+
|
|
48
|
+
After an independent mechanical failure, capability routing permits one targeted repair at the initial tier, then raises the required tier on further failures. Attempts retain the current worktree and receive the previous gate findings. Every returned candidate still passes the normal acceptance, protection, quality, review and integration gates. Critical decisions, pause/cancellation and protected-acceptance violations stop retries. Provider process failures are classified conservatively as infrastructure; they do not count as evidence that a stronger model is needed.
|
|
49
|
+
|
|
50
|
+
`routing.maxAttempts` limits implementation calls per unchanged task contract (default 5, configurable 1–8). The initial tier imposes an additional bound: light at most 5, standard 4, strong 3, frontier 2. An exhausted task blocks and requires a revised plan. These are inner implementation attempts; the outer loop's iteration count still counts task dispatches. Existing quality repair rounds and time limits remain separate bounds, and quality repairs can raise their profile tier by round. Goal execution keeps its existing global attempt, time and token budgets.
|
|
51
|
+
|
|
52
|
+
Routing observations record task class, required tier, requested and reported models, effort, independent result, duration and available consumption. Selection considers matching project/task-class/tier history from the last 30 days within the bounded registry read. At least ten matching observations are required before an observed success rate below 80% excludes a profile. History is scoped to the concrete reported model to avoid mixing changed aliases. This is a conservative exclusion rule; it does not lower the planner's safety floor or claim calibrated probabilities. Missing usage remains unknown. Financial optimization and cross-provider performance require authenticated benchmarks.
|
|
53
|
+
|
|
54
|
+
The dashboard's Now view shows the last recorded implementation profile, requested model/effort, rationale and next escalation tier. Usage & time retains reported model and role consumption, including assessment calls. Cached planning has no new model-call charge. Routing state is local runtime data and excluded from Yoke story commits.
|
|
@@ -181,3 +181,29 @@ Am 2026-09-05 gelesene Primärquellen; vor konkreten Versions-/Preisentscheidung
|
|
|
181
181
|
- https://github.com/open-gsd/gsd-core
|
|
182
182
|
|
|
183
183
|
Provenienz der ursprünglichen README-Prüfung: kein C2PA gefunden, unterstützter Scan vollständig, Verifikation/Vertrauen/Metadatenprivatsphäre unbekannt. Unicode-Befund: Emoji-Variationszeichen, kein Nachweis eines KI-Wasserzeichens. Proprietäre Wasserzeichen nicht überprüfbar. Dieses Gesprächsdokument wurde vom KI-Assistenten aus der Sitzung zusammengefasst; es enthält keine unabhängige Bestätigung der Produktthesen.
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
## Fortsetzung am 2026-09-06: Defaults und Dashboard
|
|
187
|
+
|
|
188
|
+
Der Nutzer hat die kombinierte Umsetzung von automatischem Routing/Parallelismus und den drei Dashboardansichten Jetzt, Verbrauch & Zeit sowie Ergebnisse beauftragt. Die Änderungen liegen lokal als unveröffentlichter Ausbau vor; Paketversion und zuletzt veröffentlichtes Release bleiben 1.7.0.
|
|
189
|
+
|
|
190
|
+
Umgesetzt: Routing im asynchronen Workerpfad; neue Setups mit Routing an, Parallelität auto und Isolation an; konservativ höchstens zwei Yoke-Worker bei deklarierten Schreibbereichen; Respektierung expliziter Einstellungen; dauerhafte Messhistorie getrennt von der kurzen Aktivitätsliste; Tages-/Wochen-/Monatsauswertung in UTC; Modell- und Projektvergleich; Verbrauchsdiagramm; aktuelle Aufgaben und Phasen; Abnahmen und Aufwand pro Abnahme. Verfügbare Reviewer-, Kritiker- und Reparaturnutzung wird mit erfasst. Details und Grenzen stehen in VERIFIED-PROJECTS.md.
|
|
191
|
+
|
|
192
|
+
Weiterhin keine behaupteten Fähigkeiten: dynamische gemeinsame Nutzung von Slots durch native Subagenten (native Delegation ist für Loop-Aufrufe bei allen drei Anbietern deaktiviert), exakte Generierungsgeschwindigkeit, vollständige Rekonstruktion alter Verbrauchsdaten, automatische monatliche Archivverdichtung oder kalibrierte Zeitprognosen. Bestehende Quality-Reparaturlimits bleiben erhalten; konkurrierende Kandidaten bleiben optional.
|
|
193
|
+
|
|
194
|
+
Die Umsetzung wurde mit Tests und einer lokalen Browserprüfung geprüft; authentifizierte Modellbenchmarks und Veröffentlichung waren kein Bestandteil dieser Fortsetzung. Dieses Update wurde vom KI-Assistenten aus der laufenden Umsetzung festgehalten.
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
### Releaseauftrag am 2026-09-06
|
|
198
|
+
|
|
199
|
+
Der Nutzer hat anschließend maximal drei Worker im Automatikmodus und die Veröffentlichung der Weiterentwicklung beauftragt. Releaseziel ist 1.8.0; der frühere lokale Zwischenstand mit zwei Workern ist damit überholt. Jede neue Version muss vor Veröffentlichung einen datierten Changelogeintrag erhalten; die verbindliche Regel steht in AGENTS.md. Der tatsächliche Veröffentlichungsstatus wird über GitHub Release und npm geprüft.
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
## Aufgabenbezogene Modellauswahl nach Release 1.8.0
|
|
203
|
+
|
|
204
|
+
Der Nutzer hat die Umsetzung der vorgeschlagenen Fähigkeitsauswahl ausdrücklich beauftragt: Planung mit dem Startmodell, gespeicherte Aufgabenbewertung, Modell-/Effort-Profile für Codex, Claude und Gemini, begrenzte Reparatur/Eskalation sowie nachvollziehbare Dashboardanzeige. Die Implementierung wird lokal nach 1.8.0 entwickelt. Verhalten, Migration und Grenzen stehen in CAPABILITY-ROUTING.md; die veröffentlichten 1.8.0-Defaults dürfen damit nicht verwechselt werden.
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
### Releaseauftrag 1.9.0
|
|
208
|
+
|
|
209
|
+
Der Nutzer hat die Veröffentlichung des Capability-Routing-Ausbaus ausdrücklich beauftragt. Releaseziel ist 1.9.0. Der datierte Changelog und CAPABILITY-ROUTING.md beschreiben Verhalten, Migration und Grenzen; frühere Hinweise auf den lokalen Zwischenstand bleiben historische Sitzungsnotizen.
|
|
@@ -130,6 +130,26 @@ Versioned local events record status, phase duration, attempts and available usa
|
|
|
130
130
|
|
|
131
131
|
## Local project dashboard
|
|
132
132
|
|
|
133
|
+
### Execution defaults in 1.8.0
|
|
134
|
+
|
|
135
|
+
New setups enable routing, `loop.parallel: auto` and `loop.isolate: true`. Existing explicit settings remain authoritative. At execution time, automatic routing uses configured profiles; without profiles it keeps the selected parent. Explicit `--routing` without profiles still reports a configuration error. Routing rules bypass the controller and can escalate following failed independent gates, including across worktrees and restarts. Routing now also runs asynchronously inside parallel workers. An explicit task provider affinity takes precedence over routing.
|
|
136
|
+
|
|
137
|
+
Automatic parallelism allows at most three Yoke workers when every pending task declares nonempty write scopes. Dependencies and overlapping scopes still constrain dispatch. Unknown scopes, configured tool actions and worktree recovery select serial execution. Use `--parallel=N`, `--parallel=auto`, `--no-routing` or `--no-isolate` to override defaults. Serial worktrees retain failed work and require deliberate recovery; the default does not discard an existing recovery tree. Quality repair budgets and opt-in competing candidates retain their existing policies.
|
|
138
|
+
|
|
139
|
+
Integration retains an execution slot until its candidate lands. Yoke loop runners disable native delegation so it cannot multiply the default worker budget: Codex disables multi_agent; Claude disallows Agent, Task, TeamCreate and SendMessage; Gemini uses a separate temporary system-settings copy that disables experimental agents and its always-on investigator/help overrides. Existing Gemini system policy and system-default paths are preserved; unreadable or malformed policy blocks launch. Original settings are never overwritten. The temporary copy is removed on normal exit; forced process termination may leave a private temporary directory. Explicit competing candidate counts remain a separate opt-in workload.
|
|
140
|
+
|
|
141
|
+
### Dashboard views
|
|
142
|
+
|
|
143
|
+
- **Now:** reported task and worker activity, requested provider/model, elapsed worker time, phase, integration, blockers, status age, backlog and empirical remaining-time ranges. The view refreshes every five seconds while visible and not being operated with keyboard focus. A stale status is marked rather than asserted to be live.
|
|
144
|
+
- **Usage & time:** last 24 hours, 7/30/90/365 days or custom dates, grouped by UTC day, Monday-based week or month. Displays reported input/output and cache categories, cost coverage, consumption charts, per-model time buckets, per-task usage and summed phase/call durations. The overview also compares projects for the same period.
|
|
145
|
+
- **Results:** recorded acceptances, ended attempts, explicitly successful outcomes, repair phases, rule-driven escalations and recorded tokens/time per acceptance. Saved acceptance evidence and goal attempts remain available; their timestamps may fall outside the statistics period.
|
|
146
|
+
|
|
147
|
+
The short activity list still retains at most 1,000 events. Compact measurements now also persist under `.yoke/history/YYYY-MM-DD/` independently of that retention and across runs. They contain identifiers, model/provider evidence and measurements, not prompts or full status snapshots. Runtime history is excluded from Yoke commits and added to new project ignore rules. Available reviewer, quality critic and repair usage is recorded separately; absent usage is counted as unknown. Recent and archived records are deduplicated by event ID.
|
|
148
|
+
|
|
149
|
+
Dates and buckets use UTC. The custom end date is inclusive; the API uses an exclusive upper timestamp. Usage belongs to the time the provider reports it, and durations to their end time. Tokens per elapsed minute divide recorded input/output by the full selected interval. Tokens per call minute use summed reported call duration, which can overlap across workers. Neither is measured generation speed. Cache categories are shown separately without adding them to input again.
|
|
150
|
+
|
|
151
|
+
Historical activity that was never recorded or already expired cannot be reconstructed. Queue/human waiting time that lacks measurements remains unknown; phase and attempt durations are summed worker time, not automatically elapsed project time. Queries allow at most 366 days and bounded reads (8 MiB per shard, 32 MiB overall, 50,000 records); skipped, malformed or oversized history is reported as incomplete. History currently requires local disk retention rather than automatic monthly compaction. Unknown model identity is never replaced by a requested model, and missing charges are not estimated from token counts.
|
|
152
|
+
|
|
133
153
|
```sh
|
|
134
154
|
yoke projects add /path/to/project
|
|
135
155
|
yoke projects list
|
package/gemini-extension.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.9.0",
|
|
4
4
|
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
5
|
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
6
|
}
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
import { spawn } from 'node:child_process'
|
|
2
|
+
import { readFileSync, statSync, mkdtempSync, writeFileSync, unlinkSync, rmdirSync } from 'node:fs'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
4
|
+
import { dirname, join } from 'node:path'
|
|
5
|
+
import { pathToFileURL } from 'node:url'
|
|
6
|
+
|
|
7
|
+
// Match Gemini's JSON-with-comments reader without interpreting executable content.
|
|
8
|
+
export function parseSettings(source) {
|
|
9
|
+
let clean = '', quoted = false, escaped = false, line = false, block = false
|
|
10
|
+
for (let i = 0; i < source.length; i++) {
|
|
11
|
+
const c = source[i], next = source[i + 1]
|
|
12
|
+
if (line) { if (c === '\n' || c === '\r') { line = false; clean += c } else clean += ' '; continue }
|
|
13
|
+
if (block) { if (c === '*' && next === '/') { block = false; clean += ' '; i++ } else clean += c === '\n' ? '\n' : ' '; continue }
|
|
14
|
+
if (quoted) { clean += c; if (escaped) escaped = false; else if (c === '\\') escaped = true; else if (c === '"') quoted = false; continue }
|
|
15
|
+
if (c === '"') quoted = true
|
|
16
|
+
else if (c === '/' && next === '/') { line = true; clean += ' '; i++; continue }
|
|
17
|
+
else if (c === '/' && next === '*') { block = true; clean += ' '; i++; continue }
|
|
18
|
+
clean += c
|
|
19
|
+
}
|
|
20
|
+
if (block) throw Error('Invalid settings comment')
|
|
21
|
+
const value = JSON.parse(clean)
|
|
22
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) throw Error('Settings must be an object')
|
|
23
|
+
return value
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export function prepareGeminiEnvironment(environment = process.env, platform = process.platform) {
|
|
27
|
+
const original = environment.GEMINI_CLI_SYSTEM_SETTINGS_PATH ?? (platform === 'win32' ? 'C:\\ProgramData\\gemini-cli\\settings.json' : platform === 'darwin' ? '/Library/Application Support/GeminiCli/settings.json' : '/etc/gemini-cli/settings.json')
|
|
28
|
+
let settings = {}
|
|
29
|
+
try {
|
|
30
|
+
if (statSync(original).size > 1048576) throw Error('Settings too large')
|
|
31
|
+
settings = parseSettings(readFileSync(original, 'utf8'))
|
|
32
|
+
} catch (error) {
|
|
33
|
+
if (error.code !== 'ENOENT') throw Error('Cannot safely preserve existing Gemini system settings')
|
|
34
|
+
}
|
|
35
|
+
if (settings.experimental !== undefined && (!settings.experimental || typeof settings.experimental !== 'object' || Array.isArray(settings.experimental))) throw Error('Invalid Gemini experimental settings')
|
|
36
|
+
if (settings.agents !== undefined && (!settings.agents || typeof settings.agents !== 'object' || Array.isArray(settings.agents))) throw Error('Invalid Gemini agent settings')
|
|
37
|
+
if (settings.agents?.overrides !== undefined && (!settings.agents.overrides || typeof settings.agents.overrides !== 'object' || Array.isArray(settings.agents.overrides))) throw Error('Invalid Gemini agent overrides')
|
|
38
|
+
const dir = mkdtempSync(join(tmpdir(), 'yoke-gemini-'))
|
|
39
|
+
const file = join(dir, 'settings.json')
|
|
40
|
+
const cleanup = () => { try { unlinkSync(file) } catch {} try { rmdirSync(dir) } catch {} }
|
|
41
|
+
const overrides = { ...settings.agents?.overrides }
|
|
42
|
+
// Gemini exposes these built-ins even when experimental agents are disabled.
|
|
43
|
+
for (const name of ['codebase_investigator', 'cli_help']) overrides[name] = { ...overrides[name], enabled: false }
|
|
44
|
+
try { writeFileSync(file, JSON.stringify({ ...settings, experimental: { ...settings.experimental, enableAgents: false }, agents: { ...settings.agents, overrides } }), { mode: 0o600, flag: 'wx' }) }
|
|
45
|
+
catch (error) { cleanup(); throw error }
|
|
46
|
+
return {
|
|
47
|
+
env: { ...environment, GEMINI_CLI_SYSTEM_SETTINGS_PATH: file,
|
|
48
|
+
GEMINI_CLI_SYSTEM_DEFAULTS_PATH: environment.GEMINI_CLI_SYSTEM_DEFAULTS_PATH ?? join(dirname(original), 'system-defaults.json') },
|
|
49
|
+
cleanup,
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
if (process.argv[1] && pathToFileURL(process.argv[1]).href === import.meta.url) {
|
|
54
|
+
let prepared
|
|
55
|
+
try {
|
|
56
|
+
prepared = prepareGeminiEnvironment()
|
|
57
|
+
// Keep Windows shim arguments shell-inert. Prompts use inherited stdin only.
|
|
58
|
+
const args = process.argv.slice(2)
|
|
59
|
+
const windows = process.platform === 'win32'
|
|
60
|
+
if (windows && args.some(arg => !/^[A-Za-z0-9_./:=+-]+$/u.test(arg))) throw Error('Unsafe Gemini argument')
|
|
61
|
+
const child = spawn(windows ? 'cmd.exe' : 'gemini', windows ? ['/d', '/s', '/c', ['gemini', ...args].join(' ')] : args, { stdio: 'inherit', shell: false, env: prepared.env })
|
|
62
|
+
process.once('exit', prepared.cleanup)
|
|
63
|
+
child.once('error', () => { process.stderr.write('Could not start the Gemini CLI\n'); process.exitCode = 127; prepared.cleanup() })
|
|
64
|
+
child.once('close', code => { process.exitCode ??= code === null || code < 0 ? 1 : code; prepared.cleanup() })
|
|
65
|
+
} catch {
|
|
66
|
+
prepared?.cleanup()
|
|
67
|
+
process.stderr.write('Could not establish bounded Gemini execution while preserving system settings\n')
|
|
68
|
+
process.exitCode = 1
|
|
69
|
+
}
|
|
70
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hecer/yoke",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.9.0",
|
|
4
4
|
"description": "One harness, three agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, and Gemini CLI: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|