@hecer/yoke 1.12.0 → 1.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +3 -3
- package/.codex-plugin/plugin.json +2 -2
- package/CHANGELOG.md +41 -0
- package/README.md +42 -30
- package/canon/loop/prd.schema.md +2 -2
- package/canon/manifest.yaml +2 -2
- package/canon/skills/authoring-prd/SKILL.md +1 -1
- package/canon/skills/yoke-retrofit/SKILL.md +2 -2
- package/canon/skills/yoke-workflow/SKILL.md +1 -1
- package/canon/tools/graphify.md +1 -1
- package/canon/tools/playwright-mcp.md +1 -1
- package/canon/tools/serena.md +1 -1
- package/dist/agents/catalog.js +7 -0
- package/dist/agents/contracts.js +3 -1
- package/dist/agents/host.js +4 -0
- package/dist/agents/process-streams.js +62 -0
- package/dist/agents/process.js +43 -3
- package/dist/agents/providers.js +45 -3
- package/dist/agents/telemetry.js +99 -2
- package/dist/canon/manifest.js +2 -1
- package/dist/change/inbox.js +1 -1
- package/dist/cli.js +22 -24
- package/dist/dashboard/analytics.js +193 -29
- package/dist/dashboard/contracts.js +23 -0
- package/dist/dashboard/page.js +39 -94
- package/dist/dashboard/panels.js +87 -33
- package/dist/dashboard/server.js +190 -15
- package/dist/goals/command.js +3 -2
- package/dist/loop/claims.js +2 -1
- package/dist/loop/decision.js +3 -2
- package/dist/loop/parallel-command.js +4 -2
- package/dist/loop/prd.js +2 -1
- package/dist/loop/reporter.js +1 -0
- package/dist/loop/run-command.js +31 -10
- package/dist/observability/events.js +1 -1
- package/dist/observability/history.js +1 -1
- package/dist/prd/command.js +3 -3
- package/dist/quality/candidate-comparison.js +6 -1
- package/dist/quality/command.js +17 -2
- package/dist/quality/types.js +6 -1
- package/dist/retrofit/apply.js +87 -1
- package/dist/retrofit/config.js +8 -0
- package/dist/retrofit/detect.js +6 -0
- package/dist/retrofit/plan.js +6 -0
- package/dist/retrofit/planners/kilo.js +44 -0
- package/dist/retrofit/planners/opencode.js +44 -0
- package/dist/retrofit/planners/pi.js +24 -0
- package/dist/retrofit/skill-actions.js +3 -0
- package/dist/retrofit/tools.js +8 -0
- package/dist/review/command.js +3 -2
- package/dist/review/verdict.js +1 -1
- package/dist/routing/capability.js +2 -2
- package/dist/routing/planning.js +2 -0
- package/dist/routing/registry.js +3 -1
- package/dist/routing/router.js +7 -3
- package/dist/setup/command.js +13 -3
- package/docs/DASHBOARD-EVOLUTION.md +16 -2
- package/docs/DASHBOARD-OVERHAUL.md +146 -0
- package/docs/HARNESSES.md +81 -0
- package/gemini-extension.json +2 -2
- package/package.json +6 -2
package/dist/review/command.js
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { AGENT_LIST } from '../agents/catalog.js';
|
|
1
2
|
import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runCapturedAgent, repositoryFingerprint, isAgentAvailable, } from '../loop/runner.js';
|
|
2
3
|
import { parseProviderResult } from '../agents/telemetry.js';
|
|
3
4
|
import { resolveIdleMs } from '../loop/run-command.js';
|
|
@@ -5,14 +6,14 @@ import { loadConfig } from '../retrofit/config.js';
|
|
|
5
6
|
import { parseReviewVerdict } from './verdict.js';
|
|
6
7
|
// Resolve to the first available agent, preferring a *second* model so the review
|
|
7
8
|
// is genuinely cross-model. claude last => a Claude-only box degrades to self-review.
|
|
8
|
-
const RESOLUTION_ORDER = ['codex', 'gemini', 'qwen', 'claude'];
|
|
9
|
+
const RESOLUTION_ORDER = ['codex', 'gemini', 'qwen', 'claude', 'opencode', 'kilo', 'pi'];
|
|
9
10
|
export function runReview(targetDir, opts = {}) {
|
|
10
11
|
const available = opts.isAvailable ?? isAgentAvailable;
|
|
11
12
|
const implementer = opts.implementer ?? loadConfig(targetDir)?.agents[0] ?? 'claude';
|
|
12
13
|
let reviewer = opts.reviewer;
|
|
13
14
|
if (reviewer) {
|
|
14
15
|
if (!available(reviewer)) {
|
|
15
|
-
console.error(`Reviewer agent CLI "${reviewer}" was not found on PATH. Install it, or pick another with --reviewer
|
|
16
|
+
console.error(`Reviewer agent CLI "${reviewer}" was not found on PATH. Install it, or pick another with --reviewer=<${AGENT_LIST}>.`);
|
|
16
17
|
return 2;
|
|
17
18
|
}
|
|
18
19
|
if (reviewer === implementer && !opts.allowSelfReview) {
|
package/dist/review/verdict.js
CHANGED
|
@@ -70,7 +70,7 @@ export function formatReviewContract(path, provider) {
|
|
|
70
70
|
return [
|
|
71
71
|
`Write your final verdict to this absolute path: ${path}`,
|
|
72
72
|
'The file must contain exactly one JSON object with this contract:',
|
|
73
|
-
`{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[{"id":"optional id","severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1,"actionable":true,"suggestedFix":"optional repair","evidence":["optional evidence reference"]}],"provenance":{"provider":"${provider ?? 'claude|codex|gemini'}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"safe"}}`,
|
|
73
|
+
`{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[{"id":"optional id","severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1,"actionable":true,"suggestedFix":"optional repair","evidence":["optional evidence reference"]}],"provenance":{"provider":"${provider ?? 'claude|codex|gemini|qwen|opencode|kilo|pi'}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"safe"}}`,
|
|
74
74
|
'Set approved=false when any blocking finding exists. Create the file even when the process also exits non-zero.',
|
|
75
75
|
].join('\n');
|
|
76
76
|
}
|
|
@@ -57,7 +57,7 @@ export function chooseCapability(input) {
|
|
|
57
57
|
const candidates = input.workers.filter(w => w.tier && tiers.indexOf(w.tier) >= level && (!input.maxTier || tiers.indexOf(w.tier) <= tiers.indexOf(input.maxTier)) && (!w.roles || w.roles.includes(role)) && (!input.story.agent || w.agent === input.story.agent) && (input.available?.(w.agent) ?? true));
|
|
58
58
|
const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && e.failureKind !== 'infrastructure' && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
|
|
59
59
|
const evidence = (w) => {
|
|
60
|
-
const matching = history.filter(e => e.provider === w.agent && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.actualModel);
|
|
60
|
+
const matching = history.filter(e => e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
|
|
61
61
|
const actual = matching.at(-1)?.actualModel;
|
|
62
62
|
return actual ? matching.filter(e => e.actualModel === actual) : [];
|
|
63
63
|
};
|
|
@@ -68,7 +68,7 @@ export function chooseCapability(input) {
|
|
|
68
68
|
const worker = reliable[0];
|
|
69
69
|
const blocked = !worker && (input.fallback === 'block' || input.maxTier !== undefined);
|
|
70
70
|
const provider = worker?.agent ?? input.story.agent ?? input.parent;
|
|
71
|
-
const selection = worker ? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
|
|
71
|
+
const selection = worker ? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
|
|
72
72
|
: { ...(provider === input.parent ? input.parentSelection : {}), nativeMultiAgent: false };
|
|
73
73
|
const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}`;
|
|
74
74
|
return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
|
package/dist/routing/planning.js
CHANGED
|
@@ -4,8 +4,10 @@ export function resolvePlanner(config, start, selection = {}, override) {
|
|
|
4
4
|
const inherited = agent === start ? selection : {};
|
|
5
5
|
const planning = !override || override === (config?.planning?.agent ?? start) ? config?.planning : undefined;
|
|
6
6
|
return { agent, selection: {
|
|
7
|
+
provider: planning?.provider ?? inherited.provider,
|
|
7
8
|
model: planning?.model ?? inherited.model,
|
|
8
9
|
reasoningEffort: planning?.reasoningEffort ?? inherited.reasoningEffort,
|
|
10
|
+
variant: planning?.variant ?? inherited.variant,
|
|
9
11
|
bare: inherited.bare,
|
|
10
12
|
nativeMultiAgent: false,
|
|
11
13
|
} };
|
package/dist/routing/registry.js
CHANGED
|
@@ -68,8 +68,10 @@ export function historyForWorkers(workers) {
|
|
|
68
68
|
// Capability evidence belongs to the provider/model that produced it. This
|
|
69
69
|
// prevents a reused worker id from inheriting scores from a retired model.
|
|
70
70
|
if (event.provider !== worker.agent
|
|
71
|
+
|| event.requestedProvider !== worker.provider
|
|
71
72
|
|| event.requestedModel !== worker.model
|
|
72
|
-
|| event.requestedReasoningEffort !== worker.reasoningEffort
|
|
73
|
+
|| event.requestedReasoningEffort !== worker.reasoningEffort
|
|
74
|
+
|| event.requestedVariant !== worker.variant)
|
|
73
75
|
continue;
|
|
74
76
|
if (typeof event.verificationSuccess !== 'boolean')
|
|
75
77
|
continue;
|
package/dist/routing/router.js
CHANGED
|
@@ -101,8 +101,10 @@ function callUsage(role, provider, selection, tokens, durationMs, profile) {
|
|
|
101
101
|
role,
|
|
102
102
|
provider,
|
|
103
103
|
...(profile ? { profile } : {}),
|
|
104
|
+
...(selection.provider ? { requestedProvider: selection.provider } : {}),
|
|
104
105
|
...(selection.model ? { requestedModel: selection.model } : {}),
|
|
105
106
|
...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
|
|
107
|
+
...(selection.variant ? { requestedVariant: selection.variant } : {}),
|
|
106
108
|
...(tokens?.model ? { actualModel: tokens.model } : {}),
|
|
107
109
|
inputTokens: tokens?.inputTokens ?? 0,
|
|
108
110
|
...(tokens?.cachedInputTokens !== undefined ? { cachedInputTokens: tokens.cachedInputTokens } : {}),
|
|
@@ -191,7 +193,7 @@ function routingSteps(options) {
|
|
|
191
193
|
if (!assessment)
|
|
192
194
|
return { success: false, summary: 'Routing assessment unavailable or invalid; implementation was not started', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
|
|
193
195
|
const choice = chooseCapability({ root, story: ctx.story, assessment, workers: eligibleWorkers, parent: options.parent, parentSelection: options.parentSelection, maxAttempts: options.maxAttempts, fallback: options.fallback, maxTier: options.maxTier });
|
|
194
|
-
options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, reason: choice.reason, next: choice.next, assessment });
|
|
196
|
+
options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, variant: choice.selection.variant, providerModel: choice.selection.provider, reason: choice.reason, next: choice.next, assessment });
|
|
195
197
|
if (choice.blocked)
|
|
196
198
|
return { ...blocked(choice.reason), tokens: aggregateCalls(calls) };
|
|
197
199
|
if (choice.exhausted)
|
|
@@ -209,7 +211,7 @@ function routingSteps(options) {
|
|
|
209
211
|
return;
|
|
210
212
|
recorded = true;
|
|
211
213
|
recordRoutingObservation({ projectHash: projectHash(root), storyHash: storyHash(projectHash(root), ctx.story.id), assessmentKey: routingAssessmentKey(root, ctx.story), taskClass: assessment.taskClass, requiredTier: choice.requiredTier,
|
|
212
|
-
role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort,
|
|
214
|
+
role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedProvider: choice.selection.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort, requestedVariant: choice.selection.variant,
|
|
213
215
|
actualModel: result.tokens?.model, orchestratorProvider: options.planner?.agent ?? options.parent, orchestratorModel: (options.planner?.selection ?? options.parentSelection)?.model, orchestratorDurationMs: calls.filter(c => c.role === 'orchestrator').reduce((s, c) => s + c.durationMs, 0), workerDurationMs: calls[calls.length - 1].durationMs,
|
|
214
216
|
processSuccess: result.success, verificationSuccess: infrastructureFailure ? false : verified, failureKind: infrastructureFailure ? 'infrastructure' : failureKind ?? 'implementation', usageAvailable: result.tokens !== undefined && result.tokens.measurementComplete !== false,
|
|
215
217
|
inputTokens: result.tokens?.inputTokens ?? 0, outputTokens: result.tokens?.outputTokens ?? 0, totalCostUsd: result.tokens?.totalCostUsd });
|
|
@@ -247,7 +249,7 @@ function routingSteps(options) {
|
|
|
247
249
|
return blocked('Selected routing profile exceeds configured limits; execution blocked');
|
|
248
250
|
const provider = worker?.agent ?? options.parent;
|
|
249
251
|
const selection = worker
|
|
250
|
-
? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' ? { bare: options.parentSelection?.bare } : {}) }
|
|
252
|
+
? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' ? { bare: options.parentSelection?.bare } : {}) }
|
|
251
253
|
: { ...(options.parentSelection ?? {}), nativeMultiAgent: false };
|
|
252
254
|
const workerStarted = now();
|
|
253
255
|
const result = yield () => makeWorker(provider, selection)(ctx);
|
|
@@ -296,6 +298,8 @@ function routingSteps(options) {
|
|
|
296
298
|
provider,
|
|
297
299
|
...(selection.model ? { requestedModel: selection.model } : {}),
|
|
298
300
|
...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
|
|
301
|
+
...(selection.provider ? { requestedProvider: selection.provider } : {}),
|
|
302
|
+
...(selection.variant ? { requestedVariant: selection.variant } : {}),
|
|
299
303
|
...(result.tokens?.model ? { actualModel: result.tokens.model } : {}),
|
|
300
304
|
orchestratorProvider: options.parent,
|
|
301
305
|
...(orchestratorSelection.model ? { orchestratorModel: orchestratorSelection.model } : {}),
|
package/dist/setup/command.js
CHANGED
|
@@ -7,7 +7,8 @@ import { applyActions } from '../retrofit/apply.js';
|
|
|
7
7
|
import { join } from 'node:path';
|
|
8
8
|
import { modelPresetWorkers, planModelPresets } from './model-presets.js';
|
|
9
9
|
import { runRetrofit } from '../retrofit/command.js';
|
|
10
|
-
|
|
10
|
+
import { SUPPORTED_AGENTS } from '../agents/catalog.js';
|
|
11
|
+
const ALL_AGENTS = [...SUPPORTED_AGENTS];
|
|
11
12
|
export function defaultRoutingWorkers(agents) {
|
|
12
13
|
const workers = {
|
|
13
14
|
claude: [
|
|
@@ -32,6 +33,15 @@ export function defaultRoutingWorkers(agents) {
|
|
|
32
33
|
// Respect the user's Qwen Code account/model. API presets are opt-in.
|
|
33
34
|
{ id: 'qwen-standard', agent: 'qwen', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
34
35
|
],
|
|
36
|
+
opencode: [
|
|
37
|
+
{ id: 'opencode-standard', agent: 'opencode', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
38
|
+
],
|
|
39
|
+
kilo: [
|
|
40
|
+
{ id: 'kilo-standard', agent: 'kilo', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
41
|
+
],
|
|
42
|
+
pi: [
|
|
43
|
+
{ id: 'pi-standard', agent: 'pi', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
44
|
+
],
|
|
35
45
|
};
|
|
36
46
|
return agents.flatMap(agent => workers[agent]);
|
|
37
47
|
}
|
|
@@ -85,12 +95,12 @@ export async function runSetup(targetDir, opts = {}) {
|
|
|
85
95
|
let decisionPolicy = defaultPolicy;
|
|
86
96
|
let routing = defaultRouting;
|
|
87
97
|
if (interactive && ask) {
|
|
88
|
-
agents = parseAgents(await ask(`Agents [${defaultAgents.join(',')}] (
|
|
98
|
+
agents = parseAgents(await ask(`Agents [${defaultAgents.join(',')}] (${SUPPORTED_AGENTS.join(',')}|all): `), defaultAgents);
|
|
89
99
|
const graphAnswer = (await ask(`Code graph [${defaultGraph}] (graphify|serena): `)).trim().toLowerCase();
|
|
90
100
|
if (graphAnswer === 'graphify' || graphAnswer === 'serena')
|
|
91
101
|
codeGraph = graphAnswer;
|
|
92
102
|
loop = yes(await ask(`Enable autonomous loop? [${defaultLoop ? 'yes' : 'no'}]: `), defaultLoop);
|
|
93
|
-
const runnerAnswer = (await ask(`Default runner [${runner}] (
|
|
103
|
+
const runnerAnswer = (await ask(`Default runner [${runner}] (${SUPPORTED_AGENTS.join('|')}): `)).trim().toLowerCase();
|
|
94
104
|
if (ALL_AGENTS.includes(runnerAnswer))
|
|
95
105
|
runner = runnerAnswer;
|
|
96
106
|
const policyAnswer = (await ask(`Decision mode [${decisionPolicy}] (auto|critical): `)).trim().toLowerCase();
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Dashboard evolution
|
|
2
2
|
|
|
3
|
-
The local dashboard is an actionable
|
|
3
|
+
The local dashboard is an actionable control room for registered Yoke projects. It reads the same saved project, loop, goal, acceptance, and measurement data as the CLI. Project-controlled text is rendered through `textContent`, and the server remains bound to the loopback interface with same-origin session authorization for typed controls. The full feature contract is in [DASHBOARD-OVERHAUL.md](DASHBOARD-OVERHAUL.md).
|
|
4
4
|
|
|
5
5
|
## Overview
|
|
6
6
|
|
|
@@ -10,14 +10,28 @@ Cards also show the reported current task and saved blocker reason when availabl
|
|
|
10
10
|
|
|
11
11
|
An active loop report more than 20 minutes old is labeled **unconfirmed**. This means Yoke has an old active report, not evidence that the process is still live. The overview can be searched by project name, canonical path, or goal objective and filtered to All, Active, or Needs attention. A no-match state explains the result and provides a clear action that resets both search and filter.
|
|
12
12
|
|
|
13
|
+
## Workspace control room
|
|
14
|
+
|
|
15
|
+
The overview ranks projects by attention, last activity, recorded tokens, reported cost, accepted work, or name. Search and status filters compose with the ranking, and the validated URL hash preserves the selected screen and time scope. A project row opens the live view without losing the operator’s navigation context.
|
|
16
|
+
|
|
17
|
+
The project live view shows goal and loop state independently, freshness, current worker metadata, objective progress, last successful sync, pause/resume actions at the existing safe boundary, an operator-note form, a queued-change form, and a bounded expandable event timeline. Notes are append-only events. Changes become pending inbox requests and are consumed by the existing planning boundary; the browser cannot run arbitrary commands.
|
|
18
|
+
|
|
19
|
+
## Measurement coverage
|
|
20
|
+
|
|
21
|
+
The dashboard labels recorded, partial, unknown, stale, corrupt, unavailable, and empty data separately. **Measurement coverage** is always shown alongside analytics: missing provider usage or price is not reconstructed, and a missing bucket means no recorded activity rather than a measured zero. A stale active report is shown as unconfirmed, not healthy.
|
|
22
|
+
|
|
13
23
|
## Durable navigation
|
|
14
24
|
|
|
15
|
-
The URL hash stores the current screen, project, project tab, period, UTC grouping, and complete custom date range. Supported screens are the overview, workspace
|
|
25
|
+
The URL hash stores the current screen, project, project tab, period, UTC grouping, ranking/filter state, and complete custom date range. Supported screens are the overview, workspace analytics, and project detail. Supported project tabs are Now, Usage & time, Results, and History; periods are 1, 7, 30, 90, or 365 days; groupings are day, week, or month. Custom dates must be real ISO calendar dates in chronological order and cover at most 366 inclusive days.
|
|
16
26
|
|
|
17
27
|
Invalid hash state returns to the overview with the 30-day/day defaults. Browser back and forward, a page reload, and Refresh restore the validated state. Refresh reloads data without resetting the selected view or controls. Starting any navigation aborts earlier fetches and changes a request generation, so an older response cannot replace the current screen.
|
|
18
28
|
|
|
19
29
|
The workspace comparison schedules at most three project analytics requests at once. If navigation changes, in-flight fetches are aborted and no additional obsolete project requests are scheduled. All projects and individual project links remain available in the navigation while viewing the comparison.
|
|
20
30
|
|
|
31
|
+
## Analytics and history
|
|
32
|
+
|
|
33
|
+
Workspace and project analytics expose time-bucketed recorded tokens, calls, duration, outcomes, cost state, and rankings by agent, provider, model, variant, role, phase, project, and run. The history explorer exposes at most 100 events in chronological order and identifies run, story, phase, agent/provider/model metadata, local time, and UTC time when recorded. Unknown values remain unknown; no chart or comparison converts missing measurements to zero.
|
|
34
|
+
|
|
21
35
|
## Usage comparisons
|
|
22
36
|
|
|
23
37
|
Usage & time compares the selected period with the immediately preceding period of exactly the same duration. A single time boundary is captured before either analytics request is made. The comparison covers recorded input plus output tokens, recorded acceptance events, and reported cost.
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# Dashboard Overhaul: Feature Contract
|
|
2
|
+
|
|
3
|
+
Status: approved implementation scope
|
|
4
|
+
Date: 2026-09-08
|
|
5
|
+
|
|
6
|
+
This document turns the dashboard request into executable product definitions.
|
|
7
|
+
Each feature below has a bounded behavior, source of truth, safety rule, and
|
|
8
|
+
acceptance proof. It is an implementation contract, not a list of future ideas.
|
|
9
|
+
|
|
10
|
+
## North star
|
|
11
|
+
|
|
12
|
+
Yoke Dashboard is a local-first operations console for autonomous work. A user
|
|
13
|
+
can move from workspace-level ranking to one run, understand what happened,
|
|
14
|
+
intervene safely at a loop boundary, and leave a durable note or change request
|
|
15
|
+
without interrupting the worker or inventing a second execution path.
|
|
16
|
+
|
|
17
|
+
## Feature definitions
|
|
18
|
+
|
|
19
|
+
### F1 — Unified workspace control room
|
|
20
|
+
|
|
21
|
+
The overview loads all registered projects into one ranked workspace. Each row
|
|
22
|
+
shows project identity, current state, freshness, objective, last activity,
|
|
23
|
+
latest run, token totals, cost coverage, accepted/failed outcome, and the next
|
|
24
|
+
available safe action. A project can be sorted by attention, active state, last
|
|
25
|
+
activity, token usage, cost, acceptance, or name. Search and status filters are
|
|
26
|
+
composable and persist in navigation state.
|
|
27
|
+
|
|
28
|
+
Source: registry, goal/loop snapshots, events, history measurements, and checks.
|
|
29
|
+
|
|
30
|
+
Rule: missing projects and partial telemetry remain visible; an unknown metric
|
|
31
|
+
never becomes zero.
|
|
32
|
+
|
|
33
|
+
### F2 — Historical telemetry and rankings
|
|
34
|
+
|
|
35
|
+
The dashboard exposes a selectable date range and bucket (day, week, month) for
|
|
36
|
+
workspace and project views. It aggregates input, output, cached, cache-write,
|
|
37
|
+
reasoning, total tokens, measured/unknown calls, attempts, accepted runs,
|
|
38
|
+
repairs, escalations, elapsed time, and reported cost. It ranks projects,
|
|
39
|
+
agents, providers, models, roles, phases, and runs, with measurement coverage
|
|
40
|
+
shown alongside every aggregate.
|
|
41
|
+
|
|
42
|
+
Source: durable history archives plus bounded live events.
|
|
43
|
+
|
|
44
|
+
Rule: aggregation is deterministic and bounded; corrupt archives are reported
|
|
45
|
+
as coverage gaps, not silently discarded or counted as zero.
|
|
46
|
+
|
|
47
|
+
### F3 — Run explorer and event timeline
|
|
48
|
+
|
|
49
|
+
A project detail view provides a run list and a chronological timeline. Users can
|
|
50
|
+
expand an event to inspect run/attempt/story/phase identifiers, agent metadata,
|
|
51
|
+
token fields, outcome, and source timestamp. The view supports the same date
|
|
52
|
+
range and an explicit “live” scope.
|
|
53
|
+
|
|
54
|
+
Source: event stream and history records, joined by run ID where available.
|
|
55
|
+
|
|
56
|
+
Rule: show UTC and local time clearly; bound event and archive reads; stale live
|
|
57
|
+
data is labeled unconfirmed.
|
|
58
|
+
|
|
59
|
+
### F4 — Safe live controls
|
|
60
|
+
|
|
61
|
+
The selected project has pause and resume controls. Pause uses Yoke’s existing
|
|
62
|
+
`.yoke/loop.pause` and `.yoke/goal.pause` safe-boundary mechanisms. Resume calls
|
|
63
|
+
the existing loop/goal runner with one guarded invocation and respects the loop
|
|
64
|
+
lock; it never starts a duplicate worker. The UI reports requested, running,
|
|
65
|
+
paused, blocked, and failed-to-start states separately.
|
|
66
|
+
|
|
67
|
+
Rule: all writes require loopback same-origin plus the dashboard session token.
|
|
68
|
+
No action accepts a filesystem path or shell string from the browser.
|
|
69
|
+
|
|
70
|
+
### F5 — Operator notes and queued changes
|
|
71
|
+
|
|
72
|
+
While work runs, a user can append a short operator note or queue a typed change
|
|
73
|
+
request. Notes are append-only and displayed in the timeline. Change requests
|
|
74
|
+
reuse Yoke’s existing append-only change inbox and are processed at its normal
|
|
75
|
+
safe planning boundary. The UI shows pending/applied state and the request ID.
|
|
76
|
+
|
|
77
|
+
Rule: notes and changes are never executed as shell commands. Input is bounded,
|
|
78
|
+
validated, escaped on render, and retained as project-local operational history.
|
|
79
|
+
|
|
80
|
+
### F6 — Theme, density, and responsive access
|
|
81
|
+
|
|
82
|
+
Dark and light themes use the design tokens in `DESIGN.md`; first load respects
|
|
83
|
+
system preference, then remembers the user’s choice. The shell works at desktop,
|
|
84
|
+
tablet, and narrow widths. Tables remain scannable through column priorities,
|
|
85
|
+
horizontal overflow where necessary, and accessible labels.
|
|
86
|
+
|
|
87
|
+
Rule: contrast, focus, reduced motion, keyboard navigation, and non-color status
|
|
88
|
+
signals are tested; visual polish cannot hide loading, empty, error, or stale
|
|
89
|
+
states.
|
|
90
|
+
|
|
91
|
+
### F7 — Honest operational states
|
|
92
|
+
|
|
93
|
+
Every live panel shows last successful sync, source scope, and data coverage.
|
|
94
|
+
Active-but-stale work becomes “unconfirmed”; unavailable costs/tokens are
|
|
95
|
+
explicit. Corrupt project files produce an actionable per-project error without
|
|
96
|
+
breaking the workspace view.
|
|
97
|
+
|
|
98
|
+
Rule: the dashboard never claims process health merely because a status file says
|
|
99
|
+
running; freshness and source errors are part of the displayed state.
|
|
100
|
+
|
|
101
|
+
### F8 — Stable local API contract
|
|
102
|
+
|
|
103
|
+
The server provides bounded, same-origin JSON for workspace overview, project
|
|
104
|
+
snapshot, project history/events, and typed controls. Responses are safe to
|
|
105
|
+
render as untrusted project text. Query ranges, limits, sort keys, and action
|
|
106
|
+
payloads are schema-validated.
|
|
107
|
+
|
|
108
|
+
Rule: preserve existing routes where possible, keep the server loopback-only,
|
|
109
|
+
and add tests for authorization, path safety, malformed input, concurrency, and
|
|
110
|
+
partial history.
|
|
111
|
+
|
|
112
|
+
## Delivery slices
|
|
113
|
+
|
|
114
|
+
The Yoke loop executes these slices in dependency order:
|
|
115
|
+
|
|
116
|
+
1. Contract, design tokens, shared telemetry/control types, and API helpers.
|
|
117
|
+
2. Historical aggregation, run/event projections, and workspace endpoints.
|
|
118
|
+
3. Typed operator controls, append-only notes, and guarded resume.
|
|
119
|
+
4. New shell, theme system, workspace ranking, and responsive state components.
|
|
120
|
+
5. Project live view, controls, timeline, notes, and queued changes.
|
|
121
|
+
6. Analytics/history explorer with all rankings and coverage explanations.
|
|
122
|
+
7. Accessibility, loading/error/empty/stale states, performance bounds, and
|
|
123
|
+
focused regression tests.
|
|
124
|
+
8. README/docs synchronization and final integrated verification.
|
|
125
|
+
|
|
126
|
+
## Non-goals for this release
|
|
127
|
+
|
|
128
|
+
- Remote multi-user access, accounts, or cloud storage.
|
|
129
|
+
- Arbitrary shell execution from the browser.
|
|
130
|
+
- Replacing the loop, goal runner, provider routing, or lock protocol.
|
|
131
|
+
- Fabricating historical metrics that were not recorded.
|
|
132
|
+
- A second real-time transport dependency; polling with freshness indicators is
|
|
133
|
+
sufficient for the local-first release.
|
|
134
|
+
|
|
135
|
+
## Acceptance checklist
|
|
136
|
+
|
|
137
|
+
- A workspace user can rank projects by attention, activity, tokens, cost, and
|
|
138
|
+
name and open a project without losing filters.
|
|
139
|
+
- A project user can inspect a bounded historical run/event view and identify
|
|
140
|
+
agent/provider/model/role/phase/time when recorded.
|
|
141
|
+
- A project user can pause at the existing safe boundary, resume without a
|
|
142
|
+
duplicate loop, add a note, and queue a change request from the UI.
|
|
143
|
+
- Dark/light themes, keyboard focus, responsive layout, and reduced motion work.
|
|
144
|
+
- Partial, unknown, stale, corrupt, and missing data are explicit and isolated.
|
|
145
|
+
- Existing dashboard security and tests remain green; new behavior has focused
|
|
146
|
+
tests plus build/lint coverage.
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# OpenCode, Kilo and Pi
|
|
2
|
+
|
|
3
|
+
Yoke 1.13.0 adds first-class adapters for OpenCode, Kilo and Pi coding agent. The adapters share Yoke's invocation, routing, review, quality, telemetry and retrofit contracts, while preserving the controls each harness actually provides.
|
|
4
|
+
|
|
5
|
+
This is a CLI integration, not an authentication bundle. Install the selected harness, log in or configure its API provider, and verify it independently before starting a Yoke loop.
|
|
6
|
+
|
|
7
|
+
## Setup and retrofit
|
|
8
|
+
|
|
9
|
+
Select one of the new harnesses explicitly:
|
|
10
|
+
|
|
11
|
+
```sh
|
|
12
|
+
yoke setup . --yes --agent=opencode --runner=opencode
|
|
13
|
+
yoke setup . --yes --agent=kilo --runner=kilo
|
|
14
|
+
yoke setup . --yes --agent=pi --runner=pi
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
To add one to an existing project without changing the other generated artifacts:
|
|
18
|
+
|
|
19
|
+
```sh
|
|
20
|
+
yoke retrofit . --agent=opencode
|
|
21
|
+
yoke retrofit . --agent=kilo
|
|
22
|
+
yoke retrofit . --agent=pi
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
`--agent=all` now includes all seven supported harnesses. Retrofit is merge-aware for the native config files and backs up Yoke-managed overwrites under `.yoke/backup/`.
|
|
26
|
+
|
|
27
|
+
## Native artifacts
|
|
28
|
+
|
|
29
|
+
| Harness | Generated project artifacts | Native capabilities used by Yoke |
|
|
30
|
+
|---|---|---|
|
|
31
|
+
| OpenCode | `AGENTS.md`, `.opencode/skills/`, `opencode.json`, `.opencode/agents/yoke-reviewer.md` | `run --format json`, provider/model selection, variants, plan agent, local MCP servers |
|
|
32
|
+
| Kilo | `AGENTS.md`, `.kilo/skills/`, `kilo.jsonc`, `.kilo/agents/yoke-reviewer.md` | OpenCode-compatible `run --format json`, provider/model selection, variants, plan agent, local MCP servers |
|
|
33
|
+
| Pi | `AGENTS.md`, `.pi/skills/`, `.pi/settings.json` | JSONL mode, provider/model selection, thinking level, explicit tool allowlists |
|
|
34
|
+
|
|
35
|
+
All three consume the shared `AGENTS.md` and `.yoke/context/*.md` context. OpenCode and Kilo also receive the configured local MCP servers from the Yoke code-graph choice. Pi has no native MCP or sub-agent layer, so its integration intentionally uses the portable skills and Yoke's own loop rather than pretending those features exist.
|
|
36
|
+
|
|
37
|
+
## Provider, model and variant selection
|
|
38
|
+
|
|
39
|
+
Yoke keeps the harness (`agent`) separate from the model provider (`provider`) and model identifier (`model`):
|
|
40
|
+
|
|
41
|
+
```yaml
|
|
42
|
+
runner:
|
|
43
|
+
agent: opencode
|
|
44
|
+
provider: openrouter
|
|
45
|
+
model: openai/gpt-5.6
|
|
46
|
+
variant: high
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
For OpenCode and Kilo this becomes `--model openrouter/openai/gpt-5.6 --variant high`. If no explicit provider is configured, a model string already in the harness's `provider/model` form is passed through unchanged. For Pi the equivalent is:
|
|
50
|
+
|
|
51
|
+
```yaml
|
|
52
|
+
runner:
|
|
53
|
+
agent: pi
|
|
54
|
+
provider: openai
|
|
55
|
+
model: gpt-5.6
|
|
56
|
+
reasoningEffort: high
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
This becomes `--provider openai --model gpt-5.6 --thinking high`. Pi calls `variant` and `reasoningEffort` the same underlying thinking-level selection; configuring both with different values is rejected.
|
|
60
|
+
|
|
61
|
+
The same fields are available on routing workers and quality critic/repair roles. Routing evidence is keyed by harness, provider, model, reasoning effort and variant, so a model profile does not inherit another profile's success history.
|
|
62
|
+
|
|
63
|
+
## Permission profiles and honest limits
|
|
64
|
+
|
|
65
|
+
| Yoke profile | OpenCode / Kilo | Pi |
|
|
66
|
+
|---|---|---|
|
|
67
|
+
| `safe` | Headless `--auto` run; Yoke still verifies the resulting tree and gates the commit | `read,bash,edit,write` tool allowlist so the implementer can test and modify the project |
|
|
68
|
+
| `read-only` | `--agent plan` plus JSON output | `read,grep,find,ls` only |
|
|
69
|
+
| `unsafe` | Explicit `--dangerously-skip-permissions` | Harness default tool set; no Yoke-added allowlist |
|
|
70
|
+
|
|
71
|
+
OpenCode, Kilo and Pi do not provide the same OS-level sandbox boundary as Codex or Gemini. The Yoke `safe` label therefore describes the selected harness controls and Yoke's mechanical gates, not a universal filesystem sandbox. Use `read-only` for review and do not use `unsafe` unless the project owner accepts the boundary.
|
|
72
|
+
|
|
73
|
+
The Yoke loop disables native delegation where the harness exposes it, so Yoke's worker budget remains the authority. OpenCode and Kilo native agents are installed as a read-only reviewer artifact; Yoke's review runner still validates the structured verdict itself. Pi has no native sub-agent or plan mode by design.
|
|
74
|
+
|
|
75
|
+
## Telemetry and validation limits
|
|
76
|
+
|
|
77
|
+
Yoke parses OpenCode/Kilo JSON text events and `step_finish` token/cost events, and Pi `message_end` plus `message_update.usage` events. Missing provider events produce partial or unknown measurements; they are never reported as zero-cost success. Provider-reported model and usage remain provider claims.
|
|
78
|
+
|
|
79
|
+
The release test suite covers argument construction, provider/variant propagation, routing and quality configuration, retrofit plans, host detection, JSON result parsing, and representative telemetry fixtures. It does not authenticate against every provider or claim equal model quality. Run a small project-specific smoke task after installing a harness and configuring credentials.
|
|
80
|
+
|
|
81
|
+
Official CLI references: [OpenCode CLI](https://dev.opencode.ai/docs/cli), [OpenCode configuration](https://opencode.ai/docs/config), [OpenCode skills](https://opencode.ai/docs/skills), [Kilo CLI reference](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-docs/pages/code-with-ai/platforms/cli-reference.md), and [Pi coding agent](https://github.com/badlogic/pi-mono/blob/main/packages/coding-agent/README.md).
|
package/gemini-extension.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
3
|
+
"version": "1.14.0",
|
|
4
|
+
"description": "Cross-agent coding harness for seven supported CLIs: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
5
|
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
6
|
}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hecer/yoke",
|
|
3
|
-
"version": "1.
|
|
4
|
-
"description": "One harness,
|
|
3
|
+
"version": "1.14.0",
|
|
4
|
+
"description": "One harness, seven agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo and Pi: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
7
7
|
"yoke": "dist/cli.js"
|
|
@@ -45,6 +45,10 @@
|
|
|
45
45
|
"claude-code",
|
|
46
46
|
"codex",
|
|
47
47
|
"gemini-cli",
|
|
48
|
+
"qwen-code",
|
|
49
|
+
"opencode",
|
|
50
|
+
"kilo",
|
|
51
|
+
"pi-coding-agent",
|
|
48
52
|
"agents",
|
|
49
53
|
"agentic-coding",
|
|
50
54
|
"harness",
|