@hecer/yoke 1.12.0 → 1.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/.claude-plugin/plugin.json +3 -3
  2. package/.codex-plugin/plugin.json +2 -2
  3. package/CHANGELOG.md +41 -0
  4. package/README.md +42 -30
  5. package/canon/loop/prd.schema.md +2 -2
  6. package/canon/manifest.yaml +2 -2
  7. package/canon/skills/authoring-prd/SKILL.md +1 -1
  8. package/canon/skills/yoke-retrofit/SKILL.md +2 -2
  9. package/canon/skills/yoke-workflow/SKILL.md +1 -1
  10. package/canon/tools/graphify.md +1 -1
  11. package/canon/tools/playwright-mcp.md +1 -1
  12. package/canon/tools/serena.md +1 -1
  13. package/dist/agents/catalog.js +7 -0
  14. package/dist/agents/contracts.js +3 -1
  15. package/dist/agents/host.js +4 -0
  16. package/dist/agents/process-streams.js +62 -0
  17. package/dist/agents/process.js +43 -3
  18. package/dist/agents/providers.js +45 -3
  19. package/dist/agents/telemetry.js +99 -2
  20. package/dist/canon/manifest.js +2 -1
  21. package/dist/change/inbox.js +1 -1
  22. package/dist/cli.js +22 -24
  23. package/dist/dashboard/analytics.js +193 -29
  24. package/dist/dashboard/contracts.js +23 -0
  25. package/dist/dashboard/page.js +39 -94
  26. package/dist/dashboard/panels.js +87 -33
  27. package/dist/dashboard/server.js +190 -15
  28. package/dist/goals/command.js +3 -2
  29. package/dist/loop/claims.js +2 -1
  30. package/dist/loop/decision.js +3 -2
  31. package/dist/loop/parallel-command.js +4 -2
  32. package/dist/loop/prd.js +2 -1
  33. package/dist/loop/reporter.js +1 -0
  34. package/dist/loop/run-command.js +31 -10
  35. package/dist/observability/events.js +1 -1
  36. package/dist/observability/history.js +1 -1
  37. package/dist/prd/command.js +3 -3
  38. package/dist/quality/candidate-comparison.js +6 -1
  39. package/dist/quality/command.js +17 -2
  40. package/dist/quality/types.js +6 -1
  41. package/dist/retrofit/apply.js +87 -1
  42. package/dist/retrofit/config.js +8 -0
  43. package/dist/retrofit/detect.js +6 -0
  44. package/dist/retrofit/plan.js +6 -0
  45. package/dist/retrofit/planners/kilo.js +44 -0
  46. package/dist/retrofit/planners/opencode.js +44 -0
  47. package/dist/retrofit/planners/pi.js +24 -0
  48. package/dist/retrofit/skill-actions.js +3 -0
  49. package/dist/retrofit/tools.js +8 -0
  50. package/dist/review/command.js +3 -2
  51. package/dist/review/verdict.js +1 -1
  52. package/dist/routing/capability.js +2 -2
  53. package/dist/routing/planning.js +2 -0
  54. package/dist/routing/registry.js +3 -1
  55. package/dist/routing/router.js +7 -3
  56. package/dist/setup/command.js +13 -3
  57. package/docs/DASHBOARD-EVOLUTION.md +16 -2
  58. package/docs/DASHBOARD-OVERHAUL.md +146 -0
  59. package/docs/HARNESSES.md +81 -0
  60. package/gemini-extension.json +2 -2
  61. package/package.json +6 -2
@@ -1,3 +1,4 @@
1
+ import { AGENT_LIST } from '../agents/catalog.js';
1
2
  import { agentInvocation, buildStandaloneReviewPrompt, buildWatchdogInvocation, runCapturedAgent, repositoryFingerprint, isAgentAvailable, } from '../loop/runner.js';
2
3
  import { parseProviderResult } from '../agents/telemetry.js';
3
4
  import { resolveIdleMs } from '../loop/run-command.js';
@@ -5,14 +6,14 @@ import { loadConfig } from '../retrofit/config.js';
5
6
  import { parseReviewVerdict } from './verdict.js';
6
7
  // Resolve to the first available agent, preferring a *second* model so the review
7
8
  // is genuinely cross-model. claude last => a Claude-only box degrades to self-review.
8
- const RESOLUTION_ORDER = ['codex', 'gemini', 'qwen', 'claude'];
9
+ const RESOLUTION_ORDER = ['codex', 'gemini', 'qwen', 'claude', 'opencode', 'kilo', 'pi'];
9
10
  export function runReview(targetDir, opts = {}) {
10
11
  const available = opts.isAvailable ?? isAgentAvailable;
11
12
  const implementer = opts.implementer ?? loadConfig(targetDir)?.agents[0] ?? 'claude';
12
13
  let reviewer = opts.reviewer;
13
14
  if (reviewer) {
14
15
  if (!available(reviewer)) {
15
- console.error(`Reviewer agent CLI "${reviewer}" was not found on PATH. Install it, or pick another with --reviewer=<claude|codex|gemini>.`);
16
+ console.error(`Reviewer agent CLI "${reviewer}" was not found on PATH. Install it, or pick another with --reviewer=<${AGENT_LIST}>.`);
16
17
  return 2;
17
18
  }
18
19
  if (reviewer === implementer && !opts.allowSelfReview) {
@@ -70,7 +70,7 @@ export function formatReviewContract(path, provider) {
70
70
  return [
71
71
  `Write your final verdict to this absolute path: ${path}`,
72
72
  'The file must contain exactly one JSON object with this contract:',
73
- `{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[{"id":"optional id","severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1,"actionable":true,"suggestedFix":"optional repair","evidence":["optional evidence reference"]}],"provenance":{"provider":"${provider ?? 'claude|codex|gemini'}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"safe"}}`,
73
+ `{"schemaVersion":1,"approved":boolean,"summary":"non-empty string","findings":[{"id":"optional id","severity":"blocking|warning|info","message":"non-empty string","file":"optional path","line":1,"actionable":true,"suggestedFix":"optional repair","evidence":["optional evidence reference"]}],"provenance":{"provider":"${provider ?? 'claude|codex|gemini|qwen|opencode|kilo|pi'}","model":"provider-reported model","role":"review","promptVersion":1,"permissions":"safe"}}`,
74
74
  'Set approved=false when any blocking finding exists. Create the file even when the process also exits non-zero.',
75
75
  ].join('\n');
76
76
  }
@@ -57,7 +57,7 @@ export function chooseCapability(input) {
57
57
  const candidates = input.workers.filter(w => w.tier && tiers.indexOf(w.tier) >= level && (!input.maxTier || tiers.indexOf(w.tier) <= tiers.indexOf(input.maxTier)) && (!w.roles || w.roles.includes(role)) && (!input.story.agent || w.agent === input.story.agent) && (input.available?.(w.agent) ?? true));
58
58
  const history = readRoutingObservations().filter(e => e.projectHash === projectHash(input.root) && e.taskClass === input.assessment.taskClass && e.requiredTier === baseTier && e.role === role && e.failureKind !== 'infrastructure' && Date.now() - Date.parse(e.recordedAt) < 30 * 86400000);
59
59
  const evidence = (w) => {
60
- const matching = history.filter(e => e.provider === w.agent && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.actualModel);
60
+ const matching = history.filter(e => e.provider === w.agent && e.requestedProvider === w.provider && e.requestedModel === w.model && e.requestedReasoningEffort === w.reasoningEffort && e.requestedVariant === w.variant && e.actualModel);
61
61
  const actual = matching.at(-1)?.actualModel;
62
62
  return actual ? matching.filter(e => e.actualModel === actual) : [];
63
63
  };
@@ -68,7 +68,7 @@ export function chooseCapability(input) {
68
68
  const worker = reliable[0];
69
69
  const blocked = !worker && (input.fallback === 'block' || input.maxTier !== undefined);
70
70
  const provider = worker?.agent ?? input.story.agent ?? input.parent;
71
- const selection = worker ? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
71
+ const selection = worker ? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' && input.parentSelection?.bare !== undefined ? { bare: input.parentSelection.bare } : {}) }
72
72
  : { ...(provider === input.parent ? input.parentSelection : {}), nativeMultiAgent: false };
73
73
  const reason = `${role}: ${tiers[level]}; ${input.assessment.reason}${failures.length ? `; ${failures.length} verified failure(s), ${failures.length === 1 ? 'one targeted repair' : 'escalated'}` : ''}${worker ? '' : '; no eligible profile, parent/provider fallback'}`;
74
74
  return { worker, provider, selection, reason: blocked ? `${role}: no eligible profile within routing limits; execution blocked` : reason, blocked, requiredTier: baseTier, selectedTier: tiers[level], failures: failures.length, exhausted, next: input.maxTier && level >= tiers.indexOf(input.maxTier) ? 'stop at configured tier limit' : level < 3 ? tiers[level + 1] : 'stop after bounded attempts' };
@@ -4,8 +4,10 @@ export function resolvePlanner(config, start, selection = {}, override) {
4
4
  const inherited = agent === start ? selection : {};
5
5
  const planning = !override || override === (config?.planning?.agent ?? start) ? config?.planning : undefined;
6
6
  return { agent, selection: {
7
+ provider: planning?.provider ?? inherited.provider,
7
8
  model: planning?.model ?? inherited.model,
8
9
  reasoningEffort: planning?.reasoningEffort ?? inherited.reasoningEffort,
10
+ variant: planning?.variant ?? inherited.variant,
9
11
  bare: inherited.bare,
10
12
  nativeMultiAgent: false,
11
13
  } };
@@ -68,8 +68,10 @@ export function historyForWorkers(workers) {
68
68
  // Capability evidence belongs to the provider/model that produced it. This
69
69
  // prevents a reused worker id from inheriting scores from a retired model.
70
70
  if (event.provider !== worker.agent
71
+ || event.requestedProvider !== worker.provider
71
72
  || event.requestedModel !== worker.model
72
- || event.requestedReasoningEffort !== worker.reasoningEffort)
73
+ || event.requestedReasoningEffort !== worker.reasoningEffort
74
+ || event.requestedVariant !== worker.variant)
73
75
  continue;
74
76
  if (typeof event.verificationSuccess !== 'boolean')
75
77
  continue;
@@ -101,8 +101,10 @@ function callUsage(role, provider, selection, tokens, durationMs, profile) {
101
101
  role,
102
102
  provider,
103
103
  ...(profile ? { profile } : {}),
104
+ ...(selection.provider ? { requestedProvider: selection.provider } : {}),
104
105
  ...(selection.model ? { requestedModel: selection.model } : {}),
105
106
  ...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
107
+ ...(selection.variant ? { requestedVariant: selection.variant } : {}),
106
108
  ...(tokens?.model ? { actualModel: tokens.model } : {}),
107
109
  inputTokens: tokens?.inputTokens ?? 0,
108
110
  ...(tokens?.cachedInputTokens !== undefined ? { cachedInputTokens: tokens.cachedInputTokens } : {}),
@@ -191,7 +193,7 @@ function routingSteps(options) {
191
193
  if (!assessment)
192
194
  return { success: false, summary: 'Routing assessment unavailable or invalid; implementation was not started', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
193
195
  const choice = chooseCapability({ root, story: ctx.story, assessment, workers: eligibleWorkers, parent: options.parent, parentSelection: options.parentSelection, maxAttempts: options.maxAttempts, fallback: options.fallback, maxTier: options.maxTier });
194
- options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, reason: choice.reason, next: choice.next, assessment });
196
+ options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, variant: choice.selection.variant, providerModel: choice.selection.provider, reason: choice.reason, next: choice.next, assessment });
195
197
  if (choice.blocked)
196
198
  return { ...blocked(choice.reason), tokens: aggregateCalls(calls) };
197
199
  if (choice.exhausted)
@@ -209,7 +211,7 @@ function routingSteps(options) {
209
211
  return;
210
212
  recorded = true;
211
213
  recordRoutingObservation({ projectHash: projectHash(root), storyHash: storyHash(projectHash(root), ctx.story.id), assessmentKey: routingAssessmentKey(root, ctx.story), taskClass: assessment.taskClass, requiredTier: choice.requiredTier,
212
- role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort,
214
+ role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedProvider: choice.selection.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort, requestedVariant: choice.selection.variant,
213
215
  actualModel: result.tokens?.model, orchestratorProvider: options.planner?.agent ?? options.parent, orchestratorModel: (options.planner?.selection ?? options.parentSelection)?.model, orchestratorDurationMs: calls.filter(c => c.role === 'orchestrator').reduce((s, c) => s + c.durationMs, 0), workerDurationMs: calls[calls.length - 1].durationMs,
214
216
  processSuccess: result.success, verificationSuccess: infrastructureFailure ? false : verified, failureKind: infrastructureFailure ? 'infrastructure' : failureKind ?? 'implementation', usageAvailable: result.tokens !== undefined && result.tokens.measurementComplete !== false,
215
217
  inputTokens: result.tokens?.inputTokens ?? 0, outputTokens: result.tokens?.outputTokens ?? 0, totalCostUsd: result.tokens?.totalCostUsd });
@@ -247,7 +249,7 @@ function routingSteps(options) {
247
249
  return blocked('Selected routing profile exceeds configured limits; execution blocked');
248
250
  const provider = worker?.agent ?? options.parent;
249
251
  const selection = worker
250
- ? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' ? { bare: options.parentSelection?.bare } : {}) }
252
+ ? { provider: worker.provider, model: worker.model, reasoningEffort: worker.reasoningEffort, variant: worker.variant, nativeMultiAgent: false, ...(provider !== 'gemini' && provider !== 'qwen' && provider !== 'pi' ? { bare: options.parentSelection?.bare } : {}) }
251
253
  : { ...(options.parentSelection ?? {}), nativeMultiAgent: false };
252
254
  const workerStarted = now();
253
255
  const result = yield () => makeWorker(provider, selection)(ctx);
@@ -296,6 +298,8 @@ function routingSteps(options) {
296
298
  provider,
297
299
  ...(selection.model ? { requestedModel: selection.model } : {}),
298
300
  ...(selection.reasoningEffort ? { requestedReasoningEffort: selection.reasoningEffort } : {}),
301
+ ...(selection.provider ? { requestedProvider: selection.provider } : {}),
302
+ ...(selection.variant ? { requestedVariant: selection.variant } : {}),
299
303
  ...(result.tokens?.model ? { actualModel: result.tokens.model } : {}),
300
304
  orchestratorProvider: options.parent,
301
305
  ...(orchestratorSelection.model ? { orchestratorModel: orchestratorSelection.model } : {}),
@@ -7,7 +7,8 @@ import { applyActions } from '../retrofit/apply.js';
7
7
  import { join } from 'node:path';
8
8
  import { modelPresetWorkers, planModelPresets } from './model-presets.js';
9
9
  import { runRetrofit } from '../retrofit/command.js';
10
- const ALL_AGENTS = ['claude', 'codex', 'gemini', 'qwen'];
10
+ import { SUPPORTED_AGENTS } from '../agents/catalog.js';
11
+ const ALL_AGENTS = [...SUPPORTED_AGENTS];
11
12
  export function defaultRoutingWorkers(agents) {
12
13
  const workers = {
13
14
  claude: [
@@ -32,6 +33,15 @@ export function defaultRoutingWorkers(agents) {
32
33
  // Respect the user's Qwen Code account/model. API presets are opt-in.
33
34
  { id: 'qwen-standard', agent: 'qwen', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
34
35
  ],
36
+ opencode: [
37
+ { id: 'opencode-standard', agent: 'opencode', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
38
+ ],
39
+ kilo: [
40
+ { id: 'kilo-standard', agent: 'kilo', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
41
+ ],
42
+ pi: [
43
+ { id: 'pi-standard', agent: 'pi', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
44
+ ],
35
45
  };
36
46
  return agents.flatMap(agent => workers[agent]);
37
47
  }
@@ -85,12 +95,12 @@ export async function runSetup(targetDir, opts = {}) {
85
95
  let decisionPolicy = defaultPolicy;
86
96
  let routing = defaultRouting;
87
97
  if (interactive && ask) {
88
- agents = parseAgents(await ask(`Agents [${defaultAgents.join(',')}] (claude,codex,gemini,qwen|all): `), defaultAgents);
98
+ agents = parseAgents(await ask(`Agents [${defaultAgents.join(',')}] (${SUPPORTED_AGENTS.join(',')}|all): `), defaultAgents);
89
99
  const graphAnswer = (await ask(`Code graph [${defaultGraph}] (graphify|serena): `)).trim().toLowerCase();
90
100
  if (graphAnswer === 'graphify' || graphAnswer === 'serena')
91
101
  codeGraph = graphAnswer;
92
102
  loop = yes(await ask(`Enable autonomous loop? [${defaultLoop ? 'yes' : 'no'}]: `), defaultLoop);
93
- const runnerAnswer = (await ask(`Default runner [${runner}] (claude|codex|gemini|qwen): `)).trim().toLowerCase();
103
+ const runnerAnswer = (await ask(`Default runner [${runner}] (${SUPPORTED_AGENTS.join('|')}): `)).trim().toLowerCase();
94
104
  if (ALL_AGENTS.includes(runnerAnswer))
95
105
  runner = runnerAnswer;
96
106
  const policyAnswer = (await ask(`Decision mode [${decisionPolicy}] (auto|critical): `)).trim().toLowerCase();
@@ -1,6 +1,6 @@
1
1
  # Dashboard evolution
2
2
 
3
- The local dashboard is an actionable workspace for registered Yoke projects. It reads the same saved project, loop, goal, acceptance, and measurement data as the CLI. Project-controlled text is rendered through `textContent`, and the server remains bound to the loopback interface with same-origin authorization for pause requests.
3
+ The local dashboard is an actionable control room for registered Yoke projects. It reads the same saved project, loop, goal, acceptance, and measurement data as the CLI. Project-controlled text is rendered through `textContent`, and the server remains bound to the loopback interface with same-origin session authorization for typed controls. The full feature contract is in [DASHBOARD-OVERHAUL.md](DASHBOARD-OVERHAUL.md).
4
4
 
5
5
  ## Overview
6
6
 
@@ -10,14 +10,28 @@ Cards also show the reported current task and saved blocker reason when availabl
10
10
 
11
11
  An active loop report more than 20 minutes old is labeled **unconfirmed**. This means Yoke has an old active report, not evidence that the process is still live. The overview can be searched by project name, canonical path, or goal objective and filtered to All, Active, or Needs attention. A no-match state explains the result and provides a clear action that resets both search and filter.
12
12
 
13
+ ## Workspace control room
14
+
15
+ The overview ranks projects by attention, last activity, recorded tokens, reported cost, accepted work, or name. Search and status filters compose with the ranking, and the validated URL hash preserves the selected screen and time scope. A project row opens the live view without losing the operator’s navigation context.
16
+
17
+ The project live view shows goal and loop state independently, freshness, current worker metadata, objective progress, last successful sync, pause/resume actions at the existing safe boundary, an operator-note form, a queued-change form, and a bounded expandable event timeline. Notes are append-only events. Changes become pending inbox requests and are consumed by the existing planning boundary; the browser cannot run arbitrary commands.
18
+
19
+ ## Measurement coverage
20
+
21
+ The dashboard labels recorded, partial, unknown, stale, corrupt, unavailable, and empty data separately. **Measurement coverage** is always shown alongside analytics: missing provider usage or price is not reconstructed, and a missing bucket means no recorded activity rather than a measured zero. A stale active report is shown as unconfirmed, not healthy.
22
+
13
23
  ## Durable navigation
14
24
 
15
- The URL hash stores the current screen, project, project tab, period, UTC grouping, and complete custom date range. Supported screens are the overview, workspace comparison, and project detail. Supported project tabs are Now, Usage & time, and Results; periods are 1, 7, 30, 90, or 365 days; groupings are day, week, or month. Custom dates must be real ISO calendar dates in chronological order and cover at most 366 inclusive days.
25
+ The URL hash stores the current screen, project, project tab, period, UTC grouping, ranking/filter state, and complete custom date range. Supported screens are the overview, workspace analytics, and project detail. Supported project tabs are Now, Usage & time, Results, and History; periods are 1, 7, 30, 90, or 365 days; groupings are day, week, or month. Custom dates must be real ISO calendar dates in chronological order and cover at most 366 inclusive days.
16
26
 
17
27
  Invalid hash state returns to the overview with the 30-day/day defaults. Browser back and forward, a page reload, and Refresh restore the validated state. Refresh reloads data without resetting the selected view or controls. Starting any navigation aborts earlier fetches and changes a request generation, so an older response cannot replace the current screen.
18
28
 
19
29
  The workspace comparison schedules at most three project analytics requests at once. If navigation changes, in-flight fetches are aborted and no additional obsolete project requests are scheduled. All projects and individual project links remain available in the navigation while viewing the comparison.
20
30
 
31
+ ## Analytics and history
32
+
33
+ Workspace and project analytics expose time-bucketed recorded tokens, calls, duration, outcomes, cost state, and rankings by agent, provider, model, variant, role, phase, project, and run. The history explorer exposes at most 100 events in chronological order and identifies run, story, phase, agent/provider/model metadata, local time, and UTC time when recorded. Unknown values remain unknown; no chart or comparison converts missing measurements to zero.
34
+
21
35
  ## Usage comparisons
22
36
 
23
37
  Usage & time compares the selected period with the immediately preceding period of exactly the same duration. A single time boundary is captured before either analytics request is made. The comparison covers recorded input plus output tokens, recorded acceptance events, and reported cost.
@@ -0,0 +1,146 @@
1
+ # Dashboard Overhaul: Feature Contract
2
+
3
+ Status: approved implementation scope
4
+ Date: 2026-09-08
5
+
6
+ This document turns the dashboard request into executable product definitions.
7
+ Each feature below has a bounded behavior, source of truth, safety rule, and
8
+ acceptance proof. It is an implementation contract, not a list of future ideas.
9
+
10
+ ## North star
11
+
12
+ Yoke Dashboard is a local-first operations console for autonomous work. A user
13
+ can move from workspace-level ranking to one run, understand what happened,
14
+ intervene safely at a loop boundary, and leave a durable note or change request
15
+ without interrupting the worker or inventing a second execution path.
16
+
17
+ ## Feature definitions
18
+
19
+ ### F1 — Unified workspace control room
20
+
21
+ The overview loads all registered projects into one ranked workspace. Each row
22
+ shows project identity, current state, freshness, objective, last activity,
23
+ latest run, token totals, cost coverage, accepted/failed outcome, and the next
24
+ available safe action. A project can be sorted by attention, active state, last
25
+ activity, token usage, cost, acceptance, or name. Search and status filters are
26
+ composable and persist in navigation state.
27
+
28
+ Source: registry, goal/loop snapshots, events, history measurements, and checks.
29
+
30
+ Rule: missing projects and partial telemetry remain visible; an unknown metric
31
+ never becomes zero.
32
+
33
+ ### F2 — Historical telemetry and rankings
34
+
35
+ The dashboard exposes a selectable date range and bucket (day, week, month) for
36
+ workspace and project views. It aggregates input, output, cached, cache-write,
37
+ reasoning, total tokens, measured/unknown calls, attempts, accepted runs,
38
+ repairs, escalations, elapsed time, and reported cost. It ranks projects,
39
+ agents, providers, models, roles, phases, and runs, with measurement coverage
40
+ shown alongside every aggregate.
41
+
42
+ Source: durable history archives plus bounded live events.
43
+
44
+ Rule: aggregation is deterministic and bounded; corrupt archives are reported
45
+ as coverage gaps, not silently discarded or counted as zero.
46
+
47
+ ### F3 — Run explorer and event timeline
48
+
49
+ A project detail view provides a run list and a chronological timeline. Users can
50
+ expand an event to inspect run/attempt/story/phase identifiers, agent metadata,
51
+ token fields, outcome, and source timestamp. The view supports the same date
52
+ range and an explicit “live” scope.
53
+
54
+ Source: event stream and history records, joined by run ID where available.
55
+
56
+ Rule: show UTC and local time clearly; bound event and archive reads; stale live
57
+ data is labeled unconfirmed.
58
+
59
+ ### F4 — Safe live controls
60
+
61
+ The selected project has pause and resume controls. Pause uses Yoke’s existing
62
+ `.yoke/loop.pause` and `.yoke/goal.pause` safe-boundary mechanisms. Resume calls
63
+ the existing loop/goal runner with one guarded invocation and respects the loop
64
+ lock; it never starts a duplicate worker. The UI reports requested, running,
65
+ paused, blocked, and failed-to-start states separately.
66
+
67
+ Rule: all writes require loopback same-origin plus the dashboard session token.
68
+ No action accepts a filesystem path or shell string from the browser.
69
+
70
+ ### F5 — Operator notes and queued changes
71
+
72
+ While work runs, a user can append a short operator note or queue a typed change
73
+ request. Notes are append-only and displayed in the timeline. Change requests
74
+ reuse Yoke’s existing append-only change inbox and are processed at its normal
75
+ safe planning boundary. The UI shows pending/applied state and the request ID.
76
+
77
+ Rule: notes and changes are never executed as shell commands. Input is bounded,
78
+ validated, escaped on render, and retained as project-local operational history.
79
+
80
+ ### F6 — Theme, density, and responsive access
81
+
82
+ Dark and light themes use the design tokens in `DESIGN.md`; first load respects
83
+ system preference, then remembers the user’s choice. The shell works at desktop,
84
+ tablet, and narrow widths. Tables remain scannable through column priorities,
85
+ horizontal overflow where necessary, and accessible labels.
86
+
87
+ Rule: contrast, focus, reduced motion, keyboard navigation, and non-color status
88
+ signals are tested; visual polish cannot hide loading, empty, error, or stale
89
+ states.
90
+
91
+ ### F7 — Honest operational states
92
+
93
+ Every live panel shows last successful sync, source scope, and data coverage.
94
+ Active-but-stale work becomes “unconfirmed”; unavailable costs/tokens are
95
+ explicit. Corrupt project files produce an actionable per-project error without
96
+ breaking the workspace view.
97
+
98
+ Rule: the dashboard never claims process health merely because a status file says
99
+ running; freshness and source errors are part of the displayed state.
100
+
101
+ ### F8 — Stable local API contract
102
+
103
+ The server provides bounded, same-origin JSON for workspace overview, project
104
+ snapshot, project history/events, and typed controls. Responses are safe to
105
+ render as untrusted project text. Query ranges, limits, sort keys, and action
106
+ payloads are schema-validated.
107
+
108
+ Rule: preserve existing routes where possible, keep the server loopback-only,
109
+ and add tests for authorization, path safety, malformed input, concurrency, and
110
+ partial history.
111
+
112
+ ## Delivery slices
113
+
114
+ The Yoke loop executes these slices in dependency order:
115
+
116
+ 1. Contract, design tokens, shared telemetry/control types, and API helpers.
117
+ 2. Historical aggregation, run/event projections, and workspace endpoints.
118
+ 3. Typed operator controls, append-only notes, and guarded resume.
119
+ 4. New shell, theme system, workspace ranking, and responsive state components.
120
+ 5. Project live view, controls, timeline, notes, and queued changes.
121
+ 6. Analytics/history explorer with all rankings and coverage explanations.
122
+ 7. Accessibility, loading/error/empty/stale states, performance bounds, and
123
+ focused regression tests.
124
+ 8. README/docs synchronization and final integrated verification.
125
+
126
+ ## Non-goals for this release
127
+
128
+ - Remote multi-user access, accounts, or cloud storage.
129
+ - Arbitrary shell execution from the browser.
130
+ - Replacing the loop, goal runner, provider routing, or lock protocol.
131
+ - Fabricating historical metrics that were not recorded.
132
+ - A second real-time transport dependency; polling with freshness indicators is
133
+ sufficient for the local-first release.
134
+
135
+ ## Acceptance checklist
136
+
137
+ - A workspace user can rank projects by attention, activity, tokens, cost, and
138
+ name and open a project without losing filters.
139
+ - A project user can inspect a bounded historical run/event view and identify
140
+ agent/provider/model/role/phase/time when recorded.
141
+ - A project user can pause at the existing safe boundary, resume without a
142
+ duplicate loop, add a note, and queue a change request from the UI.
143
+ - Dark/light themes, keyboard focus, responsive layout, and reduced motion work.
144
+ - Partial, unknown, stale, corrupt, and missing data are explicit and isolated.
145
+ - Existing dashboard security and tests remain green; new behavior has focused
146
+ tests plus build/lint coverage.
@@ -0,0 +1,81 @@
1
+ # OpenCode, Kilo and Pi
2
+
3
+ Yoke 1.13.0 adds first-class adapters for OpenCode, Kilo and Pi coding agent. The adapters share Yoke's invocation, routing, review, quality, telemetry and retrofit contracts, while preserving the controls each harness actually provides.
4
+
5
+ This is a CLI integration, not an authentication bundle. Install the selected harness, log in or configure its API provider, and verify it independently before starting a Yoke loop.
6
+
7
+ ## Setup and retrofit
8
+
9
+ Select one of the new harnesses explicitly:
10
+
11
+ ```sh
12
+ yoke setup . --yes --agent=opencode --runner=opencode
13
+ yoke setup . --yes --agent=kilo --runner=kilo
14
+ yoke setup . --yes --agent=pi --runner=pi
15
+ ```
16
+
17
+ To add one to an existing project without changing the other generated artifacts:
18
+
19
+ ```sh
20
+ yoke retrofit . --agent=opencode
21
+ yoke retrofit . --agent=kilo
22
+ yoke retrofit . --agent=pi
23
+ ```
24
+
25
+ `--agent=all` now includes all seven supported harnesses. Retrofit is merge-aware for the native config files and backs up Yoke-managed overwrites under `.yoke/backup/`.
26
+
27
+ ## Native artifacts
28
+
29
+ | Harness | Generated project artifacts | Native capabilities used by Yoke |
30
+ |---|---|---|
31
+ | OpenCode | `AGENTS.md`, `.opencode/skills/`, `opencode.json`, `.opencode/agents/yoke-reviewer.md` | `run --format json`, provider/model selection, variants, plan agent, local MCP servers |
32
+ | Kilo | `AGENTS.md`, `.kilo/skills/`, `kilo.jsonc`, `.kilo/agents/yoke-reviewer.md` | OpenCode-compatible `run --format json`, provider/model selection, variants, plan agent, local MCP servers |
33
+ | Pi | `AGENTS.md`, `.pi/skills/`, `.pi/settings.json` | JSONL mode, provider/model selection, thinking level, explicit tool allowlists |
34
+
35
+ All three consume the shared `AGENTS.md` and `.yoke/context/*.md` context. OpenCode and Kilo also receive the configured local MCP servers from the Yoke code-graph choice. Pi has no native MCP or sub-agent layer, so its integration intentionally uses the portable skills and Yoke's own loop rather than pretending those features exist.
36
+
37
+ ## Provider, model and variant selection
38
+
39
+ Yoke keeps the harness (`agent`) separate from the model provider (`provider`) and model identifier (`model`):
40
+
41
+ ```yaml
42
+ runner:
43
+ agent: opencode
44
+ provider: openrouter
45
+ model: openai/gpt-5.6
46
+ variant: high
47
+ ```
48
+
49
+ For OpenCode and Kilo this becomes `--model openrouter/openai/gpt-5.6 --variant high`. If no explicit provider is configured, a model string already in the harness's `provider/model` form is passed through unchanged. For Pi the equivalent is:
50
+
51
+ ```yaml
52
+ runner:
53
+ agent: pi
54
+ provider: openai
55
+ model: gpt-5.6
56
+ reasoningEffort: high
57
+ ```
58
+
59
+ This becomes `--provider openai --model gpt-5.6 --thinking high`. Pi calls `variant` and `reasoningEffort` the same underlying thinking-level selection; configuring both with different values is rejected.
60
+
61
+ The same fields are available on routing workers and quality critic/repair roles. Routing evidence is keyed by harness, provider, model, reasoning effort and variant, so a model profile does not inherit another profile's success history.
62
+
63
+ ## Permission profiles and honest limits
64
+
65
+ | Yoke profile | OpenCode / Kilo | Pi |
66
+ |---|---|---|
67
+ | `safe` | Headless `--auto` run; Yoke still verifies the resulting tree and gates the commit | `read,bash,edit,write` tool allowlist so the implementer can test and modify the project |
68
+ | `read-only` | `--agent plan` plus JSON output | `read,grep,find,ls` only |
69
+ | `unsafe` | Explicit `--dangerously-skip-permissions` | Harness default tool set; no Yoke-added allowlist |
70
+
71
+ OpenCode, Kilo and Pi do not provide the same OS-level sandbox boundary as Codex or Gemini. The Yoke `safe` label therefore describes the selected harness controls and Yoke's mechanical gates, not a universal filesystem sandbox. Use `read-only` for review and do not use `unsafe` unless the project owner accepts the boundary.
72
+
73
+ The Yoke loop disables native delegation where the harness exposes it, so Yoke's worker budget remains the authority. OpenCode and Kilo native agents are installed as a read-only reviewer artifact; Yoke's review runner still validates the structured verdict itself. Pi has no native sub-agent or plan mode by design.
74
+
75
+ ## Telemetry and validation limits
76
+
77
+ Yoke parses OpenCode/Kilo JSON text events and `step_finish` token/cost events, and Pi `message_end` plus `message_update.usage` events. Missing provider events produce partial or unknown measurements; they are never reported as zero-cost success. Provider-reported model and usage remain provider claims.
78
+
79
+ The release test suite covers argument construction, provider/variant propagation, routing and quality configuration, retrofit plans, host detection, JSON result parsing, and representative telemetry fixtures. It does not authenticate against every provider or claim equal model quality. Run a small project-specific smoke task after installing a harness and configuring credentials.
80
+
81
+ Official CLI references: [OpenCode CLI](https://dev.opencode.ai/docs/cli), [OpenCode configuration](https://opencode.ai/docs/config), [OpenCode skills](https://opencode.ai/docs/skills), [Kilo CLI reference](https://github.com/Kilo-Org/kilocode/blob/main/packages/kilo-docs/pages/code-with-ai/platforms/cli-reference.md), and [Pi coding agent](https://github.com/badlogic/pi-mono/blob/main/packages/coding-agent/README.md).
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "yoke",
3
- "version": "1.12.0",
4
- "description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
3
+ "version": "1.14.0",
4
+ "description": "Cross-agent coding harness for seven supported CLIs: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
5
5
  "contextFileName": "GEMINI-EXTENSION.md"
6
6
  }
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@hecer/yoke",
3
- "version": "1.12.0",
4
- "description": "One harness, four agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, Gemini CLI, and Qwen Code: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
3
+ "version": "1.14.0",
4
+ "description": "One harness, seven agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, Gemini CLI, Qwen Code, OpenCode, Kilo and Pi: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
5
5
  "type": "module",
6
6
  "bin": {
7
7
  "yoke": "dist/cli.js"
@@ -45,6 +45,10 @@
45
45
  "claude-code",
46
46
  "codex",
47
47
  "gemini-cli",
48
+ "qwen-code",
49
+ "opencode",
50
+ "kilo",
51
+ "pi-coding-agent",
48
52
  "agents",
49
53
  "agentic-coding",
50
54
  "harness",