@hecer/yoke 1.8.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +34 -1
- package/README.md +8 -6
- package/canon/loop/prd.schema.md +6 -0
- package/canon/skills/authoring-prd/SKILL.md +7 -0
- package/dist/agents/process-incarnation.js +1 -1
- package/dist/agents/process.js +74 -6
- package/dist/agents/supervision.js +153 -0
- package/dist/agents/windows-launch.js +80 -0
- package/dist/change/inbox.js +23 -5
- package/dist/cli.js +18 -2
- package/dist/dashboard/page.js +102 -8
- package/dist/dashboard/panels.js +84 -8
- package/dist/goals/command.js +29 -4
- package/dist/loop/git.js +12 -4
- package/dist/loop/loop.js +48 -3
- package/dist/loop/parallel-adapters.js +2 -3
- package/dist/loop/parallel-command.js +14 -7
- package/dist/loop/prd.js +4 -0
- package/dist/loop/reporter.js +8 -1
- package/dist/loop/run-command.js +25 -5
- package/dist/loop/runner.js +27 -31
- package/dist/loop/watchdog.js +87 -11
- package/dist/loop/worker.js +30 -1
- package/dist/prd/assess.js +145 -0
- package/dist/prd/command.js +59 -21
- package/dist/quality/command.js +11 -5
- package/dist/retrofit/config.js +15 -1
- package/dist/retrofit/gitignore.js +1 -1
- package/dist/routing/assessment.js +66 -0
- package/dist/routing/capability.js +91 -0
- package/dist/routing/contracts.js +60 -0
- package/dist/routing/planning.js +12 -0
- package/dist/routing/router.js +85 -2
- package/dist/setup/command.js +23 -9
- package/docs/BATCH-PLANNING-VALIDATION.md +67 -0
- package/docs/CAPABILITY-ROUTING.md +82 -0
- package/docs/DASHBOARD-EVOLUTION.md +33 -0
- package/docs/PRODUCT-DIRECTION-2026-09-05.md +28 -0
- package/docs/WINDOWS-RUNNER-VALIDATION.md +104 -0
- package/docs/assets/yoke-logo.png +0 -0
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
package/dist/routing/router.js
CHANGED
|
@@ -1,6 +1,9 @@
|
|
|
1
|
-
import { buildWatchdogInvocation, makeRunner, runCapturedAgent, runnerInvocation, } from '../loop/runner.js';
|
|
2
|
-
import { isAcceptanceCriterion } from '../loop/prd.js';
|
|
1
|
+
import { buildWatchdogInvocation, makeRunner, runCapturedAgent, runnerInvocation, contextBlockFor, } from '../loop/runner.js';
|
|
2
|
+
import { isAcceptanceCriterion, criterionCommandProblem } from '../loop/prd.js';
|
|
3
3
|
import { historyForWorkers, projectHash, readRoutingObservations, recordRoutingObservation, storyHash } from './registry.js';
|
|
4
|
+
import { assessmentInstructions, parseAssessment, tiers } from './assessment.js';
|
|
5
|
+
import { chooseCapability, readAssessment, saveAssessment, routingAssessmentKey, knownInfrastructureFailure } from './capability.js';
|
|
6
|
+
import { readPlanningFile } from './contracts.js';
|
|
4
7
|
const costRank = { low: 0, medium: 1, high: 2 };
|
|
5
8
|
export function rankWorkers(workers, strategy, maxCandidates) {
|
|
6
9
|
const history = historyForWorkers(workers);
|
|
@@ -142,6 +145,76 @@ function routingSteps(options) {
|
|
|
142
145
|
selection,
|
|
143
146
|
}));
|
|
144
147
|
return function* (ctx) {
|
|
148
|
+
const blocked = (summary) => ({ success: false, summary, routing: { blocked: true, recordOutcome: () => undefined } });
|
|
149
|
+
if (options.strategy === 'capability' && options.assessmentPolicy === 'prepared') {
|
|
150
|
+
try {
|
|
151
|
+
const criteria = ctx.story.acceptance;
|
|
152
|
+
if (criteria.length < 2 || criteria.length > 5 || criteria.some(c => !isAcceptanceCriterion(c) || criterionCommandProblem(c)))
|
|
153
|
+
return blocked('Prepared routing requires 2-5 executable acceptance criteria');
|
|
154
|
+
if (!readAssessment(options.projectRoot ?? ctx.targetDir, ctx.story, true))
|
|
155
|
+
return blocked('Task assessment is missing or stale. Run yoke prd assess before execution.');
|
|
156
|
+
}
|
|
157
|
+
catch (error) {
|
|
158
|
+
return blocked(`Cannot read prepared assessment: ${error.message}`);
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
if (options.strategy === 'capability' && !options.rules?.some(rule => (!rule.area || rule.area === ctx.story.area) && (!rule.storyId || rule.storyId === ctx.story.id))) {
|
|
162
|
+
const root = options.projectRoot ?? ctx.targetDir;
|
|
163
|
+
let assessment;
|
|
164
|
+
try {
|
|
165
|
+
assessment = readAssessment(root, ctx.story, options.assessmentPolicy === 'prepared');
|
|
166
|
+
}
|
|
167
|
+
catch (error) {
|
|
168
|
+
return blocked(`Cannot read assessment: ${error.message}`);
|
|
169
|
+
}
|
|
170
|
+
let planning;
|
|
171
|
+
const calls = [];
|
|
172
|
+
if (!assessment) {
|
|
173
|
+
const inputKey = routingAssessmentKey(root, ctx.story);
|
|
174
|
+
const prompt = [assessmentInstructions, 'Use the supplied task contract and project context to produce a bounded plan. Do not implement or change files.',
|
|
175
|
+
'Approved planning brief:', readPlanningFile(root, '.yoke/plan.md', 80_000) ?? '',
|
|
176
|
+
contextBlockFor(ctx.targetDir, ctx.story), JSON.stringify(ctx.story), 'Return exactly one line: YOKE_ASSESS {"taskClass":"implementation","difficulty":"medium","uncertainty":"low","risk":"low","scope":"low","testability":"high","reason":"evidence","approach":"steps and tests"}'].join('\n');
|
|
177
|
+
const planner = options.planner?.agent ?? options.parent;
|
|
178
|
+
if (!available(planner))
|
|
179
|
+
return blocked('Configured planning provider is unavailable');
|
|
180
|
+
const selection = { ...(options.planner?.selection ?? options.parentSelection), nativeMultiAgent: false };
|
|
181
|
+
const started = now();
|
|
182
|
+
planning = yield () => options.captureRoute ? options.captureRoute(planner, ctx, prompt, selection)
|
|
183
|
+
: runCapturedAgent(planner, buildWatchdogInvocation(runnerInvocation(planner, prompt, ctx.targetDir, true, 'read-only', selection), options.idleTimeoutMs ?? 0));
|
|
184
|
+
calls.push(callUsage('orchestrator', planner, selection, planning.tokens, now() - started));
|
|
185
|
+
if (routingAssessmentKey(root, ctx.story) !== inputKey)
|
|
186
|
+
return { ...blocked('Planning inputs changed during assessment; retry planning with the current contract'), tokens: aggregateCalls(calls) };
|
|
187
|
+
assessment = planning.success ? parseAssessment(planning.output) : undefined;
|
|
188
|
+
if (assessment)
|
|
189
|
+
saveAssessment(root, ctx.story, assessment, { provider: planner, model: planning.tokens?.model ?? selection.model });
|
|
190
|
+
}
|
|
191
|
+
if (!assessment)
|
|
192
|
+
return { success: false, summary: 'Routing assessment unavailable or invalid; implementation was not started', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
|
|
193
|
+
const choice = chooseCapability({ root, story: ctx.story, assessment, workers: eligibleWorkers, parent: options.parent, parentSelection: options.parentSelection, maxAttempts: options.maxAttempts, fallback: options.fallback, maxTier: options.maxTier });
|
|
194
|
+
options.onDecision?.(ctx.story.id, { profile: choice.worker?.id ?? 'SELF', provider: choice.provider, model: choice.selection.model, reasoningEffort: choice.selection.reasoningEffort, reason: choice.reason, next: choice.next, assessment });
|
|
195
|
+
if (choice.blocked)
|
|
196
|
+
return { ...blocked(choice.reason), tokens: aggregateCalls(calls) };
|
|
197
|
+
if (choice.exhausted)
|
|
198
|
+
return { success: false, summary: 'Routing attempt budget exhausted; replan this task before retrying', tokens: aggregateCalls(calls), routing: { recordOutcome: () => undefined, blocked: true } };
|
|
199
|
+
const started = now();
|
|
200
|
+
const result = yield () => makeWorker(choice.provider, choice.selection)({ ...ctx, attempt: choice.failures + 1, story: { ...ctx.story, assessment } });
|
|
201
|
+
calls.push(callUsage(choice.worker ? 'worker' : 'parent', choice.provider, choice.selection, result.tokens, now() - started, choice.worker?.id ?? 'SELF'));
|
|
202
|
+
let recorded = false;
|
|
203
|
+
const infrastructureFailure = result.infrastructureFailure || (!result.success && knownInfrastructureFailure(result.summary));
|
|
204
|
+
return { ...result, summary: `route=${choice.worker?.id ?? 'SELF'} (${choice.reason}); ${result.summary}`,
|
|
205
|
+
...(infrastructureFailure ? { success: false, infrastructureFailure: true } : {}),
|
|
206
|
+
tokens: { ...aggregateCalls(calls), storyId: ctx.story.id, escalated: choice.failures > 1 },
|
|
207
|
+
routing: { blocked: infrastructureFailure || undefined, canRetry: !infrastructureFailure && choice.failures + 1 < (options.maxAttempts ?? 5), recordOutcome: (verified, failureKind) => {
|
|
208
|
+
if (recorded)
|
|
209
|
+
return;
|
|
210
|
+
recorded = true;
|
|
211
|
+
recordRoutingObservation({ projectHash: projectHash(root), storyHash: storyHash(projectHash(root), ctx.story.id), assessmentKey: routingAssessmentKey(root, ctx.story), taskClass: assessment.taskClass, requiredTier: choice.requiredTier,
|
|
212
|
+
role: 'implementation', strategy: 'capability', selected: choice.worker?.id ?? 'SELF', provider: choice.provider, requestedModel: choice.selection.model, requestedReasoningEffort: choice.selection.reasoningEffort,
|
|
213
|
+
actualModel: result.tokens?.model, orchestratorProvider: options.planner?.agent ?? options.parent, orchestratorModel: (options.planner?.selection ?? options.parentSelection)?.model, orchestratorDurationMs: calls.filter(c => c.role === 'orchestrator').reduce((s, c) => s + c.durationMs, 0), workerDurationMs: calls[calls.length - 1].durationMs,
|
|
214
|
+
processSuccess: result.success, verificationSuccess: infrastructureFailure ? false : verified, failureKind: infrastructureFailure ? 'infrastructure' : failureKind ?? 'implementation', usageAvailable: result.tokens !== undefined && result.tokens.measurementComplete !== false,
|
|
215
|
+
inputTokens: result.tokens?.inputTokens ?? 0, outputTokens: result.tokens?.outputTokens ?? 0, totalCostUsd: result.tokens?.totalCostUsd });
|
|
216
|
+
} } };
|
|
217
|
+
}
|
|
145
218
|
// Re-rank per story so a long-running loop can use gate outcomes learned by
|
|
146
219
|
// earlier stories without rebuilding the runner.
|
|
147
220
|
const rule = options.rules?.find(rule => (!rule.area || rule.area === ctx.story.area) && (!rule.storyId || rule.storyId === ctx.story.id) && (rule.area || rule.storyId));
|
|
@@ -154,6 +227,8 @@ function routingSteps(options) {
|
|
|
154
227
|
const ruleWorker = rule && failedStories.has(ctx.story.id) ? rule.escalateTo ?? 'SELF' : rule?.worker;
|
|
155
228
|
const candidates = rule ? eligibleWorkers : rankWorkers(eligibleWorkers, options.strategy, options.maxCandidates);
|
|
156
229
|
if (candidates.length === 0) {
|
|
230
|
+
if (options.fallback === 'block' || options.maxTier)
|
|
231
|
+
return blocked('No eligible routing profiles; parent fallback is disabled');
|
|
157
232
|
return yield () => makeWorker(options.parent, options.parentSelection ?? {})(ctx);
|
|
158
233
|
}
|
|
159
234
|
const prompt = buildRoutingPrompt(ctx, candidates, options.strategy);
|
|
@@ -168,6 +243,8 @@ function routingSteps(options) {
|
|
|
168
243
|
: routeRun.success ? parseRouteDecision(routeRun.output, candidates.map(worker => worker.id)) : null;
|
|
169
244
|
const selected = decision?.worker ?? 'SELF';
|
|
170
245
|
const worker = selected === 'SELF' ? undefined : candidates.find(candidate => candidate.id === selected);
|
|
246
|
+
if ((!worker && (options.fallback === 'block' || options.maxTier)) || (options.maxTier && (!worker?.tier || tiers.indexOf(worker.tier) > tiers.indexOf(options.maxTier))))
|
|
247
|
+
return blocked('Selected routing profile exceeds configured limits; execution blocked');
|
|
171
248
|
const provider = worker?.agent ?? options.parent;
|
|
172
249
|
const selection = worker
|
|
173
250
|
? { model: worker.model, reasoningEffort: worker.reasoningEffort, nativeMultiAgent: false, ...(provider !== 'gemini' ? { bare: options.parentSelection?.bare } : {}) }
|
|
@@ -236,3 +313,9 @@ function routingSteps(options) {
|
|
|
236
313
|
return { ...result, summary: `${routeSummary}; ${result.summary}`, tokens, routing: { recordOutcome } };
|
|
237
314
|
};
|
|
238
315
|
}
|
|
316
|
+
function aggregateCalls(calls) {
|
|
317
|
+
return { inputTokens: calls.reduce((n, c) => n + c.inputTokens, 0), outputTokens: calls.reduce((n, c) => n + c.outputTokens, 0),
|
|
318
|
+
cachedInputTokens: calls.reduce((n, c) => n + (c.cachedInputTokens ?? 0), 0), cacheWriteInputTokens: calls.reduce((n, c) => n + (c.cacheWriteInputTokens ?? 0), 0),
|
|
319
|
+
...(calls.some(c => c.totalCostUsd !== undefined) ? { totalCostUsd: calls.reduce((n, c) => n + (c.totalCostUsd ?? 0), 0) } : {}),
|
|
320
|
+
calls, measurementComplete: calls.every(c => c.usageAvailable), costMeasurementComplete: calls.every(c => c.totalCostUsd !== undefined) };
|
|
321
|
+
}
|
package/dist/setup/command.js
CHANGED
|
@@ -7,14 +7,26 @@ import { runRetrofit } from '../retrofit/command.js';
|
|
|
7
7
|
const ALL_AGENTS = ['claude', 'codex', 'gemini'];
|
|
8
8
|
export function defaultRoutingWorkers(agents) {
|
|
9
9
|
const workers = {
|
|
10
|
-
claude:
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
10
|
+
claude: [
|
|
11
|
+
{ id: 'claude-fast', agent: 'claude', model: 'haiku', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
12
|
+
{ id: 'claude-standard', agent: 'claude', model: 'sonnet', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
13
|
+
{ id: 'claude-strong', agent: 'claude', model: 'sonnet', reasoningEffort: 'high', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
14
|
+
{ id: 'claude-frontier', agent: 'claude', model: 'opus', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
15
|
+
],
|
|
16
|
+
codex: [
|
|
17
|
+
{ id: 'codex-light', agent: 'codex', model: 'gpt-5.6-luna', reasoningEffort: 'low', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
18
|
+
{ id: 'codex-standard', agent: 'codex', model: 'gpt-5.6-terra', reasoningEffort: 'medium', tier: 'standard', costTier: 'low', capabilities: ['implementation'] },
|
|
19
|
+
{ id: 'codex-strong', agent: 'codex', model: 'gpt-5.6-sol', reasoningEffort: 'high', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
20
|
+
{ id: 'codex-frontier', agent: 'codex', model: 'gpt-6-astra', reasoningEffort: 'high', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
21
|
+
],
|
|
22
|
+
gemini: [
|
|
23
|
+
{ id: 'gemini-light', agent: 'gemini', model: 'gemini-2.5-flash', tier: 'light', costTier: 'low', capabilities: ['mechanical', 'tests'] },
|
|
24
|
+
{ id: 'gemini-standard', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'standard', costTier: 'medium', capabilities: ['implementation'] },
|
|
25
|
+
{ id: 'gemini-strong', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'strong', costTier: 'medium', capabilities: ['debugging'] },
|
|
26
|
+
{ id: 'gemini-frontier', agent: 'gemini', model: 'gemini-2.5-pro', tier: 'frontier', costTier: 'high', capabilities: ['architecture'] },
|
|
27
|
+
],
|
|
16
28
|
};
|
|
17
|
-
return agents.
|
|
29
|
+
return agents.flatMap(agent => workers[agent]);
|
|
18
30
|
}
|
|
19
31
|
function parseAgents(value, fallback) {
|
|
20
32
|
if (value.trim().toLowerCase() === 'all')
|
|
@@ -90,10 +102,12 @@ export async function runSetup(targetDir, opts = {}) {
|
|
|
90
102
|
config.routing = {
|
|
91
103
|
...config.routing,
|
|
92
104
|
enabled: routing,
|
|
93
|
-
strategy: config.routing?.strategy ?? '
|
|
105
|
+
strategy: opts.routingStrategy ?? config.routing?.strategy ?? 'capability',
|
|
94
106
|
maxCandidates: config.routing?.maxCandidates ?? 3,
|
|
107
|
+
assessmentPolicy: existing?.routing?.assessmentPolicy ?? (existing ? 'on-demand' : 'prepared'),
|
|
108
|
+
fallback: existing?.routing?.fallback ?? (existing ? 'parent' : 'block'),
|
|
95
109
|
...(config.routing?.orchestrator ? { orchestrator: config.routing.orchestrator } : {}),
|
|
96
|
-
workers: existingWorkers.length > 0 ? existingWorkers : defaultRoutingWorkers(agents),
|
|
110
|
+
workers: existingWorkers.length > 0 && !opts.routingPreset ? existingWorkers : defaultRoutingWorkers(agents),
|
|
97
111
|
};
|
|
98
112
|
saveConfig(targetDir, config);
|
|
99
113
|
console.log(`Yoke setup complete: agents=${agents.join(',')} · runner=${runner} · loop=${loop ? 'on' : 'off'} · routing=${routing ? 'on' : 'off'} · decisions=${decisionPolicy}`);
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# Batch planning validation — 2026-09-06
|
|
2
|
+
|
|
3
|
+
AI-assisted implementation record for the local development build after 1.9.0.
|
|
4
|
+
No new version has been published by this task.
|
|
5
|
+
|
|
6
|
+
## Implemented
|
|
7
|
+
|
|
8
|
+
- One bounded assessment call for a selected package, with exact output IDs,
|
|
9
|
+
executable-criteria checks, project locking and atomic PRD replacement.
|
|
10
|
+
- Separate planning provider/model/effort, complete draft/inbox assessments,
|
|
11
|
+
stale-contract detection including upstream requirements and the approved brief.
|
|
12
|
+
- Prepared routing and blocked fallback in new setups; existing settings remain
|
|
13
|
+
compatible. Automatic tier ceilings block insufficient configurations.
|
|
14
|
+
- Capability routing recognizes reported Windows process-creation/authentication
|
|
15
|
+
failures and failed gate evidence as infrastructure, stopping repair without
|
|
16
|
+
adding model-quality failures.
|
|
17
|
+
|
|
18
|
+
## Actual Yoke run
|
|
19
|
+
|
|
20
|
+
The current compiled development CLI ran in a separate `Yoke-batch` checkout with
|
|
21
|
+
capability routing, prepared assessment, blocked fallback, a strong tier ceiling
|
|
22
|
+
and one serial worker. The assigned task was limited to batch-command tests.
|
|
23
|
+
Yoke chose `codex-standard`, requested `gpt-5.6-terra` at medium effort, verified
|
|
24
|
+
the criterion tests and committed the result. The main checkout received only
|
|
25
|
+
the reviewed test file; one assertion was strengthened during review.
|
|
26
|
+
|
|
27
|
+
Recorded start: 2026-09-06T17:45:54.729Z. Terminal state: complete at
|
|
28
|
+
17:50:41.108Z, one backlog task accepted. One measured worker call and no
|
|
29
|
+
orchestrator/assessment call were recorded. Input: 546,591 tokens, including
|
|
30
|
+
496,384 cached input tokens; output: 6,294 tokens. Reported monetary cost and
|
|
31
|
+
actual model identity are unknown. This is no cost benchmark or measured saving.
|
|
32
|
+
|
|
33
|
+
This run explicitly used unsafe permissions. It does not validate the Windows
|
|
34
|
+
safe-mode execution reported in issue #5.
|
|
35
|
+
|
|
36
|
+
## Issue #5 boundary
|
|
37
|
+
|
|
38
|
+
**Follow-up:** the user subsequently requested completion of the runner correction.
|
|
39
|
+
The reproduction, safe-mode end-to-end result and supervision changes are recorded
|
|
40
|
+
in [WINDOWS-RUNNER-VALIDATION.md](WINDOWS-RUNNER-VALIDATION.md). The following paragraph
|
|
41
|
+
describes the earlier batch-planning checkpoint.
|
|
42
|
+
|
|
43
|
+
The user-provided runner handoff and [issue #5](https://github.com/HECer/yoke/issues/5)
|
|
44
|
+
were read. The infrastructure-classification correction is partial. The original
|
|
45
|
+
safe-mode shell failure has not been reproduced or diagnosed here. In particular,
|
|
46
|
+
a preflight under the actual sandbox identity, streamed tool-error handling when
|
|
47
|
+
a provider exits zero, separate heartbeat/useful-progress reporting, total process
|
|
48
|
+
budgets and argument-safe Windows launch regressions remain open. Optional MCP
|
|
49
|
+
startup warnings alone are not treated as proof of a task failure. No affected
|
|
50
|
+
DeviceLane worktree was restarted or cleaned, and the issue was not closed.
|
|
51
|
+
|
|
52
|
+
## Evidence scope
|
|
53
|
+
|
|
54
|
+
Automated tests cover batch call suppression, invalid/partial/duplicate output,
|
|
55
|
+
concurrent PRD edits, contract invalidation, planner/worker separation, prepared
|
|
56
|
+
dispatch, tier limits and infrastructure repair suppression. Injected planner
|
|
57
|
+
responses establish command behavior; they are not authenticated provider parity
|
|
58
|
+
benchmarks. The full suite executed 1,166 tests in 127 files: 1,163 passed, two
|
|
59
|
+
were skipped and one new test fixture lacked a required configuration field.
|
|
60
|
+
After correcting that fixture, all 11 tests in its file passed. Subsequent
|
|
61
|
+
focused routing checks cover the final input-freshness guard. Planning hashes
|
|
62
|
+
establish input freshness, not provenance or correctness.
|
|
63
|
+
|
|
64
|
+
Read-only provenance scan: no C2PA located, supported scan complete, verification,
|
|
65
|
+
signer trust and Markdown metadata privacy unknown. The audit's limits are:
|
|
66
|
+
"No conforming verifier was supplied." and "Keyed model-level watermarks cannot
|
|
67
|
+
be checked without the provider's key." No authorship inference or mark removal.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Routing by task requirements
|
|
2
|
+
|
|
3
|
+
Capability routing is available in Yoke 1.9.0. Batch preparation, separate planning settings and routing limits described below are local, unreleased additions.
|
|
4
|
+
|
|
5
|
+
New setups use `routing.strategy: capability`. Existing explicit strategies and profiles remain unchanged. To opt an existing project into capability routing with its current profiles:
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
yoke setup . --yes --routing --routing-strategy=capability
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Give each existing worker a `tier: light|standard|strong|frontier`. Profiles without a tier remain usable with legacy strategies but are not candidates for capability selection. To explicitly replace worker profiles with the supplied provider presets, add `--routing-preset`. This replaces customized worker profiles; omit it to retain them.
|
|
12
|
+
|
|
13
|
+
## Planning and selection
|
|
14
|
+
|
|
15
|
+
The start provider/model remains the planning default. Optional `planning.agent`, `planning.model` and `planning.reasoningEffort` select a separate planner without changing the execution model. Draft and change-inbox planning request complete assessments in the same pass that creates the tasks. The inbox still performs its separate coverage review.
|
|
16
|
+
|
|
17
|
+
New setups use `routing.assessmentPolicy: prepared` and `routing.fallback: block`. Before dispatch, each unfinished task must have 2–5 executable criteria and a current assessment. No per-task planning call runs in this mode. Existing configurations retain `on-demand` and `parent` unless explicitly changed; on-demand routing makes a read-only planning call for an unassessed task and caches its result.
|
|
18
|
+
|
|
19
|
+
Run `yoke prd assess .` to assess all missing or stale unfinished tasks together. One invocation makes at most one read-only planner call, with exact task IDs and validated output, then atomically replaces the PRD. Invalid, incomplete or duplicate output leaves the PRD unchanged; concurrent input edits are preserved and the result is rejected. A current package uses zero model calls. `--story=ID` selects one unfinished task; `--reassess` also includes already-current assessments. The default package limit is 20 tasks (`planning.maxTasks`, range 1–50) and the prompt limit is 60,000 characters; oversized input is rejected before calling the provider.
|
|
20
|
+
|
|
21
|
+
Yoke writes `assessmentFor` bindings over requirements, declared write scope/provider, transitive dependency contracts and `.yoke/plan.md`. Changes invalidate affected unfinished tasks; progress or priority changes do not. A changed brief invalidates all unfinished tasks. These hashes detect stale input, not authorship or semantic correctness. Source-code changes outside these contracts require explicit reassessment when relevant.
|
|
22
|
+
|
|
23
|
+
Example settings alongside the project's existing worker profiles:
|
|
24
|
+
|
|
25
|
+
```yaml
|
|
26
|
+
runner:
|
|
27
|
+
agent: codex
|
|
28
|
+
model: gpt-5.6-terra
|
|
29
|
+
planning:
|
|
30
|
+
agent: codex
|
|
31
|
+
model: gpt-6-astra
|
|
32
|
+
reasoningEffort: high
|
|
33
|
+
maxTasks: 20
|
|
34
|
+
routing:
|
|
35
|
+
enabled: true
|
|
36
|
+
strategy: capability
|
|
37
|
+
assessmentPolicy: prepared
|
|
38
|
+
fallback: block
|
|
39
|
+
maxTier: strong
|
|
40
|
+
# Keep the existing workers list here.
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
`maxTier` limits automatic execution and escalation, including routing-rule selections. If a task needs frontier while the ceiling is strong, it blocks; Yoke does not lower the required capability. Planning itself may still use Astra. Explicit quality-role model overrides retain precedence. Goal execution retains its own protected manifest and budgets, uses the configured planner on demand, and honors routing fallback/tier limits; PRD preparation policy does not apply to synthetic goal tasks.
|
|
44
|
+
|
|
45
|
+
An assessment is a planning judgment, not a measured success probability. High testability means executable checks can detect an incorrect implementation. High uncertainty, architecture work or high risk require the frontier tier; difficult or broadly coupled work requires strong; routine implementation requires standard. Light is reserved for clear, low-risk mechanical work with strong checks. Weak testability raises the minimum tier. Reviews and critics have a standard minimum even for light tasks.
|
|
46
|
+
|
|
47
|
+
```yaml
|
|
48
|
+
assessment:
|
|
49
|
+
taskClass: implementation
|
|
50
|
+
difficulty: medium
|
|
51
|
+
uncertainty: low
|
|
52
|
+
risk: low
|
|
53
|
+
scope: low
|
|
54
|
+
testability: high
|
|
55
|
+
reason: Existing handler pattern and executable contract tests
|
|
56
|
+
approach: Extend the handler, cover the boundary cases, run contract tests
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Yoke chooses an eligible profile at or above the required tier, then compares declared cost tiers. Optional `roles: [implementation, reviewer, critic, repair]` limits a profile's uses. Task `agent` affinity restricts implementation to that provider. Explicit routing rules and explicit quality role models retain precedence. With legacy `fallback: parent` and no tier ceiling, a missing suitable profile falls back to the start model (or the explicitly bound provider's default) and labels the fallback; it does not prove sufficient capability. `fallback: block` or a configured tier ceiling prevents that fallback. An invalid assessment blocks implementation.
|
|
60
|
+
|
|
61
|
+
## Initial profiles
|
|
62
|
+
|
|
63
|
+
| Tier | Codex | Claude | Gemini |
|
|
64
|
+
| --- | --- | --- | --- |
|
|
65
|
+
| light | gpt-5.6-luna, low | haiku | gemini-2.5-flash |
|
|
66
|
+
| standard | gpt-5.6-terra, medium | sonnet | gemini-2.5-pro |
|
|
67
|
+
| strong | gpt-5.6-sol, high | sonnet, high effort | gemini-2.5-pro |
|
|
68
|
+
| frontier | gpt-6-astra, high | opus | gemini-2.5-pro |
|
|
69
|
+
|
|
70
|
+
These are editable starting hypotheses, not measured equivalences or price claims. The Codex names follow the requested profile family. Account access is not established by finding an installed CLI. Gemini uses documented explicit model IDs and receives no unsupported reasoning-effort parameter. Several Gemini tiers deliberately share Pro; moving between those tiers alone is not a stronger-model transition. Adjust the presets to the models available to your account. Claude aliases can resolve to different concrete models over time. Provider-reported model identity remains separate from requested identity.
|
|
71
|
+
|
|
72
|
+
Provider references: [Claude model configuration](https://code.claude.com/docs/en/model-config), [Gemini model selection](https://geminicli.com/docs/cli/model/).
|
|
73
|
+
|
|
74
|
+
## Repair, escalation and evidence
|
|
75
|
+
|
|
76
|
+
After an independent mechanical failure, capability routing permits one targeted repair at the initial tier, then raises the required tier on further failures. Attempts retain the current worktree and receive the previous gate findings. Every returned candidate still passes the normal acceptance, protection, quality, review and integration gates. Critical decisions, pause/cancellation and protected-acceptance violations stop retries. Provider process failures are classified conservatively as infrastructure; they do not count as evidence that a stronger model is needed.
|
|
77
|
+
|
|
78
|
+
`routing.maxAttempts` limits implementation calls per unchanged task contract (default 5, configurable 1–8). The initial tier imposes an additional bound: light at most 5, standard 4, strong 3, frontier 2. An exhausted task blocks and requires a revised plan. These are inner implementation attempts; the outer loop's iteration count still counts task dispatches. Existing quality repair rounds and time limits remain separate bounds, and quality repairs can raise their profile tier by round. Goal execution keeps its existing global attempt, time and token budgets.
|
|
79
|
+
|
|
80
|
+
Routing observations record task class, required tier, requested and reported models, effort, independent result, duration and available consumption. Selection considers matching project/task-class/tier history from the last 30 days within the bounded registry read. At least ten matching observations are required before an observed success rate below 80% excludes a profile. History is scoped to the concrete reported model to avoid mixing changed aliases. This is a conservative exclusion rule; it does not lower the planner's safety floor or claim calibrated probabilities. Missing usage remains unknown. Financial optimization and cross-provider performance require authenticated benchmarks.
|
|
81
|
+
|
|
82
|
+
The dashboard's Now view shows the last recorded implementation profile, requested model/effort, rationale and next escalation tier. Usage & time retains reported model and role consumption, including assessment calls. Cached planning has no new model-call charge. Routing state is local runtime data and excluded from Yoke story commits.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Dashboard evolution
|
|
2
|
+
|
|
3
|
+
The local dashboard is an actionable workspace for registered Yoke projects. It reads the same saved project, loop, goal, acceptance, and measurement data as the CLI. Project-controlled text is rendered through `textContent`, and the server remains bound to the loopback interface with same-origin authorization for pause requests.
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
Project status is derived from both the saved goal and the latest loop report. A blocked or running loop is not hidden by a completed goal. When goal and loop states differ, the card displays both. Blocked, failed, paused, unavailable, and stale active reports need attention; those projects are ordered before active projects, followed by the remaining projects.
|
|
8
|
+
|
|
9
|
+
Cards also show the reported current task and saved blocker reason when available, so the overview explains why a project needs attention before opening it.
|
|
10
|
+
|
|
11
|
+
An active loop report more than 20 minutes old is labeled **unconfirmed**. This means Yoke has an old active report, not evidence that the process is still live. The overview can be searched by project name, canonical path, or goal objective and filtered to All, Active, or Needs attention. A no-match state explains the result and provides a clear action that resets both search and filter.
|
|
12
|
+
|
|
13
|
+
## Durable navigation
|
|
14
|
+
|
|
15
|
+
The URL hash stores the current screen, project, project tab, period, UTC grouping, and complete custom date range. Supported screens are the overview, workspace comparison, and project detail. Supported project tabs are Now, Usage & time, and Results; periods are 1, 7, 30, 90, or 365 days; groupings are day, week, or month. Custom dates must be real ISO calendar dates in chronological order and cover at most 366 inclusive days.
|
|
16
|
+
|
|
17
|
+
Invalid hash state returns to the overview with the 30-day/day defaults. Browser back and forward, a page reload, and Refresh restore the validated state. Refresh reloads data without resetting the selected view or controls. Starting any navigation aborts earlier fetches and changes a request generation, so an older response cannot replace the current screen.
|
|
18
|
+
|
|
19
|
+
The workspace comparison schedules at most three project analytics requests at once. If navigation changes, in-flight fetches are aborted and no additional obsolete project requests are scheduled. All projects and individual project links remain available in the navigation while viewing the comparison.
|
|
20
|
+
|
|
21
|
+
## Usage comparisons
|
|
22
|
+
|
|
23
|
+
Usage & time compares the selected period with the immediately preceding period of exactly the same duration. A single time boundary is captured before either analytics request is made. The comparison covers recorded input plus output tokens, recorded acceptance events, and reported cost.
|
|
24
|
+
|
|
25
|
+
When the preceding value is zero, the dashboard describes no change or new recorded activity instead of calculating an infinite percentage. A cost percentage is shown only when both periods have fully measured cost. Otherwise the dashboard says the percentage is unavailable. Current and previous missing-usage coverage is displayed from calls with unknown usage and unmeasured attempts. Recorded portions remain recorded portions; missing tokens or charges are not estimated as zero.
|
|
26
|
+
|
|
27
|
+
Charts and their tables include periods that contain recorded events. Empty UTC calendar buckets are omitted and the chart explains that a gap means no recorded activity, not a measured zero. Detailed provider/model/role, model timeline, task, and phase tables remain below the comparison.
|
|
28
|
+
|
|
29
|
+
## Design findings and limitations
|
|
30
|
+
|
|
31
|
+
Goal state and loop state describe different durable facts and need independent presentation. Freshness is also separate from state: a saved `running` value can become unconfirmed without being rewritten. Navigation state belongs in the URL because the dashboard has several independently useful views and time controls. Analytics fan-out needs cancellation as well as a concurrency bound because registered workspaces may contain many projects.
|
|
32
|
+
|
|
33
|
+
The dashboard combines local saved evidence with process-identity checks for supervised providers (see [Windows runner validation](WINDOWS-RUNNER-VALIDATION.md)). Unverified process identity remains unknown. It does not reconstruct activity that predates retained measurements, estimate missing provider usage or prices, or turn requested model names into proof of models used. Parallel call durations can overlap, so summed call time is consumption intensity rather than generation speed. The 20-minute freshness threshold is a presentation rule, not a process-health guarantee. No provider performance benchmark is implied by these views.
|
|
@@ -197,3 +197,31 @@ Die Umsetzung wurde mit Tests und einer lokalen Browserprüfung geprüft; authen
|
|
|
197
197
|
### Releaseauftrag am 2026-09-06
|
|
198
198
|
|
|
199
199
|
Der Nutzer hat anschließend maximal drei Worker im Automatikmodus und die Veröffentlichung der Weiterentwicklung beauftragt. Releaseziel ist 1.8.0; der frühere lokale Zwischenstand mit zwei Workern ist damit überholt. Jede neue Version muss vor Veröffentlichung einen datierten Changelogeintrag erhalten; die verbindliche Regel steht in AGENTS.md. Der tatsächliche Veröffentlichungsstatus wird über GitHub Release und npm geprüft.
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
## Aufgabenbezogene Modellauswahl nach Release 1.8.0
|
|
203
|
+
|
|
204
|
+
Der Nutzer hat die Umsetzung der vorgeschlagenen Fähigkeitsauswahl ausdrücklich beauftragt: Planung mit dem Startmodell, gespeicherte Aufgabenbewertung, Modell-/Effort-Profile für Codex, Claude und Gemini, begrenzte Reparatur/Eskalation sowie nachvollziehbare Dashboardanzeige. Die Implementierung wird lokal nach 1.8.0 entwickelt. Verhalten, Migration und Grenzen stehen in CAPABILITY-ROUTING.md; die veröffentlichten 1.8.0-Defaults dürfen damit nicht verwechselt werden.
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
### Releaseauftrag 1.9.0
|
|
208
|
+
|
|
209
|
+
Der Nutzer hat die Veröffentlichung des Capability-Routing-Ausbaus ausdrücklich beauftragt. Releaseziel ist 1.9.0. Der datierte Changelog und CAPABILITY-ROUTING.md beschreiben Verhalten, Migration und Grenzen; frühere Hinweise auf den lokalen Zwischenstand bleiben historische Sitzungsnotizen.
|
|
210
|
+
|
|
211
|
+
## Dashboard-Eigengebrauch am 2026-09-06 nach Release 1.9.0
|
|
212
|
+
|
|
213
|
+
Der Nutzer hat eine vollständige Weiterentwicklung des Dashboards durch tatsächlichen Einsatz der neuesten Yoke-Version mit Routing beauftragt. Die global installierte Version 1.9.0 wurde verifiziert und im separaten Checkout `G:/NN-Developed/Yoke-dashboard` mit Capability-Routing eingesetzt; die zusammenhängende UI-Aufgabe wurde anhand ihrer gespeicherten Bewertung auf `codex-strong` / `gpt-5.6-sol` mit hoher Denktiefe geroutet. Ein einzelner serieller Auftrag vermeidet konkurrierende Änderungen an derselben Frontend-Navigation; das allgemeine Automatikmaximum bleibt drei Worker.
|
|
214
|
+
|
|
215
|
+
Der lokale Ausbau ergänzt Statusvorrang für blockierte/laufende Loops trotz abgeschlossenem Ziel, Kennzeichnung veralteter Meldungen, Projektsuche und Statusfilter, aktuelle Aufgabe und Blocker auf Projektkarten, wiederherstellbare URL-Ansichten und UTC-Zeiträume, abbrechbare Projektvergleiche mit maximal drei gleichzeitigen Anfragen sowie Vorperiodenvergleiche mit sichtbaren Messlücken. Der Eigengebrauch deckte außerdem Fehler beim Umgang mit Git-Laufzeitdateien auf; Status/Sperren werden ausgeschlossen und Implementierungsdateien auch bei ignorierten Historienordnern sicher gestagt. Bedienung und Grenzen: [DASHBOARD-EVOLUTION.md](DASHBOARD-EVOLUTION.md).
|
|
216
|
+
|
|
217
|
+
Unabhängige Browserprüfungen verwendeten synthetische Projekte gegen den echten lokalen HTTP-Server, einschließlich mobiler Darstellung, History-Navigation und verzögerter Antworten. Dieser Auftrag umfasst lokale Entwicklung; eine neue Paketversion oder Veröffentlichung wurde dabei nicht beauftragt. Kein behaupteter Modellbenchmark, keine berechnete Kostenersparnis und keine Rekonstruktion unbekannter Verbrauchsdaten. Diese Sitzungsnotiz wurde vom KI-Assistenten erstellt.
|
|
218
|
+
|
|
219
|
+
## Batch-Planung nach 1.9.0 am 2026-09-06
|
|
220
|
+
|
|
221
|
+
Der Nutzer hat den nächsten Ausbau beauftragt: vollständig vorbereitete Aufgabenpakete, getrennte Planungs-/Ausführungsmodelle, begrenzter Fallback und gezielte Neubewertung geänderter Verträge. Lokal implementiert sind prd assess, assessmentFor-Bindungen einschließlich Abhängigkeiten/Plan, planning-Einstellungen und Routing-Grenzen. Neue Setups verlangen vorbereitete Assessments und blockieren fehlende Profile; bestehende Konfigurationen bleiben kompatibel. Ein tatsächlicher Yoke-Lauf prüfte die Batch-Befehle mit einem auf Terra gerouteten Worker. Details, Messwerte und Grenzen: [BATCH-PLANNING-VALIDATION.md](BATCH-PLANNING-VALIDATION.md).
|
|
222
|
+
|
|
223
|
+
Der zusätzliche Windows-Runner-Handoff wurde gelesen, Issue #5 geprüft und eine Teilkorrektur der Infrastrukturklassifikation samt begrenztem Abbruch ergänzt. Sandbox-Preflight und weitergehende Prozessaufsicht bleiben ausdrücklich offen; keine Reproduktion oder Behebung der ursprünglichen Windows-Ursache wird behauptet. Keine neue Version veröffentlicht. Diese Notiz wurde vom KI-Assistenten erstellt.
|
|
224
|
+
|
|
225
|
+
## Vollständiger Runner-Folgeauftrag am 2026-09-06
|
|
226
|
+
|
|
227
|
+
Auf ausdrücklichen Nutzerauftrag wurde Issue #5 weiterbearbeitet. Der Store-PowerShell-Fehler wurde modellfrei mit dem exakten Fehlercode reproduziert; native PowerShell bestand denselben Sandbox-Test. Der lokale Ausbau prüft den Shell-Start vor dem Modell, entfernt ungeeignete Store-Aliase nur aus der Provider-Umgebung, nutzt argv-sichere Windows-Starts und überwacht Laufzeit, Ausgabe, erfolgreichen Tool-Fortschritt und Prozessidentität getrennt. Ein echter Yoke-Lauf mit Luna, Routing und sicherer Isolation endete nach 3m8s erfolgreich einschließlich Abnahmetests und Commit-Integration. Protokoll, Grenzen und Konfiguration: [WINDOWS-RUNNER-VALIDATION.md](WINDOWS-RUNNER-VALIDATION.md). Die alte DeviceLane-Instanz wurde nicht verändert; Veröffentlichung oder rückwirkende Instrumentierung wird nicht behauptet. Diese Notiz wurde vom KI-Assistenten erstellt.
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
# Windows runner correction — 2026-09-06
|
|
2
|
+
|
|
3
|
+
AI-assisted implementation and validation record for issue #5. These changes are
|
|
4
|
+
included in 1.10.0; installed 1.9.0 packages and existing processes must be updated/restarted to use them.
|
|
5
|
+
|
|
6
|
+
## Reproduction and correction
|
|
7
|
+
|
|
8
|
+
On this machine, Codex 0.153.4 running the Microsoft Store PowerShell executable
|
|
9
|
+
under `codex sandbox -P :workspace` reproduced `CreateProcessAsUserW failed:
|
|
10
|
+
-1073283067`. The System32 Windows PowerShell executable succeeded under the
|
|
11
|
+
same sandbox profile. A first end-to-end probe exposed a second entry point:
|
|
12
|
+
the WindowsApps `pwsh.exe` app-execution alias failed with access denied.
|
|
13
|
+
|
|
14
|
+
Yoke now filters Store PowerShell directories and their aliases from the provider's
|
|
15
|
+
own PATH, prefers an available native PowerShell executable and tests it in the
|
|
16
|
+
actual isolated working directory before starting the model. The same PATH is
|
|
17
|
+
passed to Codex and its shell environment. This changes neither machine/user PATH
|
|
18
|
+
nor sandbox privileges. Native PowerShell 7 is preferred; Windows PowerShell is
|
|
19
|
+
the fallback when no native `pwsh.exe` is available. Projects requiring PowerShell
|
|
20
|
+
7 should install a native distribution and make it available on PATH.
|
|
21
|
+
|
|
22
|
+
The preflight uses the installed CLI's built-in `:workspace` or `:read-only`
|
|
23
|
+
permission profile. An unsupported CLI/profile or failing shell blocks execution
|
|
24
|
+
with an actionable message. No permission escalation or unsandboxed retry occurs.
|
|
25
|
+
The preflight itself has a 30-second execution deadline, followed by bounded
|
|
26
|
+
process-tree cleanup; an outer 90-second guard bounds an unresponsive supervisor.
|
|
27
|
+
Environment values, prompts and raw authentication diagnostics are not stored in
|
|
28
|
+
the supervision record.
|
|
29
|
+
|
|
30
|
+
Windows npm launchers are resolved to their JavaScript entry point and invoked
|
|
31
|
+
with a native Node argv array; executable providers are launched directly. Unknown
|
|
32
|
+
batch launcher formats are rejected. This removes the provider/watchdog use of
|
|
33
|
+
`shell: true` and preserves spaces, quotes, percent signs and shell metacharacters
|
|
34
|
+
as literal arguments.
|
|
35
|
+
|
|
36
|
+
## Supervision and failure behavior
|
|
37
|
+
|
|
38
|
+
Both serial and parallel implementation providers have independent output,
|
|
39
|
+
successful-tool/edit progress and overall timers. Defaults:
|
|
40
|
+
|
|
41
|
+
```yaml
|
|
42
|
+
loop:
|
|
43
|
+
timeoutMinutes: 20 # output inactivity; existing setting
|
|
44
|
+
progressTimeoutMinutes: 20 # no observed successful tool result or edit
|
|
45
|
+
maxCallMinutes: 30 # overall provider call
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
The two new settings accept positive values up to 1,440 minutes. Disabling the
|
|
49
|
+
old output timeout does not disable the other bounds. Successful tool/edit events
|
|
50
|
+
are evidence of activity, not proof that the product requirement has been met.
|
|
51
|
+
|
|
52
|
+
Codex structured failed-tool events and its actual `exec_command failed:
|
|
53
|
+
CreateProcess` stderr diagnostic stop the worker immediately. Terminal provider
|
|
54
|
+
authentication errors are separate from optional MCP startup warnings; the latter
|
|
55
|
+
alone do not fail the task. Failed infrastructure cannot produce a candidate or
|
|
56
|
+
story success merely because existing tests happen to pass, and does not train
|
|
57
|
+
capability escalation. Explicit `bare` startup settings now survive capability
|
|
58
|
+
profile selection.
|
|
59
|
+
|
|
60
|
+
Process records under `.yoke/supervision/` expose PID, process identity, supervisor
|
|
61
|
+
heartbeat, current attempt, last output, last successful tool/edit and terminal reason. Status CLI
|
|
62
|
+
and dashboard show these separately. Backlog ratios are labeled as backlog, not
|
|
63
|
+
overall product completion. Live identity checks distinguish an existing PID from
|
|
64
|
+
the recorded process. Tree termination checks that identity before reaping;
|
|
65
|
+
unconfirmed termination retains ownership/evidence and blocks a subsequent worker.
|
|
66
|
+
No retry/restart is triggered merely by an observer timing out. Failed worktrees
|
|
67
|
+
remain available for inspection.
|
|
68
|
+
|
|
69
|
+
## Validation
|
|
70
|
+
|
|
71
|
+
- Model-free reproduction: Store executable failed with the reported code;
|
|
72
|
+
native System32 PowerShell passed in the same permission profile.
|
|
73
|
+
- Both safe and read-only preflights passed with a 30,707 UTF-16-character inherited
|
|
74
|
+
environment in a path containing spaces. This simulates a large environment;
|
|
75
|
+
it does not reconstruct every variable in the original Visual Studio session.
|
|
76
|
+
- Actual Yoke isolated safe-mode run: `codex-light`, requested `gpt-5.6-luna` at low
|
|
77
|
+
effort; task started 18:49:44 UTC and completed 18:52:52 UTC. Real shell commands,
|
|
78
|
+
file creation, two independent acceptance commands, verification and commit
|
|
79
|
+
integration succeeded. One model call, no assessment call. Recorded input
|
|
80
|
+
345,821 tokens (321,024 cached subset), output 2,353; monetary cost and actual
|
|
81
|
+
reported model identity remain unknown.
|
|
82
|
+
- Regression tests cover literal argv, packaged aliases, streaming failure,
|
|
83
|
+
optional MCP diagnostics, total timeout despite heartbeat output, retained
|
|
84
|
+
redacted proof, no new worker after unconfirmed termination, and an actual
|
|
85
|
+
child failure propagated through capability routing and the worker gate.
|
|
86
|
+
- Full suite: 1,170 passed, two skipped across 128 files at the recorded full-run
|
|
87
|
+
checkpoint. Final focused validation: 119 passed across five files, plus a
|
|
88
|
+
successful build and documentation consistency check.
|
|
89
|
+
|
|
90
|
+
The original DeviceLane process was observed alive. It was not restarted or
|
|
91
|
+
terminated, its retained worktree was not modified, and its application task was
|
|
92
|
+
not claimed complete. Issue #5 was not closed or externally commented on. The fix
|
|
93
|
+
must be installed before it can supervise a newly started DeviceLane run; it cannot
|
|
94
|
+
retrofit supervision into an already running old process.
|
|
95
|
+
|
|
96
|
+
Primary implementation reference consulted:
|
|
97
|
+
[OpenAI shell detection](https://github.com/openai/codex/blob/main/codex-rs/shell-command/src/shell_detect.rs).
|
|
98
|
+
The local executable probes above, rather than assumptions about upstream release
|
|
99
|
+
contents, establish the behavior observed here.
|
|
100
|
+
|
|
101
|
+
Read-only provenance audit: supported scan complete, no C2PA located; verification,
|
|
102
|
+
signer trust and Markdown metadata privacy unknown. The audit states: "No conforming
|
|
103
|
+
verifier was supplied." and "Keyed model-level watermarks cannot be checked without
|
|
104
|
+
the provider's key." No authorship inference or watermark removal was performed.
|
|
Binary file
|
package/gemini-extension.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "yoke",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.10.0",
|
|
4
4
|
"description": "Cross-agent coding harness: curated skill canon, mechanical safety gates, autonomous loop with proof artifacts. CLI: npm i -g @hecer/yoke",
|
|
5
5
|
"contextFileName": "GEMINI-EXTENSION.md"
|
|
6
6
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hecer/yoke",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.10.0",
|
|
4
4
|
"description": "One harness, three agents, zero trust in \"done\" — cross-agent coding harness for Claude Code, Codex CLI, and Gemini CLI: one skill canon, mechanical safety gates, an autonomous loop with screenshot/video proofs.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|