@hecer/yoke 1.21.1 → 1.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/.codex-plugin/plugin.json +1 -1
- package/CHANGELOG.md +27 -0
- package/README.md +6 -1
- package/bench/analyze-codex-comparison.mjs +90 -17
- package/bench/compare-codex.mjs +159 -36
- package/bench/result-schema.mjs +132 -0
- package/canon/manifest.yaml +1 -1
- package/dist/agents/pi-telemetry.js +2 -1
- package/dist/agents/process-streams.js +12 -64
- package/dist/agents/provider-selection.js +12 -0
- package/dist/agents/telemetry.js +52 -52
- package/dist/change/inbox.js +8 -3
- package/dist/check/command.js +69 -17
- package/dist/check/delivery.js +121 -0
- package/dist/cli.js +12 -3
- package/dist/code-intelligence/budgets.js +138 -0
- package/dist/code-intelligence/contracts.js +2 -0
- package/dist/code-intelligence/coordinator.js +156 -84
- package/dist/code-intelligence/evidence.js +87 -34
- package/dist/code-intelligence/mcp-client.js +10 -2
- package/dist/code-intelligence/mcp-server.js +10 -10
- package/dist/dashboard/analytics.js +5 -3
- package/dist/goals/command.js +183 -53
- package/dist/goals/usage.js +87 -0
- package/dist/loop/candidate-cleanup.js +47 -17
- package/dist/loop/candidates.js +17 -11
- package/dist/loop/dispatcher.js +89 -26
- package/dist/loop/failure.js +104 -0
- package/dist/loop/gate-snapshot.js +19 -0
- package/dist/loop/git.js +1 -1
- package/dist/loop/loop.js +100 -66
- package/dist/loop/parallel-adapters.js +22 -4
- package/dist/loop/parallel-command.js +32 -5
- package/dist/loop/recovery.js +23 -5
- package/dist/loop/reporter.js +21 -4
- package/dist/loop/run-command.js +95 -46
- package/dist/loop/runner.js +5 -4
- package/dist/loop/worker.js +145 -91
- package/dist/observability/history.js +1 -1
- package/dist/observability/invocation.js +42 -0
- package/dist/observability/usage.js +17 -0
- package/dist/prd/command.js +20 -7
- package/dist/prd/decompose.js +5 -2
- package/dist/retrofit/config.js +26 -1
- package/dist/retrofit/gitignore.js +2 -0
- package/dist/routing/attempts.js +241 -0
- package/dist/routing/capability.js +13 -9
- package/dist/routing/optimization.js +73 -0
- package/dist/routing/registry.js +7 -1
- package/dist/routing/router.js +280 -127
- package/dist/setup/command.js +8 -2
- package/dist/smoke/command.js +302 -74
- package/docs/BENCHMARK-MANIFEST.md +131 -0
- package/docs/CODE-INTELLIGENCE.md +43 -1
- package/docs/CODEX-COMPARISON-2026-09-29.md +15 -0
- package/docs/DELIVERY-JOURNEYS.md +199 -0
- package/docs/ECONOMIC-ROUTING.md +180 -0
- package/docs/GOALS.md +61 -4
- package/docs/RELEASE-VALIDATION-1.22.0.md +115 -0
- package/docs/parallel-execution.md +37 -9
- package/gemini-extension.json +1 -1
- package/package.json +1 -1
package/dist/loop/reporter.js
CHANGED
|
@@ -4,6 +4,7 @@ import { join } from 'node:path';
|
|
|
4
4
|
import { randomUUID } from 'node:crypto';
|
|
5
5
|
import { appendEvent } from '../observability/events.js';
|
|
6
6
|
import { estimateDurations, validDuration } from '../estimation/durations.js';
|
|
7
|
+
import { recordActiveRoutingUsage, markActiveRoutingUsageIncomplete } from '../routing/attempts.js';
|
|
7
8
|
export const LOG_CAP_BYTES = 256 * 1024;
|
|
8
9
|
// Append a line to .yoke/loop.log, keeping the file bounded: once it exceeds
|
|
9
10
|
// capBytes, truncate to the recent tail (starting at a line boundary) so the log
|
|
@@ -225,15 +226,15 @@ export function makeReporter(dir, opts = {}, now = () => new Date()) {
|
|
|
225
226
|
const base = phase === 'exploring' || phase === 'waiting-exploration' || phase === 'waiting-recovery'
|
|
226
227
|
? (({ story: _story, storyTitle: _storyTitle, ...withoutStory }) => withoutStory)(status)
|
|
227
228
|
: status;
|
|
228
|
-
persist({ ...base, state: 'running', phase, ...(progress ? { progress } : {}), ...(reason ? { reason } : { reason: undefined }), updatedAt: now().toISOString() }, phase, ` · ${phase}…`);
|
|
229
|
+
persist({ ...base, state: 'running', phase, failure: undefined, ...(progress ? { progress } : {}), ...(reason ? { reason } : { reason: undefined }), updatedAt: now().toISOString() }, phase, ` · ${phase}…`);
|
|
229
230
|
},
|
|
230
|
-
blocked(reason) {
|
|
231
|
+
blocked(reason, failure) {
|
|
231
232
|
const base = current ?? emptyStatus(now().toISOString());
|
|
232
|
-
persist({ ...withoutParallel(base), state: 'blocked', reason, updatedAt: now().toISOString() }, 'blocked', `■ blocked on ${base.story ?? '?'}: ${reason}`);
|
|
233
|
+
persist({ ...withoutParallel(base), state: 'blocked', reason, failure, updatedAt: now().toISOString() }, 'blocked', `■ blocked on ${base.story ?? '?'}: ${reason}`);
|
|
233
234
|
},
|
|
234
235
|
complete(progress) {
|
|
235
236
|
persist({ ...withoutParallel(current ?? emptyStatus(now().toISOString())), state: 'complete', phase: undefined,
|
|
236
|
-
progress, reason: undefined, updatedAt: now().toISOString() }, 'complete', `✔ loop complete — ${progress.passed}/${progress.total}`);
|
|
237
|
+
progress, reason: undefined, failure: undefined, updatedAt: now().toISOString() }, 'complete', `✔ loop complete — ${progress.passed}/${progress.total}`);
|
|
237
238
|
},
|
|
238
239
|
capReached(progress) {
|
|
239
240
|
persist({ ...withoutParallel(current ?? emptyStatus(now().toISOString())), state: 'cap-reached', phase: undefined,
|
|
@@ -345,6 +346,22 @@ export function makeReporter(dir, opts = {}, now = () => new Date()) {
|
|
|
345
346
|
if (value !== undefined && (!Number.isFinite(value) || value < 0))
|
|
346
347
|
delete usage[key];
|
|
347
348
|
}
|
|
349
|
+
if (!usage.calls?.length && !usage.callId)
|
|
350
|
+
usage.callId = randomUUID();
|
|
351
|
+
const costRole = usage.role && ['reviewer', 'critic', 'repair', 'quality-critic', 'quality-repair', 'candidate-selection'].includes(usage.role);
|
|
352
|
+
if (usage.storyId && (usage.routingAttemptId || (costRole && !usage.calls?.length))) {
|
|
353
|
+
try {
|
|
354
|
+
recordActiveRoutingUsage(dir, usage.storyId, usage);
|
|
355
|
+
}
|
|
356
|
+
catch {
|
|
357
|
+
// Accounting failures must disqualify economic evidence without losing
|
|
358
|
+
// the independent event below or interrupting a useful implementation.
|
|
359
|
+
try {
|
|
360
|
+
markActiveRoutingUsageIncomplete(dir, usage.storyId);
|
|
361
|
+
}
|
|
362
|
+
catch { /* Router finalization also fails closed on corrupt accounting. */ }
|
|
363
|
+
}
|
|
364
|
+
}
|
|
348
365
|
const calls = usage.calls?.length ? usage.calls : [{ usageAvailable: usage.measurementComplete !== false, totalCostUsd: usage.totalCostUsd }];
|
|
349
366
|
measuredCalls += calls.filter(call => call.usageAvailable !== false).length;
|
|
350
367
|
unknownCalls += calls.filter(call => call.usageAvailable === false).length;
|
package/dist/loop/run-command.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { roleSelection } from "../routing/capability.js";
|
|
2
|
+
import { createHash } from 'node:crypto';
|
|
2
3
|
import { join } from 'node:path';
|
|
3
4
|
import { existsSync, unlinkSync } from 'node:fs';
|
|
4
5
|
import { loadConfig, saveConfig, defaultConfig, resolveOutputPolicy, resolveVerifyCommand } from '../retrofit/config.js';
|
|
@@ -31,6 +32,15 @@ import { MAX_PROJECT_WORKERS, sharedPoolStatus, withSharedWorkerSync } from './r
|
|
|
31
32
|
import { runPrdExplore } from '../prd/explore.js';
|
|
32
33
|
export const DEFAULT_IDLE_MINUTES = 20;
|
|
33
34
|
const STALE_MINUTES = 20; // a running status older than this likely means the loop died
|
|
35
|
+
const nativeReporters = new WeakSet();
|
|
36
|
+
function executionPolicyFingerprint(policy) {
|
|
37
|
+
const canonical = JSON.stringify(policy, (_key, value) => {
|
|
38
|
+
if (!value || typeof value !== 'object' || Array.isArray(value))
|
|
39
|
+
return value;
|
|
40
|
+
return Object.fromEntries(Object.keys(value).sort().map(key => [key, value[key]]));
|
|
41
|
+
});
|
|
42
|
+
return createHash('sha256').update(canonical).digest('hex');
|
|
43
|
+
}
|
|
34
44
|
export function relativeTime(fromIso, now) {
|
|
35
45
|
const ms = Math.max(0, now.getTime() - Date.parse(fromIso));
|
|
36
46
|
const s = Math.floor(ms / 1000);
|
|
@@ -241,6 +251,8 @@ async function runContinuousExploration(targetDir, options) {
|
|
|
241
251
|
}
|
|
242
252
|
const limitDeadline = options.savedRun?.exploreDeadline ?? (options.exploreLimitMs === undefined ? undefined : Date.now() + options.exploreLimitMs);
|
|
243
253
|
const reporter = options.reporter ?? makeReporter(targetDir, { json: options.json, quiet: true });
|
|
254
|
+
if (!options.reporter)
|
|
255
|
+
nativeReporters.add(reporter);
|
|
244
256
|
let config;
|
|
245
257
|
try {
|
|
246
258
|
config = loadConfig(targetDir);
|
|
@@ -261,6 +273,7 @@ async function runContinuousExploration(targetDir, options) {
|
|
|
261
273
|
let reviewerAgent = options.reviewer ?? (reviewRequested ? SUPPORTED_AGENTS.find(agent => agent !== defaultImplementationAgent && available(agent)) : undefined);
|
|
262
274
|
let explorationAgent = defaultExplorer;
|
|
263
275
|
let retryWorktree = false;
|
|
276
|
+
let retryFeedback;
|
|
264
277
|
const safeBatchSize = () => {
|
|
265
278
|
if (retryWorktree)
|
|
266
279
|
return 1;
|
|
@@ -365,6 +378,7 @@ async function runContinuousExploration(targetDir, options) {
|
|
|
365
378
|
resultCode = await Promise.resolve(runLoopCommand(targetDir, {
|
|
366
379
|
...innerOptions,
|
|
367
380
|
agent: implementationAgent,
|
|
381
|
+
recoveryFeedback: retryFeedback,
|
|
368
382
|
...(reviewRequested && reviewerAgent ? { reviewer: reviewerAgent } : {}),
|
|
369
383
|
...(batchLimit !== undefined ? { maxIterations: batchLimit } : {}),
|
|
370
384
|
...(retryWorktree ? { resumeWorktree: true, parallel: 1, candidates: 1 } : {}),
|
|
@@ -455,7 +469,11 @@ async function runContinuousExploration(targetDir, options) {
|
|
|
455
469
|
}
|
|
456
470
|
continue;
|
|
457
471
|
}
|
|
458
|
-
if (afterStatus?.
|
|
472
|
+
if (afterStatus?.failure?.kind === 'no-progress') {
|
|
473
|
+
reporter.blocked(afterStatus.reason ?? 'Automatic continuation stopped after unchanged failures', afterStatus.failure);
|
|
474
|
+
return 1;
|
|
475
|
+
}
|
|
476
|
+
if (afterStatus?.failure?.kind === 'completion-failed' && allCurrentStoriesPass(targetDir)) {
|
|
459
477
|
retryWorktree = false;
|
|
460
478
|
reporter.phase('exploring', `all planned tasks pass, but ${afterStatus.reason}; looking for work that can resolve the completion gate`, currentProgress(targetDir));
|
|
461
479
|
const scan = await scanForWork(`The integrated completion gate is still failing: ${afterStatus.reason}`);
|
|
@@ -478,6 +496,7 @@ async function runContinuousExploration(targetDir, options) {
|
|
|
478
496
|
continue;
|
|
479
497
|
}
|
|
480
498
|
retryCount++;
|
|
499
|
+
retryFeedback = afterStatus?.reason;
|
|
481
500
|
advanceRecoveryProviders();
|
|
482
501
|
retryWorktree = (afterStatus?.parallel?.reopened ?? 0) === 0;
|
|
483
502
|
const wait = retryDelay(retryCount);
|
|
@@ -579,9 +598,10 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
579
598
|
}
|
|
580
599
|
}
|
|
581
600
|
const outputPolicy = resolveOutputPolicy(config);
|
|
601
|
+
const configuredVerifyCommand = opts.verify ? undefined : resolveVerifyCommand(targetDir, config);
|
|
582
602
|
let verify = opts.verify;
|
|
583
603
|
if (!verify) {
|
|
584
|
-
const command =
|
|
604
|
+
const command = configuredVerifyCommand;
|
|
585
605
|
if (!command) {
|
|
586
606
|
console.error('No verify command configured. Set verify.command in .yoke/config.yaml (e.g. "npm test") so the loop can confirm tests pass before marking work done.');
|
|
587
607
|
return 2;
|
|
@@ -752,6 +772,73 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
752
772
|
selection: runnerSelection,
|
|
753
773
|
commit: (_path, request) => commitPaths(targetDir, ['.yoke/prd.yaml'], `yoke: plan change ${request.id}`, commitIdentity),
|
|
754
774
|
}));
|
|
775
|
+
const reviewerProviders = !opts.reviewRunner && (opts.review || opts.reviewer) ? SUPPORTED_AGENTS.filter(available) : [];
|
|
776
|
+
let review = opts.reviewRunner;
|
|
777
|
+
let reviewProvider = 'unknown';
|
|
778
|
+
if (!review && (opts.review || opts.reviewer)) {
|
|
779
|
+
const reviewerAgent = opts.reviewer ?? reviewerProviders.find(agent => agent !== runnerAgent);
|
|
780
|
+
if (!reviewerAgent) {
|
|
781
|
+
if (!opts.allowSelfReview) {
|
|
782
|
+
console.error('No independent reviewer CLI is available. Install or select a second agent, or pass --allow-self-review explicitly.');
|
|
783
|
+
return 2;
|
|
784
|
+
}
|
|
785
|
+
}
|
|
786
|
+
const resolvedReviewer = reviewerAgent ?? runnerAgent;
|
|
787
|
+
reviewProvider = resolvedReviewer;
|
|
788
|
+
if (resolvedReviewer === runnerAgent && !opts.allowSelfReview) {
|
|
789
|
+
console.error(`Reviewer "${resolvedReviewer}" is also the implementer. Pick another agent or pass --allow-self-review explicitly.`);
|
|
790
|
+
return 2;
|
|
791
|
+
}
|
|
792
|
+
if (!available(resolvedReviewer)) {
|
|
793
|
+
console.error(`Reviewer agent CLI "${resolvedReviewer}" was not found on PATH. Install it, or pick another with --reviewer=<${AGENT_LIST}>.`);
|
|
794
|
+
return 2;
|
|
795
|
+
}
|
|
796
|
+
review = context => {
|
|
797
|
+
const implementer = readStatus(targetDir)?.routingDecisions?.[context.story.id]?.provider ?? runnerAgent;
|
|
798
|
+
const selectedReviewer = !opts.reviewer && resolvedReviewer === implementer
|
|
799
|
+
? reviewerProviders.find(agent => agent !== implementer) ?? resolvedReviewer : resolvedReviewer;
|
|
800
|
+
if (selectedReviewer === implementer && !opts.allowSelfReview)
|
|
801
|
+
return { success: false, summary: "Independent review requires a provider distinct from the routed implementer", reviewOutcome: { kind: "infrastructure", summary: "Routed implementation and reviewer share a provider" } };
|
|
802
|
+
reviewProvider = selectedReviewer;
|
|
803
|
+
return makeReviewRunner(selectedReviewer, idleMs, undefined, routingEnabled ? roleSelection(targetDir, config, context.story, selectedReviewer, "reviewer") : undefined)(context);
|
|
804
|
+
};
|
|
805
|
+
}
|
|
806
|
+
if (review) {
|
|
807
|
+
const reviewRunner = review;
|
|
808
|
+
review = context => {
|
|
809
|
+
const started = Date.now();
|
|
810
|
+
let result;
|
|
811
|
+
try {
|
|
812
|
+
result = reviewRunner(context);
|
|
813
|
+
return result;
|
|
814
|
+
}
|
|
815
|
+
finally {
|
|
816
|
+
executionReporter?.addTokens({ inputTokens: 0, outputTokens: 0, measurementComplete: result?.tokens !== undefined, ...result?.tokens, provider: reviewProvider, role: 'reviewer', storyId: context.story.id, durationMs: Date.now() - started });
|
|
817
|
+
}
|
|
818
|
+
};
|
|
819
|
+
}
|
|
820
|
+
// Only native, attributable roles justify comparisons of complete attempts.
|
|
821
|
+
// Concurrent candidates have ambiguous story-wide role attribution and remain
|
|
822
|
+
// conservative until their costs can be joined to an individual alternative.
|
|
823
|
+
const completeAccounting = candidates === 1 && !opts.runner && !opts.verify && !opts.design && !opts.perf && !opts.audit
|
|
824
|
+
&& !opts.reviewRunner && !opts.qualityRuntime && !opts.git && !opts.intake
|
|
825
|
+
&& (!opts.reporter || nativeReporters.has(opts.reporter));
|
|
826
|
+
const accountingScope = completeAccounting ? 'execution-attempt' : undefined;
|
|
827
|
+
const executionPolicyKey = completeAccounting ? executionPolicyFingerprint({
|
|
828
|
+
version: 1,
|
|
829
|
+
mode: useParallelDispatcher ? candidates > 1 ? 'candidates' : 'parallel' : 'serial',
|
|
830
|
+
parallel, candidates, isolate: useParallelDispatcher || isolate,
|
|
831
|
+
verify: { command: configuredVerifyCommand, retries: config.verify?.retries ?? 1, requireCriteria: config.verify?.requireCriteria ?? false },
|
|
832
|
+
design: design ? { max: config.design?.max } : false,
|
|
833
|
+
perf: perf ? { command: config.perf?.command, retries: config.perf?.retries ?? 1 } : false,
|
|
834
|
+
audit: audit ? config.audit : false,
|
|
835
|
+
completion: completion ? { command: config.completion?.command, retries: config.completion?.retries ?? 1 } : false,
|
|
836
|
+
quality: quality ? { defaults: config.quality, overrides: qualityOverrides, limits: quality.repairLimits, runnerAgent, runner: config.runner, agents: config.agents } : false,
|
|
837
|
+
review: review ? { provider: reviewProvider, explicitProvider: opts.reviewer, eligibleProviders: reviewerProviders, allowSelfReview: opts.allowSelfReview ?? false } : false,
|
|
838
|
+
roleRouting: routingEnabled && (quality || review) ? config.routing : false,
|
|
839
|
+
permissions: { implementation: permissions, orchestrator: 'read-only', reviewer: 'read-only', critic: 'read-only', repair: 'safe' },
|
|
840
|
+
idleMs, ambiguityPolicy,
|
|
841
|
+
}) : undefined;
|
|
755
842
|
let runner = opts.runner;
|
|
756
843
|
if (!runner) {
|
|
757
844
|
const requiredProviders = useParallelDispatcher
|
|
@@ -783,6 +870,9 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
783
870
|
strategy: config.routing.strategy,
|
|
784
871
|
maxCandidates: config.routing.maxCandidates,
|
|
785
872
|
maxAttempts: config.routing.maxAttempts,
|
|
873
|
+
optimization: config.routing.optimization,
|
|
874
|
+
accountingScope,
|
|
875
|
+
executionPolicyKey,
|
|
786
876
|
planner: resolvePlanner(config, runnerAgent, runnerSelection),
|
|
787
877
|
assessmentPolicy: config.routing.assessmentPolicy,
|
|
788
878
|
fallback: config.routing.fallback,
|
|
@@ -805,50 +895,6 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
805
895
|
}
|
|
806
896
|
runner = makeActionRunner(config.actions, runner);
|
|
807
897
|
}
|
|
808
|
-
let review = opts.reviewRunner;
|
|
809
|
-
let reviewProvider = 'unknown';
|
|
810
|
-
if (!review && (opts.review || opts.reviewer)) {
|
|
811
|
-
const reviewerAgent = opts.reviewer ?? SUPPORTED_AGENTS.find(agent => agent !== runnerAgent && available(agent));
|
|
812
|
-
if (!reviewerAgent) {
|
|
813
|
-
if (!opts.allowSelfReview) {
|
|
814
|
-
console.error('No independent reviewer CLI is available. Install or select a second agent, or pass --allow-self-review explicitly.');
|
|
815
|
-
return 2;
|
|
816
|
-
}
|
|
817
|
-
}
|
|
818
|
-
const resolvedReviewer = reviewerAgent ?? runnerAgent;
|
|
819
|
-
reviewProvider = resolvedReviewer;
|
|
820
|
-
if (resolvedReviewer === runnerAgent && !opts.allowSelfReview) {
|
|
821
|
-
console.error(`Reviewer "${resolvedReviewer}" is also the implementer. Pick another agent or pass --allow-self-review explicitly.`);
|
|
822
|
-
return 2;
|
|
823
|
-
}
|
|
824
|
-
if (!available(resolvedReviewer)) {
|
|
825
|
-
console.error(`Reviewer agent CLI "${resolvedReviewer}" was not found on PATH. Install it, or pick another with --reviewer=<${AGENT_LIST}>.`);
|
|
826
|
-
return 2;
|
|
827
|
-
}
|
|
828
|
-
review = context => {
|
|
829
|
-
const implementer = readStatus(targetDir)?.routingDecisions?.[context.story.id]?.provider ?? runnerAgent;
|
|
830
|
-
const selectedReviewer = !opts.reviewer && resolvedReviewer === implementer
|
|
831
|
-
? SUPPORTED_AGENTS.find(agent => agent !== implementer && available(agent)) ?? resolvedReviewer : resolvedReviewer;
|
|
832
|
-
if (selectedReviewer === implementer && !opts.allowSelfReview)
|
|
833
|
-
return { success: false, summary: "Independent review requires a provider distinct from the routed implementer", reviewOutcome: { kind: "infrastructure", summary: "Routed implementation and reviewer share a provider" } };
|
|
834
|
-
reviewProvider = selectedReviewer;
|
|
835
|
-
return makeReviewRunner(selectedReviewer, idleMs, undefined, routingEnabled ? roleSelection(targetDir, config, context.story, selectedReviewer, "reviewer") : undefined)(context);
|
|
836
|
-
};
|
|
837
|
-
}
|
|
838
|
-
if (review) {
|
|
839
|
-
const reviewRunner = review;
|
|
840
|
-
review = context => {
|
|
841
|
-
const started = Date.now();
|
|
842
|
-
let result;
|
|
843
|
-
try {
|
|
844
|
-
result = reviewRunner(context);
|
|
845
|
-
return result;
|
|
846
|
-
}
|
|
847
|
-
finally {
|
|
848
|
-
executionReporter?.addTokens({ inputTokens: 0, outputTokens: 0, measurementComplete: result?.tokens !== undefined, ...result?.tokens, provider: reviewProvider, role: 'reviewer', storyId: context.story.id, durationMs: Date.now() - started });
|
|
849
|
-
}
|
|
850
|
-
};
|
|
851
|
-
}
|
|
852
898
|
if (!useParallelDispatcher) {
|
|
853
899
|
if (runner) {
|
|
854
900
|
const unpooled = runner;
|
|
@@ -951,6 +997,8 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
951
997
|
providers: parallelProviders,
|
|
952
998
|
affinityProviders: parallelAffinityProviders,
|
|
953
999
|
routing: routingEnabled ? config.routing : undefined,
|
|
1000
|
+
accountingScope,
|
|
1001
|
+
executionPolicyKey,
|
|
954
1002
|
planning: config.planning,
|
|
955
1003
|
isAvailable: available,
|
|
956
1004
|
onAmbiguity: ambiguityPolicy,
|
|
@@ -976,6 +1024,7 @@ export function runLoopCommand(targetDir, opts) {
|
|
|
976
1024
|
prdPath: path,
|
|
977
1025
|
targetDir,
|
|
978
1026
|
runner,
|
|
1027
|
+
feedback: opts.recoveryFeedback,
|
|
979
1028
|
git,
|
|
980
1029
|
commitIdentity,
|
|
981
1030
|
verify,
|
package/dist/loop/runner.js
CHANGED
|
@@ -9,6 +9,7 @@ import { loadContext, formatForPrompt, contextDir } from '../context/context.js'
|
|
|
9
9
|
import { contextPacket } from '../context/packet.js';
|
|
10
10
|
import { buildProviderInvocation, startProviderProcess } from '../agents/providers.js';
|
|
11
11
|
import { parseProviderResult, parseProviderTelemetry } from '../agents/telemetry.js';
|
|
12
|
+
import { providerTelemetryUsage } from '../observability/usage.js';
|
|
12
13
|
import { formatReviewContract, formatReviewStdoutContract, parseReviewVerdict } from '../review/verdict.js';
|
|
13
14
|
import { prepareWindowsInvocation } from '../agents/windows-launch.js';
|
|
14
15
|
import { readSupervision } from '../agents/supervision.js';
|
|
@@ -243,7 +244,7 @@ function processFailureSummary(error) {
|
|
|
243
244
|
export function runCapturedAgent(agent, inv) {
|
|
244
245
|
try {
|
|
245
246
|
const output = runCliCapture(inv);
|
|
246
|
-
return { success: true, output, summary: 'exited 0', tokens: parseProviderTelemetry(agent, output.split(/\r?\n/))
|
|
247
|
+
return { success: true, output, summary: 'exited 0', tokens: providerTelemetryUsage(parseProviderTelemetry(agent, output.split(/\r?\n/))) };
|
|
247
248
|
}
|
|
248
249
|
catch (error) {
|
|
249
250
|
const partial = error.stdout;
|
|
@@ -252,7 +253,7 @@ export function runCapturedAgent(agent, inv) {
|
|
|
252
253
|
success: false,
|
|
253
254
|
output,
|
|
254
255
|
summary: error.message,
|
|
255
|
-
tokens: output ? parseProviderTelemetry(agent, output.split(/\r?\n/))
|
|
256
|
+
tokens: output ? providerTelemetryUsage(parseProviderTelemetry(agent, output.split(/\r?\n/))) : undefined,
|
|
256
257
|
};
|
|
257
258
|
}
|
|
258
259
|
}
|
|
@@ -316,12 +317,12 @@ export function makeRunner(agent, idleTimeoutMs = 0, opts = {}) {
|
|
|
316
317
|
try {
|
|
317
318
|
const out = capture(inv);
|
|
318
319
|
const telemetry = parseProviderTelemetry(agent, out.split(/\r?\n/));
|
|
319
|
-
return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: attributed(telemetry
|
|
320
|
+
return { success: true, summary: `${agent} implemented ${ctx.story.id}`, tokens: attributed(providerTelemetryUsage(telemetry)) };
|
|
320
321
|
}
|
|
321
322
|
catch (e) {
|
|
322
323
|
// Salvage usage from whatever the agent streamed before dying — those tokens were spent.
|
|
323
324
|
const partial = e.stdout;
|
|
324
|
-
const tokens = partial == null ? undefined : parseProviderTelemetry(agent, String(partial).split(/\r?\n/))
|
|
325
|
+
const tokens = partial == null ? undefined : providerTelemetryUsage(parseProviderTelemetry(agent, String(partial).split(/\r?\n/)));
|
|
325
326
|
const reason = readSupervision(ctx.targetDir, new Date(started).toISOString())[0]?.reason;
|
|
326
327
|
return { success: false, infrastructureFailure: true, summary: `${agent} failed on ${ctx.story.id}: ${reason ?? e.message}`, tokens: attributed(tokens) };
|
|
327
328
|
}
|
package/dist/loop/worker.js
CHANGED
|
@@ -3,6 +3,9 @@ import { existsSync } from "node:fs";
|
|
|
3
3
|
import { join } from "node:path";
|
|
4
4
|
import { acceptanceProtectionProblem } from "../check/command.js";
|
|
5
5
|
import { isAcceptanceCriterion } from './prd.js';
|
|
6
|
+
import { observeFailure } from './failure.js';
|
|
7
|
+
import { consumeAmbiguity } from './loop.js';
|
|
8
|
+
import { gateIdentity, snapshotGates, reuseGates } from './gate-snapshot.js';
|
|
6
9
|
import { runQualityRepairLoop } from '../quality/loop.js';
|
|
7
10
|
function emptyEvidence() { return { criteria: [] }; }
|
|
8
11
|
function errorMessage(error) {
|
|
@@ -40,6 +43,7 @@ function reviewOutcome(result) {
|
|
|
40
43
|
return { kind: 'malformed', summary: result.summary };
|
|
41
44
|
}
|
|
42
45
|
function runMechanicalGates(input, context, evidence) {
|
|
46
|
+
evidence.criteria = [];
|
|
43
47
|
const criteria = context.story.acceptance.filter(isAcceptanceCriterion);
|
|
44
48
|
if (criteria.length === 0) {
|
|
45
49
|
if (input.requireCriterionEvidence)
|
|
@@ -170,8 +174,13 @@ export async function runStoryWorker(input) {
|
|
|
170
174
|
return finalResult(input, { ...baseResult(input, evidence, preflight.summary), kind: 'mechanical-failure', stage: 'quality' });
|
|
171
175
|
}
|
|
172
176
|
let implementation;
|
|
177
|
+
let verified;
|
|
178
|
+
let failure;
|
|
173
179
|
try {
|
|
174
|
-
|
|
180
|
+
const outcome = await runWorkerImplementation(input, context, evidence);
|
|
181
|
+
implementation = outcome.result;
|
|
182
|
+
verified = outcome.gates;
|
|
183
|
+
failure = outcome.failure;
|
|
175
184
|
}
|
|
176
185
|
catch (error) {
|
|
177
186
|
return finalResult(input, {
|
|
@@ -185,110 +194,155 @@ export async function runStoryWorker(input) {
|
|
|
185
194
|
if (implementation.infrastructureFailure)
|
|
186
195
|
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
187
196
|
if (implementation.infrastructureFailure || implementation.routing?.blocked)
|
|
188
|
-
return finalResult(input, { ...baseResult(input, evidence, implementation.summary), kind: "mechanical-failure", stage: "implementation" });
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
...baseResult(input, evidence,
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
197
|
+
return finalResult(input, { ...baseResult(input, evidence, implementation.summary), kind: "mechanical-failure", stage: "implementation", ...(failure ? { failure } : {}) });
|
|
198
|
+
try {
|
|
199
|
+
const afterImplementationCancellation = cancellationReason(input.cancellation);
|
|
200
|
+
if (afterImplementationCancellation) {
|
|
201
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
202
|
+
return finalResult(input, { ...baseResult(input, evidence, afterImplementationCancellation), kind: 'cancelled' });
|
|
203
|
+
}
|
|
204
|
+
const ambiguity = consumeAmbiguity(context.targetDir);
|
|
205
|
+
if (ambiguity) {
|
|
206
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
207
|
+
return finalResult(input, {
|
|
208
|
+
...baseResult(input, evidence, `story ${context.story.id} stopped: ambiguous acceptance criteria — ${ambiguity}`),
|
|
209
|
+
kind: 'paused', reason: 'ambiguity',
|
|
210
|
+
});
|
|
211
|
+
}
|
|
212
|
+
const decision = input.beforeGates?.(context);
|
|
213
|
+
if (decision) {
|
|
214
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
215
|
+
return finalResult(input, { ...baseResult(input, evidence, decision), kind: 'paused', reason: 'decision' });
|
|
216
|
+
}
|
|
217
|
+
if (input.pause?.()) {
|
|
218
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
219
|
+
return finalResult(input, { ...baseResult(input, evidence, 'worker paused before acceptance checks'), kind: 'paused' });
|
|
220
|
+
}
|
|
221
|
+
const gates = reuseGates(context.targetDir, context.story, verified) ?? runMechanicalGates(input, context, evidence);
|
|
222
|
+
if (gates.kind === 'cancelled') {
|
|
223
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
224
|
+
return finalResult(input, { ...baseResult(input, evidence, gates.summary), kind: 'cancelled' });
|
|
225
|
+
}
|
|
226
|
+
if (gates.kind === 'failed') {
|
|
227
|
+
implementation.routing?.recordOutcome(false);
|
|
228
|
+
const observed = input.failureRoot ? observeFailure({ root: input.failureRoot, scope: input.failureScope, directory: context.targetDir, story: context.story, stage: gates.stage, summary: gates.summary }) : undefined;
|
|
229
|
+
return finalResult(input, { ...baseResult(input, evidence, observed?.feedback ?? gates.summary), kind: 'mechanical-failure', stage: gates.stage, ...(observed ? { failure: observed.failure } : {}) });
|
|
230
|
+
}
|
|
231
|
+
const summary = implementation.success
|
|
232
|
+
? implementation.summary
|
|
233
|
+
: `${implementation.summary} (runner exited non-zero but verify is green)`;
|
|
234
|
+
if (!qualityEnabled && !input.review) {
|
|
235
|
+
return finalResult(input, {
|
|
236
|
+
...baseResult(input, evidence, summary),
|
|
237
|
+
kind: 'candidate',
|
|
238
|
+
routing: { outcome: 'pending-integration', ...(implementation.routing ? { recordOutcome: implementation.routing.recordOutcome } : {}) },
|
|
239
|
+
...(implementation.tokens ? { tokens: implementation.tokens } : {}),
|
|
240
|
+
});
|
|
241
|
+
}
|
|
242
|
+
const beforeQualityCancellation = cancellationReason(input.cancellation);
|
|
243
|
+
if (beforeQualityCancellation) {
|
|
244
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
245
|
+
return finalResult(input, { ...baseResult(input, evidence, beforeQualityCancellation), kind: 'cancelled' });
|
|
246
|
+
}
|
|
247
|
+
const qualityStage = qualityEnabled ? input.qualityStage : undefined;
|
|
248
|
+
const review = input.review;
|
|
249
|
+
let qualityFailure;
|
|
250
|
+
const quality = runQualityRepairLoop({
|
|
251
|
+
...(qualityStage
|
|
252
|
+
? { quality: (round) => {
|
|
253
|
+
input.reporter?.phase('comparing');
|
|
254
|
+
return qualityStage(context, round);
|
|
255
|
+
} }
|
|
256
|
+
: {}),
|
|
257
|
+
...(review
|
|
258
|
+
? { review: () => {
|
|
259
|
+
input.reporter?.phase('reviewing');
|
|
260
|
+
return reviewOutcome(review(context));
|
|
261
|
+
} }
|
|
262
|
+
: {}),
|
|
263
|
+
repair: request => {
|
|
264
|
+
if (!input.repair)
|
|
265
|
+
return { kind: 'blocked', summary: 'repair callback is not configured' };
|
|
266
|
+
input.reporter?.phase('repairing');
|
|
267
|
+
const observed = input.failureRoot ? observeFailure({ root: input.failureRoot, scope: input.failureScope, directory: context.targetDir, story: context.story, stage: 'quality', summary: JSON.stringify(request.finding) }) : undefined;
|
|
268
|
+
if (observed?.action === 'blocked') {
|
|
269
|
+
qualityFailure = observed.failure;
|
|
270
|
+
return { kind: 'blocked', summary: observed.feedback };
|
|
271
|
+
}
|
|
272
|
+
const targetedRequest = observed?.action === 'diagnose' ? { ...request, finding: { ...request.finding, message: observed.feedback } } : request;
|
|
273
|
+
const repair = input.repair(context, targetedRequest);
|
|
274
|
+
return repair.success ? { kind: 'repaired' } : { kind: 'blocked', summary: repair.summary };
|
|
275
|
+
},
|
|
276
|
+
rerunGates: () => {
|
|
277
|
+
const rerun = runMechanicalGates(input, context, evidence);
|
|
278
|
+
if (rerun.kind === 'passed')
|
|
279
|
+
return { kind: 'passed' };
|
|
280
|
+
if (rerun.kind === 'cancelled')
|
|
281
|
+
return rerun;
|
|
282
|
+
return { kind: 'failed', stage: rerun.stage, summary: rerun.summary };
|
|
283
|
+
},
|
|
284
|
+
...(input.repairLimits ? { limits: input.repairLimits } : {}),
|
|
285
|
+
pause: input.pause,
|
|
286
|
+
onStatus: status => {
|
|
287
|
+
const metadata = input.qualityMetadata?.(context);
|
|
288
|
+
input.reporter?.quality({ ...status, ...(metadata ?? { policy: 'blocking' }) });
|
|
289
|
+
},
|
|
214
290
|
});
|
|
291
|
+
evidence.quality = quality;
|
|
292
|
+
const afterQualityCancellation = cancellationReason(input.cancellation);
|
|
293
|
+
if (afterQualityCancellation) {
|
|
294
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
295
|
+
return finalResult(input, { ...baseResult(input, evidence, afterQualityCancellation), kind: 'cancelled' });
|
|
296
|
+
}
|
|
297
|
+
const result = { ...resultFromQuality(input, evidence, summary, quality), ...(qualityFailure ? { failure: qualityFailure } : {}) };
|
|
298
|
+
if (result.kind === 'paused' || result.kind === 'cancelled' || result.kind === 'review-failure' || result.kind === 'quality-failure')
|
|
299
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
300
|
+
else if (result.kind !== 'candidate')
|
|
301
|
+
implementation.routing?.recordOutcome(false);
|
|
302
|
+
if (result.kind === 'candidate') {
|
|
303
|
+
return finalResult(input, {
|
|
304
|
+
...result,
|
|
305
|
+
routing: { outcome: 'pending-integration', ...(implementation.routing ? { recordOutcome: implementation.routing.recordOutcome } : {}) },
|
|
306
|
+
...(implementation.tokens ? { tokens: implementation.tokens } : {}),
|
|
307
|
+
});
|
|
308
|
+
}
|
|
309
|
+
return finalResult(input, result);
|
|
215
310
|
}
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
}
|
|
220
|
-
const qualityStage = qualityEnabled ? input.qualityStage : undefined;
|
|
221
|
-
const review = input.review;
|
|
222
|
-
const quality = runQualityRepairLoop({
|
|
223
|
-
...(qualityStage
|
|
224
|
-
? { quality: (round) => {
|
|
225
|
-
input.reporter?.phase('comparing');
|
|
226
|
-
return qualityStage(context, round);
|
|
227
|
-
} }
|
|
228
|
-
: {}),
|
|
229
|
-
...(review
|
|
230
|
-
? { review: () => {
|
|
231
|
-
input.reporter?.phase('reviewing');
|
|
232
|
-
return reviewOutcome(review(context));
|
|
233
|
-
} }
|
|
234
|
-
: {}),
|
|
235
|
-
repair: request => {
|
|
236
|
-
if (!input.repair)
|
|
237
|
-
return { kind: 'blocked', summary: 'repair callback is not configured' };
|
|
238
|
-
input.reporter?.phase('repairing');
|
|
239
|
-
const repair = input.repair(context, request);
|
|
240
|
-
return repair.success ? { kind: 'repaired' } : { kind: 'blocked', summary: repair.summary };
|
|
241
|
-
},
|
|
242
|
-
rerunGates: () => {
|
|
243
|
-
const rerun = runMechanicalGates(input, context, evidence);
|
|
244
|
-
if (rerun.kind === 'passed')
|
|
245
|
-
return { kind: 'passed' };
|
|
246
|
-
if (rerun.kind === 'cancelled')
|
|
247
|
-
return rerun;
|
|
248
|
-
return { kind: 'failed', stage: rerun.stage, summary: rerun.summary };
|
|
249
|
-
},
|
|
250
|
-
...(input.repairLimits ? { limits: input.repairLimits } : {}),
|
|
251
|
-
pause: input.pause,
|
|
252
|
-
onStatus: status => {
|
|
253
|
-
const metadata = input.qualityMetadata?.(context);
|
|
254
|
-
input.reporter?.quality({ ...status, ...(metadata ?? { policy: 'blocking' }) });
|
|
255
|
-
},
|
|
256
|
-
});
|
|
257
|
-
evidence.quality = quality;
|
|
258
|
-
const afterQualityCancellation = cancellationReason(input.cancellation);
|
|
259
|
-
if (afterQualityCancellation) {
|
|
260
|
-
return finalResult(input, { ...baseResult(input, evidence, afterQualityCancellation), kind: 'cancelled' });
|
|
261
|
-
}
|
|
262
|
-
const result = resultFromQuality(input, evidence, summary, quality);
|
|
263
|
-
if (result.kind !== 'candidate' && result.kind !== 'paused')
|
|
264
|
-
implementation.routing?.recordOutcome(false);
|
|
265
|
-
if (result.kind === 'candidate') {
|
|
266
|
-
return finalResult(input, {
|
|
267
|
-
...result,
|
|
268
|
-
routing: { outcome: 'pending-integration', ...(implementation.routing ? { recordOutcome: implementation.routing.recordOutcome } : {}) },
|
|
269
|
-
...(implementation.tokens ? { tokens: implementation.tokens } : {}),
|
|
270
|
-
});
|
|
311
|
+
catch (error) {
|
|
312
|
+
implementation.routing?.recordOutcome(false, 'infrastructure');
|
|
313
|
+
throw error;
|
|
271
314
|
}
|
|
272
|
-
return finalResult(input, result);
|
|
273
315
|
}
|
|
274
316
|
async function runWorkerImplementation(input, context, evidence) {
|
|
275
|
-
let feedback;
|
|
317
|
+
let feedback = input.feedback;
|
|
276
318
|
for (let attempt = 0;; attempt++) {
|
|
277
319
|
const result = await input.runner({ ...context, feedback });
|
|
278
320
|
if (!result.routing?.canRetry || result.routing.blocked || attempt >= 7 || cancellationReason(input.cancellation) || input.pause?.())
|
|
279
|
-
return result;
|
|
321
|
+
return { result };
|
|
280
322
|
if (["decision-request.yaml", "ambiguity.md", "loop.pause"].some(name => existsSync(join(context.targetDir, ".yoke", name))) || acceptanceProtectionProblem(context.targetDir))
|
|
281
|
-
return result;
|
|
282
|
-
const
|
|
283
|
-
|
|
284
|
-
|
|
323
|
+
return { result };
|
|
324
|
+
const before = gateIdentity(context.targetDir, context.story);
|
|
325
|
+
let gates;
|
|
326
|
+
try {
|
|
327
|
+
gates = runMechanicalGates(input, context, evidence);
|
|
328
|
+
}
|
|
329
|
+
catch (error) {
|
|
330
|
+
result.routing.recordOutcome(false, 'infrastructure');
|
|
331
|
+
throw error;
|
|
332
|
+
}
|
|
333
|
+
if (gates.kind !== 'failed') {
|
|
334
|
+
return { result, ...(gates.kind === 'passed' ? { gates: snapshotGates(context.targetDir, context.story, before, gates) } : {}) };
|
|
335
|
+
}
|
|
285
336
|
if (knownInfrastructureFailure(gates.summary)) {
|
|
286
337
|
result.routing.recordOutcome(false, 'infrastructure');
|
|
287
|
-
return { ...result, success: false, infrastructureFailure: true, summary: gates.summary, routing: { ...result.routing, blocked: true, canRetry: false } };
|
|
338
|
+
return { result: { ...result, success: false, infrastructureFailure: true, summary: gates.summary, routing: { ...result.routing, blocked: true, canRetry: false } } };
|
|
288
339
|
}
|
|
289
340
|
result.routing.recordOutcome(false);
|
|
341
|
+
const observed = input.failureRoot ? observeFailure({ root: input.failureRoot, scope: input.failureScope, directory: context.targetDir, story: context.story, stage: gates.stage, summary: gates.summary }) : undefined;
|
|
342
|
+
if (observed?.action === 'blocked')
|
|
343
|
+
return { result: { ...result, success: false, summary: observed.feedback, routing: { ...result.routing, blocked: true, canRetry: false } }, failure: observed.failure };
|
|
290
344
|
if (result.tokens)
|
|
291
345
|
input.reporter?.addTokens(result.tokens);
|
|
292
|
-
feedback = gates.summary;
|
|
346
|
+
feedback = observed?.feedback ?? gates.summary;
|
|
293
347
|
}
|
|
294
348
|
}
|
|
@@ -30,7 +30,7 @@ export function archiveMeasurement(root, event) {
|
|
|
30
30
|
throw error;
|
|
31
31
|
}
|
|
32
32
|
const data = event.data ?? {};
|
|
33
|
-
const allowed = ['inputTokens', 'outputTokens', 'cachedInputTokens', 'cacheWriteInputTokens', 'reasoningOutputTokens', 'totalCostUsd', 'agent', 'provider', 'model', 'actualModel', 'requestedModel', 'variant', 'role', 'calls', 'measurementComplete', 'costMeasurementComplete', 'usageAvailable', 'prediction', 'errorMs', 'withinObservedRange', 'escalated'];
|
|
33
|
+
const allowed = ['inputTokens', 'outputTokens', 'cachedInputTokens', 'cacheWriteInputTokens', 'reasoningOutputTokens', 'totalCostUsd', 'agent', 'provider', 'model', 'actualModel', 'requestedProvider', 'requestedModel', 'requestedReasoningEffort', 'requestedVariant', 'variant', 'role', 'callId', 'parentCallId', 'calls', 'measurementComplete', 'costMeasurementComplete', 'usageAvailable', 'prediction', 'errorMs', 'withinObservedRange', 'escalated'];
|
|
34
34
|
const compact = { ...event, data: Object.fromEntries(allowed.filter(key => data[key] !== undefined).map(key => [key, data[key]])) };
|
|
35
35
|
appendFileSync(file, JSON.stringify(compact) + '\n');
|
|
36
36
|
}
|