newmark-agent 0.6.4 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config.example.json +13 -3
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +53 -2
- package/dist/cli-discovery.js +3 -0
- package/dist/conversation-utility-host.bundle.cjs +2880 -1069
- package/dist/conversation-utility-host.js +94 -7
- package/dist/core/agent.d.ts +126 -12
- package/dist/core/agent.js +783 -172
- package/dist/core/agentKernelRunner.d.ts +7 -1
- package/dist/core/agentKernelRunner.js +156 -20
- package/dist/core/autoRouter.d.ts +49 -44
- package/dist/core/autoRouter.js +117 -302
- package/dist/core/config.js +27 -11
- package/dist/core/continuation/contracts.d.ts +1 -1
- package/dist/core/continuation/store.d.ts +91 -1
- package/dist/core/continuation/store.js +291 -61
- package/dist/core/conversationKernel.d.ts +49 -7
- package/dist/core/conversationKernel.js +303 -51
- package/dist/core/conversationStateDocument.d.ts +18 -0
- package/dist/core/conversationStateDocument.js +88 -0
- package/dist/core/conversationVisualBudget.d.ts +74 -0
- package/dist/core/conversationVisualBudget.js +153 -0
- package/dist/core/electronUtilityAgentClient.d.ts +3 -2
- package/dist/core/electronUtilityAgentClient.js +32 -8
- package/dist/core/electronUtilityRuntimePool.d.ts +28 -3
- package/dist/core/electronUtilityRuntimePool.js +93 -24
- package/dist/core/hostRuntimeHooks.d.ts +23 -0
- package/dist/core/hostRuntimeHooks.js +17 -0
- package/dist/core/jevDecision.d.ts +144 -0
- package/dist/core/jevDecision.js +226 -0
- package/dist/core/memoryProbe.d.ts +24 -0
- package/dist/core/memoryProbe.js +104 -0
- package/dist/core/mobilePairing.d.ts +7 -1
- package/dist/core/mobilePairing.js +9 -1
- package/dist/core/performanceDiagnostics.d.ts +1 -1
- package/dist/core/routeDecisionValidator.d.ts +86 -0
- package/dist/core/routeDecisionValidator.js +249 -0
- package/dist/core/routeEligibility.d.ts +52 -0
- package/dist/core/routeEligibility.js +108 -0
- package/dist/core/runtimeMemoryBudget.d.ts +6 -0
- package/dist/core/runtimeMemoryBudget.js +12 -0
- package/dist/core/utilityAgentProtocol.d.ts +8 -10
- package/dist/core/utilityHostToolRouter.js +2 -0
- package/dist/core/visualDownscale.d.ts +6 -0
- package/dist/core/visualDownscale.js +137 -0
- package/dist/core/wslAgentClient.d.ts +1 -2
- package/dist/core/wslAgentClient.js +0 -8
- package/dist/core/wslAgentProtocol.d.ts +1 -10
- package/dist/core/wslAgentRuntimePool.d.ts +1 -3
- package/dist/core/wslAgentRuntimePool.js +17 -22
- package/dist/llm/provider.d.ts +3 -1
- package/dist/llm/provider.js +53 -9
- package/dist/main.js +107 -30
- package/dist/preload.js +3 -1
- package/dist/server.js +30 -18
- package/dist/tools/computerUse.d.ts +4 -0
- package/dist/tools/computerUse.js +202 -4
- package/dist/tools/computerUsePowerShellHost.d.ts +6 -1
- package/dist/tools/computerUsePowerShellHost.js +105 -31
- package/dist/tools/index.js +4 -1
- package/dist/tui/src/adapters/core-runtime-adapter.js +13 -1
- package/dist/tui/src/i18n.js +12 -1
- package/dist/tui/src/render.js +9 -2
- package/dist/tui/src/settings-schema.js +19 -0
- package/dist/tui/src/state.js +69 -1
- package/dist/ui/index.html +651 -233
- package/dist/ui/lucide-sprite.svg +0 -8
- package/dist/wsl-agent-host.bundle.cjs +2685 -1044
- package/dist/wsl-agent-host.js +0 -3
- package/package.json +36 -9
package/dist/core/autoRouter.js
CHANGED
|
@@ -1,18 +1,40 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.
|
|
3
|
+
exports.RouteExecutionController = exports.ROUTE_POLICY_VERSION = void 0;
|
|
4
4
|
exports.normalizeAutoPreference = normalizeAutoPreference;
|
|
5
5
|
exports.defaultRoutePolicy = defaultRoutePolicy;
|
|
6
6
|
exports.classifyTaskClasses = classifyTaskClasses;
|
|
7
7
|
exports.classifyRouteFailure = classifyRouteFailure;
|
|
8
|
+
exports.createRouteId = createRouteId;
|
|
9
|
+
exports.catalogSnapshotHash = catalogSnapshotHash;
|
|
10
|
+
exports.validationEvidenceIsStale = validationEvidenceIsStale;
|
|
11
|
+
/**
|
|
12
|
+
* Auto Router v2 — shared route contracts, failure classification, operational
|
|
13
|
+
* endpoint health and provider-local recovery planning.
|
|
14
|
+
*
|
|
15
|
+
* This file no longer selects models. Model selection is the exclusive
|
|
16
|
+
* responsibility of the JEV decision engine (`jevDecision.ts`) inside the
|
|
17
|
+
* candidate space produced by the deterministic eligibility guard
|
|
18
|
+
* (`routeEligibility.ts`). What remains here is execution plumbing:
|
|
19
|
+
*
|
|
20
|
+
* - shared route types (`RouteDecision`, `RouteAttempt`, `RoutePolicy`, …)
|
|
21
|
+
* - provider failure classification
|
|
22
|
+
* - endpoint health / circuit breaker (operational protection only)
|
|
23
|
+
* - build-scoped transaction pinning (execution consistency inside one Build)
|
|
24
|
+
* - provider-local recovery planning (retry / equivalent / fallback / JEV alternate)
|
|
25
|
+
*
|
|
26
|
+
* Removed on purpose: the feedback map, EWMA preference learning,
|
|
27
|
+
* `quality_by_task` scoring, cache affinity, switch thresholds, and every
|
|
28
|
+
* weighted cost/reliability/speed utility pass. Nothing in this file may
|
|
29
|
+
* reconstruct a second model selector.
|
|
30
|
+
*/
|
|
8
31
|
const crypto_1 = require("crypto");
|
|
32
|
+
const routeEligibility_1 = require("./routeEligibility");
|
|
9
33
|
const DAY_MS = 24 * 60 * 60 * 1_000;
|
|
10
34
|
const VALIDATION_TTL_MS = 7 * DAY_MS;
|
|
11
|
-
const AFFINITY_TTL_MS = 5 * 60_000;
|
|
12
35
|
const HEALTH_WINDOW_MS = 30 * 60_000;
|
|
13
36
|
const CIRCUIT_COOLDOWN_MS = 60_000;
|
|
14
|
-
|
|
15
|
-
const FEEDBACK_HALF_LIFE_MS = 30 * DAY_MS;
|
|
37
|
+
exports.ROUTE_POLICY_VERSION = 'newmark-auto-v2';
|
|
16
38
|
function finite(value) {
|
|
17
39
|
const number = Number(value);
|
|
18
40
|
return Number.isFinite(number) ? number : undefined;
|
|
@@ -23,32 +45,17 @@ function clamp(value, min = 0, max = 1) {
|
|
|
23
45
|
function deploymentKey(deployment) {
|
|
24
46
|
return `${deployment.providerId}\u0000${deployment.modelId}\u0000${deployment.logicalModelGroupId || ''}`;
|
|
25
47
|
}
|
|
26
|
-
function sameDeployment(left, right) {
|
|
27
|
-
return left.providerId === right.providerId && left.modelId === right.modelId;
|
|
28
|
-
}
|
|
29
|
-
function inSubset(deployment, subset) {
|
|
30
|
-
if (!subset?.length)
|
|
31
|
-
return true;
|
|
32
|
-
return subset.some(item => sameDeployment(item, deployment));
|
|
33
|
-
}
|
|
34
|
-
function inScope(deployment, scope) {
|
|
35
|
-
return scope.kind === 'global' || deployment.providerId === scope.providerId;
|
|
36
|
-
}
|
|
37
48
|
function normalizeAutoPreference(value) {
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
default: return 'balanced';
|
|
45
|
-
}
|
|
49
|
+
const normalized = String(value || '').trim().toLowerCase();
|
|
50
|
+
if (normalized === 'conservative' || normalized === 'balanced' || normalized === 'aggressive')
|
|
51
|
+
return normalized;
|
|
52
|
+
// Previous values selected fixed quality/cost/speed objectives. They do not
|
|
53
|
+
// have a reliable one-to-one mapping to a model decision tendency.
|
|
54
|
+
return 'balanced';
|
|
46
55
|
}
|
|
47
56
|
function defaultRoutePolicy(mode = 'balanced') {
|
|
48
|
-
const maxQualityLoss = mode === 'quality' ? 0 : mode === 'balanced' ? 0.02 : mode === 'cost' ? 0.06 : 0.04;
|
|
49
57
|
return {
|
|
50
58
|
mode,
|
|
51
|
-
maxQualityLoss,
|
|
52
59
|
allowPreview: false,
|
|
53
60
|
privacy: 'default',
|
|
54
61
|
requiredCapabilities: [],
|
|
@@ -134,132 +141,76 @@ function classifyRouteFailure(error) {
|
|
|
134
141
|
}
|
|
135
142
|
return { type: 'execution_error', retryable: false, switchAllowed: false, statusCode };
|
|
136
143
|
}
|
|
137
|
-
|
|
144
|
+
function createRouteId() {
|
|
145
|
+
return (0, crypto_1.randomUUID)();
|
|
146
|
+
}
|
|
147
|
+
function catalogSnapshotHash(candidates) {
|
|
148
|
+
const snapshot = candidates.map(candidate => ({
|
|
149
|
+
deployment: candidate.deployment,
|
|
150
|
+
enabled: candidate.enabled,
|
|
151
|
+
unavailableReason: candidate.unavailableReason,
|
|
152
|
+
validation: candidate.validation,
|
|
153
|
+
capabilities: [...candidate.capabilities].sort(),
|
|
154
|
+
maxContextTokens: candidate.maxContextTokens,
|
|
155
|
+
preview: candidate.preview,
|
|
156
|
+
privacy: [...candidate.privacy].sort(),
|
|
157
|
+
dataRegions: [...(candidate.dataRegions || [])].sort(),
|
|
158
|
+
supportedProtocolParameters: [...(candidate.supportedProtocolParameters || [])].sort(),
|
|
159
|
+
expectedInputCostUsdPerM: candidate.expectedInputCostUsdPerM,
|
|
160
|
+
expectedOutputCostUsdPerM: candidate.expectedOutputCostUsdPerM,
|
|
161
|
+
latencyMs: candidate.latencyMs,
|
|
162
|
+
reliability: candidate.reliability,
|
|
163
|
+
toolValidity: candidate.toolValidity,
|
|
164
|
+
throughput: candidate.throughput,
|
|
165
|
+
circuit: candidate.circuit,
|
|
166
|
+
preference: candidate.preference,
|
|
167
|
+
fallbackOnly: !!candidate.fallbackOnly,
|
|
168
|
+
})).sort((left, right) => deploymentKey(left.deployment).localeCompare(deploymentKey(right.deployment)));
|
|
169
|
+
return (0, crypto_1.createHash)('sha256').update(JSON.stringify(snapshot)).digest('hex');
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* Operational execution controller.
|
|
173
|
+
*
|
|
174
|
+
* It owns endpoint health, the circuit breaker, build-scoped transaction pins
|
|
175
|
+
* and the provider-local recovery ladder. It never ranks models and never
|
|
176
|
+
* learns a preference; health observations only decide whether an endpoint is
|
|
177
|
+
* attempted right now.
|
|
178
|
+
*/
|
|
179
|
+
class RouteExecutionController {
|
|
138
180
|
now;
|
|
139
181
|
policyVersion;
|
|
140
|
-
affinityTtlMs;
|
|
141
|
-
switchThreshold;
|
|
142
182
|
transactionPins = new Map();
|
|
143
|
-
affinities = new Map();
|
|
144
183
|
endpointHealth = new Map();
|
|
145
|
-
feedback = new Map();
|
|
146
184
|
constructor(options = {}) {
|
|
147
185
|
this.now = options.now || Date.now;
|
|
148
|
-
this.policyVersion = options.policyVersion ||
|
|
149
|
-
this.affinityTtlMs = Math.max(0, options.affinityTtlMs ?? AFFINITY_TTL_MS);
|
|
150
|
-
this.switchThreshold = Math.max(0, options.switchThreshold ?? 0.15);
|
|
186
|
+
this.policyVersion = options.policyVersion || exports.ROUTE_POLICY_VERSION;
|
|
151
187
|
}
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
finalStatus: 'no_candidate',
|
|
165
|
-
retryBudgetMs: request.batch ? 15_000 : 5_000,
|
|
166
|
-
};
|
|
167
|
-
if (selection.kind === 'fixed') {
|
|
168
|
-
const fixed = candidates.find(item => sameDeployment(item.deployment, selection.deployment));
|
|
169
|
-
if (!fixed || fixed.enabled === false) {
|
|
170
|
-
decision.finalStatus = 'fixed_unavailable';
|
|
171
|
-
return decision;
|
|
172
|
-
}
|
|
173
|
-
const ranked = this.rankCandidate(fixed, taskClasses, policy, request, false, now);
|
|
174
|
-
decision.rankedCandidates = [ranked];
|
|
175
|
-
decision.resolvedDeployment = { ...fixed.deployment };
|
|
176
|
-
decision.attempts = [{ deployment: { ...fixed.deployment }, kind: 'initial', status: 'planned' }];
|
|
177
|
-
decision.finalStatus = 'resolved';
|
|
178
|
-
this.claimEndpointAttempt(fixed.deployment);
|
|
179
|
-
return decision;
|
|
180
|
-
}
|
|
181
|
-
const eligible = [];
|
|
182
|
-
for (const candidate of candidates) {
|
|
183
|
-
const reasons = [];
|
|
184
|
-
if (!candidate.enabled)
|
|
185
|
-
reasons.push('disabled');
|
|
186
|
-
if (!inScope(candidate.deployment, selection.scope))
|
|
187
|
-
reasons.push('outside_scope');
|
|
188
|
-
if (!inSubset(candidate.deployment, selection.subset))
|
|
189
|
-
reasons.push('outside_subset');
|
|
190
|
-
if (candidate.fallbackOnly)
|
|
191
|
-
reasons.push('fallback_only');
|
|
192
|
-
if (candidate.preview && !policy.allowPreview)
|
|
193
|
-
reasons.push('preview_disallowed');
|
|
194
|
-
if (request.estimatedInputTokens + request.expectedOutputTokens > Math.max(0, candidate.maxContextTokens || 0))
|
|
195
|
-
reasons.push('context_too_small');
|
|
196
|
-
if (policy.privacy !== 'default' && !candidate.privacy.includes(policy.privacy))
|
|
197
|
-
reasons.push(`privacy:${policy.privacy}`);
|
|
198
|
-
if (policy.dataRegion) {
|
|
199
|
-
const requiredRegion = policy.dataRegion.toLowerCase();
|
|
200
|
-
const regions = new Set((candidate.dataRegions || []).map(item => String(item).toLowerCase()));
|
|
201
|
-
if (!regions.has(requiredRegion))
|
|
202
|
-
reasons.push(`data_region:${policy.dataRegion}`);
|
|
203
|
-
}
|
|
204
|
-
const supportedParameters = new Set((candidate.supportedProtocolParameters || []).map(item => String(item).toLowerCase()));
|
|
205
|
-
for (const parameter of policy.requiredProtocolParameters || []) {
|
|
206
|
-
const requiredParameter = String(parameter).toLowerCase();
|
|
207
|
-
if (requiredParameter && !supportedParameters.has(requiredParameter))
|
|
208
|
-
reasons.push(`protocol_parameter:${requiredParameter}`);
|
|
209
|
-
}
|
|
210
|
-
const expectedCost = expectedRequestCost(candidate, request);
|
|
211
|
-
if (policy.maxExpectedCostUsd !== undefined && (expectedCost === undefined || expectedCost > policy.maxExpectedCostUsd)) {
|
|
212
|
-
reasons.push(expectedCost === undefined ? 'unknown_cost' : 'budget_exceeded');
|
|
213
|
-
}
|
|
214
|
-
if (reasons.length)
|
|
215
|
-
decision.excludedCandidates.push({ deployment: { ...candidate.deployment }, reasons: [...new Set(reasons)] });
|
|
216
|
-
else
|
|
217
|
-
eligible.push(candidate);
|
|
218
|
-
}
|
|
219
|
-
if (!eligible.length)
|
|
220
|
-
return decision;
|
|
221
|
-
const affinityKey = this.affinityKey(request.affinityKey, selection.scope, policy.mode);
|
|
222
|
-
let affinity = this.affinities.get(affinityKey);
|
|
223
|
-
if (affinity && affinity.expiresAt <= now) {
|
|
224
|
-
this.affinities.delete(affinityKey);
|
|
225
|
-
affinity = undefined;
|
|
226
|
-
}
|
|
227
|
-
const affinityDeployment = affinity?.deployment;
|
|
228
|
-
const transactionPin = this.transactionPins.get(request.transactionId);
|
|
229
|
-
if (transactionPin) {
|
|
230
|
-
const pinned = eligible.find(item => sameDeployment(item.deployment, transactionPin));
|
|
231
|
-
if (pinned) {
|
|
232
|
-
decision.rankedCandidates = this.rankEligible(eligible, taskClasses, policy, request, now, affinityDeployment);
|
|
233
|
-
decision.resolvedDeployment = { ...pinned.deployment };
|
|
234
|
-
decision.pinReason = 'transaction';
|
|
235
|
-
decision.attempts = [{ deployment: { ...pinned.deployment }, kind: 'initial', status: 'planned' }];
|
|
236
|
-
decision.finalStatus = 'resolved';
|
|
237
|
-
this.claimEndpointAttempt(pinned.deployment);
|
|
238
|
-
return decision;
|
|
239
|
-
}
|
|
240
|
-
this.transactionPins.delete(request.transactionId);
|
|
241
|
-
}
|
|
242
|
-
const ranked = this.rankEligible(eligible, taskClasses, policy, request, now, affinityDeployment);
|
|
243
|
-
decision.rankedCandidates = ranked;
|
|
244
|
-
let selected = ranked[0];
|
|
245
|
-
if (affinity) {
|
|
246
|
-
const incumbent = ranked.find(item => sameDeployment(item.deployment, affinity.deployment));
|
|
247
|
-
if (incumbent && selected.utility - incumbent.utility < this.switchThreshold) {
|
|
248
|
-
selected = incumbent;
|
|
249
|
-
decision.pinReason = 'cache_affinity';
|
|
250
|
-
}
|
|
251
|
-
}
|
|
252
|
-
decision.resolvedDeployment = { ...selected.deployment };
|
|
253
|
-
decision.attempts = [{ deployment: { ...selected.deployment }, kind: 'initial', status: 'planned' }];
|
|
254
|
-
decision.finalStatus = 'resolved';
|
|
255
|
-
this.claimEndpointAttempt(selected.deployment);
|
|
256
|
-
this.transactionPins.set(request.transactionId, { ...selected.deployment });
|
|
257
|
-
this.affinities.set(affinityKey, { deployment: { ...selected.deployment }, expiresAt: now + this.affinityTtlMs });
|
|
258
|
-
return decision;
|
|
188
|
+
version() {
|
|
189
|
+
return this.policyVersion;
|
|
190
|
+
}
|
|
191
|
+
/** Execution-consistency pin for one Build / route transaction. */
|
|
192
|
+
pinTransaction(transactionId, deployment) {
|
|
193
|
+
if (!transactionId || !deployment.providerId || !deployment.modelId)
|
|
194
|
+
return;
|
|
195
|
+
this.transactionPins.set(transactionId, { ...deployment });
|
|
196
|
+
}
|
|
197
|
+
transactionPin(transactionId) {
|
|
198
|
+
const pinned = this.transactionPins.get(transactionId);
|
|
199
|
+
return pinned ? { ...pinned } : undefined;
|
|
259
200
|
}
|
|
260
201
|
endTransaction(transactionId) {
|
|
261
202
|
this.transactionPins.delete(transactionId);
|
|
262
203
|
}
|
|
204
|
+
/**
|
|
205
|
+
* Provider-local recovery ladder.
|
|
206
|
+
*
|
|
207
|
+
* Order of preference: retry the same deployment (when the failure is
|
|
208
|
+
* retryable inside the retry budget), then an equivalent deployment of the
|
|
209
|
+
* same logical model group, then the alternates the decision engine already
|
|
210
|
+
* validated (in the order it produced them), then explicit fallback-only
|
|
211
|
+
* models. Recovery never crosses the failed provider boundary and never
|
|
212
|
+
* re-runs model selection.
|
|
213
|
+
*/
|
|
263
214
|
planAttempts(decision, candidates, failure) {
|
|
264
215
|
const current = decision.resolvedDeployment;
|
|
265
216
|
if (!current || !failure.error.switchAllowed || failure.streamCommitted || failure.sideEffectCommitted)
|
|
@@ -270,7 +221,7 @@ class AutoRouter {
|
|
|
270
221
|
const attempts = [];
|
|
271
222
|
const retryDelayMs = failure.error.retryAfterMs ?? 250;
|
|
272
223
|
const alreadyRetriedCurrent = decision.attempts.some(attempt => attempt.kind === 'retry_same_deployment'
|
|
273
|
-
&&
|
|
224
|
+
&& (0, routeEligibility_1.sameDeploymentRef)(attempt.deployment, current));
|
|
274
225
|
if (failure.error.retryable && !alreadyRetriedCurrent && retryDelayMs <= (decision.retryBudgetMs ?? 5_000)) {
|
|
275
226
|
attempts.push({
|
|
276
227
|
deployment: { ...current },
|
|
@@ -287,11 +238,9 @@ class AutoRouter {
|
|
|
287
238
|
const selection = decision.requestedSelection;
|
|
288
239
|
if (selection.kind === 'fixed')
|
|
289
240
|
return attempts.slice(0, remainingAttempts);
|
|
290
|
-
const scope = selection.kind === 'auto' ? selection.scope : { kind: 'provider', providerId: current.providerId };
|
|
291
241
|
const subset = selection.kind === 'auto' ? selection.subset : undefined;
|
|
292
242
|
const currentGroup = current.logicalModelGroupId;
|
|
293
243
|
const currentProviderId = current.providerId;
|
|
294
|
-
const now = this.now();
|
|
295
244
|
const attemptedDeployments = decision.attempts.map(attempt => attempt.deployment);
|
|
296
245
|
// Fallback is a recovery operation, not a new global routing decision.
|
|
297
246
|
// It must never cross the provider boundary, even when the original Auto
|
|
@@ -299,26 +248,33 @@ class AutoRouter {
|
|
|
299
248
|
// switching credentials/endpoints (for example provider A/model X to
|
|
300
249
|
// provider B/model X).
|
|
301
250
|
const eligible = candidates.filter(candidate => candidate.enabled
|
|
251
|
+
&& !candidate.unavailableReason
|
|
302
252
|
&& candidate.deployment.providerId === currentProviderId
|
|
303
|
-
&&
|
|
304
|
-
&&
|
|
305
|
-
&& !
|
|
306
|
-
&& !attemptedDeployments.some(attempted =>
|
|
253
|
+
&& (0, routeEligibility_1.inAutoScope)(candidate.deployment, { kind: 'provider', providerId: currentProviderId })
|
|
254
|
+
&& (0, routeEligibility_1.inAutoSubset)(candidate.deployment, subset)
|
|
255
|
+
&& !(0, routeEligibility_1.sameDeploymentRef)(candidate.deployment, current)
|
|
256
|
+
&& !attemptedDeployments.some(attempted => (0, routeEligibility_1.sameDeploymentRef)(candidate.deployment, attempted))
|
|
307
257
|
&& this.passedInitialHardFilters(decision, candidate));
|
|
308
258
|
const equivalent = currentGroup
|
|
309
259
|
? eligible.find(candidate => candidate.deployment.logicalModelGroupId === currentGroup && !candidate.fallbackOnly)
|
|
310
260
|
: undefined;
|
|
311
261
|
const fallback = eligible.find(candidate => candidate.fallbackOnly);
|
|
312
|
-
const
|
|
313
|
-
|
|
314
|
-
.
|
|
262
|
+
const orderedAlternates = [];
|
|
263
|
+
for (const alternate of decision.alternatives) {
|
|
264
|
+
const match = eligible.find(candidate => (0, routeEligibility_1.sameDeploymentRef)(candidate.deployment, alternate));
|
|
265
|
+
if (!match || match === equivalent || match === fallback || match.fallbackOnly)
|
|
266
|
+
continue;
|
|
267
|
+
if (orderedAlternates.some(existing => (0, routeEligibility_1.sameDeploymentRef)(existing.deployment, match.deployment)))
|
|
268
|
+
continue;
|
|
269
|
+
orderedAlternates.push(match);
|
|
270
|
+
}
|
|
315
271
|
for (const candidate of eligible) {
|
|
316
272
|
if (candidate === equivalent || candidate === fallback || candidate.fallbackOnly
|
|
317
|
-
||
|
|
273
|
+
|| orderedAlternates.some(existing => (0, routeEligibility_1.sameDeploymentRef)(existing.deployment, candidate.deployment)))
|
|
318
274
|
continue;
|
|
319
|
-
|
|
275
|
+
orderedAlternates.push(candidate);
|
|
320
276
|
}
|
|
321
|
-
for (const next of [equivalent,
|
|
277
|
+
for (const next of [equivalent, ...orderedAlternates, fallback]) {
|
|
322
278
|
if (!next || attempts.length >= 2)
|
|
323
279
|
continue;
|
|
324
280
|
attempts.push({
|
|
@@ -396,96 +352,6 @@ class AutoRouter {
|
|
|
396
352
|
circuit: this.circuitState(deployment, now, false),
|
|
397
353
|
};
|
|
398
354
|
}
|
|
399
|
-
recordFeedback(event) {
|
|
400
|
-
const at = event.at ?? this.now();
|
|
401
|
-
const safe = {
|
|
402
|
-
deployment: { ...event.deployment },
|
|
403
|
-
taskClass: event.taskClass,
|
|
404
|
-
score: clamp(Number(event.score) || 0, -1, 1),
|
|
405
|
-
source: event.source,
|
|
406
|
-
at,
|
|
407
|
-
};
|
|
408
|
-
const key = feedbackKey(safe.deployment, safe.taskClass);
|
|
409
|
-
const events = (this.feedback.get(key) || []).filter(item => at - Number(item.at || 0) <= FEEDBACK_WINDOW_MS);
|
|
410
|
-
events.push(safe);
|
|
411
|
-
this.feedback.set(key, events);
|
|
412
|
-
return events.length >= 3;
|
|
413
|
-
}
|
|
414
|
-
clearLearnedPreferences() {
|
|
415
|
-
this.feedback.clear();
|
|
416
|
-
}
|
|
417
|
-
learnedPreference(deployment, taskClasses) {
|
|
418
|
-
const now = this.now();
|
|
419
|
-
const values = [];
|
|
420
|
-
for (const taskClass of taskClasses) {
|
|
421
|
-
const events = (this.feedback.get(feedbackKey(deployment, taskClass)) || [])
|
|
422
|
-
.filter(event => now - Number(event.at || 0) <= FEEDBACK_WINDOW_MS)
|
|
423
|
-
.sort((left, right) => Number(left.at || 0) - Number(right.at || 0));
|
|
424
|
-
if (events.length < 3)
|
|
425
|
-
continue;
|
|
426
|
-
let ewma = events[0].score;
|
|
427
|
-
let previousAt = Number(events[0].at || now);
|
|
428
|
-
for (const event of events.slice(1)) {
|
|
429
|
-
const eventAt = Number(event.at || previousAt);
|
|
430
|
-
const betweenDecay = Math.pow(0.5, Math.max(0, eventAt - previousAt) / FEEDBACK_HALF_LIFE_MS);
|
|
431
|
-
ewma = 0.2 * event.score + 0.8 * ewma * betweenDecay;
|
|
432
|
-
previousAt = eventAt;
|
|
433
|
-
}
|
|
434
|
-
const newestAt = Number(events[events.length - 1].at || now);
|
|
435
|
-
const decay = Math.pow(0.5, Math.max(0, now - newestAt) / FEEDBACK_HALF_LIFE_MS);
|
|
436
|
-
values.push(clamp(ewma * decay, -1, 1));
|
|
437
|
-
}
|
|
438
|
-
return values.length ? values.reduce((sum, value) => sum + value, 0) / values.length : 0;
|
|
439
|
-
}
|
|
440
|
-
rankEligible(candidates, taskClasses, policy, request, now, affinityDeployment) {
|
|
441
|
-
const qualities = candidates.map(candidate => ({ candidate, quality: qualityScore(candidate, taskClasses) }));
|
|
442
|
-
const bestQuality = Math.max(...qualities.map(item => item.quality));
|
|
443
|
-
const band = qualities.filter(item => bestQuality - item.quality <= policy.maxQualityLoss + Number.EPSILON);
|
|
444
|
-
return band.map(item => this.rankCandidate(item.candidate, taskClasses, policy, request, !!affinityDeployment && sameDeployment(item.candidate.deployment, affinityDeployment), now, item.quality))
|
|
445
|
-
.sort((left, right) => right.utility - left.utility || right.quality - left.quality || deploymentKey(left.deployment).localeCompare(deploymentKey(right.deployment)));
|
|
446
|
-
}
|
|
447
|
-
rankCandidate(candidate, taskClasses, policy, request, cache, _now, quality = qualityScore(candidate, taskClasses)) {
|
|
448
|
-
const expectedCostUsd = expectedRequestCost(candidate, request);
|
|
449
|
-
const cost = expectedCostUsd === undefined ? 0 : 1 / (1 + Math.max(0, expectedCostUsd) * 1_000);
|
|
450
|
-
const endpointReliability = candidate.reliability === undefined ? 0.5 : clamp(candidate.reliability);
|
|
451
|
-
const toolRelevant = taskClasses.some(taskClass => taskClass === 'tool_use' || taskClass === 'computer_use');
|
|
452
|
-
const reliability = toolRelevant && candidate.toolValidity !== undefined
|
|
453
|
-
? (endpointReliability + clamp(candidate.toolValidity)) / 2
|
|
454
|
-
: endpointReliability;
|
|
455
|
-
const latency = candidate.latencyMs === undefined ? 0.5 : 1 / (1 + Math.max(0, candidate.latencyMs) / 1_000);
|
|
456
|
-
const throughput = candidate.throughput === undefined ? 0.5 : Math.max(0, candidate.throughput) / (Math.max(0, candidate.throughput) + 40);
|
|
457
|
-
const speed = (latency + throughput) / 2;
|
|
458
|
-
const learned = this.learnedPreference(candidate.deployment, taskClasses);
|
|
459
|
-
const configuredPreference = clamp(Number(candidate.preference) || 0, -1, 1);
|
|
460
|
-
const effectivePreference = configuredPreference === 0 ? learned : configuredPreference;
|
|
461
|
-
const preference = clamp((effectivePreference + 1) / 2);
|
|
462
|
-
const components = { cost, reliability, speed, cache: cache ? 1 : 0, preference };
|
|
463
|
-
let utility;
|
|
464
|
-
switch (policy.mode) {
|
|
465
|
-
case 'quality':
|
|
466
|
-
utility = quality * 0.55 + reliability * 0.25 + speed * 0.15 + cost * 0.05;
|
|
467
|
-
break;
|
|
468
|
-
case 'cost':
|
|
469
|
-
utility = cost * 0.65 + reliability * 0.15 + speed * 0.10 + components.cache * 0.05 + preference * 0.05;
|
|
470
|
-
break;
|
|
471
|
-
case 'speed':
|
|
472
|
-
utility = speed * 0.60 + reliability * 0.20 + cost * 0.10 + components.cache * 0.05 + preference * 0.05;
|
|
473
|
-
break;
|
|
474
|
-
default:
|
|
475
|
-
utility = cost * 0.40 + reliability * 0.25 + speed * 0.20 + components.cache * 0.10 + preference * 0.05;
|
|
476
|
-
break;
|
|
477
|
-
}
|
|
478
|
-
return {
|
|
479
|
-
deployment: { ...candidate.deployment },
|
|
480
|
-
quality,
|
|
481
|
-
utility: clamp(utility),
|
|
482
|
-
expectedCostUsd,
|
|
483
|
-
components,
|
|
484
|
-
};
|
|
485
|
-
}
|
|
486
|
-
affinityKey(key, scope, mode) {
|
|
487
|
-
return `${key}\u0000${scope.kind === 'global' ? 'global' : `provider:${scope.providerId}`}\u0000${mode}`;
|
|
488
|
-
}
|
|
489
355
|
healthFor(deployment, now) {
|
|
490
356
|
const key = deploymentKey(deployment);
|
|
491
357
|
let health = this.endpointHealth.get(key);
|
|
@@ -497,12 +363,10 @@ class AutoRouter {
|
|
|
497
363
|
return health;
|
|
498
364
|
}
|
|
499
365
|
passedInitialHardFilters(decision, candidate) {
|
|
500
|
-
const exclusion = decision.excludedCandidates.find(entry =>
|
|
366
|
+
const exclusion = decision.excludedCandidates.find(entry => (0, routeEligibility_1.sameDeploymentRef)(entry.deployment, candidate.deployment));
|
|
501
367
|
if (!exclusion)
|
|
502
368
|
return true;
|
|
503
|
-
return !!candidate.fallbackOnly
|
|
504
|
-
&& exclusion.reasons.length > 0
|
|
505
|
-
&& exclusion.reasons.every(reason => reason === 'fallback_only');
|
|
369
|
+
return !!candidate.fallbackOnly && (0, routeEligibility_1.excludedOnlyForFallbackPool)(exclusion);
|
|
506
370
|
}
|
|
507
371
|
circuitState(deployment, now, claimHalfOpen = true) {
|
|
508
372
|
const health = this.healthFor(deployment, now);
|
|
@@ -519,61 +383,12 @@ class AutoRouter {
|
|
|
519
383
|
return 'closed';
|
|
520
384
|
}
|
|
521
385
|
}
|
|
522
|
-
exports.
|
|
523
|
-
function
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
scope: selection.scope.kind === 'global' ? { kind: 'global' } : { kind: 'provider', providerId: selection.scope.providerId },
|
|
529
|
-
policyId: selection.policyId,
|
|
530
|
-
subset: selection.subset?.map(item => ({ ...item })),
|
|
531
|
-
};
|
|
532
|
-
}
|
|
533
|
-
function qualityScore(candidate, taskClasses) {
|
|
534
|
-
const scores = taskClasses.map(taskClass => {
|
|
535
|
-
const metric = candidate.qualityByTask?.[taskClass];
|
|
536
|
-
if (!metric)
|
|
537
|
-
return 0.5;
|
|
538
|
-
const attempts = Math.max(0, Math.floor(Number(metric.attempts) || 0));
|
|
539
|
-
const successes = clamp(Number(metric.successes) || 0, 0, attempts);
|
|
540
|
-
return (successes + 2) / (attempts + 4);
|
|
541
|
-
});
|
|
542
|
-
return scores.reduce((sum, value) => sum + value, 0) / Math.max(1, scores.length);
|
|
543
|
-
}
|
|
544
|
-
function expectedRequestCost(candidate, request) {
|
|
545
|
-
const input = finite(candidate.expectedInputCostUsdPerM);
|
|
546
|
-
const output = finite(candidate.expectedOutputCostUsdPerM);
|
|
547
|
-
if (input === undefined || output === undefined)
|
|
548
|
-
return undefined;
|
|
549
|
-
return input * Math.max(0, request.estimatedInputTokens) / 1_000_000
|
|
550
|
-
+ output * Math.max(0, request.expectedOutputTokens) / 1_000_000;
|
|
551
|
-
}
|
|
552
|
-
function catalogHash(candidates) {
|
|
553
|
-
const snapshot = candidates.map(candidate => ({
|
|
554
|
-
deployment: candidate.deployment,
|
|
555
|
-
enabled: candidate.enabled,
|
|
556
|
-
validation: candidate.validation,
|
|
557
|
-
capabilities: [...candidate.capabilities].sort(),
|
|
558
|
-
maxContextTokens: candidate.maxContextTokens,
|
|
559
|
-
preview: candidate.preview,
|
|
560
|
-
privacy: [...candidate.privacy].sort(),
|
|
561
|
-
dataRegions: [...(candidate.dataRegions || [])].sort(),
|
|
562
|
-
supportedProtocolParameters: [...(candidate.supportedProtocolParameters || [])].sort(),
|
|
563
|
-
expectedInputCostUsdPerM: candidate.expectedInputCostUsdPerM,
|
|
564
|
-
expectedOutputCostUsdPerM: candidate.expectedOutputCostUsdPerM,
|
|
565
|
-
latencyMs: candidate.latencyMs,
|
|
566
|
-
reliability: candidate.reliability,
|
|
567
|
-
toolValidity: candidate.toolValidity,
|
|
568
|
-
throughput: candidate.throughput,
|
|
569
|
-
qualityByTask: candidate.qualityByTask,
|
|
570
|
-
preference: candidate.preference,
|
|
571
|
-
fallbackOnly: !!candidate.fallbackOnly,
|
|
572
|
-
})).sort((left, right) => deploymentKey(left.deployment).localeCompare(deploymentKey(right.deployment)));
|
|
573
|
-
return (0, crypto_1.createHash)('sha256').update(JSON.stringify(snapshot)).digest('hex');
|
|
574
|
-
}
|
|
575
|
-
function feedbackKey(deployment, taskClass) {
|
|
576
|
-
return `${deploymentKey(deployment)}\u0000${taskClass}`;
|
|
386
|
+
exports.RouteExecutionController = RouteExecutionController;
|
|
387
|
+
function validationEvidenceIsStale(checkedAt, now = Date.now()) {
|
|
388
|
+
const checked = Date.parse(String(checkedAt || ''));
|
|
389
|
+
if (!Number.isFinite(checked))
|
|
390
|
+
return true;
|
|
391
|
+
return now - checked > VALIDATION_TTL_MS;
|
|
577
392
|
}
|
|
578
393
|
function trimHealth(health, now) {
|
|
579
394
|
health.events = health.events.filter(event => now - event.at <= HEALTH_WINDOW_MS).slice(-100);
|
package/dist/core/config.js
CHANGED
|
@@ -90,9 +90,10 @@ class ConfigManager {
|
|
|
90
90
|
}
|
|
91
91
|
const providerIdsMigrated = migrateProviderIdsInConfig(normalized);
|
|
92
92
|
const marqueeConfigRemoved = removeDeprecatedMarqueeConfig(normalized);
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
//
|
|
93
|
+
const legacyJevRuntimeRemoved = removeLegacyJevDecisionRuntimeConfig(normalized);
|
|
94
|
+
if (providerIdsMigrated || marqueeConfigRemoved || legacyJevRuntimeRemoved) {
|
|
95
|
+
// Persist migrations immediately so obsolete JEV slot settings are not
|
|
96
|
+
// left in the active user config until an unrelated settings save.
|
|
96
97
|
try {
|
|
97
98
|
if (!this.readOnly)
|
|
98
99
|
fs.writeFileSync(cp, JSON.stringify(normalized, null, 2), 'utf-8');
|
|
@@ -356,13 +357,7 @@ class ConfigManager {
|
|
|
356
357
|
autoSwitchEnabled() { return this.getBool('models', 'auto_switch'); }
|
|
357
358
|
autoSwitchPreference() {
|
|
358
359
|
const value = this.getStr('models', 'auto_switch_preference');
|
|
359
|
-
|
|
360
|
-
return 'quality';
|
|
361
|
-
if (value === 'cheap_save')
|
|
362
|
-
return 'cost';
|
|
363
|
-
if (value === 'speed')
|
|
364
|
-
return 'speed';
|
|
365
|
-
return value === 'quality' || value === 'cost' || value === 'balanced' ? value : 'balanced';
|
|
360
|
+
return value === 'conservative' || value === 'balanced' || value === 'aggressive' ? value : 'balanced';
|
|
366
361
|
}
|
|
367
362
|
autoSwitchScope() { return this.getStr('models', 'auto_switch_scope') || 'all'; }
|
|
368
363
|
autoSwitchAnchorProvider() { return this.getStr('models', 'auto_switch_anchor_provider'); }
|
|
@@ -478,6 +473,25 @@ function removeDeprecatedMarqueeConfig(config) {
|
|
|
478
473
|
}
|
|
479
474
|
return changed;
|
|
480
475
|
}
|
|
476
|
+
function removeLegacyJevDecisionRuntimeConfig(config) {
|
|
477
|
+
const models = config.models;
|
|
478
|
+
if (!models)
|
|
479
|
+
return false;
|
|
480
|
+
const retiredKeys = [
|
|
481
|
+
'jev_runtime', 'jev_model_id', 'jev_model_path',
|
|
482
|
+
'jev_fallback_strategy', 'jev_fallback_provider_id', 'jev_fallback_model_id',
|
|
483
|
+
];
|
|
484
|
+
const hadRetiredInterface = retiredKeys.some(key => Object.prototype.hasOwnProperty.call(models, key));
|
|
485
|
+
if (!hadRetiredInterface)
|
|
486
|
+
return false;
|
|
487
|
+
const legacyTimeout = models.jev_timeout_ms?.value;
|
|
488
|
+
if (legacyTimeout === 100 && models.jev_timeout_ms) {
|
|
489
|
+
models.jev_timeout_ms.value = 3000;
|
|
490
|
+
}
|
|
491
|
+
for (const key of retiredKeys)
|
|
492
|
+
delete models[key];
|
|
493
|
+
return true;
|
|
494
|
+
}
|
|
481
495
|
function isConfigEntry(value) {
|
|
482
496
|
return !!value && typeof value === 'object' && !Array.isArray(value) && Object.prototype.hasOwnProperty.call(value, 'value');
|
|
483
497
|
}
|
|
@@ -879,7 +893,7 @@ function defaultConfig() {
|
|
|
879
893
|
default_intelligence: { _description: "Default reasoning effort", _type: "choice", _values: ["low", "medium", "high", "xhigh", "max", "ultra"], value: "medium" },
|
|
880
894
|
agent_engine: { _description: "Agent engine", _type: "choice", _values: ["builtin", "codex", "opencode"], value: "builtin" },
|
|
881
895
|
auto_switch: { _description: "Auto-switch models", _type: "boolean", value: false },
|
|
882
|
-
auto_switch_preference: { _description: "Auto
|
|
896
|
+
auto_switch_preference: { _description: "Qualitative tendency for current-model Auto decisions", _type: "choice", _values: ["conservative", "balanced", "aggressive"], value: "balanced" },
|
|
883
897
|
auto_switch_scope: { _description: "Auto-switch scope", _type: "choice", _values: ["all", "provider"], value: "all" },
|
|
884
898
|
auto_switch_anchor_provider: { _description: "Stable provider id anchor for provider-scoped Auto routing", _type: "string", value: "" },
|
|
885
899
|
auto_switch_subset: { _description: "Explicit deployment allowlist for Auto; new catalog models are never added automatically", _type: "array", value: [] },
|
|
@@ -888,6 +902,8 @@ function defaultConfig() {
|
|
|
888
902
|
auto_data_region: { _description: "Required Auto data region; empty disables the region filter", _type: "string", value: "" },
|
|
889
903
|
auto_required_protocol_parameters: { _description: "Protocol parameters every Auto candidate must support", _type: "array", value: [] },
|
|
890
904
|
auto_max_expected_cost_usd: { _description: "Hard expected request cost ceiling; 0 disables the ceiling", _type: "number", value: 0 },
|
|
905
|
+
jev_enabled: { _description: "Use the current eligible conversation model for strict-JSON Auto decisions", _type: "boolean", value: true },
|
|
906
|
+
jev_timeout_ms: { _description: "Hard timeout for one current-model JEV decision in milliseconds", _type: "number", value: 3000 },
|
|
891
907
|
fallback_on_unavailable: { _description: "Fallback when model unavailable", _type: "boolean", value: true },
|
|
892
908
|
openai_api_mode: { _description: "OpenAI-compatible API mode", _type: "choice", _values: ["chat_stream", "chat", "responses"], value: "chat_stream" },
|
|
893
909
|
openai_streaming: { _description: "Legacy streaming flag for OpenAI-compatible chat completions", _type: "boolean", value: true },
|
|
@@ -146,7 +146,7 @@ export interface FinalRecord {
|
|
|
146
146
|
export interface ContinuationEvent {
|
|
147
147
|
eventId: string;
|
|
148
148
|
cursor: number;
|
|
149
|
-
type: 'BuildAccepted' | 'AttemptStarted' | 'AttemptHeartbeat' | 'BuildCommitted' | 'BuildFailed' | 'BuildCancelled' | 'BranchForked' | 'BranchPaused' | 'BranchResumed' | 'BuildQueueRepaired' | 'StaleWriteRejected';
|
|
149
|
+
type: 'BuildAccepted' | 'AttemptStarted' | 'AttemptHeartbeat' | 'BuildCommitted' | 'BuildFailed' | 'BuildCancelled' | 'BranchForked' | 'BranchPaused' | 'BranchResumed' | 'BuildQueueRepaired' | 'AttemptOrphanRecovered' | 'StaleWriteRejected';
|
|
150
150
|
workspaceId: WorkspaceId;
|
|
151
151
|
rootId: RootId;
|
|
152
152
|
branchId: BranchId;
|