@north-light/crouter 0.3.161 → 0.3.162

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/dist/api/client.d.ts +3 -1
  2. package/dist/api/client.js +6 -0
  3. package/dist/api/dto/modelauth.d.ts +32 -0
  4. package/dist/api/dto/modelauth.js +2 -1
  5. package/dist/api/routes.d.ts +1 -0
  6. package/dist/api/routes.js +1 -0
  7. package/dist/builtin-pi-packages/pi-crtr-extensions/extensions/__tests__/provider-rotation.test.ts +55 -395
  8. package/dist/builtin-pi-packages/pi-crtr-extensions/extensions/provider-rotation.ts +204 -419
  9. package/dist/clients/attach/viewer.js +592 -592
  10. package/dist/core/__tests__/broker-launch-candidates.test.js +13 -3
  11. package/dist/core/__tests__/broker-sdk-wiring.test.js +53 -25
  12. package/dist/core/__tests__/full/broker-provider-retry.test.js +6 -122
  13. package/dist/core/__tests__/model-pin-durability.test.js +7 -5
  14. package/dist/core/__tests__/model-routes.test.d.ts +1 -0
  15. package/dist/core/__tests__/model-routes.test.js +41 -0
  16. package/dist/core/__tests__/review-model-floor.test.js +2 -2
  17. package/dist/core/auth-file.d.ts +3 -2
  18. package/dist/core/auth-file.js +5 -3
  19. package/dist/core/canvas/types.d.ts +6 -1
  20. package/dist/core/config.d.ts +3 -1
  21. package/dist/core/config.js +88 -2
  22. package/dist/core/model-routes.d.ts +37 -0
  23. package/dist/core/model-routes.js +140 -0
  24. package/dist/core/runtime/broker.d.ts +0 -4
  25. package/dist/core/runtime/broker.js +116 -372
  26. package/dist/core/runtime/launch.d.ts +9 -11
  27. package/dist/core/runtime/launch.js +15 -21
  28. package/dist/core/runtime/managed-provider-cooling.d.ts +0 -14
  29. package/dist/core/runtime/managed-provider-cooling.js +0 -19
  30. package/dist/core/runtime/model-swap.js +7 -1
  31. package/dist/core/runtime/promote.js +1 -0
  32. package/dist/core/subscription-state.d.ts +12 -36
  33. package/dist/core/subscription-state.js +19 -203
  34. package/dist/daemon/api/__tests__/full/api-server.test.js +9 -6
  35. package/dist/daemon/api/__tests__/full/b10-attach-modelauth.test.js +123 -1
  36. package/dist/daemon/api/handlers/modelauth.js +51 -2
  37. package/dist/daemon/api/handlers/nodes.js +2 -0
  38. package/dist/types.d.ts +17 -0
  39. package/dist/web-client/assets/{index-B76ZKfT_.js → index-NIuSCOHM.js} +1 -1
  40. package/dist/web-client/index.html +1 -1
  41. package/dist/web-client/sw.js +1 -1
  42. package/package.json +1 -1
  43. package/runtime.lock.json +2 -2
@@ -36,14 +36,14 @@ import { buildRefInventory } from '../memory/inline-ref-inventory.js';
36
36
  import { buildGuidance } from '../memory/inline-ref-guidance.js';
37
37
  import { clearSessionCache } from '../substrate/session-cache.js';
38
38
  import { FRONT_DOOR_ENV } from './front-door.js';
39
- import { CANVAS_EXTENSIONS, configuredLadderProviders, equivalentOtherProviderModel, isProviderPinnedModelToken, nextLadderModel, normalizeModel, piInvocationToSdkConfig, } from './launch.js';
39
+ import { CANVAS_EXTENSIONS, configuredLadderProviders, isProviderPinnedModelToken, nextLadderModel, normalizeModel, piInvocationToSdkConfig, } from './launch.js';
40
40
  import { assertEngineVersion, loadBrokerEngine } from './broker-sdk.js';
41
41
  import { resolveBundledPiPackageDir } from './pi-cli.js';
42
42
  import { classify } from '../fault-classifier.js';
43
43
  import { PoolCredentialStore } from '../pool-credential-store.js';
44
44
  import { clearFault, isModelNotFoundError, isProviderUnconfiguredError, readFault, recordFault, recordFaultForCurrentNode } from './fault.js';
45
45
  import { clearCleanAbort, markBusy, markCleanAbort } from './busy.js';
46
- import { extractCoolingDeadline, MANAGED_PROVIDER_ROTATION_POLICY_UI, } from './managed-provider-cooling.js';
46
+ import { extractCoolingDeadline } from './managed-provider-cooling.js';
47
47
  import { AUTH_FAULT_RECOVERY_BODY, CONNECTION_FAULT_RECOVERY_BODY, PROVIDER_FAULT_RECOVERY_BODY } from './fault-recovery-nudge.js';
48
48
  import { markFatalOnExhaust, nextFaultRetry, policyFor } from './fault-recovery.js';
49
49
  import { probeOnline as probeOnlineReal } from './connectivity.js';
@@ -51,7 +51,8 @@ import { CRTR_CYCLE_CUSTOM_TYPE, cycleAwareMessages } from './session-cycles.js'
51
51
  import { BUILTIN_SLASH_COMMANDS, piSessionsRoot } from './pi-vendored.js';
52
52
  import { StreamWatchdog, resolveStreamWatchdogMs, releaseKnownStreamWait, KNOWN_STREAM_WAIT_UI } from './stream-watchdog.js';
53
53
  import { isDeterministicCommandExpansion } from './command-expansion.js';
54
- import { clearPreferredModel, isManagedProvider, readRotationConfig, readSubscriptionPool, rememberPreferredModel, resolveProviderCandidates } from '../subscription-state.js';
54
+ import { isManagedProvider, readSubscriptionPool, resolveProviderCandidates } from '../subscription-state.js';
55
+ import { expandModelCandidates, modelRequestFromConfig, probeRouteAvailability } from '../model-routes.js';
55
56
  import { encodeFrame, FrameDecoder, FrameOverflowError, BROKER_READ_CAPS, } from './broker-protocol.js';
56
57
  const THINKING_LEVELS = new Set(['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max']);
57
58
  /**
@@ -780,7 +781,7 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
780
781
  clearTimeout(faultRetryTimer);
781
782
  faultRetryTimer = null;
782
783
  };
783
- const failedRetryProviders = new Set();
784
+ const failedRetryRoutes = new Set();
784
785
  let providerFallbackInFlight = null;
785
786
  // Models that have already produced a 404 not_found this session. Guards the
786
787
  // not-found fallback below against an infinite swap→re-drive→404 loop when the
@@ -839,122 +840,43 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
839
840
  return;
840
841
  const candidateSession = session;
841
842
  const candidateServices = services;
842
- // An exact-consult launch (cfg.modelExact) locks the session to the single
843
- // captured model — never cross-provider-retry, not even the
844
- // modelProviderPinned restore-toward-pin exception below. Unavailability
845
- // must fail the turn, not substitute a different model.
843
+ // An exact-consult launch locks the session to its one captured model.
844
+ // Other requests advance only through their declared route expansion.
846
845
  if (cfg.modelExact === true)
847
846
  return;
848
847
  const currentSpec = formatModelSpec(session.model, session.thinkingLevel);
849
848
  if (currentSpec === null)
850
849
  return;
851
- const next = equivalentOtherProviderModel(currentSpec, failedRetryProviders);
852
- if (next === null)
853
- return;
854
- // An explicit provider pin forbids turn-time rotation AWAY to another
855
- // provider — EXCEPT a switch back TOWARD the recorded restoration pin (a
856
- // launch that fell back off a cooling pinned provider records the requested
857
- // managed model). Restoring to the pinned provider honors the pin rather
858
- // than violating it; every other rotation stays suppressed.
859
- if (cfg.modelProviderPinned === true) {
860
- const pin = readRotationConfig().preferredModel;
861
- const targetSpec = parseModelSpec(next.model).modelSpec;
862
- const targetProvider = targetSpec.slice(0, targetSpec.indexOf('/'));
863
- const targetId = targetSpec.slice(targetSpec.indexOf('/') + 1);
864
- if (pin?.providerId !== targetProvider || pin.modelId !== targetId)
850
+ const routeRequest = modelRequestFromConfig(cfg.model, process.env['CRTR_MODEL_INTENT'], false);
851
+ if (routeRequest !== undefined) {
852
+ const routes = expandModelCandidates(routeRequest, process.env['CRTR_NODE_CWD'] ?? cfg.cwd, process.env['CRTR_PROFILE_ID']);
853
+ const current = parseModelSpec(currentSpec);
854
+ const liveRank = routes.find((candidate) => `${candidate.providerId}/${candidate.modelId}` === current.modelSpec && candidate.thinkingLevel === current.thinkingLevel)?.rank;
855
+ const nextRoute = liveRank === undefined ? undefined : routes.slice(liveRank + 1).find((candidate) => !failedRetryRoutes.has(candidate.routeId) && resolveProviderCandidates([{ ...candidate, ...probeRouteAvailability(candidate, registryOf(candidateServices)), automaticFallback: true }]).candidates.length > 0);
856
+ if (nextRoute === undefined)
865
857
  return;
866
- }
867
- const fallbackPromise = (async () => {
868
- const requestedSpec = parseModelSpec(next.model).modelSpec;
869
- const slash = requestedSpec.indexOf('/');
870
- const model = slash > 0
871
- ? registryOf(candidateServices).find(requestedSpec.slice(0, slash), requestedSpec.slice(slash + 1))
872
- : undefined;
873
- const requested = parseModelSpec(next.model);
874
- const pool = isManagedProvider(requested.modelSpec.slice(0, requested.modelSpec.indexOf('/')))
875
- ? readSubscriptionPool(requested.modelSpec.slice(0, requested.modelSpec.indexOf('/')))
876
- : [];
877
- const resolved = resolveProviderCandidates([{
878
- providerId: requested.modelSpec.slice(0, requested.modelSpec.indexOf('/')),
879
- modelId: requested.modelSpec.slice(requested.modelSpec.indexOf('/') + 1),
880
- thinkingLevel: requested.thinkingLevel,
881
- registered: model !== undefined,
882
- authenticated: model !== undefined && (isManagedProvider(requested.modelSpec.slice(0, requested.modelSpec.indexOf('/')))
883
- ? pool.length > 0
884
- : registryOf(candidateServices).hasConfiguredAuth(model)),
885
- coolingUntil: pool.length > 0 && pool.every((entry) => (entry.rateLimitedUntil || 0) > Date.now())
886
- ? Math.min(...pool.map((entry) => entry.rateLimitedUntil || 0))
887
- : undefined,
888
- }]);
889
- if (resolved.candidates.length === 0 || model === undefined) {
890
- emitEvent({
891
- level: 'debug',
892
- event: 'broker.runtime.diagnostic',
893
- fields: { message: `provider fallback skipped: ${next.model} is ${resolved.rejected[0]?.reason ?? 'unavailable'}` },
894
- });
895
- return;
896
- }
897
- // Snapshot the restoration intent before the asynchronous model change;
898
- // this retry must not replace a launch-recorded requested pin.
899
- const preferred = readRotationConfig().preferredModel;
900
- await candidateSession.setModel(model);
901
- if (installedGeneration !== generation || session !== candidateSession)
902
- return;
903
- if (requested.thinkingLevel !== undefined)
904
- candidateSession.setThinkingLevel(requested.thinkingLevel);
905
- const currentProvider = currentSpec.slice(0, currentSpec.indexOf('/'));
906
- const currentModelId = parseModelSpec(currentSpec).modelSpec.slice(currentSpec.indexOf('/') + 1);
907
- // A launch fallback already records its requested managed model as the
908
- // restoration pin. Keep that intent through turn-time retries; reaching
909
- // it makes the temporary pin fulfilled. A normal retry has no such pin,
910
- // so retain the failed managed model for provider-rotation to restore.
911
- const switchedProvider = requested.modelSpec.slice(0, requested.modelSpec.indexOf('/'));
912
- const switchedModelId = requested.modelSpec.slice(switchedProvider.length + 1);
913
- if (preferred?.providerId === switchedProvider && preferred.modelId === switchedModelId) {
914
- clearPreferredModel();
915
- }
916
- else if (preferred === undefined && isManagedProvider(currentProvider)) {
917
- rememberPreferredModel({ providerId: currentProvider, modelId: currentModelId });
918
- }
919
- failedRetryProviders.add(next.fromProvider);
920
- broadcastModelChanged(false);
921
- emitEvent({
922
- level: 'debug',
923
- event: 'broker.runtime.diagnostic',
924
- fields: { message: `provider fallback after ${kind}: ${currentSpec} → ${formatModelSpec(candidateSession.model, candidateSession.thinkingLevel) ?? next.model}` },
925
- });
926
- })()
927
- .catch((err) => {
928
- emitEvent({
929
- level: 'debug',
930
- event: 'broker.runtime.diagnostic',
931
- fields: { message: `provider fallback failed: ${err instanceof Error ? err.message : String(err)}` },
858
+ const fallbackPromise = (async () => {
859
+ const target = registryOf(candidateServices).find(nextRoute.providerId, nextRoute.modelId);
860
+ if (!target)
861
+ return;
862
+ await candidateSession.setModel(target);
863
+ if (installedGeneration !== generation || session !== candidateSession)
864
+ return;
865
+ candidateSession.setThinkingLevel((nextRoute.thinkingLevel ?? 'off'));
866
+ failedRetryRoutes.add(nextRoute.routeId);
867
+ broadcastModelChanged(false);
868
+ })().catch((err) => { emitEvent({ level: 'debug', event: 'broker.runtime.diagnostic', fields: { message: `provider route fallback failed: ${err instanceof Error ? err.message : String(err)}` } }); }).finally(() => {
869
+ if (installedGeneration === generation)
870
+ providerFallbackInFlight = null;
932
871
  });
933
- })
934
- .finally(() => {
935
- if (installedGeneration === generation)
936
- providerFallbackInFlight = null;
937
- });
938
- providerFallbackInFlight = fallbackPromise;
872
+ providerFallbackInFlight = fallbackPromise;
873
+ return;
874
+ }
939
875
  };
940
- // A 404 not_found is NOT one of pi's auto-retried kinds (it retries 429/5xx
941
- // only) and never resolves on its own — the configured model id is wrong or
942
- // decommissioned (e.g. the default `anthropic/ultra` pointing at a model the
943
- // account can't reach). Left alone the node errors out, the stophook parks it
944
- // as a stall, and the daemon force-recycles it onto the SAME bad model in a
945
- // loop. Instead: fail the node over to the strong-anthropic ladder model and
946
- // re-drive the turn so the work continues. broadcastModelChanged persists the
947
- // swap into the durable launch recipe, so a later revive uses the good model
948
- // too. Fires on the terminal-error message (stopReason 'error'), the signal a
949
- // non-retried request surfaces.
950
- //
951
- // The SAME treatment covers a provider whose credential could not be resolved
952
- // (`Provider is not configured: <id>`). That error is classified `other`, i.e.
953
- // fatal, so without this the node parks and every wake re-fails on the same
954
- // unusable provider — the only escape being a human running `crtr node config
955
- // --model`. Here it substitutes an equivalent model on the OTHER provider (the
956
- // same ladder cell `maybeSwitchProviderForRetry` uses), proves that target is
957
- // actually authenticated before switching, and re-drives.
876
+ // A 404 or unconfigured provider is terminal to pi's ordinary retry loop.
877
+ // Advance once through the request's declared route expansion, then re-drive
878
+ // the turn on the selected viable route. The raw registry never supplies this
879
+ // recovery path: it is an explicit routing decision, not a last-resort guess.
958
880
  const maybeFallbackOnUnusableModel = (event) => {
959
881
  if (event.type !== 'message_end')
960
882
  return;
@@ -964,10 +886,8 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
964
886
  const unconfigured = isProviderUnconfiguredError(msg.errorMessage ?? '');
965
887
  if (!unconfigured && !isModelNotFoundError(msg.errorMessage ?? ''))
966
888
  return;
967
- if (cfg.modelProviderPinned === true)
968
- return;
969
- // An exact-consult launch must fail closed on a 404 for its locked model,
970
- // never auto-fail-over to the strong-anthropic ladder.
889
+ // Exact-consult launches fail closed. A concrete request's expansion already
890
+ // contains only the identical model through its eligible credential sources.
971
891
  if (cfg.modelExact === true)
972
892
  return;
973
893
  if (notFoundFallbackInFlight !== null)
@@ -980,30 +900,22 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
980
900
  const currentSpec = formatModelSpec(session.model, session.thinkingLevel);
981
901
  if (currentSpec === null)
982
902
  return;
983
- notFoundModels.add(currentSpec);
984
- // The strong-anthropic ladder cell — "whatever the current strong anthropic
985
- // provider model is" (config modelLadders.anthropic.strong), not hardcoded.
986
- // An unusable PROVIDER (as opposed to a missing model id) prefers the
987
- // same-strength cell on the other provider, so a node keeps the capability
988
- // it was launched with instead of being promoted or demoted by the swap.
989
- const equivalent = unconfigured ? equivalentOtherProviderModel(currentSpec, failedRetryProviders)?.model : undefined;
990
- const target = normalizeModel(equivalent ?? 'anthropic/strong');
991
- // Don't loop: the failed model already IS the target, or the target itself
992
- // 404'd earlier this session. Leave it to the normal stall/recycle path.
993
- if (target === currentSpec || notFoundModels.has(target)) {
994
- emitEvent({
995
- level: 'debug',
996
- event: 'broker.runtime.diagnostic',
997
- fields: { message: `not_found fallback skipped: ${currentSpec} has no healthy strong-anthropic target (${target})` },
998
- });
903
+ const current = parseModelSpec(currentSpec);
904
+ // A 404 rejects the provider/model itself, not only one thinking setting.
905
+ notFoundModels.add(current.modelSpec);
906
+ const routeRequest = modelRequestFromConfig(cfg.model, process.env['CRTR_MODEL_INTENT'], false);
907
+ const routes = routeRequest ? expandModelCandidates(routeRequest, process.env['CRTR_NODE_CWD'] ?? cfg.cwd, process.env['CRTR_PROFILE_ID']) : [];
908
+ const liveRank = routes.find((candidate) => `${candidate.providerId}/${candidate.modelId}` === current.modelSpec && candidate.thinkingLevel === current.thinkingLevel)?.rank;
909
+ const nextRoute = liveRank === undefined ? undefined : routes.slice(liveRank + 1).find((candidate) => {
910
+ if (notFoundModels.has(`${candidate.providerId}/${candidate.modelId}`))
911
+ return false;
912
+ return resolveProviderCandidates([{ ...candidate, ...probeRouteAvailability(candidate, registryOf(candidateServices)), automaticFallback: true }]).candidates.length > 0;
913
+ });
914
+ if (!nextRoute)
999
915
  return;
1000
- }
916
+ const target = `${nextRoute.providerId}/${nextRoute.modelId}${nextRoute.thinkingLevel ? `:${nextRoute.thinkingLevel}` : ''}`;
1001
917
  const fallbackPromise = (async () => {
1002
- const targetSpec = parseModelSpec(target).modelSpec;
1003
- const slash = targetSpec.indexOf('/');
1004
- const model = slash > 0
1005
- ? registryOf(candidateServices).find(targetSpec.slice(0, slash), targetSpec.slice(slash + 1))
1006
- : undefined;
918
+ const model = registryOf(candidateServices).find(nextRoute.providerId, nextRoute.modelId);
1007
919
  if (model === undefined) {
1008
920
  emitEvent({
1009
921
  level: 'debug',
@@ -1012,22 +924,10 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
1012
924
  });
1013
925
  return;
1014
926
  }
1015
- // Swapping onto a second provider that ALSO has no credential would just
1016
- // repeat the fault under a different name, so prove this one resolves.
1017
- if (unconfigured && !registryOf(candidateServices).hasConfiguredAuth(model)) {
1018
- emitEvent({
1019
- level: 'debug',
1020
- event: 'broker.runtime.diagnostic',
1021
- fields: { message: `unconfigured-provider fallback skipped: ${target} is not authenticated either` },
1022
- });
1023
- return;
1024
- }
1025
- const requested = parseModelSpec(target);
1026
927
  await candidateSession.setModel(model);
1027
928
  if (installedGeneration !== generation || session !== candidateSession)
1028
929
  return;
1029
- if (requested.thinkingLevel !== undefined)
1030
- candidateSession.setThinkingLevel(requested.thinkingLevel);
930
+ candidateSession.setThinkingLevel((nextRoute.thinkingLevel ?? 'off'));
1031
931
  broadcastModelChanged();
1032
932
  const resolved = formatModelSpec(candidateSession.model, candidateSession.thinkingLevel) ?? target;
1033
933
  emitEvent({
@@ -1320,25 +1220,11 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
1320
1220
  // buildPiArgv replays on revive). Best-effort: a degenerate/ephemeral node
1321
1221
  // with no canvas row just skips persistence. CRTR_NODE_ID is set in the
1322
1222
  // broker's own env (merged by piInvocationToSdkConfig at boot).
1323
- // `pinnedOverride`, when passed, is the durable-pin decision for THIS model
1324
- // change — decided by the caller from the change's RAW pre-normalize token
1325
- // (an explicit set_model). Omitted — an algorithmic model change (cycle_model,
1326
- // cycle_ladder, a provider-fallback/not-found/cooling re-drive) — the existing
1327
- // pin is carried through UNCHANGED via the `meta.launch` spread, exactly as
1328
- // before: those changes never re-decide a pin the user didn't ask to change.
1329
- //
1330
- // `cfg.modelProviderPinned` is this broker's LIVE authoritative pin state —
1331
- // every in-process policy read (the retry-fallback gate, the model-not-found
1332
- // gate, and pinnedToFirst on a reload_auth re-drive) consults it directly, not
1333
- // the boot-time launch recipe. Mutate it here, synchronously, BEFORE the
1334
- // best-effort durable write below, so an explicit set_model takes effect for
1335
- // the very next policy decision even when the canvas row is unreachable
1336
- // (a degenerate/ephemeral node). Only ever reached from the success path of a
1337
- // completed `set_model` — never before the engine call it decided from has
1338
- // actually succeeded.
1339
- const persistModelChoice = (pinnedOverride) => {
1340
- if (pinnedOverride !== undefined)
1341
- cfg.modelProviderPinned = pinnedOverride;
1223
+ // `pinnedOverride`, when passed, is the durable launch-recipe decision for
1224
+ // an explicit model change, derived from its raw pre-normalize token. Omitted
1225
+ // algorithmic changes retain the existing recipe field through the spread;
1226
+ // route expansion, rather than a mutable broker pin, selects retry recovery.
1227
+ const persistModelChoice = (pinnedOverride, explicit = false) => {
1342
1228
  const nodeId = process.env['CRTR_NODE_ID'];
1343
1229
  const spec = formatModelSpec(session.model, session.thinkingLevel);
1344
1230
  if (nodeId === undefined || nodeId === '' || spec === null)
@@ -1347,9 +1233,24 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
1347
1233
  const meta = getNode(nodeId);
1348
1234
  if (meta === null)
1349
1235
  return;
1350
- const launch = meta.launch !== undefined
1351
- ? { ...meta.launch, model: spec, ...(pinnedOverride !== undefined ? { modelProviderPinned: pinnedOverride } : {}) }
1352
- : undefined;
1236
+ const launch = meta.launch === undefined
1237
+ ? undefined
1238
+ : explicit
1239
+ ? (() => {
1240
+ const { modelIntent: _modelIntent, env, ...rest } = meta.launch;
1241
+ const { CRTR_MODEL_INTENT: _intentEnv, ...remainingEnv } = env;
1242
+ return {
1243
+ ...rest,
1244
+ model: spec,
1245
+ ...(pinnedOverride !== undefined ? { modelProviderPinned: pinnedOverride } : {}),
1246
+ env: remainingEnv,
1247
+ };
1248
+ })()
1249
+ : { ...meta.launch, model: spec, ...(pinnedOverride !== undefined ? { modelProviderPinned: pinnedOverride } : {}) };
1250
+ if (explicit) {
1251
+ delete process.env['CRTR_MODEL_INTENT'];
1252
+ cfg.model = spec;
1253
+ }
1353
1254
  updateNode(nodeId, { model_override: spec, ...(launch !== undefined ? { launch } : {}) });
1354
1255
  }
1355
1256
  catch {
@@ -1360,9 +1261,9 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
1360
1261
  // set_model/cycle_model the broker announces it itself — otherwise the new
1361
1262
  // model reaches no viewer (the requester gets only a bare ack) and every
1362
1263
  // footer shows the stale model until the next unrelated event.
1363
- const broadcastModelChanged = (persist = true, pinnedOverride) => {
1264
+ const broadcastModelChanged = (persist = true, pinnedOverride, explicit = false) => {
1364
1265
  if (persist)
1365
- persistModelChoice(pinnedOverride);
1266
+ persistModelChoice(pinnedOverride, explicit);
1366
1267
  broadcast({ type: 'model_changed', model: isUnknownModel(session.model) ? undefined : session.model, spec: formatModelSpec(session.model, session.thinkingLevel) });
1367
1268
  };
1368
1269
  const buildSnapshot = () => ({
@@ -1506,11 +1407,6 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
1506
1407
  forward: (client, request) => sendFrame(client, request),
1507
1408
  pending: pendingDialogs,
1508
1409
  broadcast: broadcastUi,
1509
- providerRotationPolicy: () => cfg.modelExact === true
1510
- ? { allowCrossProviderFallback: false, reason: 'model-exact' }
1511
- : cfg.modelProviderPinned === true
1512
- ? { allowCrossProviderFallback: false, reason: 'provider-pinned' }
1513
- : { allowCrossProviderFallback: true },
1514
1410
  // A deliberate managed-provider cooldown wait is known liveness, not a wedged
1515
1411
  // stream: pause the dead-stream watchdog for its duration. Only pause for a
1516
1412
  // finite future deadline, and resume ONLY while the same turn is still
@@ -2436,7 +2332,7 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
2436
2332
  // setModelLive), carries the RAW-token pin decision explicitly.
2437
2333
  // Otherwise frame.model itself IS the raw/explicit token (a
2438
2334
  // slash-command arg, or a concrete picker pick) and derives it.
2439
- broadcastModelChanged(true, frame.pinned !== undefined ? frame.pinned : isProviderPinnedModelToken(frame.model));
2335
+ broadcastModelChanged(true, frame.pinned !== undefined ? frame.pinned : isProviderPinnedModelToken(frame.model), true);
2440
2336
  })
2441
2337
  .catch(engineErrorTo(client));
2442
2338
  break;
@@ -2862,8 +2758,7 @@ export async function runBroker(nodeId, startupAt = process.hrtime.bigint(), sta
2862
2758
  await authSession.setModel(resolved.model);
2863
2759
  if (installedGeneration !== authGeneration || session !== authSession)
2864
2760
  return;
2865
- if (resolved.thinkingLevel !== undefined)
2866
- authSession.setThinkingLevel(resolved.thinkingLevel);
2761
+ authSession.setThinkingLevel((resolved.thinkingLevel ?? 'off'));
2867
2762
  broadcastModelChanged();
2868
2763
  drivePendingKickoff();
2869
2764
  }
@@ -3122,199 +3017,49 @@ export function snapshotMessages(session) {
3122
3017
  // alone on a failed re-drive attempt).
3123
3018
  // ---------------------------------------------------------------------------
3124
3019
  async function resolveLaunchModel(services, cfg) {
3125
- // Resolve one viable model against the SERVICES registry, which includes
3126
- // extension-provided providers. `resolveProviderCandidates` is the SOLE
3127
- // launch selection path — the same viability policy governs turn-time retry
3128
- // and rotation fallback — so launch selection can never diverge from
3129
- // rotation. A requested model gets first refusal (its provider pinned when
3130
- // the caller pinned it), then the authenticated registry order decides.
3131
- const requested = cfg.model ? parseModelSpec(cfg.model) : undefined;
3132
- const faultLaunchModel = (message) => {
3133
- recordFaultForCurrentNode({
3134
- link: 'crtr→pi',
3135
- op: 'model resolution at launch',
3136
- kind: 'model-not-found',
3137
- retry: { disposition: 'fatal' },
3138
- message,
3020
+ // Route declarations own preference order; the raw registry remains only a
3021
+ // legacy tail for models a user did not declare.
3022
+ const routeRequest = modelRequestFromConfig(cfg.model, process.env['CRTR_MODEL_INTENT'], cfg.modelExact === true);
3023
+ if (routeRequest !== undefined) {
3024
+ const registry = registryOf(services);
3025
+ const routes = expandModelCandidates(routeRequest, process.env['CRTR_NODE_CWD'] ?? cfg.cwd, process.env['CRTR_PROFILE_ID']);
3026
+ const routeKeys = new Set(routes.map((candidate) => `${candidate.providerId}/${candidate.modelId}`));
3027
+ // A concrete request may advance only through its declared same-model
3028
+ // credential-source routes. The raw registry tail is logical-request legacy.
3029
+ const tail = routeRequest.kind === 'logical'
3030
+ ? registry.getAll()
3031
+ .filter((model) => !routeKeys.has(`${model.provider}/${model.id}`))
3032
+ .map((model, index) => ({ routeId: `registry:${model.provider}/${model.id}`, credentialSource: 'registry', family: model.provider, strength: 'strong', providerId: model.provider, modelId: model.id, rank: routes.length + index }))
3033
+ : [];
3034
+ const hasManagedAuthenticated = [...routes, ...tail].some((candidate) => {
3035
+ const availability = probeRouteAvailability(candidate, registry);
3036
+ return isManagedProvider(candidate.providerId) && availability.registered && availability.authenticated;
3139
3037
  });
3140
- throw new Error(`[broker] FATAL: ${message}`);
3141
- };
3142
- if (requested !== undefined && requested.modelSpec.indexOf('/') <= 0) {
3143
- faultLaunchModel(`model '${cfg.model}' has no 'provider/id' form and cannot be resolved`);
3144
- }
3145
- const requestedProvider = requested?.modelSpec.slice(0, requested.modelSpec.indexOf('/'));
3146
- const requestedId = requested?.modelSpec.slice(requested.modelSpec.indexOf('/') + 1);
3147
- const registry = registryOf(services);
3148
- // The requested model's SAME-STRENGTH cell on the OTHER provider — the one
3149
- // substitution that preserves the caller's intent, since a persona asks for a
3150
- // STRENGTH (`explore: openai/light`) and the ladder only picks the id. It is
3151
- // candidate #2, ahead of raw registry order: without it, an unavailable
3152
- // `openai/light` fell through to whatever the registry happened to list first
3153
- // (`anthropic/claude-fable-5` — the most expensive model there is) for a scout
3154
- // that asked for the cheapest. This is the SAME mapping turn-time rotation
3155
- // already falls back through (`equivalentOtherProviderModel`), so launch
3156
- // selection and rotation now degrade identically instead of divergently.
3157
- const ladderFallback = cfg.modelExact === true || cfg.model === undefined
3158
- ? null
3159
- : equivalentOtherProviderModel(cfg.model);
3160
- const ladderCandidate = (() => {
3161
- if (ladderFallback === null)
3162
- return undefined;
3163
- const spec = parseModelSpec(ladderFallback.model);
3164
- const slash = spec.modelSpec.indexOf('/');
3165
- if (slash <= 0)
3166
- return undefined;
3167
- const provider = spec.modelSpec.slice(0, slash);
3168
- const id = spec.modelSpec.slice(slash + 1);
3169
- if (provider === requestedProvider && id === requestedId)
3170
- return undefined;
3171
- return { provider, id, thinkingLevel: spec.thinkingLevel };
3172
- })();
3173
- const requestedModel = requestedProvider !== undefined && requestedId !== undefined
3174
- ? registry.find(requestedProvider, requestedId)
3175
- : undefined;
3176
- // Enumerate every REGISTERED model (not just the authenticated set) so the
3177
- // resolver itself classifies each candidate's registered/authenticated/
3178
- // cooling state and reports a per-candidate rejection reason. Requested
3179
- // first, then the rest of the registry in order.
3180
- // cfg.modelExact locks launch selection to ONLY the requested candidate —
3181
- // the rest of the registry is never spread in, so no cross-model/cross-
3182
- // provider substitution is even reachable below.
3183
- const orderedModels = [
3184
- ...(requestedProvider !== undefined && requestedId !== undefined
3185
- ? [{ provider: requestedProvider, id: requestedId, model: requestedModel, thinkingLevel: requested?.thinkingLevel }]
3186
- : []),
3187
- ...(ladderCandidate !== undefined
3188
- ? [{
3189
- provider: ladderCandidate.provider,
3190
- id: ladderCandidate.id,
3191
- model: registry.find(ladderCandidate.provider, ladderCandidate.id),
3192
- thinkingLevel: ladderCandidate.thinkingLevel,
3193
- }]
3194
- : []),
3195
- ...(cfg.modelExact === true ? [] : registry.getAll()
3196
- .filter((candidate) => (candidate.provider !== requestedProvider || candidate.id !== requestedId)
3197
- && (ladderCandidate === undefined || candidate.provider !== ladderCandidate.provider || candidate.id !== ladderCandidate.id))
3198
- .map((candidate) => ({
3199
- provider: candidate.provider,
3200
- id: candidate.id,
3201
- model: candidate,
3202
- thinkingLevel: undefined,
3203
- }))),
3204
- ];
3205
- // Tier 1/2 policy: a non-managed candidate is only automatic-fallback-eligible
3206
- // when NO managed provider has an authenticated (pool-occupied) credential
3207
- // ANYWHERE in the ordered set — regardless of that managed candidate's
3208
- // current cooling state. A cooling-but-authenticated managed pool still
3209
- // blocks automatic fallback to a non-managed provider (preserves 4be8be6's
3210
- // intent: don't silently abandon a paid managed subscription for a
3211
- // pay-per-token key just because it's rate-limited right now). Only when
3212
- // there is NO managed-authenticated candidate at all does a registered +
3213
- // authenticated non-managed model become an acceptable automatic launch
3214
- // fallback (tier 2); only when even that is unavailable does launch fall all
3215
- // the way through to a model-less boot (tier 3, below).
3216
- const hasManagedAuthenticated = orderedModels.some((candidate) => {
3217
- if (!isManagedProvider(candidate.provider))
3218
- return false;
3219
- if (candidate.model === undefined)
3220
- return false; // not registered
3221
- return readSubscriptionPool(candidate.provider).length > 0;
3222
- });
3223
- const selection = resolveProviderCandidates(orderedModels.map((candidate, index) => {
3224
- // Managed providers authenticate off their subscription pool's accounts, not
3225
- // static credentials; a fully-cooling pool marks the candidate cooling.
3226
- const pool = isManagedProvider(candidate.provider) ? readSubscriptionPool(candidate.provider) : [];
3227
- const explicit = index === 0 && requestedProvider !== undefined && requestedId !== undefined;
3228
- return {
3229
- providerId: candidate.provider,
3230
- modelId: candidate.id,
3231
- thinkingLevel: candidate.thinkingLevel,
3232
- registered: candidate.model !== undefined,
3233
- // Non-managed models are valid automatic fallback when explicitly
3234
- // requested, when the candidate IS a managed provider itself, or (tier
3235
- // 2) when no managed-authenticated candidate exists anywhere in this
3236
- // launch's candidate set.
3237
- automaticFallback: explicit || isManagedProvider(candidate.provider) || !hasManagedAuthenticated,
3238
- authenticated: candidate.model !== undefined && (isManagedProvider(candidate.provider)
3239
- ? pool.length > 0
3240
- : registry.hasConfiguredAuth(candidate.model)),
3241
- coolingUntil: pool.length > 0 && pool.every((entry) => (entry.rateLimitedUntil || 0) > Date.now())
3242
- ? Math.min(...pool.map((entry) => entry.rateLimitedUntil || 0))
3243
- : undefined,
3244
- };
3245
- }), Date.now(),
3246
- // An explicit provider pin forbids cross-provider fallback for an
3247
- // unregistered/unauthenticated request; only a transient managed cooldown
3248
- // on the pinned provider still boots a viable alternative.
3249
- { pinnedToFirst: cfg.modelProviderPinned === true && requested !== undefined });
3250
- const selected = selection.candidates[0];
3251
- if (selected === undefined) {
3252
- const rejectionText = selection.rejected.length > 0
3253
- ? selection.rejected.map(({ candidate, reason }) => `'${candidate.providerId}/${candidate.modelId}' ${reason === 'non-managed-fallback' ? 'non-managed provider is not an automatic fallback' : reason}`).join('; ')
3254
- : 'no candidates were supplied';
3255
- // An exact-consult launch must FAIL rather than fall through to a
3256
- // model-less session — an unavailable captured model faults the launch
3257
- // (crash), never a silent degrade/substitution.
3038
+ const selection = resolveProviderCandidates([...routes, ...tail].map((candidate) => ({
3039
+ ...candidate,
3040
+ ...probeRouteAvailability(candidate, registry),
3041
+ automaticFallback: candidate.routeId.startsWith('registry:')
3042
+ ? (isManagedProvider(candidate.providerId) || !hasManagedAuthenticated)
3043
+ : true,
3044
+ })), Date.now(), { pinnedToFirst: routeRequest.kind === 'exact' });
3045
+ const selected = selection.candidates[0];
3046
+ if (selected !== undefined) {
3047
+ const model = registry.find(selected.providerId, selected.modelId);
3048
+ if (model !== undefined) {
3049
+ if ((selected.rank ?? 0) > 0)
3050
+ emitEvent({ level: 'warn', event: 'broker.model.launch_fallback', fields: { requested: cfg.model, selected: `${selected.providerId}/${selected.modelId}`, route_id: selected.routeId, credential_source: selected.credentialSource, rank: selected.rank } });
3051
+ return { model, thinkingLevel: selected.thinkingLevel };
3052
+ }
3053
+ }
3258
3054
  if (cfg.modelExact === true) {
3259
- return faultLaunchModel(`exact-consult model '${cfg.model}' is unavailable — refusing to substitute (${rejectionText})`);
3055
+ const rejection = selection.rejected.map(({ candidate, reason }) => `'${candidate.providerId}/${candidate.modelId}' ${reason}`).join('; ');
3056
+ recordFaultForCurrentNode({ link: 'crtr→pi', op: 'model resolution at launch', kind: 'model-not-found', retry: { disposition: 'fatal' }, message: `exact-consult model '${cfg.model}' is unavailable — refusing to substitute (${rejection})` });
3057
+ throw new Error(`[broker] FATAL: exact-consult model '${cfg.model}' is unavailable`);
3260
3058
  }
3261
- // Tier 3: no viable candidate at all (genuine zero-auth, OR every managed
3262
- // pool is cooling AND no non-managed provider is authenticated). This is
3263
- // NOT a fault — the caller constructs a model-less session so the
3264
- // socket/listener still comes up and the user can /login; reload_auth
3265
- // re-invokes this same helper to re-drive once auth changes.
3266
- emitEvent({
3267
- level: 'warn',
3268
- event: 'broker.model.none_viable',
3269
- fields: {
3270
- rejected_candidates: selection.rejected.slice(0, 16).map(({ candidate, reason }) => ({
3271
- provider: candidate.providerId,
3272
- model: candidate.modelId,
3273
- reason,
3274
- })),
3275
- rejected_candidates_omitted: Math.max(0, selection.rejected.length - 16),
3276
- },
3277
- });
3059
+ emitEvent({ level: 'warn', event: 'broker.model.none_viable', fields: { rejected_candidates: selection.rejected.slice(0, 16).map(({ candidate, reason }) => ({ provider: candidate.providerId, model: candidate.modelId, reason, route_id: candidate.routeId, credential_source: candidate.credentialSource })), rejected_candidates_omitted: Math.max(0, selection.rejected.length - 16) } });
3278
3060
  return undefined;
3279
3061
  }
3280
- const model = registry.find(selected.providerId, selected.modelId);
3281
- if (model === undefined) {
3282
- return faultLaunchModel(`selected model '${selected.providerId}/${selected.modelId}' disappeared during launch`);
3283
- }
3284
- const thinkingLevel = selected.thinkingLevel;
3285
- // A launch that does not get the model it asked for is a SUBSTITUTION, and it
3286
- // used to be entirely silent: the session simply came up on another model and
3287
- // the only trace was the pi `model_change` entry in the .jsonl. One line, only
3288
- // on the rare path, carrying the per-candidate rejection reasons that decided
3289
- // it — so "why is this node on the wrong model" is answerable from the log.
3290
- if (requestedProvider !== undefined &&
3291
- requestedId !== undefined &&
3292
- (selected.providerId !== requestedProvider || selected.modelId !== requestedId)) {
3293
- emitEvent({
3294
- level: 'warn',
3295
- event: 'broker.model.launch_fallback',
3296
- fields: {
3297
- requested: `${requestedProvider}/${requestedId}`,
3298
- selected: `${selected.providerId}/${selected.modelId}`,
3299
- ladder_equivalent: ladderFallback === null ? null : ladderFallback.model,
3300
- rejected_candidates: selection.rejected.slice(0, 8).map(({ candidate, reason }) => ({
3301
- provider: candidate.providerId,
3302
- model: candidate.modelId,
3303
- reason,
3304
- })),
3305
- },
3306
- });
3307
- }
3308
- if (requestedProvider !== undefined &&
3309
- requestedId !== undefined &&
3310
- isManagedProvider(requestedProvider) &&
3311
- requestedModel !== undefined &&
3312
- (selected.providerId !== requestedProvider || selected.modelId !== requestedId)) {
3313
- // This launch fallback is temporary: retain the requested managed pin for
3314
- // provider-rotation's before_agent_start restoration once it is ready.
3315
- rememberPreferredModel({ providerId: requestedProvider, modelId: requestedId });
3316
- }
3317
- return { model, thinkingLevel };
3062
+ return undefined;
3318
3063
  }
3319
3064
  /**
3320
3065
  * Wrap a live ModelRuntime so a model-less `createAgentSessionFromServices`
@@ -3631,7 +3376,6 @@ export function makeBrokerUiContext(deps) {
3631
3376
  };
3632
3377
  const noop = () => { };
3633
3378
  const ctx = {
3634
- [MANAGED_PROVIDER_ROTATION_POLICY_UI]: deps.providerRotationPolicy ?? (() => ({ allowCrossProviderFallback: true })),
3635
3379
  // The broker-owned known-stream-wait control (see BrokerDialogDeps): a
3636
3380
  // deliberate provider cooldown is KNOWN liveness, so the provider extension
3637
3381
  // reaches this through the stable symbol to pause the dead-stream watchdog for