@caupulican/pi-adaptative 0.81.15 → 0.81.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/CHANGELOG.md +23 -0
  2. package/dist/bundled-resources/runtimes/hf-transformers-openai-server.py +427 -0
  3. package/dist/bundled-resources/skills/tool-call-repair/references/repair-catalogue.md +18 -0
  4. package/dist/bundled-resources/skills/tool-call-repair/references/text-protocol-grammar.md +14 -7
  5. package/dist/core/agent-session.d.ts +15 -3
  6. package/dist/core/agent-session.d.ts.map +1 -1
  7. package/dist/core/agent-session.js +201 -25
  8. package/dist/core/agent-session.js.map +1 -1
  9. package/dist/core/local-runtime-controller.d.ts +17 -9
  10. package/dist/core/local-runtime-controller.d.ts.map +1 -1
  11. package/dist/core/local-runtime-controller.js +124 -20
  12. package/dist/core/local-runtime-controller.js.map +1 -1
  13. package/dist/core/models/adaptation-store.d.ts +5 -0
  14. package/dist/core/models/adaptation-store.d.ts.map +1 -1
  15. package/dist/core/models/adaptation-store.js +16 -0
  16. package/dist/core/models/adaptation-store.js.map +1 -1
  17. package/dist/core/models/context-sizing.d.ts +30 -0
  18. package/dist/core/models/context-sizing.d.ts.map +1 -0
  19. package/dist/core/models/context-sizing.js +90 -0
  20. package/dist/core/models/context-sizing.js.map +1 -0
  21. package/dist/core/models/default-model-suggestions.d.ts +4 -4
  22. package/dist/core/models/default-model-suggestions.d.ts.map +1 -1
  23. package/dist/core/models/default-model-suggestions.js +9 -0
  24. package/dist/core/models/default-model-suggestions.js.map +1 -1
  25. package/dist/core/models/local-registration.d.ts +12 -0
  26. package/dist/core/models/local-registration.d.ts.map +1 -1
  27. package/dist/core/models/local-registration.js +68 -0
  28. package/dist/core/models/local-registration.js.map +1 -1
  29. package/dist/core/models/local-runtime.d.ts +96 -1
  30. package/dist/core/models/local-runtime.d.ts.map +1 -1
  31. package/dist/core/models/local-runtime.js +338 -4
  32. package/dist/core/models/local-runtime.js.map +1 -1
  33. package/dist/core/models/model-ref.d.ts +4 -0
  34. package/dist/core/models/model-ref.d.ts.map +1 -1
  35. package/dist/core/models/model-ref.js +12 -3
  36. package/dist/core/models/model-ref.js.map +1 -1
  37. package/dist/core/models/perf-profile.d.ts +36 -0
  38. package/dist/core/models/perf-profile.d.ts.map +1 -0
  39. package/dist/core/models/perf-profile.js +163 -0
  40. package/dist/core/models/perf-profile.js.map +1 -0
  41. package/dist/core/models/runtime-arbiter.d.ts +57 -0
  42. package/dist/core/models/runtime-arbiter.d.ts.map +1 -0
  43. package/dist/core/models/runtime-arbiter.js +49 -0
  44. package/dist/core/models/runtime-arbiter.js.map +1 -0
  45. package/dist/core/tool-repair-health.d.ts.map +1 -1
  46. package/dist/core/tool-repair-health.js +2 -1
  47. package/dist/core/tool-repair-health.js.map +1 -1
  48. package/dist/modes/interactive/interactive-mode.d.ts +1 -0
  49. package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
  50. package/dist/modes/interactive/interactive-mode.js +4 -0
  51. package/dist/modes/interactive/interactive-mode.js.map +1 -1
  52. package/dist/modes/interactive/local-model-commands.d.ts +3 -1
  53. package/dist/modes/interactive/local-model-commands.d.ts.map +1 -1
  54. package/dist/modes/interactive/local-model-commands.js +163 -22
  55. package/dist/modes/interactive/local-model-commands.js.map +1 -1
  56. package/docs/models.md +19 -1
  57. package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
  58. package/examples/extensions/custom-provider-anthropic/package.json +1 -1
  59. package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
  60. package/examples/extensions/sandbox/package-lock.json +2 -2
  61. package/examples/extensions/sandbox/package.json +1 -1
  62. package/examples/extensions/with-deps/package-lock.json +2 -2
  63. package/examples/extensions/with-deps/package.json +1 -1
  64. package/npm-shrinkwrap.json +12 -12
  65. package/package.json +4 -4
@@ -1,6 +1,7 @@
1
- import { readFileSync } from "node:fs";
1
+ import { readFileSync, rmSync, writeFileSync } from "node:fs";
2
+ import { tmpdir } from "node:os";
2
3
  import { basename, dirname, join } from "node:path";
3
- import { classifyFailure, compactToolResultDetailsForRetention, computeRetryDelayMs, createCustomMessage, DEFAULT_RETRY_POLICY, RetryController, sleepAbortable, withStreamIdleWatchdog, } from "@caupulican/pi-agent-core";
4
+ import { classifyFailure, compactToolResultDetailsForRetention, computeRetryDelayMs, createCustomMessage, DEFAULT_RETRY_POLICY, DEFAULT_STREAM_IDLE, RetryController, sleepAbortable, withStreamIdleWatchdog, } from "@caupulican/pi-agent-core";
4
5
  import { calculateContextTokens, compact, createDeterministicCompaction, estimateContextTokens, getLatestCompactionEntry, prepareCompaction, runCompactionLoop, shouldCompact, } from "@caupulican/pi-agent-core/node";
5
6
  import { cleanupSessionResources, formatToolRepairStandingRule, generateTextToolProtocolPrimer, isContextOverflow, parseTextToolCalls, streamSimple, } from "@caupulican/pi-ai";
6
7
  import { Type } from "typebox";
@@ -30,6 +31,7 @@ import { deriveModelCapabilityProfile, filterToolNamesForCapability, } from "./m
30
31
  import { formatModelRouterModel, ModelRouterController } from "./model-router-controller.js";
31
32
  import { ModelSelectionController } from "./model-selection-controller.js";
32
33
  import { ModelAdaptationStore } from "./models/adaptation-store.js";
34
+ import { DEFAULT_ADAPTIVE_STREAM_IDLE_CEILING_MS, estimateContextPromptTokens, resolveAdaptiveStreamIdleOptions, withModelPerfProfile, } from "./models/perf-profile.js";
33
35
  import { ProfileFilterController } from "./profile-filter-controller.js";
34
36
  import { expandPromptTemplate } from "./prompt-templates.js";
35
37
  import { ReflectionController } from "./reflection-controller.js";
@@ -60,12 +62,22 @@ const MODEL_ADAPTATION_REPAIR_THRESHOLD = 3;
60
62
  const TEXT_TOOL_PROTOCOL_VERSION = 1;
61
63
  const TEXT_TOOL_PROTOCOL_TRIALS_PER_VARIANT = 2;
62
64
  const TEXT_TOOL_PROTOCOL_PARSE_FAILURE_THRESHOLD = 3;
63
- const TEXT_TOOL_PROTOCOL_VARIANTS = ["tool-tag", "tool-call", "fenced-json"];
65
+ const TEXT_TOOL_PROTOCOL_VARIANTS = [
66
+ "tool-tag",
67
+ "tool-call",
68
+ "fenced-json",
69
+ "function-xml",
70
+ ];
64
71
  const TEXT_TOOL_PROTOCOL_ECHO_TOOL = {
65
72
  name: "echo",
66
73
  description: "Echo calibration data",
67
74
  parameters: Type.Object({ data: Type.String() }),
68
75
  };
76
+ const NATIVE_TOOL_PROBE_READ_TOOL = {
77
+ name: "read",
78
+ description: "Read file contents",
79
+ parameters: Type.Object({ path: Type.String() }),
80
+ };
69
81
  /** Test-only override of the stream-idle bounds. Read per-request by the wiring's resolver. */
70
82
  let streamIdleOptionsOverride;
71
83
  /**
@@ -148,6 +160,8 @@ export class AgentSession {
148
160
  _collectWorkspaceSources;
149
161
  _localRuntimeController;
150
162
  _modelAdaptationStore;
163
+ _prefixWarmer;
164
+ _completedPrefixWarms = new Set();
151
165
  _repairModeSessionCounts = new Map();
152
166
  _textProtocolParseFailures = new Map();
153
167
  _textProtocolParseObservedThisTurn = false;
@@ -229,14 +243,35 @@ export class AgentSession {
229
243
  // withStreamIdleWatchdog's contract), so no extra drain is added at this wiring site.
230
244
  // Wrapping also breaks the `streamFn === streamSimple` identity the auth-injection checks
231
245
  // use, so the wrapper carries a rawness marker that _isRawStreamSimple reads.
246
+ const agentDir = config.agentDir ?? getAgentDir();
247
+ const modelAdaptationStore = ModelAdaptationStore.forAgentDir(agentDir);
232
248
  const baseStreamFn = this.agent.streamFn;
249
+ const profiledStreamFn = withModelPerfProfile(baseStreamFn, {
250
+ modelKey: (model) => formatModelRouterModel(model),
251
+ recordSample: (modelKey, sample) => {
252
+ modelAdaptationStore.recordPerfSample(modelKey, sample);
253
+ },
254
+ });
233
255
  // `this.settingsManager` is assigned below; the resolver closes over the config reference
234
256
  // because the wrapper must be installed before that assignment runs.
235
257
  const stallSettingsSource = config.settingsManager;
236
- this.agent.streamFn = tagRawness(withStreamIdleWatchdog(baseStreamFn, () => ({
237
- ...stallSettingsSource.getStreamStallSettings(),
238
- ...streamIdleOptionsOverride,
239
- })), baseStreamFn === streamSimple);
258
+ this.agent.streamFn = tagRawness(withStreamIdleWatchdog(profiledStreamFn, (model, context) => {
259
+ const configured = {
260
+ ...stallSettingsSource.getStreamStallSettings(),
261
+ ...streamIdleOptionsOverride,
262
+ };
263
+ const httpIdleTimeoutMs = stallSettingsSource.getHttpIdleTimeoutMs();
264
+ const profile = modelAdaptationStore.get(formatModelRouterModel(model)).perf;
265
+ const adaptive = resolveAdaptiveStreamIdleOptions({
266
+ base: { ...DEFAULT_STREAM_IDLE, ...configured },
267
+ profile,
268
+ promptTokens: estimateContextPromptTokens(context),
269
+ ceilingMs: httpIdleTimeoutMs === 0
270
+ ? DEFAULT_ADAPTIVE_STREAM_IDLE_CEILING_MS
271
+ : Math.max(DEFAULT_STREAM_IDLE.quietIdleMs, httpIdleTimeoutMs - 60_000),
272
+ });
273
+ return { ...configured, ...adaptive };
274
+ }), baseStreamFn === streamSimple);
240
275
  this.sessionManager = config.sessionManager;
241
276
  this.settingsManager = config.settingsManager;
242
277
  // Auto-retry rides the reliability kernel: the controller owns the attempt counter and the
@@ -260,8 +295,8 @@ export class AgentSession {
260
295
  this._resourceLoader = config.resourceLoader;
261
296
  this._customTools = config.customTools ?? [];
262
297
  this._cwd = config.cwd;
263
- this._agentDir = config.agentDir ?? getAgentDir();
264
- this._modelAdaptationStore = ModelAdaptationStore.forAgentDir(this._agentDir);
298
+ this._agentDir = agentDir;
299
+ this._modelAdaptationStore = modelAdaptationStore;
265
300
  this.agent.onTextToolProtocolParse = (event) => this._handleTextToolProtocolParse(event);
266
301
  this._applyToolRepairLayerSettings();
267
302
  this._collectWorkspaceSources = config.collectWorkspaceSources ?? collectWorkspaceSources;
@@ -578,11 +613,82 @@ export class AgentSession {
578
613
  activeToolNames: this._initialActiveToolNames,
579
614
  includeAllExtensionTools: true,
580
615
  });
616
+ this._scheduleLocalPrefixWarm(this.agent.state.model, "session-start");
581
617
  }
582
618
  /** Model registry for API key resolution and model discovery */
583
619
  get modelRegistry() {
584
620
  return this._modelRegistry;
585
621
  }
622
+ _scheduleLocalPrefixWarm(model, _reason) {
623
+ if (!model || !this._isWarmableLocalModel(model))
624
+ return;
625
+ const modelKey = formatModelRouterModel(model);
626
+ if (this._completedPrefixWarms.has(modelKey) || this._prefixWarmer?.modelKey === modelKey)
627
+ return;
628
+ this._cancelPrefixWarm();
629
+ const controller = new AbortController();
630
+ const timer = setTimeout(() => {
631
+ const warmer = this._prefixWarmer;
632
+ if (!warmer || warmer.controller !== controller || controller.signal.aborted)
633
+ return;
634
+ warmer.timer = undefined;
635
+ void this._runLocalPrefixWarm(model, modelKey, controller);
636
+ }, 0);
637
+ timer.unref?.();
638
+ this._prefixWarmer = { modelKey, controller, timer };
639
+ }
640
+ _cancelPrefixWarm() {
641
+ const warmer = this._prefixWarmer;
642
+ if (!warmer)
643
+ return;
644
+ if (warmer.timer)
645
+ clearTimeout(warmer.timer);
646
+ warmer.controller.abort(new Error("prefix warmer preempted"));
647
+ this._prefixWarmer = undefined;
648
+ }
649
+ async _runLocalPrefixWarm(model, modelKey, controller) {
650
+ try {
651
+ const options = {
652
+ maxTokens: 1,
653
+ signal: controller.signal,
654
+ onPayload: this.agent.onPayload,
655
+ onResponse: this.agent.onResponse,
656
+ };
657
+ if (this._isRawStreamSimple(this.agent.streamFn)) {
658
+ const auth = await this._getRequiredRequestAuth(model);
659
+ options.apiKey = auth.apiKey;
660
+ options.headers = auth.headers;
661
+ }
662
+ if (controller.signal.aborted)
663
+ return;
664
+ const stream = await this.agent.streamFn(model, {
665
+ systemPrompt: this._baseSystemPrompt,
666
+ tools: this.agent.state.tools,
667
+ messages: [],
668
+ }, options);
669
+ await stream.result();
670
+ if (!controller.signal.aborted)
671
+ this._completedPrefixWarms.add(modelKey);
672
+ }
673
+ catch {
674
+ // Best-effort cache warm only; a miss must never affect the real turn.
675
+ }
676
+ finally {
677
+ if (this._prefixWarmer?.controller === controller)
678
+ this._prefixWarmer = undefined;
679
+ }
680
+ }
681
+ _isWarmableLocalModel(model) {
682
+ if (model.api !== "openai-completions")
683
+ return false;
684
+ try {
685
+ const hostname = new URL(model.baseUrl).hostname.toLowerCase();
686
+ return hostname === "localhost" || hostname === "127.0.0.1" || hostname === "::1" || hostname === "[::1]";
687
+ }
688
+ catch {
689
+ return false;
690
+ }
691
+ }
586
692
  /**
587
693
  * True when the session's stream fn is the raw `streamSimple` provider entry (directly, or as the
588
694
  * base wrapped by the idle watchdog at construction). Callers use this to decide whether request
@@ -841,18 +947,56 @@ export class AgentSession {
841
947
  messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
842
948
  };
843
949
  }
844
- _messageHasEchoProbe(message, token) {
845
- return message.content.some((block) => block.type === "toolCall" && block.name === "echo" && block.arguments.data === token);
950
+ _messageHasToolCallWithStringArgument(message, toolName, argName, argValue) {
951
+ return message.content.some((block) => {
952
+ if (block.type !== "toolCall" || block.name !== toolName)
953
+ return false;
954
+ const args = block.arguments;
955
+ return (typeof args === "object" &&
956
+ args !== null &&
957
+ !Array.isArray(args) &&
958
+ args[argName] === argValue);
959
+ });
960
+ }
961
+ _nativeToolProbeSystemPrompt(instruction) {
962
+ const base = (this.agent.state.systemPrompt ?? "").trim();
963
+ return base ? `${base}\n\n${instruction}` : instruction;
964
+ }
965
+ async _runNativeReadTaskProbeTrial(model, path) {
966
+ const instruction = `Native tool-call capability probe: task-scale read. Use provider-native tool calling, not prose. ` +
967
+ `Call read exactly once with path exactly "${path}".`;
968
+ const stream = await this._streamForToolProbe(model, {
969
+ systemPrompt: this._nativeToolProbeSystemPrompt(instruction),
970
+ messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
971
+ tools: [NATIVE_TOOL_PROBE_READ_TOOL],
972
+ }, { textToolCallProtocol: false, maxRetries: 0, temperature: 0, maxTokens: 768 });
973
+ return this._messageHasToolCallWithStringArgument(await stream.result(), "read", "path", path);
846
974
  }
847
- async _runNativeToolProbeTrial(model, token) {
848
- const instruction = `Native tool-call capability probe. Use provider-native tool calling, not prose. ` +
975
+ async _runNativeEchoToolProbeTrial(model, token) {
976
+ const instruction = `Native tool-call capability probe: echo-only. Use provider-native tool calling, not prose. ` +
849
977
  `Call echo with data exactly "${token}".`;
850
978
  const stream = await this._streamForToolProbe(model, {
851
- systemPrompt: instruction,
979
+ systemPrompt: this._nativeToolProbeSystemPrompt(instruction),
852
980
  messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
853
981
  tools: [TEXT_TOOL_PROTOCOL_ECHO_TOOL],
854
982
  }, { textToolCallProtocol: false, maxRetries: 0, temperature: 0, maxTokens: 256 });
855
- return this._messageHasEchoProbe(await stream.result(), token);
983
+ return this._messageHasToolCallWithStringArgument(await stream.result(), "echo", "data", token);
984
+ }
985
+ async _gradeNativeToolCallingForModel(model, token) {
986
+ const path = join(tmpdir(), `pi-native-probe-${process.pid}-${Date.now()}.txt`);
987
+ writeFileSync(path, token, "utf-8");
988
+ try {
989
+ const taskPassed = await this._runNativeReadTaskProbeTrial(model, path);
990
+ if (taskPassed)
991
+ return "task";
992
+ const echoPassed = await this._runNativeEchoToolProbeTrial(model, token);
993
+ if (echoPassed)
994
+ return "echo-only";
995
+ return "absent";
996
+ }
997
+ finally {
998
+ rmSync(path, { force: true });
999
+ }
856
1000
  }
857
1001
  async _runTextProtocolTrial(model, variant, token) {
858
1002
  const stream = await this._streamForToolProbe(model, this._textProtocolCalibrationContext(variant, token), {
@@ -933,12 +1077,17 @@ export class AgentSession {
933
1077
  return `${model.provider}/${model.id}`;
934
1078
  }
935
1079
  _formatToolProbeReport(results) {
936
- const lines = ["Tool probe results:", "Model | Verdict | Variant | Diagnostic", "--- | --- | --- | ---"];
1080
+ const lines = [
1081
+ "Tool probe results:",
1082
+ "Model | Verdict | Variant | Native grade | Diagnostic",
1083
+ "--- | --- | --- | --- | ---",
1084
+ ];
937
1085
  for (const result of results) {
938
1086
  lines.push([
939
1087
  result.model,
940
1088
  result.verdict,
941
1089
  result.variant ?? "-",
1090
+ result.nativeGrade ?? "-",
942
1091
  result.diagnostic ? result.diagnostic.replace(/\s+/g, " ").slice(0, 160) : "-",
943
1092
  ].join(" | "));
944
1093
  }
@@ -950,12 +1099,23 @@ export class AgentSession {
950
1099
  async _probeToolCallingForModel(model) {
951
1100
  const modelKey = this._modelRef(model);
952
1101
  const probedAt = new Date().toISOString();
1102
+ let nativeGrade = "absent";
953
1103
  let diagnostic;
954
1104
  try {
955
- if (await this._runNativeToolProbeTrial(model, "pi-native-probe")) {
956
- this._storeToolProbe(modelKey, { version: TEXT_TOOL_PROTOCOL_VERSION, status: "native", probedAt });
957
- return { model: modelKey, verdict: "native" };
1105
+ nativeGrade = await this._gradeNativeToolCallingForModel(model, "pi-native-probe");
1106
+ if (nativeGrade === "task") {
1107
+ this._storeToolProbe(modelKey, {
1108
+ version: TEXT_TOOL_PROTOCOL_VERSION,
1109
+ status: "native",
1110
+ probedAt,
1111
+ nativeGrade,
1112
+ });
1113
+ return { model: modelKey, verdict: "native", nativeGrade };
958
1114
  }
1115
+ diagnostic =
1116
+ nativeGrade === "echo-only"
1117
+ ? "Native echo probe passed but task-scale read probe failed."
1118
+ : "Native task-scale read and echo probes did not produce provider-native tool calls.";
959
1119
  }
960
1120
  catch (error) {
961
1121
  diagnostic = error instanceof Error ? error.message : String(error);
@@ -968,16 +1128,24 @@ export class AgentSession {
968
1128
  status: "text-protocol",
969
1129
  probedAt: calibrated.calibratedAt,
970
1130
  variant: calibrated.variant,
1131
+ nativeGrade,
1132
+ diagnostic,
971
1133
  });
972
- return { model: modelKey, verdict: "text-protocol", variant: calibrated.variant };
1134
+ return { model: modelKey, verdict: "text-protocol", variant: calibrated.variant, nativeGrade, diagnostic };
973
1135
  }
974
- diagnostic ??= `Text protocol variants failed: ${calibrated.variantsTried.join(", ")}`;
1136
+ diagnostic = `${diagnostic ? `${diagnostic} ` : ""}Text protocol variants failed: ${calibrated.variantsTried.join(", ")}`;
975
1137
  }
976
1138
  catch (error) {
977
1139
  diagnostic = error instanceof Error ? error.message : String(error);
978
1140
  }
979
- this._storeToolProbe(modelKey, { version: TEXT_TOOL_PROTOCOL_VERSION, status: "none", probedAt, diagnostic });
980
- return { model: modelKey, verdict: "none", diagnostic };
1141
+ this._storeToolProbe(modelKey, {
1142
+ version: TEXT_TOOL_PROTOCOL_VERSION,
1143
+ status: "none",
1144
+ probedAt,
1145
+ nativeGrade,
1146
+ diagnostic,
1147
+ });
1148
+ return { model: modelKey, verdict: "none", nativeGrade, diagnostic };
981
1149
  }
982
1150
  async _resolveToolProbeModels(target) {
983
1151
  const trimmed = target?.trim();
@@ -1687,6 +1855,7 @@ export class AgentSession {
1687
1855
  this.abortCompaction();
1688
1856
  this.abortBranchSummary();
1689
1857
  this.abortBash();
1858
+ this._cancelPrefixWarm();
1690
1859
  this.agent.abort();
1691
1860
  // R8: stop any deployment-registered gateway channels / schedulers.
1692
1861
  void this._gatewayRegistry.stop().catch(() => { });
@@ -1935,6 +2104,9 @@ export class AgentSession {
1935
2104
  getLocalRuntime(baseUrl) {
1936
2105
  return this._localRuntimeController.getLocalRuntime(baseUrl);
1937
2106
  }
2107
+ getTransformersRuntime(modelId, baseUrl) {
2108
+ return this._localRuntimeController.getTransformersRuntime(modelId, baseUrl);
2109
+ }
1938
2110
  /** models.json registers a local model's baseUrl as `<server>/v1` (OpenAI-compat); the runtime's
1939
2111
  * own health/boot endpoints are on the Ollama-native server root. Delegates to
1940
2112
  * {@link LocalRuntimeController}; kept here for `_warnIfManualModelChoiceIsRisky`'s own use. */
@@ -2010,6 +2182,7 @@ export class AgentSession {
2010
2182
  }
2011
2183
  async _promptUnserialized(text, options) {
2012
2184
  this._applyToolRepairLayerSettings();
2185
+ this._cancelPrefixWarm();
2013
2186
  const expandPromptTemplates = options?.expandPromptTemplates ?? true;
2014
2187
  const processSlashCommands = options?.processSlashCommands ?? expandPromptTemplates;
2015
2188
  const preflightResult = options?.preflightResult;
@@ -2521,10 +2694,13 @@ export class AgentSession {
2521
2694
  // Model Management
2522
2695
  // =========================================================================
2523
2696
  async setModel(model, options = {}) {
2524
- return this._modelSelection.setModel(model, options);
2697
+ await this._modelSelection.setModel(model, options);
2698
+ this._scheduleLocalPrefixWarm(this.agent.state.model, "selection");
2525
2699
  }
2526
2700
  async cycleModel(direction = "forward") {
2527
- return this._modelSelection.cycleModel(direction);
2701
+ const result = await this._modelSelection.cycleModel(direction);
2702
+ this._scheduleLocalPrefixWarm(result?.model, "selection");
2703
+ return result;
2528
2704
  }
2529
2705
  // =========================================================================
2530
2706
  // Thinking Level Management