@caupulican/pi-adaptative 0.81.15 → 0.81.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/dist/bundled-resources/runtimes/hf-transformers-openai-server.py +427 -0
- package/dist/bundled-resources/skills/tool-call-repair/references/repair-catalogue.md +18 -0
- package/dist/bundled-resources/skills/tool-call-repair/references/text-protocol-grammar.md +14 -7
- package/dist/core/agent-session.d.ts +15 -3
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +201 -25
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/local-runtime-controller.d.ts +17 -9
- package/dist/core/local-runtime-controller.d.ts.map +1 -1
- package/dist/core/local-runtime-controller.js +124 -20
- package/dist/core/local-runtime-controller.js.map +1 -1
- package/dist/core/models/adaptation-store.d.ts +5 -0
- package/dist/core/models/adaptation-store.d.ts.map +1 -1
- package/dist/core/models/adaptation-store.js +16 -0
- package/dist/core/models/adaptation-store.js.map +1 -1
- package/dist/core/models/context-sizing.d.ts +30 -0
- package/dist/core/models/context-sizing.d.ts.map +1 -0
- package/dist/core/models/context-sizing.js +90 -0
- package/dist/core/models/context-sizing.js.map +1 -0
- package/dist/core/models/default-model-suggestions.d.ts +4 -4
- package/dist/core/models/default-model-suggestions.d.ts.map +1 -1
- package/dist/core/models/default-model-suggestions.js +9 -0
- package/dist/core/models/default-model-suggestions.js.map +1 -1
- package/dist/core/models/local-registration.d.ts +12 -0
- package/dist/core/models/local-registration.d.ts.map +1 -1
- package/dist/core/models/local-registration.js +68 -0
- package/dist/core/models/local-registration.js.map +1 -1
- package/dist/core/models/local-runtime.d.ts +96 -1
- package/dist/core/models/local-runtime.d.ts.map +1 -1
- package/dist/core/models/local-runtime.js +338 -4
- package/dist/core/models/local-runtime.js.map +1 -1
- package/dist/core/models/model-ref.d.ts +4 -0
- package/dist/core/models/model-ref.d.ts.map +1 -1
- package/dist/core/models/model-ref.js +12 -3
- package/dist/core/models/model-ref.js.map +1 -1
- package/dist/core/models/perf-profile.d.ts +36 -0
- package/dist/core/models/perf-profile.d.ts.map +1 -0
- package/dist/core/models/perf-profile.js +163 -0
- package/dist/core/models/perf-profile.js.map +1 -0
- package/dist/core/models/runtime-arbiter.d.ts +57 -0
- package/dist/core/models/runtime-arbiter.d.ts.map +1 -0
- package/dist/core/models/runtime-arbiter.js +49 -0
- package/dist/core/models/runtime-arbiter.js.map +1 -0
- package/dist/core/tool-repair-health.d.ts.map +1 -1
- package/dist/core/tool-repair-health.js +2 -1
- package/dist/core/tool-repair-health.js.map +1 -1
- package/dist/modes/interactive/interactive-mode.d.ts +1 -0
- package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
- package/dist/modes/interactive/interactive-mode.js +4 -0
- package/dist/modes/interactive/interactive-mode.js.map +1 -1
- package/dist/modes/interactive/local-model-commands.d.ts +3 -1
- package/dist/modes/interactive/local-model-commands.d.ts.map +1 -1
- package/dist/modes/interactive/local-model-commands.js +163 -22
- package/dist/modes/interactive/local-model-commands.js.map +1 -1
- package/docs/models.md +19 -1
- package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/sandbox/package-lock.json +2 -2
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/with-deps/package-lock.json +2 -2
- package/examples/extensions/with-deps/package.json +1 -1
- package/npm-shrinkwrap.json +12 -12
- package/package.json +4 -4
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
import { readFileSync } from "node:fs";
|
|
1
|
+
import { readFileSync, rmSync, writeFileSync } from "node:fs";
|
|
2
|
+
import { tmpdir } from "node:os";
|
|
2
3
|
import { basename, dirname, join } from "node:path";
|
|
3
|
-
import { classifyFailure, compactToolResultDetailsForRetention, computeRetryDelayMs, createCustomMessage, DEFAULT_RETRY_POLICY, RetryController, sleepAbortable, withStreamIdleWatchdog, } from "@caupulican/pi-agent-core";
|
|
4
|
+
import { classifyFailure, compactToolResultDetailsForRetention, computeRetryDelayMs, createCustomMessage, DEFAULT_RETRY_POLICY, DEFAULT_STREAM_IDLE, RetryController, sleepAbortable, withStreamIdleWatchdog, } from "@caupulican/pi-agent-core";
|
|
4
5
|
import { calculateContextTokens, compact, createDeterministicCompaction, estimateContextTokens, getLatestCompactionEntry, prepareCompaction, runCompactionLoop, shouldCompact, } from "@caupulican/pi-agent-core/node";
|
|
5
6
|
import { cleanupSessionResources, formatToolRepairStandingRule, generateTextToolProtocolPrimer, isContextOverflow, parseTextToolCalls, streamSimple, } from "@caupulican/pi-ai";
|
|
6
7
|
import { Type } from "typebox";
|
|
@@ -30,6 +31,7 @@ import { deriveModelCapabilityProfile, filterToolNamesForCapability, } from "./m
|
|
|
30
31
|
import { formatModelRouterModel, ModelRouterController } from "./model-router-controller.js";
|
|
31
32
|
import { ModelSelectionController } from "./model-selection-controller.js";
|
|
32
33
|
import { ModelAdaptationStore } from "./models/adaptation-store.js";
|
|
34
|
+
import { DEFAULT_ADAPTIVE_STREAM_IDLE_CEILING_MS, estimateContextPromptTokens, resolveAdaptiveStreamIdleOptions, withModelPerfProfile, } from "./models/perf-profile.js";
|
|
33
35
|
import { ProfileFilterController } from "./profile-filter-controller.js";
|
|
34
36
|
import { expandPromptTemplate } from "./prompt-templates.js";
|
|
35
37
|
import { ReflectionController } from "./reflection-controller.js";
|
|
@@ -60,12 +62,22 @@ const MODEL_ADAPTATION_REPAIR_THRESHOLD = 3;
|
|
|
60
62
|
const TEXT_TOOL_PROTOCOL_VERSION = 1;
|
|
61
63
|
const TEXT_TOOL_PROTOCOL_TRIALS_PER_VARIANT = 2;
|
|
62
64
|
const TEXT_TOOL_PROTOCOL_PARSE_FAILURE_THRESHOLD = 3;
|
|
63
|
-
const TEXT_TOOL_PROTOCOL_VARIANTS = [
|
|
65
|
+
const TEXT_TOOL_PROTOCOL_VARIANTS = [
|
|
66
|
+
"tool-tag",
|
|
67
|
+
"tool-call",
|
|
68
|
+
"fenced-json",
|
|
69
|
+
"function-xml",
|
|
70
|
+
];
|
|
64
71
|
const TEXT_TOOL_PROTOCOL_ECHO_TOOL = {
|
|
65
72
|
name: "echo",
|
|
66
73
|
description: "Echo calibration data",
|
|
67
74
|
parameters: Type.Object({ data: Type.String() }),
|
|
68
75
|
};
|
|
76
|
+
const NATIVE_TOOL_PROBE_READ_TOOL = {
|
|
77
|
+
name: "read",
|
|
78
|
+
description: "Read file contents",
|
|
79
|
+
parameters: Type.Object({ path: Type.String() }),
|
|
80
|
+
};
|
|
69
81
|
/** Test-only override of the stream-idle bounds. Read per-request by the wiring's resolver. */
|
|
70
82
|
let streamIdleOptionsOverride;
|
|
71
83
|
/**
|
|
@@ -148,6 +160,8 @@ export class AgentSession {
|
|
|
148
160
|
_collectWorkspaceSources;
|
|
149
161
|
_localRuntimeController;
|
|
150
162
|
_modelAdaptationStore;
|
|
163
|
+
_prefixWarmer;
|
|
164
|
+
_completedPrefixWarms = new Set();
|
|
151
165
|
_repairModeSessionCounts = new Map();
|
|
152
166
|
_textProtocolParseFailures = new Map();
|
|
153
167
|
_textProtocolParseObservedThisTurn = false;
|
|
@@ -229,14 +243,35 @@ export class AgentSession {
|
|
|
229
243
|
// withStreamIdleWatchdog's contract), so no extra drain is added at this wiring site.
|
|
230
244
|
// Wrapping also breaks the `streamFn === streamSimple` identity the auth-injection checks
|
|
231
245
|
// use, so the wrapper carries a rawness marker that _isRawStreamSimple reads.
|
|
246
|
+
const agentDir = config.agentDir ?? getAgentDir();
|
|
247
|
+
const modelAdaptationStore = ModelAdaptationStore.forAgentDir(agentDir);
|
|
232
248
|
const baseStreamFn = this.agent.streamFn;
|
|
249
|
+
const profiledStreamFn = withModelPerfProfile(baseStreamFn, {
|
|
250
|
+
modelKey: (model) => formatModelRouterModel(model),
|
|
251
|
+
recordSample: (modelKey, sample) => {
|
|
252
|
+
modelAdaptationStore.recordPerfSample(modelKey, sample);
|
|
253
|
+
},
|
|
254
|
+
});
|
|
233
255
|
// `this.settingsManager` is assigned below; the resolver closes over the config reference
|
|
234
256
|
// because the wrapper must be installed before that assignment runs.
|
|
235
257
|
const stallSettingsSource = config.settingsManager;
|
|
236
|
-
this.agent.streamFn = tagRawness(withStreamIdleWatchdog(
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
258
|
+
this.agent.streamFn = tagRawness(withStreamIdleWatchdog(profiledStreamFn, (model, context) => {
|
|
259
|
+
const configured = {
|
|
260
|
+
...stallSettingsSource.getStreamStallSettings(),
|
|
261
|
+
...streamIdleOptionsOverride,
|
|
262
|
+
};
|
|
263
|
+
const httpIdleTimeoutMs = stallSettingsSource.getHttpIdleTimeoutMs();
|
|
264
|
+
const profile = modelAdaptationStore.get(formatModelRouterModel(model)).perf;
|
|
265
|
+
const adaptive = resolveAdaptiveStreamIdleOptions({
|
|
266
|
+
base: { ...DEFAULT_STREAM_IDLE, ...configured },
|
|
267
|
+
profile,
|
|
268
|
+
promptTokens: estimateContextPromptTokens(context),
|
|
269
|
+
ceilingMs: httpIdleTimeoutMs === 0
|
|
270
|
+
? DEFAULT_ADAPTIVE_STREAM_IDLE_CEILING_MS
|
|
271
|
+
: Math.max(DEFAULT_STREAM_IDLE.quietIdleMs, httpIdleTimeoutMs - 60_000),
|
|
272
|
+
});
|
|
273
|
+
return { ...configured, ...adaptive };
|
|
274
|
+
}), baseStreamFn === streamSimple);
|
|
240
275
|
this.sessionManager = config.sessionManager;
|
|
241
276
|
this.settingsManager = config.settingsManager;
|
|
242
277
|
// Auto-retry rides the reliability kernel: the controller owns the attempt counter and the
|
|
@@ -260,8 +295,8 @@ export class AgentSession {
|
|
|
260
295
|
this._resourceLoader = config.resourceLoader;
|
|
261
296
|
this._customTools = config.customTools ?? [];
|
|
262
297
|
this._cwd = config.cwd;
|
|
263
|
-
this._agentDir =
|
|
264
|
-
this._modelAdaptationStore =
|
|
298
|
+
this._agentDir = agentDir;
|
|
299
|
+
this._modelAdaptationStore = modelAdaptationStore;
|
|
265
300
|
this.agent.onTextToolProtocolParse = (event) => this._handleTextToolProtocolParse(event);
|
|
266
301
|
this._applyToolRepairLayerSettings();
|
|
267
302
|
this._collectWorkspaceSources = config.collectWorkspaceSources ?? collectWorkspaceSources;
|
|
@@ -578,11 +613,82 @@ export class AgentSession {
|
|
|
578
613
|
activeToolNames: this._initialActiveToolNames,
|
|
579
614
|
includeAllExtensionTools: true,
|
|
580
615
|
});
|
|
616
|
+
this._scheduleLocalPrefixWarm(this.agent.state.model, "session-start");
|
|
581
617
|
}
|
|
582
618
|
/** Model registry for API key resolution and model discovery */
|
|
583
619
|
get modelRegistry() {
|
|
584
620
|
return this._modelRegistry;
|
|
585
621
|
}
|
|
622
|
+
_scheduleLocalPrefixWarm(model, _reason) {
|
|
623
|
+
if (!model || !this._isWarmableLocalModel(model))
|
|
624
|
+
return;
|
|
625
|
+
const modelKey = formatModelRouterModel(model);
|
|
626
|
+
if (this._completedPrefixWarms.has(modelKey) || this._prefixWarmer?.modelKey === modelKey)
|
|
627
|
+
return;
|
|
628
|
+
this._cancelPrefixWarm();
|
|
629
|
+
const controller = new AbortController();
|
|
630
|
+
const timer = setTimeout(() => {
|
|
631
|
+
const warmer = this._prefixWarmer;
|
|
632
|
+
if (!warmer || warmer.controller !== controller || controller.signal.aborted)
|
|
633
|
+
return;
|
|
634
|
+
warmer.timer = undefined;
|
|
635
|
+
void this._runLocalPrefixWarm(model, modelKey, controller);
|
|
636
|
+
}, 0);
|
|
637
|
+
timer.unref?.();
|
|
638
|
+
this._prefixWarmer = { modelKey, controller, timer };
|
|
639
|
+
}
|
|
640
|
+
_cancelPrefixWarm() {
|
|
641
|
+
const warmer = this._prefixWarmer;
|
|
642
|
+
if (!warmer)
|
|
643
|
+
return;
|
|
644
|
+
if (warmer.timer)
|
|
645
|
+
clearTimeout(warmer.timer);
|
|
646
|
+
warmer.controller.abort(new Error("prefix warmer preempted"));
|
|
647
|
+
this._prefixWarmer = undefined;
|
|
648
|
+
}
|
|
649
|
+
async _runLocalPrefixWarm(model, modelKey, controller) {
|
|
650
|
+
try {
|
|
651
|
+
const options = {
|
|
652
|
+
maxTokens: 1,
|
|
653
|
+
signal: controller.signal,
|
|
654
|
+
onPayload: this.agent.onPayload,
|
|
655
|
+
onResponse: this.agent.onResponse,
|
|
656
|
+
};
|
|
657
|
+
if (this._isRawStreamSimple(this.agent.streamFn)) {
|
|
658
|
+
const auth = await this._getRequiredRequestAuth(model);
|
|
659
|
+
options.apiKey = auth.apiKey;
|
|
660
|
+
options.headers = auth.headers;
|
|
661
|
+
}
|
|
662
|
+
if (controller.signal.aborted)
|
|
663
|
+
return;
|
|
664
|
+
const stream = await this.agent.streamFn(model, {
|
|
665
|
+
systemPrompt: this._baseSystemPrompt,
|
|
666
|
+
tools: this.agent.state.tools,
|
|
667
|
+
messages: [],
|
|
668
|
+
}, options);
|
|
669
|
+
await stream.result();
|
|
670
|
+
if (!controller.signal.aborted)
|
|
671
|
+
this._completedPrefixWarms.add(modelKey);
|
|
672
|
+
}
|
|
673
|
+
catch {
|
|
674
|
+
// Best-effort cache warm only; a miss must never affect the real turn.
|
|
675
|
+
}
|
|
676
|
+
finally {
|
|
677
|
+
if (this._prefixWarmer?.controller === controller)
|
|
678
|
+
this._prefixWarmer = undefined;
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
_isWarmableLocalModel(model) {
|
|
682
|
+
if (model.api !== "openai-completions")
|
|
683
|
+
return false;
|
|
684
|
+
try {
|
|
685
|
+
const hostname = new URL(model.baseUrl).hostname.toLowerCase();
|
|
686
|
+
return hostname === "localhost" || hostname === "127.0.0.1" || hostname === "::1" || hostname === "[::1]";
|
|
687
|
+
}
|
|
688
|
+
catch {
|
|
689
|
+
return false;
|
|
690
|
+
}
|
|
691
|
+
}
|
|
586
692
|
/**
|
|
587
693
|
* True when the session's stream fn is the raw `streamSimple` provider entry (directly, or as the
|
|
588
694
|
* base wrapped by the idle watchdog at construction). Callers use this to decide whether request
|
|
@@ -841,18 +947,56 @@ export class AgentSession {
|
|
|
841
947
|
messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
|
|
842
948
|
};
|
|
843
949
|
}
|
|
844
|
-
|
|
845
|
-
return message.content.some((block) =>
|
|
950
|
+
_messageHasToolCallWithStringArgument(message, toolName, argName, argValue) {
|
|
951
|
+
return message.content.some((block) => {
|
|
952
|
+
if (block.type !== "toolCall" || block.name !== toolName)
|
|
953
|
+
return false;
|
|
954
|
+
const args = block.arguments;
|
|
955
|
+
return (typeof args === "object" &&
|
|
956
|
+
args !== null &&
|
|
957
|
+
!Array.isArray(args) &&
|
|
958
|
+
args[argName] === argValue);
|
|
959
|
+
});
|
|
960
|
+
}
|
|
961
|
+
_nativeToolProbeSystemPrompt(instruction) {
|
|
962
|
+
const base = (this.agent.state.systemPrompt ?? "").trim();
|
|
963
|
+
return base ? `${base}\n\n${instruction}` : instruction;
|
|
964
|
+
}
|
|
965
|
+
async _runNativeReadTaskProbeTrial(model, path) {
|
|
966
|
+
const instruction = `Native tool-call capability probe: task-scale read. Use provider-native tool calling, not prose. ` +
|
|
967
|
+
`Call read exactly once with path exactly "${path}".`;
|
|
968
|
+
const stream = await this._streamForToolProbe(model, {
|
|
969
|
+
systemPrompt: this._nativeToolProbeSystemPrompt(instruction),
|
|
970
|
+
messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
|
|
971
|
+
tools: [NATIVE_TOOL_PROBE_READ_TOOL],
|
|
972
|
+
}, { textToolCallProtocol: false, maxRetries: 0, temperature: 0, maxTokens: 768 });
|
|
973
|
+
return this._messageHasToolCallWithStringArgument(await stream.result(), "read", "path", path);
|
|
846
974
|
}
|
|
847
|
-
async
|
|
848
|
-
const instruction = `Native tool-call capability probe. Use provider-native tool calling, not prose. ` +
|
|
975
|
+
async _runNativeEchoToolProbeTrial(model, token) {
|
|
976
|
+
const instruction = `Native tool-call capability probe: echo-only. Use provider-native tool calling, not prose. ` +
|
|
849
977
|
`Call echo with data exactly "${token}".`;
|
|
850
978
|
const stream = await this._streamForToolProbe(model, {
|
|
851
|
-
systemPrompt: instruction,
|
|
979
|
+
systemPrompt: this._nativeToolProbeSystemPrompt(instruction),
|
|
852
980
|
messages: [{ role: "user", content: [{ type: "text", text: instruction }], timestamp: Date.now() }],
|
|
853
981
|
tools: [TEXT_TOOL_PROTOCOL_ECHO_TOOL],
|
|
854
982
|
}, { textToolCallProtocol: false, maxRetries: 0, temperature: 0, maxTokens: 256 });
|
|
855
|
-
return this.
|
|
983
|
+
return this._messageHasToolCallWithStringArgument(await stream.result(), "echo", "data", token);
|
|
984
|
+
}
|
|
985
|
+
async _gradeNativeToolCallingForModel(model, token) {
|
|
986
|
+
const path = join(tmpdir(), `pi-native-probe-${process.pid}-${Date.now()}.txt`);
|
|
987
|
+
writeFileSync(path, token, "utf-8");
|
|
988
|
+
try {
|
|
989
|
+
const taskPassed = await this._runNativeReadTaskProbeTrial(model, path);
|
|
990
|
+
if (taskPassed)
|
|
991
|
+
return "task";
|
|
992
|
+
const echoPassed = await this._runNativeEchoToolProbeTrial(model, token);
|
|
993
|
+
if (echoPassed)
|
|
994
|
+
return "echo-only";
|
|
995
|
+
return "absent";
|
|
996
|
+
}
|
|
997
|
+
finally {
|
|
998
|
+
rmSync(path, { force: true });
|
|
999
|
+
}
|
|
856
1000
|
}
|
|
857
1001
|
async _runTextProtocolTrial(model, variant, token) {
|
|
858
1002
|
const stream = await this._streamForToolProbe(model, this._textProtocolCalibrationContext(variant, token), {
|
|
@@ -933,12 +1077,17 @@ export class AgentSession {
|
|
|
933
1077
|
return `${model.provider}/${model.id}`;
|
|
934
1078
|
}
|
|
935
1079
|
_formatToolProbeReport(results) {
|
|
936
|
-
const lines = [
|
|
1080
|
+
const lines = [
|
|
1081
|
+
"Tool probe results:",
|
|
1082
|
+
"Model | Verdict | Variant | Native grade | Diagnostic",
|
|
1083
|
+
"--- | --- | --- | --- | ---",
|
|
1084
|
+
];
|
|
937
1085
|
for (const result of results) {
|
|
938
1086
|
lines.push([
|
|
939
1087
|
result.model,
|
|
940
1088
|
result.verdict,
|
|
941
1089
|
result.variant ?? "-",
|
|
1090
|
+
result.nativeGrade ?? "-",
|
|
942
1091
|
result.diagnostic ? result.diagnostic.replace(/\s+/g, " ").slice(0, 160) : "-",
|
|
943
1092
|
].join(" | "));
|
|
944
1093
|
}
|
|
@@ -950,12 +1099,23 @@ export class AgentSession {
|
|
|
950
1099
|
async _probeToolCallingForModel(model) {
|
|
951
1100
|
const modelKey = this._modelRef(model);
|
|
952
1101
|
const probedAt = new Date().toISOString();
|
|
1102
|
+
let nativeGrade = "absent";
|
|
953
1103
|
let diagnostic;
|
|
954
1104
|
try {
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
1105
|
+
nativeGrade = await this._gradeNativeToolCallingForModel(model, "pi-native-probe");
|
|
1106
|
+
if (nativeGrade === "task") {
|
|
1107
|
+
this._storeToolProbe(modelKey, {
|
|
1108
|
+
version: TEXT_TOOL_PROTOCOL_VERSION,
|
|
1109
|
+
status: "native",
|
|
1110
|
+
probedAt,
|
|
1111
|
+
nativeGrade,
|
|
1112
|
+
});
|
|
1113
|
+
return { model: modelKey, verdict: "native", nativeGrade };
|
|
958
1114
|
}
|
|
1115
|
+
diagnostic =
|
|
1116
|
+
nativeGrade === "echo-only"
|
|
1117
|
+
? "Native echo probe passed but task-scale read probe failed."
|
|
1118
|
+
: "Native task-scale read and echo probes did not produce provider-native tool calls.";
|
|
959
1119
|
}
|
|
960
1120
|
catch (error) {
|
|
961
1121
|
diagnostic = error instanceof Error ? error.message : String(error);
|
|
@@ -968,16 +1128,24 @@ export class AgentSession {
|
|
|
968
1128
|
status: "text-protocol",
|
|
969
1129
|
probedAt: calibrated.calibratedAt,
|
|
970
1130
|
variant: calibrated.variant,
|
|
1131
|
+
nativeGrade,
|
|
1132
|
+
diagnostic,
|
|
971
1133
|
});
|
|
972
|
-
return { model: modelKey, verdict: "text-protocol", variant: calibrated.variant };
|
|
1134
|
+
return { model: modelKey, verdict: "text-protocol", variant: calibrated.variant, nativeGrade, diagnostic };
|
|
973
1135
|
}
|
|
974
|
-
diagnostic
|
|
1136
|
+
diagnostic = `${diagnostic ? `${diagnostic} ` : ""}Text protocol variants failed: ${calibrated.variantsTried.join(", ")}`;
|
|
975
1137
|
}
|
|
976
1138
|
catch (error) {
|
|
977
1139
|
diagnostic = error instanceof Error ? error.message : String(error);
|
|
978
1140
|
}
|
|
979
|
-
this._storeToolProbe(modelKey, {
|
|
980
|
-
|
|
1141
|
+
this._storeToolProbe(modelKey, {
|
|
1142
|
+
version: TEXT_TOOL_PROTOCOL_VERSION,
|
|
1143
|
+
status: "none",
|
|
1144
|
+
probedAt,
|
|
1145
|
+
nativeGrade,
|
|
1146
|
+
diagnostic,
|
|
1147
|
+
});
|
|
1148
|
+
return { model: modelKey, verdict: "none", nativeGrade, diagnostic };
|
|
981
1149
|
}
|
|
982
1150
|
async _resolveToolProbeModels(target) {
|
|
983
1151
|
const trimmed = target?.trim();
|
|
@@ -1687,6 +1855,7 @@ export class AgentSession {
|
|
|
1687
1855
|
this.abortCompaction();
|
|
1688
1856
|
this.abortBranchSummary();
|
|
1689
1857
|
this.abortBash();
|
|
1858
|
+
this._cancelPrefixWarm();
|
|
1690
1859
|
this.agent.abort();
|
|
1691
1860
|
// R8: stop any deployment-registered gateway channels / schedulers.
|
|
1692
1861
|
void this._gatewayRegistry.stop().catch(() => { });
|
|
@@ -1935,6 +2104,9 @@ export class AgentSession {
|
|
|
1935
2104
|
getLocalRuntime(baseUrl) {
|
|
1936
2105
|
return this._localRuntimeController.getLocalRuntime(baseUrl);
|
|
1937
2106
|
}
|
|
2107
|
+
getTransformersRuntime(modelId, baseUrl) {
|
|
2108
|
+
return this._localRuntimeController.getTransformersRuntime(modelId, baseUrl);
|
|
2109
|
+
}
|
|
1938
2110
|
/** models.json registers a local model's baseUrl as `<server>/v1` (OpenAI-compat); the runtime's
|
|
1939
2111
|
* own health/boot endpoints are on the Ollama-native server root. Delegates to
|
|
1940
2112
|
* {@link LocalRuntimeController}; kept here for `_warnIfManualModelChoiceIsRisky`'s own use. */
|
|
@@ -2010,6 +2182,7 @@ export class AgentSession {
|
|
|
2010
2182
|
}
|
|
2011
2183
|
async _promptUnserialized(text, options) {
|
|
2012
2184
|
this._applyToolRepairLayerSettings();
|
|
2185
|
+
this._cancelPrefixWarm();
|
|
2013
2186
|
const expandPromptTemplates = options?.expandPromptTemplates ?? true;
|
|
2014
2187
|
const processSlashCommands = options?.processSlashCommands ?? expandPromptTemplates;
|
|
2015
2188
|
const preflightResult = options?.preflightResult;
|
|
@@ -2521,10 +2694,13 @@ export class AgentSession {
|
|
|
2521
2694
|
// Model Management
|
|
2522
2695
|
// =========================================================================
|
|
2523
2696
|
async setModel(model, options = {}) {
|
|
2524
|
-
|
|
2697
|
+
await this._modelSelection.setModel(model, options);
|
|
2698
|
+
this._scheduleLocalPrefixWarm(this.agent.state.model, "selection");
|
|
2525
2699
|
}
|
|
2526
2700
|
async cycleModel(direction = "forward") {
|
|
2527
|
-
|
|
2701
|
+
const result = await this._modelSelection.cycleModel(direction);
|
|
2702
|
+
this._scheduleLocalPrefixWarm(result?.model, "selection");
|
|
2703
|
+
return result;
|
|
2528
2704
|
}
|
|
2529
2705
|
// =========================================================================
|
|
2530
2706
|
// Thinking Level Management
|