turbollm 1.8.4 → 1.8.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +30 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -17492,6 +17492,9 @@ function mapToOpenAI(req) {
|
|
|
17492
17492
|
const content = parts.length === 1 && parts[0].type === "text" ? parts[0].text : parts;
|
|
17493
17493
|
messages.push({ role: "user", content });
|
|
17494
17494
|
}
|
|
17495
|
+
} else if (msg2.role === "system") {
|
|
17496
|
+
const text = raw.filter((b) => b.type === "text").map((b) => b.text ?? "").join("\n");
|
|
17497
|
+
if (text) messages.push({ role: "system", content: text });
|
|
17495
17498
|
} else {
|
|
17496
17499
|
let text = "";
|
|
17497
17500
|
const toolCalls = [];
|
|
@@ -17537,6 +17540,15 @@ function mapToOpenAI(req) {
|
|
|
17537
17540
|
if (tc)
|
|
17538
17541
|
oai.tool_choice = tc.type === "auto" ? "auto" : tc.type === "any" ? "required" : { type: "function", function: { name: tc.name } };
|
|
17539
17542
|
}
|
|
17543
|
+
if (req.thinking) {
|
|
17544
|
+
const requested = req.thinking.budget_tokens;
|
|
17545
|
+
const hasMaxTokens = Number.isFinite(req.max_tokens) && (req.max_tokens ?? 0) > 0;
|
|
17546
|
+
if (Number.isFinite(requested) && (requested ?? 0) > 0) {
|
|
17547
|
+
oai.thinking_budget_tokens = hasMaxTokens ? Math.min(requested, Math.floor(req.max_tokens / 2)) : requested;
|
|
17548
|
+
} else if (hasMaxTokens) {
|
|
17549
|
+
oai.thinking_budget_tokens = Math.floor(req.max_tokens / 2);
|
|
17550
|
+
}
|
|
17551
|
+
}
|
|
17540
17552
|
return oai;
|
|
17541
17553
|
}
|
|
17542
17554
|
function extractThinkTag(content) {
|
|
@@ -17944,7 +17956,24 @@ function registerGateway(app2, d) {
|
|
|
17944
17956
|
init.duplex = "half";
|
|
17945
17957
|
}
|
|
17946
17958
|
}
|
|
17947
|
-
|
|
17959
|
+
let res;
|
|
17960
|
+
try {
|
|
17961
|
+
res = await fetch(upstream, init);
|
|
17962
|
+
} catch (e) {
|
|
17963
|
+
const err6 = e;
|
|
17964
|
+
const isAbort = err6.name === "AbortError" || ac.signal.aborted;
|
|
17965
|
+
const cause = err6.cause instanceof Error ? `: ${err6.cause.message}` : "";
|
|
17966
|
+
return c.json(
|
|
17967
|
+
{
|
|
17968
|
+
error: {
|
|
17969
|
+
message: isAbort ? "Client disconnected before the engine responded." : `${err6.message || "Engine unreachable."}${cause}`,
|
|
17970
|
+
type: "api_error",
|
|
17971
|
+
code: isAbort ? "client_disconnected" : "engine_unreachable"
|
|
17972
|
+
}
|
|
17973
|
+
},
|
|
17974
|
+
500
|
|
17975
|
+
);
|
|
17976
|
+
}
|
|
17948
17977
|
if (res.ok && res.body && isChat) {
|
|
17949
17978
|
try {
|
|
17950
17979
|
const [a, b] = res.body.tee();
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "turbollm",
|
|
3
|
-
"version": "1.8.
|
|
3
|
+
"version": "1.8.5",
|
|
4
4
|
"description": "TurboLLM — local LLM platform: run any inference engine auto-tuned to your GPU, with a web UI and OpenAI/Anthropic-compatible API. Point Claude Code at your own machine in one command.",
|
|
5
5
|
"license": "FSL-1.1-ALv2",
|
|
6
6
|
"author": "Mohit Soni",
|