pi-llama-cpp 0.8.2 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -4
- package/package.json +2 -1
- package/src/constants.ts +2 -1
- package/src/managers/events.ts +1 -1
- package/src/models/baseModel.ts +1 -0
- package/src/resolver.ts +5 -5
- package/tests/events.test.ts +5 -4
- package/tests/resolver.test.ts +1 -1
- package/src/interfaces/levels.ts +0 -7
package/README.md
CHANGED
|
@@ -164,7 +164,7 @@ When browsing models via the `/models` command, you can:
|
|
|
164
164
|
### Thinking Budgets
|
|
165
165
|
|
|
166
166
|
The extension supports configurable **thinking budgets** that control how many tokens the model allocates to its reasoning/thinking process.
|
|
167
|
-
This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh).
|
|
167
|
+
This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh, max).
|
|
168
168
|
|
|
169
169
|
| Level | Tokens | Description |
|
|
170
170
|
| --------- | ------ | ---------------------------- |
|
|
@@ -173,7 +173,8 @@ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, x
|
|
|
173
173
|
| `low` | 2,048 | Light reasoning |
|
|
174
174
|
| `medium` | 8,192 | Balanced reasoning (default) |
|
|
175
175
|
| `high` | 16,384 | Extended reasoning |
|
|
176
|
-
| `xhigh` |
|
|
176
|
+
| `xhigh` | 32,768 | Deep reasoning |
|
|
177
|
+
| `max` | -1 | Unlimited reasoning |
|
|
177
178
|
|
|
178
179
|
User-defined budgets can override the defaults by adding a `thinkingBudgets` object to `~/.pi/agent/settings.json` (global) or `.pi/settings.json` (per-project):
|
|
179
180
|
|
|
@@ -183,12 +184,13 @@ User-defined budgets can override the defaults by adding a `thinkingBudgets` obj
|
|
|
183
184
|
"minimal": 256,
|
|
184
185
|
"low": 1024,
|
|
185
186
|
"medium": 2048,
|
|
186
|
-
"high": 4096
|
|
187
|
+
"high": 4096,
|
|
188
|
+
"xhigh": 8192
|
|
187
189
|
}
|
|
188
190
|
}
|
|
189
191
|
```
|
|
190
192
|
|
|
191
|
-
Only `minimal`, `low`, `medium`, and `
|
|
193
|
+
Only `minimal`, `low`, `medium`, `high` and `xhigh` are configurable — `off` (0) and `max` (-1, unlimited) are fixed.
|
|
192
194
|
The extension automatically injects the appropriate `thinking_budget_tokens` into each request payload based on the selected level.
|
|
193
195
|
|
|
194
196
|
### Model Selection Event
|
|
@@ -219,5 +221,6 @@ Each model exposed to Pi includes the following defaults:
|
|
|
219
221
|
|
|
220
222
|
| Peer dependency | Purpose |
|
|
221
223
|
| --------------------------------- | ------------------- |
|
|
224
|
+
| `@earendil-works/pi-ai` | Pi AI SDK |
|
|
222
225
|
| `@earendil-works/pi-coding-agent` | Pi Coding Agent SDK |
|
|
223
226
|
| `@earendil-works/pi-tui` | Pi TUI SDK |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
]
|
|
33
33
|
},
|
|
34
34
|
"peerDependencies": {
|
|
35
|
+
"@earendil-works/pi-ai": "*",
|
|
35
36
|
"@earendil-works/pi-coding-agent": "*",
|
|
36
37
|
"@earendil-works/pi-tui": "*"
|
|
37
38
|
},
|
package/src/constants.ts
CHANGED
package/src/managers/events.ts
CHANGED
|
@@ -94,7 +94,7 @@ export class EventManager {
|
|
|
94
94
|
if (level === "off")
|
|
95
95
|
return { ...payload, chat_template_kwargs: { enable_thinking: false } };
|
|
96
96
|
|
|
97
|
-
if (level === "
|
|
97
|
+
if (level === "max") return payload;
|
|
98
98
|
|
|
99
99
|
return { ...payload, thinking_budget_tokens };
|
|
100
100
|
}
|
package/src/models/baseModel.ts
CHANGED
package/src/resolver.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { ApiKeyCredential, ModelThinkingLevel } from "@earendil-works/pi-ai";
|
|
1
2
|
import {
|
|
2
3
|
getAgentDir,
|
|
3
4
|
readStoredCredential,
|
|
@@ -10,7 +11,6 @@ import {
|
|
|
10
11
|
DEFAULT_LLAMA_SERVER_URL,
|
|
11
12
|
DEFAULT_THINKING_BUDGETS,
|
|
12
13
|
} from "./constants";
|
|
13
|
-
import { ThinkingLevel } from "./interfaces/levels";
|
|
14
14
|
|
|
15
15
|
export class ConfigResolver {
|
|
16
16
|
private warnings: string[] = [];
|
|
@@ -109,8 +109,8 @@ export class ConfigResolver {
|
|
|
109
109
|
* Resolves API key for the provider ID using Pi's stored credentials
|
|
110
110
|
*/
|
|
111
111
|
resolveApiKey(providerId: string): string {
|
|
112
|
-
const credential = readStoredCredential(providerId);
|
|
113
|
-
return credential?.
|
|
112
|
+
const credential = readStoredCredential(providerId) as ApiKeyCredential;
|
|
113
|
+
return credential?.key ?? API_KEY_PLACEHOLDER;
|
|
114
114
|
}
|
|
115
115
|
|
|
116
116
|
/**
|
|
@@ -128,7 +128,7 @@ export class ConfigResolver {
|
|
|
128
128
|
*
|
|
129
129
|
* @returns Selected level
|
|
130
130
|
*/
|
|
131
|
-
resolveThinkingLevel():
|
|
131
|
+
resolveThinkingLevel(): ModelThinkingLevel | undefined {
|
|
132
132
|
return this.settingsManager.getDefaultThinkingLevel();
|
|
133
133
|
}
|
|
134
134
|
|
|
@@ -137,7 +137,7 @@ export class ConfigResolver {
|
|
|
137
137
|
*
|
|
138
138
|
* @returns Thinking budgets
|
|
139
139
|
*/
|
|
140
|
-
resolveThinkingBudgets(): Record<
|
|
140
|
+
resolveThinkingBudgets(): Record<ModelThinkingLevel, number> {
|
|
141
141
|
const settingsBudgets = this.settingsManager.getThinkingBudgets() ?? {};
|
|
142
142
|
const availableBudgets = {
|
|
143
143
|
...DEFAULT_THINKING_BUDGETS,
|
package/tests/events.test.ts
CHANGED
|
@@ -55,7 +55,8 @@ describe("EventManager.onBeforeProviderRequest", () => {
|
|
|
55
55
|
{ level: "low", expected: { thinking_budget_tokens: 2048 } },
|
|
56
56
|
{ level: "medium", expected: { thinking_budget_tokens: 8192 } },
|
|
57
57
|
{ level: "high", expected: { thinking_budget_tokens: 16384 } },
|
|
58
|
-
{ level: "xhigh", expected: {} },
|
|
58
|
+
{ level: "xhigh", expected: { thinking_budget_tokens: 32768 } },
|
|
59
|
+
{ level: "max", expected: {} },
|
|
59
60
|
])(
|
|
60
61
|
'level "$level" should return $expected',
|
|
61
62
|
async ({ level, expected }) => {
|
|
@@ -216,10 +217,10 @@ describe("EventManager.onBeforeProviderRequest", () => {
|
|
|
216
217
|
expect(result).not.toHaveProperty("thinking_budget_tokens");
|
|
217
218
|
});
|
|
218
219
|
|
|
219
|
-
it("should not
|
|
220
|
-
mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("
|
|
220
|
+
it("should not inject budget for 'max' — unlimited reasoning", async () => {
|
|
221
|
+
mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("max");
|
|
221
222
|
mockSettingsManager.getThinkingBudgets.mockReturnValue({
|
|
222
|
-
|
|
223
|
+
max: 1,
|
|
223
224
|
} as any);
|
|
224
225
|
|
|
225
226
|
const server = createMockServer({
|
package/tests/resolver.test.ts
CHANGED
|
@@ -163,7 +163,7 @@ describe("API key resolution", () => {
|
|
|
163
163
|
});
|
|
164
164
|
|
|
165
165
|
it("should return the apiKey when present in credential", () => {
|
|
166
|
-
mockReadStoredCredential.mockReturnValue({
|
|
166
|
+
mockReadStoredCredential.mockReturnValue({ key: "test-api-key" });
|
|
167
167
|
|
|
168
168
|
const resolver = new ConfigResolver();
|
|
169
169
|
const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|