pi-llama-cpp 0.8.1 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -4
- package/package.json +4 -3
- package/src/constants.ts +2 -1
- package/src/managers/events.ts +1 -1
- package/src/models/baseModel.ts +1 -0
- package/src/resolver.ts +8 -11
- package/tests/events.test.ts +5 -4
- package/tests/resolver.test.ts +20 -30
- package/src/interfaces/levels.ts +0 -7
package/README.md
CHANGED
|
@@ -164,7 +164,7 @@ When browsing models via the `/models` command, you can:
|
|
|
164
164
|
### Thinking Budgets
|
|
165
165
|
|
|
166
166
|
The extension supports configurable **thinking budgets** that control how many tokens the model allocates to its reasoning/thinking process.
|
|
167
|
-
This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh).
|
|
167
|
+
This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh, max).
|
|
168
168
|
|
|
169
169
|
| Level | Tokens | Description |
|
|
170
170
|
| --------- | ------ | ---------------------------- |
|
|
@@ -173,7 +173,8 @@ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, x
|
|
|
173
173
|
| `low` | 2,048 | Light reasoning |
|
|
174
174
|
| `medium` | 8,192 | Balanced reasoning (default) |
|
|
175
175
|
| `high` | 16,384 | Extended reasoning |
|
|
176
|
-
| `xhigh` |
|
|
176
|
+
| `xhigh` | 32,768 | Deep reasoning |
|
|
177
|
+
| `max` | -1 | Unlimited reasoning |
|
|
177
178
|
|
|
178
179
|
User-defined budgets can override the defaults by adding a `thinkingBudgets` object to `~/.pi/agent/settings.json` (global) or `.pi/settings.json` (per-project):
|
|
179
180
|
|
|
@@ -183,12 +184,13 @@ User-defined budgets can override the defaults by adding a `thinkingBudgets` obj
|
|
|
183
184
|
"minimal": 256,
|
|
184
185
|
"low": 1024,
|
|
185
186
|
"medium": 2048,
|
|
186
|
-
"high": 4096
|
|
187
|
+
"high": 4096,
|
|
188
|
+
"xhigh": 8192
|
|
187
189
|
}
|
|
188
190
|
}
|
|
189
191
|
```
|
|
190
192
|
|
|
191
|
-
Only `minimal`, `low`, `medium`, and `
|
|
193
|
+
Only `minimal`, `low`, `medium`, `high` and `xhigh` are configurable — `off` (0) and `max` (-1, unlimited) are fixed.
|
|
192
194
|
The extension automatically injects the appropriate `thinking_budget_tokens` into each request payload based on the selected level.
|
|
193
195
|
|
|
194
196
|
### Model Selection Event
|
|
@@ -219,5 +221,6 @@ Each model exposed to Pi includes the following defaults:
|
|
|
219
221
|
|
|
220
222
|
| Peer dependency | Purpose |
|
|
221
223
|
| --------------------------------- | ------------------- |
|
|
224
|
+
| `@earendil-works/pi-ai` | Pi AI SDK |
|
|
222
225
|
| `@earendil-works/pi-coding-agent` | Pi Coding Agent SDK |
|
|
223
226
|
| `@earendil-works/pi-tui` | Pi TUI SDK |
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -32,12 +32,13 @@
|
|
|
32
32
|
]
|
|
33
33
|
},
|
|
34
34
|
"peerDependencies": {
|
|
35
|
+
"@earendil-works/pi-ai": "*",
|
|
35
36
|
"@earendil-works/pi-coding-agent": "*",
|
|
36
37
|
"@earendil-works/pi-tui": "*"
|
|
37
38
|
},
|
|
38
39
|
"devDependencies": {
|
|
39
|
-
"@types/node": "^26.
|
|
40
|
+
"@types/node": "^26.1.1",
|
|
40
41
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
41
|
-
"vitest": "^4.1.
|
|
42
|
+
"vitest": "^4.1.10"
|
|
42
43
|
}
|
|
43
44
|
}
|
package/src/constants.ts
CHANGED
package/src/managers/events.ts
CHANGED
|
@@ -94,7 +94,7 @@ export class EventManager {
|
|
|
94
94
|
if (level === "off")
|
|
95
95
|
return { ...payload, chat_template_kwargs: { enable_thinking: false } };
|
|
96
96
|
|
|
97
|
-
if (level === "
|
|
97
|
+
if (level === "max") return payload;
|
|
98
98
|
|
|
99
99
|
return { ...payload, thinking_budget_tokens };
|
|
100
100
|
}
|
package/src/models/baseModel.ts
CHANGED
package/src/resolver.ts
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
|
+
import { ApiKeyCredential, ModelThinkingLevel } from "@earendil-works/pi-ai";
|
|
1
2
|
import {
|
|
2
|
-
AuthStorage,
|
|
3
3
|
getAgentDir,
|
|
4
|
+
readStoredCredential,
|
|
4
5
|
SettingsManager,
|
|
5
6
|
} from "@earendil-works/pi-coding-agent";
|
|
6
7
|
import { readFile } from "node:fs/promises";
|
|
@@ -10,13 +11,11 @@ import {
|
|
|
10
11
|
DEFAULT_LLAMA_SERVER_URL,
|
|
11
12
|
DEFAULT_THINKING_BUDGETS,
|
|
12
13
|
} from "./constants";
|
|
13
|
-
import { ThinkingLevel } from "./interfaces/levels";
|
|
14
14
|
|
|
15
15
|
export class ConfigResolver {
|
|
16
16
|
private warnings: string[] = [];
|
|
17
17
|
|
|
18
18
|
private cachedUrls: string[] = [];
|
|
19
|
-
private authStorage = AuthStorage.create(join(getAgentDir(), "auth.json"));
|
|
20
19
|
private settingsManager = SettingsManager.create(
|
|
21
20
|
process.cwd(),
|
|
22
21
|
getAgentDir(),
|
|
@@ -107,13 +106,11 @@ export class ConfigResolver {
|
|
|
107
106
|
}
|
|
108
107
|
|
|
109
108
|
/**
|
|
110
|
-
* Resolves API key for the provider ID using Pi's
|
|
109
|
+
* Resolves API key for the provider ID using Pi's stored credentials
|
|
111
110
|
*/
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
return apiKey ?? API_KEY_PLACEHOLDER;
|
|
111
|
+
resolveApiKey(providerId: string): string {
|
|
112
|
+
const credential = readStoredCredential(providerId) as ApiKeyCredential;
|
|
113
|
+
return credential?.key ?? API_KEY_PLACEHOLDER;
|
|
117
114
|
}
|
|
118
115
|
|
|
119
116
|
/**
|
|
@@ -131,7 +128,7 @@ export class ConfigResolver {
|
|
|
131
128
|
*
|
|
132
129
|
* @returns Selected level
|
|
133
130
|
*/
|
|
134
|
-
resolveThinkingLevel():
|
|
131
|
+
resolveThinkingLevel(): ModelThinkingLevel | undefined {
|
|
135
132
|
return this.settingsManager.getDefaultThinkingLevel();
|
|
136
133
|
}
|
|
137
134
|
|
|
@@ -140,7 +137,7 @@ export class ConfigResolver {
|
|
|
140
137
|
*
|
|
141
138
|
* @returns Thinking budgets
|
|
142
139
|
*/
|
|
143
|
-
resolveThinkingBudgets(): Record<
|
|
140
|
+
resolveThinkingBudgets(): Record<ModelThinkingLevel, number> {
|
|
144
141
|
const settingsBudgets = this.settingsManager.getThinkingBudgets() ?? {};
|
|
145
142
|
const availableBudgets = {
|
|
146
143
|
...DEFAULT_THINKING_BUDGETS,
|
package/tests/events.test.ts
CHANGED
|
@@ -55,7 +55,8 @@ describe("EventManager.onBeforeProviderRequest", () => {
|
|
|
55
55
|
{ level: "low", expected: { thinking_budget_tokens: 2048 } },
|
|
56
56
|
{ level: "medium", expected: { thinking_budget_tokens: 8192 } },
|
|
57
57
|
{ level: "high", expected: { thinking_budget_tokens: 16384 } },
|
|
58
|
-
{ level: "xhigh", expected: {} },
|
|
58
|
+
{ level: "xhigh", expected: { thinking_budget_tokens: 32768 } },
|
|
59
|
+
{ level: "max", expected: {} },
|
|
59
60
|
])(
|
|
60
61
|
'level "$level" should return $expected',
|
|
61
62
|
async ({ level, expected }) => {
|
|
@@ -216,10 +217,10 @@ describe("EventManager.onBeforeProviderRequest", () => {
|
|
|
216
217
|
expect(result).not.toHaveProperty("thinking_budget_tokens");
|
|
217
218
|
});
|
|
218
219
|
|
|
219
|
-
it("should not
|
|
220
|
-
mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("
|
|
220
|
+
it("should not inject budget for 'max' — unlimited reasoning", async () => {
|
|
221
|
+
mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("max");
|
|
221
222
|
mockSettingsManager.getThinkingBudgets.mockReturnValue({
|
|
222
|
-
|
|
223
|
+
max: 1,
|
|
223
224
|
} as any);
|
|
224
225
|
|
|
225
226
|
const server = createMockServer({
|
package/tests/resolver.test.ts
CHANGED
|
@@ -5,22 +5,18 @@ import {
|
|
|
5
5
|
} from "../src/constants";
|
|
6
6
|
|
|
7
7
|
// Hoisted mock instances — survives vi.resetModules()
|
|
8
|
-
const
|
|
9
|
-
reload: vi.fn(),
|
|
10
|
-
getApiKey: vi.fn(),
|
|
11
|
-
}));
|
|
8
|
+
const mockReadStoredCredential = vi.hoisted(() => vi.fn());
|
|
12
9
|
|
|
13
10
|
const mockSettingsManager = vi.hoisted(() => ({
|
|
14
11
|
getProjectSettings: vi.fn(),
|
|
15
12
|
getGlobalSettings: vi.fn(),
|
|
16
13
|
}));
|
|
17
14
|
|
|
18
|
-
// Mock getAgentDir,
|
|
15
|
+
// Mock getAgentDir, readStoredCredential, and SettingsManager before importing resolver
|
|
19
16
|
vi.mock("@earendil-works/pi-coding-agent", () => ({
|
|
20
17
|
getAgentDir: vi.fn().mockReturnValue("/fake/agent/dir"),
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
},
|
|
18
|
+
readStoredCredential: (...args: unknown[]) =>
|
|
19
|
+
mockReadStoredCredential(...args),
|
|
24
20
|
SettingsManager: {
|
|
25
21
|
create: vi.fn().mockReturnValue(mockSettingsManager),
|
|
26
22
|
},
|
|
@@ -145,50 +141,44 @@ describe("API key resolution", () => {
|
|
|
145
141
|
beforeEach(() => {
|
|
146
142
|
vi.clearAllMocks();
|
|
147
143
|
mockGetAgentDir.mockReturnValue("/fake/agent/dir");
|
|
148
|
-
|
|
149
|
-
mockAuthStorage.getApiKey.mockResolvedValue(undefined);
|
|
144
|
+
mockReadStoredCredential.mockReturnValue(undefined);
|
|
150
145
|
});
|
|
151
146
|
|
|
152
|
-
it("should return placeholder when
|
|
153
|
-
|
|
147
|
+
it("should return placeholder when credential is not found", () => {
|
|
148
|
+
mockReadStoredCredential.mockReturnValue(undefined);
|
|
154
149
|
|
|
155
150
|
const resolver = new ConfigResolver();
|
|
156
|
-
const result =
|
|
157
|
-
"llama-server=http://127.0.0.1:8080",
|
|
158
|
-
);
|
|
151
|
+
const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|
|
159
152
|
|
|
160
153
|
expect(result).toEqual(API_KEY_PLACEHOLDER);
|
|
161
154
|
});
|
|
162
155
|
|
|
163
|
-
it("should return placeholder when
|
|
164
|
-
|
|
156
|
+
it("should return placeholder when apiKey is missing from credential", () => {
|
|
157
|
+
mockReadStoredCredential.mockReturnValue({});
|
|
165
158
|
|
|
166
159
|
const resolver = new ConfigResolver();
|
|
167
|
-
const result =
|
|
168
|
-
"llama-server=http://127.0.0.1:8080",
|
|
169
|
-
);
|
|
160
|
+
const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|
|
170
161
|
|
|
171
162
|
expect(result).toEqual(API_KEY_PLACEHOLDER);
|
|
172
163
|
});
|
|
173
164
|
|
|
174
|
-
it("should return the
|
|
175
|
-
|
|
165
|
+
it("should return the apiKey when present in credential", () => {
|
|
166
|
+
mockReadStoredCredential.mockReturnValue({ key: "test-api-key" });
|
|
176
167
|
|
|
177
168
|
const resolver = new ConfigResolver();
|
|
178
|
-
const result =
|
|
179
|
-
"llama-server=http://127.0.0.1:8080",
|
|
180
|
-
);
|
|
169
|
+
const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|
|
181
170
|
|
|
182
171
|
expect(result).toEqual("test-api-key");
|
|
183
172
|
});
|
|
184
173
|
|
|
185
|
-
it("should call
|
|
186
|
-
|
|
174
|
+
it("should call readStoredCredential with the provider ID", () => {
|
|
175
|
+
mockReadStoredCredential.mockReturnValue({ apiKey: "test-key" });
|
|
187
176
|
|
|
188
177
|
const resolver = new ConfigResolver();
|
|
189
|
-
|
|
190
|
-
await resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|
|
178
|
+
resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
|
|
191
179
|
|
|
192
|
-
expect(
|
|
180
|
+
expect(mockReadStoredCredential).toHaveBeenCalledWith(
|
|
181
|
+
"llama-server=http://127.0.0.1:8080",
|
|
182
|
+
);
|
|
193
183
|
});
|
|
194
184
|
});
|