pi-llama-cpp 0.8.1 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -164,7 +164,7 @@ When browsing models via the `/models` command, you can:
164
164
  ### Thinking Budgets
165
165
 
166
166
  The extension supports configurable **thinking budgets** that control how many tokens the model allocates to its reasoning/thinking process.
167
- This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh).
167
+ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh, max).
168
168
 
169
169
  | Level | Tokens | Description |
170
170
  | --------- | ------ | ---------------------------- |
@@ -173,7 +173,8 @@ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, x
173
173
  | `low` | 2,048 | Light reasoning |
174
174
  | `medium` | 8,192 | Balanced reasoning (default) |
175
175
  | `high` | 16,384 | Extended reasoning |
176
- | `xhigh` | -1 | Unlimited reasoning |
176
+ | `xhigh` | 32,768 | Deep reasoning |
177
+ | `max` | -1 | Unlimited reasoning |
177
178
 
178
179
  User-defined budgets can override the defaults by adding a `thinkingBudgets` object to `~/.pi/agent/settings.json` (global) or `.pi/settings.json` (per-project):
179
180
 
@@ -183,12 +184,13 @@ User-defined budgets can override the defaults by adding a `thinkingBudgets` obj
183
184
  "minimal": 256,
184
185
  "low": 1024,
185
186
  "medium": 2048,
186
- "high": 4096
187
+ "high": 4096,
188
+ "xhigh": 8192
187
189
  }
188
190
  }
189
191
  ```
190
192
 
191
- Only `minimal`, `low`, `medium`, and `high` are configurable — `off` (0) and `xhigh` (-1, unlimited) are fixed.
193
+ Only `minimal`, `low`, `medium`, `high` and `xhigh` are configurable — `off` (0) and `max` (-1, unlimited) are fixed.
192
194
  The extension automatically injects the appropriate `thinking_budget_tokens` into each request payload based on the selected level.
193
195
 
194
196
  ### Model Selection Event
@@ -219,5 +221,6 @@ Each model exposed to Pi includes the following defaults:
219
221
 
220
222
  | Peer dependency | Purpose |
221
223
  | --------------------------------- | ------------------- |
224
+ | `@earendil-works/pi-ai` | Pi AI SDK |
222
225
  | `@earendil-works/pi-coding-agent` | Pi Coding Agent SDK |
223
226
  | `@earendil-works/pi-tui` | Pi TUI SDK |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.8.1",
3
+ "version": "0.9.0",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -32,12 +32,13 @@
32
32
  ]
33
33
  },
34
34
  "peerDependencies": {
35
+ "@earendil-works/pi-ai": "*",
35
36
  "@earendil-works/pi-coding-agent": "*",
36
37
  "@earendil-works/pi-tui": "*"
37
38
  },
38
39
  "devDependencies": {
39
- "@types/node": "^26.0.1",
40
+ "@types/node": "^26.1.1",
40
41
  "prettier-plugin-organize-imports": "^4.3.0",
41
- "vitest": "^4.1.9"
42
+ "vitest": "^4.1.10"
42
43
  }
43
44
  }
package/src/constants.ts CHANGED
@@ -57,5 +57,6 @@ export const DEFAULT_THINKING_BUDGETS = {
57
57
  low: 2048,
58
58
  medium: 8192,
59
59
  high: 16384,
60
- xhigh: -1,
60
+ xhigh: 32768,
61
+ max: -1,
61
62
  };
@@ -94,7 +94,7 @@ export class EventManager {
94
94
  if (level === "off")
95
95
  return { ...payload, chat_template_kwargs: { enable_thinking: false } };
96
96
 
97
- if (level === "xhigh") return payload;
97
+ if (level === "max") return payload;
98
98
 
99
99
  return { ...payload, thinking_budget_tokens };
100
100
  }
@@ -180,6 +180,7 @@ export abstract class BaseModel {
180
180
  medium: "medium",
181
181
  high: "high",
182
182
  xhigh: "xhigh",
183
+ max: "max",
183
184
  },
184
185
  input: await this.getCapabilities(),
185
186
  contextWindow: await this.getContextSize(),
package/src/resolver.ts CHANGED
@@ -1,6 +1,7 @@
1
+ import { ApiKeyCredential, ModelThinkingLevel } from "@earendil-works/pi-ai";
1
2
  import {
2
- AuthStorage,
3
3
  getAgentDir,
4
+ readStoredCredential,
4
5
  SettingsManager,
5
6
  } from "@earendil-works/pi-coding-agent";
6
7
  import { readFile } from "node:fs/promises";
@@ -10,13 +11,11 @@ import {
10
11
  DEFAULT_LLAMA_SERVER_URL,
11
12
  DEFAULT_THINKING_BUDGETS,
12
13
  } from "./constants";
13
- import { ThinkingLevel } from "./interfaces/levels";
14
14
 
15
15
  export class ConfigResolver {
16
16
  private warnings: string[] = [];
17
17
 
18
18
  private cachedUrls: string[] = [];
19
- private authStorage = AuthStorage.create(join(getAgentDir(), "auth.json"));
20
19
  private settingsManager = SettingsManager.create(
21
20
  process.cwd(),
22
21
  getAgentDir(),
@@ -107,13 +106,11 @@ export class ConfigResolver {
107
106
  }
108
107
 
109
108
  /**
110
- * Resolves API key for the provider ID using Pi's AuthStorage
109
+ * Resolves API key for the provider ID using Pi's stored credentials
111
110
  */
112
- async resolveApiKey(providerId: string): Promise<string> {
113
- this.authStorage.reload();
114
- const apiKey = await this.authStorage.getApiKey(providerId);
115
-
116
- return apiKey ?? API_KEY_PLACEHOLDER;
111
+ resolveApiKey(providerId: string): string {
112
+ const credential = readStoredCredential(providerId) as ApiKeyCredential;
113
+ return credential?.key ?? API_KEY_PLACEHOLDER;
117
114
  }
118
115
 
119
116
  /**
@@ -131,7 +128,7 @@ export class ConfigResolver {
131
128
  *
132
129
  * @returns Selected level
133
130
  */
134
- resolveThinkingLevel(): ThinkingLevel | undefined {
131
+ resolveThinkingLevel(): ModelThinkingLevel | undefined {
135
132
  return this.settingsManager.getDefaultThinkingLevel();
136
133
  }
137
134
 
@@ -140,7 +137,7 @@ export class ConfigResolver {
140
137
  *
141
138
  * @returns Thinking budgets
142
139
  */
143
- resolveThinkingBudgets(): Record<ThinkingLevel, number> {
140
+ resolveThinkingBudgets(): Record<ModelThinkingLevel, number> {
144
141
  const settingsBudgets = this.settingsManager.getThinkingBudgets() ?? {};
145
142
  const availableBudgets = {
146
143
  ...DEFAULT_THINKING_BUDGETS,
@@ -55,7 +55,8 @@ describe("EventManager.onBeforeProviderRequest", () => {
55
55
  { level: "low", expected: { thinking_budget_tokens: 2048 } },
56
56
  { level: "medium", expected: { thinking_budget_tokens: 8192 } },
57
57
  { level: "high", expected: { thinking_budget_tokens: 16384 } },
58
- { level: "xhigh", expected: {} },
58
+ { level: "xhigh", expected: { thinking_budget_tokens: 32768 } },
59
+ { level: "max", expected: {} },
59
60
  ])(
60
61
  'level "$level" should return $expected',
61
62
  async ({ level, expected }) => {
@@ -216,10 +217,10 @@ describe("EventManager.onBeforeProviderRequest", () => {
216
217
  expect(result).not.toHaveProperty("thinking_budget_tokens");
217
218
  });
218
219
 
219
- it("should not allow overriding 'xhigh' — no budget is injected", async () => {
220
- mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("xhigh");
220
+ it("should not inject budget for 'max' — unlimited reasoning", async () => {
221
+ mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("max");
221
222
  mockSettingsManager.getThinkingBudgets.mockReturnValue({
222
- xhigh: 1,
223
+ max: 1,
223
224
  } as any);
224
225
 
225
226
  const server = createMockServer({
@@ -5,22 +5,18 @@ import {
5
5
  } from "../src/constants";
6
6
 
7
7
  // Hoisted mock instances — survives vi.resetModules()
8
- const mockAuthStorage = vi.hoisted(() => ({
9
- reload: vi.fn(),
10
- getApiKey: vi.fn(),
11
- }));
8
+ const mockReadStoredCredential = vi.hoisted(() => vi.fn());
12
9
 
13
10
  const mockSettingsManager = vi.hoisted(() => ({
14
11
  getProjectSettings: vi.fn(),
15
12
  getGlobalSettings: vi.fn(),
16
13
  }));
17
14
 
18
- // Mock getAgentDir, AuthStorage, and SettingsManager before importing resolver
15
+ // Mock getAgentDir, readStoredCredential, and SettingsManager before importing resolver
19
16
  vi.mock("@earendil-works/pi-coding-agent", () => ({
20
17
  getAgentDir: vi.fn().mockReturnValue("/fake/agent/dir"),
21
- AuthStorage: {
22
- create: vi.fn().mockReturnValue(mockAuthStorage),
23
- },
18
+ readStoredCredential: (...args: unknown[]) =>
19
+ mockReadStoredCredential(...args),
24
20
  SettingsManager: {
25
21
  create: vi.fn().mockReturnValue(mockSettingsManager),
26
22
  },
@@ -145,50 +141,44 @@ describe("API key resolution", () => {
145
141
  beforeEach(() => {
146
142
  vi.clearAllMocks();
147
143
  mockGetAgentDir.mockReturnValue("/fake/agent/dir");
148
- mockAuthStorage.reload.mockReturnValue(undefined);
149
- mockAuthStorage.getApiKey.mockResolvedValue(undefined);
144
+ mockReadStoredCredential.mockReturnValue(undefined);
150
145
  });
151
146
 
152
- it("should return placeholder when auth file does not exist", async () => {
153
- mockAuthStorage.getApiKey.mockResolvedValue(undefined);
147
+ it("should return placeholder when credential is not found", () => {
148
+ mockReadStoredCredential.mockReturnValue(undefined);
154
149
 
155
150
  const resolver = new ConfigResolver();
156
- const result = await resolver.resolveApiKey(
157
- "llama-server=http://127.0.0.1:8080",
158
- );
151
+ const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
159
152
 
160
153
  expect(result).toEqual(API_KEY_PLACEHOLDER);
161
154
  });
162
155
 
163
- it("should return placeholder when provider key is missing", async () => {
164
- mockAuthStorage.getApiKey.mockResolvedValue(undefined);
156
+ it("should return placeholder when apiKey is missing from credential", () => {
157
+ mockReadStoredCredential.mockReturnValue({});
165
158
 
166
159
  const resolver = new ConfigResolver();
167
- const result = await resolver.resolveApiKey(
168
- "llama-server=http://127.0.0.1:8080",
169
- );
160
+ const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
170
161
 
171
162
  expect(result).toEqual(API_KEY_PLACEHOLDER);
172
163
  });
173
164
 
174
- it("should return the provider key when present", async () => {
175
- mockAuthStorage.getApiKey.mockResolvedValue("test-api-key");
165
+ it("should return the apiKey when present in credential", () => {
166
+ mockReadStoredCredential.mockReturnValue({ key: "test-api-key" });
176
167
 
177
168
  const resolver = new ConfigResolver();
178
- const result = await resolver.resolveApiKey(
179
- "llama-server=http://127.0.0.1:8080",
180
- );
169
+ const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
181
170
 
182
171
  expect(result).toEqual("test-api-key");
183
172
  });
184
173
 
185
- it("should call reload before each getApiKey", async () => {
186
- mockAuthStorage.getApiKey.mockResolvedValue("cached-key");
174
+ it("should call readStoredCredential with the provider ID", () => {
175
+ mockReadStoredCredential.mockReturnValue({ apiKey: "test-key" });
187
176
 
188
177
  const resolver = new ConfigResolver();
189
- await resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
190
- await resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
178
+ resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
191
179
 
192
- expect(mockAuthStorage.reload).toHaveBeenCalledTimes(2);
180
+ expect(mockReadStoredCredential).toHaveBeenCalledWith(
181
+ "llama-server=http://127.0.0.1:8080",
182
+ );
193
183
  });
194
184
  });
@@ -1,7 +0,0 @@
1
- export type ThinkingLevel =
2
- | "off"
3
- | "minimal"
4
- | "low"
5
- | "medium"
6
- | "high"
7
- | "xhigh";