pi-llama-cpp 0.8.2 → 0.9.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -164,7 +164,7 @@ When browsing models via the `/models` command, you can:
164
164
  ### Thinking Budgets
165
165
 
166
166
  The extension supports configurable **thinking budgets** that control how many tokens the model allocates to its reasoning/thinking process.
167
- This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh).
167
+ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, xhigh, max).
168
168
 
169
169
  | Level | Tokens | Description |
170
170
  | --------- | ------ | ---------------------------- |
@@ -173,7 +173,8 @@ This is tied to Pi's thinking level selector (off, minimal, low, medium, high, x
173
173
  | `low` | 2,048 | Light reasoning |
174
174
  | `medium` | 8,192 | Balanced reasoning (default) |
175
175
  | `high` | 16,384 | Extended reasoning |
176
- | `xhigh` | -1 | Unlimited reasoning |
176
+ | `xhigh` | 32,768 | Deep reasoning |
177
+ | `max` | -1 | Unlimited reasoning |
177
178
 
178
179
  User-defined budgets can override the defaults by adding a `thinkingBudgets` object to `~/.pi/agent/settings.json` (global) or `.pi/settings.json` (per-project):
179
180
 
@@ -183,12 +184,13 @@ User-defined budgets can override the defaults by adding a `thinkingBudgets` obj
183
184
  "minimal": 256,
184
185
  "low": 1024,
185
186
  "medium": 2048,
186
- "high": 4096
187
+ "high": 4096,
188
+ "xhigh": 8192
187
189
  }
188
190
  }
189
191
  ```
190
192
 
191
- Only `minimal`, `low`, `medium`, and `high` are configurable — `off` (0) and `xhigh` (-1, unlimited) are fixed.
193
+ Only `minimal`, `low`, `medium`, `high` and `xhigh` are configurable — `off` (0) and `max` (-1, unlimited) are fixed.
192
194
  The extension automatically injects the appropriate `thinking_budget_tokens` into each request payload based on the selected level.
193
195
 
194
196
  ### Model Selection Event
@@ -219,5 +221,6 @@ Each model exposed to Pi includes the following defaults:
219
221
 
220
222
  | Peer dependency | Purpose |
221
223
  | --------------------------------- | ------------------- |
224
+ | `@earendil-works/pi-ai` | Pi AI SDK |
222
225
  | `@earendil-works/pi-coding-agent` | Pi Coding Agent SDK |
223
226
  | `@earendil-works/pi-tui` | Pi TUI SDK |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-llama-cpp",
3
- "version": "0.8.2",
3
+ "version": "0.9.1",
4
4
  "description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
5
5
  "keywords": [
6
6
  "pi",
@@ -32,6 +32,7 @@
32
32
  ]
33
33
  },
34
34
  "peerDependencies": {
35
+ "@earendil-works/pi-ai": "*",
35
36
  "@earendil-works/pi-coding-agent": "*",
36
37
  "@earendil-works/pi-tui": "*"
37
38
  },
package/src/constants.ts CHANGED
@@ -57,5 +57,6 @@ export const DEFAULT_THINKING_BUDGETS = {
57
57
  low: 2048,
58
58
  medium: 8192,
59
59
  high: 16384,
60
- xhigh: -1,
60
+ xhigh: 32768,
61
+ max: -1,
61
62
  };
@@ -177,15 +177,20 @@ export class CommandManager {
177
177
  }
178
178
  };
179
179
 
180
+ const onFinished = async () => {
181
+ cleanupProgress();
182
+ EventManager.resetInflightModel();
183
+
184
+ // Re-scan providers to ensure accuracy of loaded models
185
+ await this.serverManager.update(pi);
186
+
187
+ // Force TUI refresh so Pi picks up the updated model states
188
+ ctx.ui.setStatus(PROVIDER_NAME, " ");
189
+ ctx.ui.setStatus(PROVIDER_NAME, undefined);
190
+ };
191
+
180
192
  // Load the model without blocking the UI
181
- model
182
- .load()
183
- .then(onSuccess)
184
- .catch(onFailure)
185
- .finally(() => {
186
- cleanupProgress();
187
- EventManager.resetInflightModel();
188
- });
193
+ model.load().then(onSuccess).catch(onFailure).finally(onFinished);
189
194
  }
190
195
  }
191
196
 
@@ -94,7 +94,7 @@ export class EventManager {
94
94
  if (level === "off")
95
95
  return { ...payload, chat_template_kwargs: { enable_thinking: false } };
96
96
 
97
- if (level === "xhigh") return payload;
97
+ if (level === "max") return payload;
98
98
 
99
99
  return { ...payload, thinking_budget_tokens };
100
100
  }
@@ -180,6 +180,7 @@ export abstract class BaseModel {
180
180
  medium: "medium",
181
181
  high: "high",
182
182
  xhigh: "xhigh",
183
+ max: "max",
183
184
  },
184
185
  input: await this.getCapabilities(),
185
186
  contextWindow: await this.getContextSize(),
package/src/resolver.ts CHANGED
@@ -1,3 +1,4 @@
1
+ import { ApiKeyCredential, ModelThinkingLevel } from "@earendil-works/pi-ai";
1
2
  import {
2
3
  getAgentDir,
3
4
  readStoredCredential,
@@ -10,7 +11,6 @@ import {
10
11
  DEFAULT_LLAMA_SERVER_URL,
11
12
  DEFAULT_THINKING_BUDGETS,
12
13
  } from "./constants";
13
- import { ThinkingLevel } from "./interfaces/levels";
14
14
 
15
15
  export class ConfigResolver {
16
16
  private warnings: string[] = [];
@@ -109,8 +109,8 @@ export class ConfigResolver {
109
109
  * Resolves API key for the provider ID using Pi's stored credentials
110
110
  */
111
111
  resolveApiKey(providerId: string): string {
112
- const credential = readStoredCredential(providerId);
113
- return credential?.apiKey ?? API_KEY_PLACEHOLDER;
112
+ const credential = readStoredCredential(providerId) as ApiKeyCredential;
113
+ return credential?.key ?? API_KEY_PLACEHOLDER;
114
114
  }
115
115
 
116
116
  /**
@@ -128,7 +128,7 @@ export class ConfigResolver {
128
128
  *
129
129
  * @returns Selected level
130
130
  */
131
- resolveThinkingLevel(): ThinkingLevel | undefined {
131
+ resolveThinkingLevel(): ModelThinkingLevel | undefined {
132
132
  return this.settingsManager.getDefaultThinkingLevel();
133
133
  }
134
134
 
@@ -137,7 +137,7 @@ export class ConfigResolver {
137
137
  *
138
138
  * @returns Thinking budgets
139
139
  */
140
- resolveThinkingBudgets(): Record<ThinkingLevel, number> {
140
+ resolveThinkingBudgets(): Record<ModelThinkingLevel, number> {
141
141
  const settingsBudgets = this.settingsManager.getThinkingBudgets() ?? {};
142
142
  const availableBudgets = {
143
143
  ...DEFAULT_THINKING_BUDGETS,
@@ -55,7 +55,8 @@ describe("EventManager.onBeforeProviderRequest", () => {
55
55
  { level: "low", expected: { thinking_budget_tokens: 2048 } },
56
56
  { level: "medium", expected: { thinking_budget_tokens: 8192 } },
57
57
  { level: "high", expected: { thinking_budget_tokens: 16384 } },
58
- { level: "xhigh", expected: {} },
58
+ { level: "xhigh", expected: { thinking_budget_tokens: 32768 } },
59
+ { level: "max", expected: {} },
59
60
  ])(
60
61
  'level "$level" should return $expected',
61
62
  async ({ level, expected }) => {
@@ -216,10 +217,10 @@ describe("EventManager.onBeforeProviderRequest", () => {
216
217
  expect(result).not.toHaveProperty("thinking_budget_tokens");
217
218
  });
218
219
 
219
- it("should not allow overriding 'xhigh' — no budget is injected", async () => {
220
- mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("xhigh");
220
+ it("should not inject budget for 'max' — unlimited reasoning", async () => {
221
+ mockSettingsManager.getDefaultThinkingLevel.mockReturnValue("max");
221
222
  mockSettingsManager.getThinkingBudgets.mockReturnValue({
222
- xhigh: 1,
223
+ max: 1,
223
224
  } as any);
224
225
 
225
226
  const server = createMockServer({
@@ -163,7 +163,7 @@ describe("API key resolution", () => {
163
163
  });
164
164
 
165
165
  it("should return the apiKey when present in credential", () => {
166
- mockReadStoredCredential.mockReturnValue({ apiKey: "test-api-key" });
166
+ mockReadStoredCredential.mockReturnValue({ key: "test-api-key" });
167
167
 
168
168
  const resolver = new ConfigResolver();
169
169
  const result = resolver.resolveApiKey("llama-server=http://127.0.0.1:8080");
@@ -1,7 +0,0 @@
1
- export type ThinkingLevel =
2
- | "off"
3
- | "minimal"
4
- | "low"
5
- | "medium"
6
- | "high"
7
- | "xhigh";