pi-llama-cpp 0.9.1 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +97 -22
- package/package.json +7 -6
- package/src/constants.ts +13 -3
- package/src/index.ts +4 -10
- package/src/interfaces/settings.ts +60 -0
- package/src/managers/events.ts +33 -9
- package/src/managers/server.ts +6 -4
- package/src/managers/settings.ts +206 -0
- package/src/models/baseModel.ts +32 -9
- package/src/models/legacyModel.ts +2 -2
- package/src/models/routerModel.ts +3 -3
- package/src/server.ts +33 -9
- package/src/sse/manager.ts +2 -1
- package/tests/events.test.ts +186 -11
- package/tests/mocks.ts +1 -1
- package/tests/server.test.ts +46 -0
- package/tests/settings.test.ts +636 -0
- package/tests/sseManager.test.ts +21 -0
- package/src/resolver.ts +0 -149
- package/tests/resolver.test.ts +0 -184
package/README.md
CHANGED
|
@@ -9,9 +9,9 @@ A [Pi Coding Agent](https://pi.dev/) extension that integrates with running [lla
|
|
|
9
9
|
- **Load / unload / switch** — manage models directly from the Pi command palette
|
|
10
10
|
- **Multi-model router support** — works with both single-model and multi-model llama.cpp server configurations
|
|
11
11
|
- **Image capabilities detection** — detects multimodal models automatically
|
|
12
|
-
- **Flexible URL resolution** — configures the server
|
|
12
|
+
- **Flexible URL resolution** — configures the server via `llamaSettings` (project/global), environment variable, or legacy `llamaServerUrl`
|
|
13
13
|
- **Auth support** — allows to login into a llama.cpp server that was secured with an API key
|
|
14
|
-
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously
|
|
14
|
+
- **Multiple server support** — connect to multiple llama.cpp servers simultaneously via `llamaSettings.servers` or semicolon-separated URLs
|
|
15
15
|
- **Thinking budget support** — configurable token budgets for model reasoning/thinking, mapped to Pi's thinking levels
|
|
16
16
|
- **Real-time progress tracking** — live loading progress via SSE (falls back to polling)
|
|
17
17
|
|
|
@@ -48,34 +48,107 @@ pi install https://github.com/gsanhueza/pi-llama-cpp
|
|
|
48
48
|
|
|
49
49
|
## Configuration
|
|
50
50
|
|
|
51
|
-
The extension resolves the llama.cpp server
|
|
51
|
+
The extension resolves the llama.cpp server configuration using the following priority order:
|
|
52
52
|
|
|
53
|
-
1. **
|
|
53
|
+
1. **Environment variable** — `LLAMA_SERVER_URL`
|
|
54
|
+
2. **`llamaSettings`** — Main configuration format in `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global)
|
|
55
|
+
3. **`llamaServerUrl`** — Legacy format in `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global)
|
|
56
|
+
4. **Default** — `http://127.0.0.1:8080`
|
|
54
57
|
|
|
55
|
-
|
|
56
|
-
{
|
|
57
|
-
"llamaServerUrl": "http://127.0.0.1:8080"
|
|
58
|
-
}
|
|
59
|
-
```
|
|
58
|
+
### Server configuration
|
|
60
59
|
|
|
61
|
-
|
|
60
|
+
The recommended way to configure the extension is using the `llamaSettings` key. This provides a structured way to define multiple servers with custom names and IDs, plus additional behavior options.
|
|
62
61
|
|
|
63
|
-
|
|
62
|
+
Add this to your `.pi/settings.json` (project) or `~/.pi/agent/settings.json` (global):
|
|
64
63
|
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
64
|
+
```json
|
|
65
|
+
{
|
|
66
|
+
"llamaSettings": {
|
|
67
|
+
"servers": [
|
|
68
|
+
{
|
|
69
|
+
"url": "http://127.0.0.1:8080",
|
|
70
|
+
"id": "local",
|
|
71
|
+
"name": "Local Server"
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"url": "http://10.0.0.5:8080",
|
|
75
|
+
"name": "Remote Server"
|
|
76
|
+
}
|
|
77
|
+
],
|
|
78
|
+
"reactToModelSelect": true,
|
|
79
|
+
"autoloadOnMessage": false,
|
|
80
|
+
"pollingTimeout": 60000,
|
|
81
|
+
"serverTimeout": 1000
|
|
82
|
+
}
|
|
83
|
+
}
|
|
84
|
+
```
|
|
70
85
|
|
|
71
|
-
|
|
86
|
+
With this config, the servers will appear in Pi as **Llama.cpp (Local Server)** and **Llama.cpp (Remote Server)**.
|
|
87
|
+
|
|
88
|
+
#### Server Options
|
|
89
|
+
|
|
90
|
+
| Option | Type | Required | Description |
|
|
91
|
+
| ------ | ------ | -------- | ---------------------------------------------------------------------------- |
|
|
92
|
+
| `url` | string | Yes | The URL of the llama.cpp server |
|
|
93
|
+
| `id` | string | No | Custom provider ID (used for API key auth). Defaults to `llama-server=<url>` |
|
|
94
|
+
| `name` | string | No | Display name for the server in the UI (shown as `Llama.cpp — <name>`) |
|
|
95
|
+
|
|
96
|
+
> **Note:** If you set a custom `id`, you can use it in `~/.pi/agent/auth.json`. The extension will also fall back to the URL-based ID if no key is found for the custom `id`.
|
|
97
|
+
|
|
98
|
+
#### Settings Options
|
|
99
|
+
|
|
100
|
+
| Option | Type | Default | Description |
|
|
101
|
+
| -------------------- | ------- | ------- | ------------------------------------------------------------- |
|
|
102
|
+
| `reactToModelSelect` | boolean | `true` | Load the model when you switch via Pi's model picker. |
|
|
103
|
+
| `autoloadOnMessage` | boolean | `false` | Automatically load an unloaded model before sending a message |
|
|
104
|
+
| `pollingTimeout` | number | `60000` | Max time (ms) to wait for model loading before giving up |
|
|
105
|
+
| `serverTimeout` | number | `1000` | Timeout (ms) for server health checks and SSE probes |
|
|
106
|
+
|
|
107
|
+
> **Note:** `serverTimeout` controls individual HTTP request timeouts (health checks, SSE probe). `pollingTimeout` controls the total wait time for a model to finish loading. Increase `serverTimeout` for slow/high-latency servers, and `pollingTimeout` for large models or slow hardware.
|
|
108
|
+
|
|
109
|
+
#### Environment variable
|
|
110
|
+
|
|
111
|
+
For a quick setup, you can use the `LLAMA_SERVER_URL` environment variable instead of the JSON config:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
export LLAMA_SERVER_URL="http://127.0.0.1:8080"
|
|
115
|
+
```
|
|
72
116
|
|
|
73
|
-
|
|
117
|
+
This is equivalent to defining a single server in `llamaSettings.servers` with just a URL.
|
|
74
118
|
|
|
75
|
-
|
|
119
|
+
### Legacy configuration
|
|
120
|
+
|
|
121
|
+
For a simpler setup, you can use the legacy `llamaServerUrl` key:
|
|
122
|
+
|
|
123
|
+
```json
|
|
124
|
+
{
|
|
125
|
+
"llamaServerUrl": "http://127.0.0.1:8080"
|
|
126
|
+
}
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
This is equivalent to defining a single server in `llamaSettings.servers` with just a URL.
|
|
130
|
+
|
|
131
|
+
### Multiple servers
|
|
132
|
+
|
|
133
|
+
To connect to multiple llama.cpp servers simultaneously:
|
|
134
|
+
|
|
135
|
+
**Using `llamaSettings` (recommended):**
|
|
136
|
+
|
|
137
|
+
```json
|
|
138
|
+
{
|
|
139
|
+
"llamaSettings": {
|
|
140
|
+
"servers": [
|
|
141
|
+
{ "url": "http://127.0.0.1:8080" },
|
|
142
|
+
{ "url": "http://127.0.0.1:8081" },
|
|
143
|
+
{ "url": "http://10.0.0.5:8080" }
|
|
144
|
+
]
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
**Using the environment variable:**
|
|
76
150
|
|
|
77
151
|
```bash
|
|
78
|
-
# Example for env, but you can use any of the other methods
|
|
79
152
|
LLAMA_SERVER_URL="http://127.0.0.1:8080;http://127.0.0.1:8081;http://10.0.0.5:8080"
|
|
80
153
|
```
|
|
81
154
|
|
|
@@ -86,7 +159,7 @@ Each server gets its own provider (e.g., **Llama.cpp (http://127.0.0.1:8080)**)
|
|
|
86
159
|
If your llama.cpp server requires authentication, use `/login` in Pi, select the "API key" option, and choose the provider from the list that correlates with the server needing the API key.
|
|
87
160
|
|
|
88
161
|
Alternatively, configure the API key in `~/.pi/agent/auth.json`:
|
|
89
|
-
Use the provider ID `llama-server=<url
|
|
162
|
+
Use the provider ID `llama-server=<url>` (or your custom `id` if you set one in `llamaSettings.servers`):
|
|
90
163
|
|
|
91
164
|
```json
|
|
92
165
|
{
|
|
@@ -199,13 +272,15 @@ When you switch models via Pi's model picker (instead of using the `/models` com
|
|
|
199
272
|
|
|
200
273
|
This keeps the server in sync with the active model in Pi, regardless of how the switch was initiated — you don't need to manually load models before using them.
|
|
201
274
|
|
|
275
|
+
You can disable this behavior by setting `reactToModelSelect` to `false` in `llamaSettings`.
|
|
276
|
+
|
|
202
277
|
> **Note:** If you switch sessions while a model load is in-flight, you'll see a warning, but the load continues in the background. Use `/models` in the new session to verify the model status.
|
|
203
278
|
|
|
204
279
|
### Loading Models
|
|
205
280
|
|
|
206
281
|
When you trigger a load, switch, or retry action, the extension uses SSE (Server-Sent Events) to receive real-time progress updates from the server. If SSE is not available, it falls back to polling.
|
|
207
282
|
|
|
208
|
-
If loading takes longer than **60 seconds
|
|
283
|
+
If loading takes longer than **60 seconds** (configurable via `pollingTimeout`), the operation times out with an error.
|
|
209
284
|
|
|
210
285
|
> **Note:** The timeout only applies to the progress detection. The model might still be loading in the background.
|
|
211
286
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-llama-cpp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "Pi extension for llama.cpp integration. Supports router, single and legacy models. Supports multiple servers.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"pi",
|
|
@@ -32,13 +32,14 @@
|
|
|
32
32
|
]
|
|
33
33
|
},
|
|
34
34
|
"peerDependencies": {
|
|
35
|
-
"@earendil-works/pi-ai": "
|
|
36
|
-
"@earendil-works/pi-coding-agent": "
|
|
37
|
-
"@earendil-works/pi-tui": "
|
|
35
|
+
"@earendil-works/pi-ai": ">=0.84.0",
|
|
36
|
+
"@earendil-works/pi-coding-agent": ">=0.84.0",
|
|
37
|
+
"@earendil-works/pi-tui": ">=0.84.0"
|
|
38
38
|
},
|
|
39
|
+
"type": "module",
|
|
39
40
|
"devDependencies": {
|
|
40
|
-
"@types/node": "^26.
|
|
41
|
+
"@types/node": "^26.4.0",
|
|
41
42
|
"prettier-plugin-organize-imports": "^4.3.0",
|
|
42
|
-
"vitest": "^4.1.
|
|
43
|
+
"vitest": "^4.1.11"
|
|
43
44
|
}
|
|
44
45
|
}
|
package/src/constants.ts
CHANGED
|
@@ -21,12 +21,12 @@ export const API_KEY_PLACEHOLDER = "sk-placeholder";
|
|
|
21
21
|
/**
|
|
22
22
|
* The default URL if the resolver couldn't find it
|
|
23
23
|
*/
|
|
24
|
-
export const
|
|
24
|
+
export const LLAMA_SERVER_URL = "http://127.0.0.1:8080";
|
|
25
25
|
|
|
26
26
|
/**
|
|
27
27
|
* The default context if the server didn't expose it
|
|
28
28
|
*/
|
|
29
|
-
export const
|
|
29
|
+
export const FALLBACK_CTX = 128000;
|
|
30
30
|
|
|
31
31
|
/**
|
|
32
32
|
* Polling interval (ms) for checking model load status
|
|
@@ -48,10 +48,20 @@ export const READABLE_TIMEOUT = 15000;
|
|
|
48
48
|
*/
|
|
49
49
|
export const SERVER_TIMEOUT = 1000;
|
|
50
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Default value for reactToModelSelect setting.
|
|
53
|
+
*/
|
|
54
|
+
export const REACT_TO_MODEL_SELECT = true;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Default value for autoloadOnMessage setting.
|
|
58
|
+
*/
|
|
59
|
+
export const AUTOLOAD_ON_MESSAGE = false;
|
|
60
|
+
|
|
51
61
|
/**
|
|
52
62
|
* Thinking budgets to send to the server, depending on user-selected level in Pi.
|
|
53
63
|
*/
|
|
54
|
-
export const
|
|
64
|
+
export const THINKING_BUDGETS = {
|
|
55
65
|
off: 0,
|
|
56
66
|
minimal: 1024,
|
|
57
67
|
low: 2048,
|
package/src/index.ts
CHANGED
|
@@ -11,13 +11,10 @@ import { ModelSelectEvent } from "./interfaces/events";
|
|
|
11
11
|
import { CommandManager } from "./managers/command";
|
|
12
12
|
import { EventManager } from "./managers/events";
|
|
13
13
|
import { ServerManager } from "./managers/server";
|
|
14
|
-
import {
|
|
15
|
-
import { Server } from "./server";
|
|
14
|
+
import { settings } from "./managers/settings";
|
|
16
15
|
|
|
17
16
|
export default async function (pi: ExtensionAPI) {
|
|
18
|
-
const
|
|
19
|
-
const urls = await resolver.resolveUrls();
|
|
20
|
-
const servers = urls.map((url) => new Server(url));
|
|
17
|
+
const servers = await settings.resolveServers();
|
|
21
18
|
|
|
22
19
|
const eventManager = new EventManager(servers);
|
|
23
20
|
const serverManager = new ServerManager(servers);
|
|
@@ -40,15 +37,12 @@ export default async function (pi: ExtensionAPI) {
|
|
|
40
37
|
if (event.reason !== "startup") return;
|
|
41
38
|
for (const warning of serverManager.getWarnings())
|
|
42
39
|
ctx.ui.notify(warning, "warning");
|
|
43
|
-
|
|
44
|
-
for (const warning of resolver.getWarnings())
|
|
45
|
-
ctx.ui.notify(warning, "warning");
|
|
46
40
|
});
|
|
47
41
|
|
|
48
42
|
pi.on(
|
|
49
43
|
"before_provider_request",
|
|
50
|
-
async (event: BeforeProviderRequestEvent) =>
|
|
51
|
-
await eventManager.onBeforeProviderRequest(event),
|
|
44
|
+
async (event: BeforeProviderRequestEvent, ctx: ExtensionContext) =>
|
|
45
|
+
await eventManager.onBeforeProviderRequest(event, ctx),
|
|
52
46
|
);
|
|
53
47
|
|
|
54
48
|
pi.on(
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A description of a server in the "llamaSettings" key
|
|
3
|
+
*/
|
|
4
|
+
interface LlamaServer {
|
|
5
|
+
/**
|
|
6
|
+
* The URL of the llama.cpp server.
|
|
7
|
+
*/
|
|
8
|
+
url: string;
|
|
9
|
+
/**
|
|
10
|
+
* Custom provider ID for this server.
|
|
11
|
+
*/
|
|
12
|
+
id?: string;
|
|
13
|
+
/**
|
|
14
|
+
* Custom display name for this server.
|
|
15
|
+
*/
|
|
16
|
+
name?: string;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* The main configuration interface for this extension
|
|
21
|
+
*
|
|
22
|
+
* E.g.:
|
|
23
|
+
*
|
|
24
|
+
* {
|
|
25
|
+
* "servers": [{
|
|
26
|
+
* "id": "server-a",
|
|
27
|
+
* "name": "Server A",
|
|
28
|
+
* "url": "http://localhost:8080"
|
|
29
|
+
* }],
|
|
30
|
+
* "reactToModelSelect": true
|
|
31
|
+
* "autoloadOnMessage": false
|
|
32
|
+
* }
|
|
33
|
+
}
|
|
34
|
+
*/
|
|
35
|
+
export interface LlamaSettings {
|
|
36
|
+
/**
|
|
37
|
+
* List of servers to connect to.
|
|
38
|
+
*/
|
|
39
|
+
servers: LlamaServer[];
|
|
40
|
+
/**
|
|
41
|
+
* Whether to react to model selection events by loading the model.
|
|
42
|
+
* @default true
|
|
43
|
+
*/
|
|
44
|
+
reactToModelSelect?: boolean;
|
|
45
|
+
/**
|
|
46
|
+
* Whether to auto-load models when a message is sent.
|
|
47
|
+
* @default false
|
|
48
|
+
*/
|
|
49
|
+
autoloadOnMessage?: boolean;
|
|
50
|
+
/**
|
|
51
|
+
* Maximum time (ms) to wait for model loading before giving up.
|
|
52
|
+
* @default 60000
|
|
53
|
+
*/
|
|
54
|
+
pollingTimeout?: number;
|
|
55
|
+
/**
|
|
56
|
+
* Timeout (ms) for server verification and SSE support probe.
|
|
57
|
+
* @default 1000
|
|
58
|
+
*/
|
|
59
|
+
serverTimeout?: number;
|
|
60
|
+
}
|
package/src/managers/events.ts
CHANGED
|
@@ -3,9 +3,10 @@ import {
|
|
|
3
3
|
type ExtensionContext,
|
|
4
4
|
} from "@earendil-works/pi-coding-agent";
|
|
5
5
|
import { READABLE_TIMEOUT } from "../constants";
|
|
6
|
+
import { Status } from "../enums/status";
|
|
6
7
|
import { ModelSelectEvent } from "../interfaces/events";
|
|
8
|
+
import { settings } from "../managers/settings";
|
|
7
9
|
import { BaseModel } from "../models/baseModel";
|
|
8
|
-
import { ConfigResolver } from "../resolver";
|
|
9
10
|
import { Server } from "../server";
|
|
10
11
|
|
|
11
12
|
export class EventManager {
|
|
@@ -27,6 +28,9 @@ export class EventManager {
|
|
|
27
28
|
* @param ctx Pi context
|
|
28
29
|
*/
|
|
29
30
|
async onModelSelect(event: ModelSelectEvent, ctx: ExtensionContext) {
|
|
31
|
+
// Check if the model_select event should be used
|
|
32
|
+
if (!settings.resolveReactToModelSelect()) return;
|
|
33
|
+
|
|
30
34
|
for (const { providerId, models } of this.servers) {
|
|
31
35
|
if (event.model.provider !== providerId) continue;
|
|
32
36
|
|
|
@@ -44,6 +48,20 @@ export class EventManager {
|
|
|
44
48
|
}
|
|
45
49
|
}
|
|
46
50
|
|
|
51
|
+
/**
|
|
52
|
+
* Loads the model if auto-loading is enabled and the model is unloaded.
|
|
53
|
+
*
|
|
54
|
+
* @param model The model to potentially auto-load
|
|
55
|
+
*/
|
|
56
|
+
private async autoLoadIfNeeded(model: BaseModel): Promise<void> {
|
|
57
|
+
if (!settings.resolveAutoloadOnMessage()) return;
|
|
58
|
+
|
|
59
|
+
const status = await model.getStatus();
|
|
60
|
+
if (status !== Status.UNLOADED) return;
|
|
61
|
+
|
|
62
|
+
await model.load();
|
|
63
|
+
}
|
|
64
|
+
|
|
47
65
|
/**
|
|
48
66
|
* Session-switch handler. Registered once at extension init.
|
|
49
67
|
* Only notifies if a model load is actually in-flight.
|
|
@@ -72,22 +90,28 @@ export class EventManager {
|
|
|
72
90
|
* @param event Request event
|
|
73
91
|
* @returns Updated payload
|
|
74
92
|
*/
|
|
75
|
-
async onBeforeProviderRequest(
|
|
93
|
+
async onBeforeProviderRequest(
|
|
94
|
+
event: BeforeProviderRequestEvent,
|
|
95
|
+
ctx: ExtensionContext,
|
|
96
|
+
) {
|
|
76
97
|
const payload = event.payload as { model?: string };
|
|
77
98
|
const { model } = payload;
|
|
78
99
|
if (!model) return payload;
|
|
79
100
|
|
|
80
101
|
// Check if this model belongs to one of our servers
|
|
81
|
-
const
|
|
82
|
-
|
|
83
|
-
|
|
102
|
+
const serverModel = this.servers
|
|
103
|
+
.flatMap((s) => s.models)
|
|
104
|
+
.find((m) => m.id === model);
|
|
105
|
+
|
|
106
|
+
if (!serverModel) return payload;
|
|
84
107
|
|
|
85
|
-
if
|
|
108
|
+
// Auto-load if enabled and model is unloaded
|
|
109
|
+
await this.autoLoadIfNeeded(serverModel);
|
|
86
110
|
|
|
87
111
|
// Retrieve pi's current thinking level, so we can setup a budget
|
|
88
|
-
const
|
|
89
|
-
|
|
90
|
-
const budgets =
|
|
112
|
+
const level =
|
|
113
|
+
ctx.thinkingLevel ?? settings.resolveThinkingLevel() ?? "medium";
|
|
114
|
+
const budgets = settings.resolveThinkingBudgets();
|
|
91
115
|
const thinking_budget_tokens = budgets[level];
|
|
92
116
|
|
|
93
117
|
// Setup payload
|
package/src/managers/server.ts
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { API_TYPE, PROVIDER_NAME
|
|
2
|
+
import { API_TYPE, PROVIDER_NAME } from "../constants";
|
|
3
3
|
import { ServerStatus } from "../enums/serverStatus";
|
|
4
4
|
import { BaseModel } from "../models/baseModel";
|
|
5
5
|
import { Server } from "../server";
|
|
6
|
+
import { settings } from "./settings";
|
|
6
7
|
|
|
7
8
|
export class ServerManager {
|
|
8
9
|
readonly failedUrls: string[] = [];
|
|
@@ -16,8 +17,9 @@ export class ServerManager {
|
|
|
16
17
|
* @param pi The Pi extension API
|
|
17
18
|
*/
|
|
18
19
|
async initialize(pi: ExtensionAPI) {
|
|
19
|
-
// Register the providers with
|
|
20
|
-
|
|
20
|
+
// Register the providers with the configured server timeout
|
|
21
|
+
const { serverTimeout } = settings.resolveTimeouts();
|
|
22
|
+
await this.update(pi, serverTimeout);
|
|
21
23
|
}
|
|
22
24
|
|
|
23
25
|
/**
|
|
@@ -67,7 +69,7 @@ export class ServerManager {
|
|
|
67
69
|
} else if (status === ServerStatus.TIMEOUT) {
|
|
68
70
|
const message = [
|
|
69
71
|
"[pi-llama-cpp]",
|
|
70
|
-
`${PROVIDER_NAME} server initialization for '${server.baseUrl}' took more than ${
|
|
72
|
+
`${PROVIDER_NAME} server initialization for '${server.baseUrl}' took more than ${timeout} ms, so it has been skipped.`,
|
|
71
73
|
"Run `/models` to retry without timeout and see all models.",
|
|
72
74
|
].join("\n");
|
|
73
75
|
this.warnings.push(message);
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
import { ApiKeyCredential, ModelThinkingLevel } from "@earendil-works/pi-ai";
|
|
2
|
+
import {
|
|
3
|
+
readStoredCredential,
|
|
4
|
+
SettingsManager,
|
|
5
|
+
} from "@earendil-works/pi-coding-agent";
|
|
6
|
+
import {
|
|
7
|
+
API_KEY_PLACEHOLDER,
|
|
8
|
+
AUTOLOAD_ON_MESSAGE,
|
|
9
|
+
LLAMA_SERVER_URL,
|
|
10
|
+
POLLING_TIMEOUT,
|
|
11
|
+
REACT_TO_MODEL_SELECT,
|
|
12
|
+
SERVER_TIMEOUT,
|
|
13
|
+
THINKING_BUDGETS,
|
|
14
|
+
} from "../constants";
|
|
15
|
+
import { LlamaSettings } from "../interfaces/settings";
|
|
16
|
+
import { Server } from "../server";
|
|
17
|
+
|
|
18
|
+
const SETTINGS_KEY = "llamaSettings";
|
|
19
|
+
|
|
20
|
+
export class LlamaSettingsManager {
|
|
21
|
+
private settingsManager = SettingsManager.create(process.cwd());
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Convenience getter for merged project/global settings
|
|
25
|
+
*/
|
|
26
|
+
private get mergedSettings(): Record<string, any> {
|
|
27
|
+
const merged = {
|
|
28
|
+
...this.settingsManager.getGlobalSettings(),
|
|
29
|
+
...this.settingsManager.getProjectSettings(),
|
|
30
|
+
} as Record<string, any>;
|
|
31
|
+
return merged;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Convenience getter for the `llamaSettings` key
|
|
36
|
+
*/
|
|
37
|
+
private get llamaSettings(): LlamaSettings {
|
|
38
|
+
return this.mergedSettings[SETTINGS_KEY] ?? {};
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Resolves the server URLs to use in the following order:
|
|
43
|
+
*
|
|
44
|
+
* - `LLAMA_SERVER_URL` env variable
|
|
45
|
+
* - `llamaSettings` key (current - project, then global)
|
|
46
|
+
* - `llamaServerUrl` key (legacy - project, then global)
|
|
47
|
+
* - Default URL
|
|
48
|
+
*
|
|
49
|
+
* @returns The list of URLs to use
|
|
50
|
+
*/
|
|
51
|
+
resolveUrls(): string[] {
|
|
52
|
+
let response = this.resolveEnvUrls();
|
|
53
|
+
if (response.length > 0) return response;
|
|
54
|
+
|
|
55
|
+
response = this.resolveServerUrls();
|
|
56
|
+
if (response.length > 0) return response;
|
|
57
|
+
|
|
58
|
+
response = this.resolveLegacyUrls();
|
|
59
|
+
if (response.length > 0) return response;
|
|
60
|
+
|
|
61
|
+
return [LLAMA_SERVER_URL];
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Resolves the llama-server URLs from the environment variable.
|
|
66
|
+
*
|
|
67
|
+
* @returns A list of detected URLs
|
|
68
|
+
*/
|
|
69
|
+
private resolveEnvUrls(): string[] {
|
|
70
|
+
const raw = process.env.LLAMA_SERVER_URL;
|
|
71
|
+
if (!raw) return [];
|
|
72
|
+
|
|
73
|
+
return this.parseUrls(raw);
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* Resolves the llama-server URLs from `llamaSettings.servers`.
|
|
78
|
+
* Settings are merged, prioritizing project over global settings.
|
|
79
|
+
*
|
|
80
|
+
* @returns A list of detected URLs
|
|
81
|
+
*/
|
|
82
|
+
private resolveServerUrls(): string[] {
|
|
83
|
+
const { servers = [] } = this.llamaSettings;
|
|
84
|
+
return servers.map((s) => this.parseUrls(s.url)).flat();
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* Resolves the llama-server URLs from `llamaSettings.servers`.
|
|
89
|
+
* Settings are merged, prioritizing project over global settings.
|
|
90
|
+
*
|
|
91
|
+
* @returns A list of detected URLs
|
|
92
|
+
*/
|
|
93
|
+
private resolveLegacyUrls(): string[] {
|
|
94
|
+
const { llamaServerUrl = null } = this.mergedSettings;
|
|
95
|
+
if (!llamaServerUrl) return [];
|
|
96
|
+
|
|
97
|
+
return this.parseUrls(llamaServerUrl);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Parses a raw URL string into an array of cleaned URLs.
|
|
102
|
+
* Splits on semicolons, trims whitespace, filters empty strings,
|
|
103
|
+
* and strips trailing slashes.
|
|
104
|
+
*
|
|
105
|
+
* @returns A sanitized URL
|
|
106
|
+
*/
|
|
107
|
+
private parseUrls(raw: string): string[] {
|
|
108
|
+
return raw
|
|
109
|
+
.split(";")
|
|
110
|
+
.map((u) => u.trim())
|
|
111
|
+
.filter((u) => u.length > 0)
|
|
112
|
+
.map((u) => u.replace(/\/+$/, ""));
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* Resolves the servers that this extension will use.
|
|
117
|
+
* Uses `resolveUrls()` as the source of truth for URLs (env > settings >
|
|
118
|
+
* legacy > default), then applies `id`/`name` from `llamaSettings.servers`
|
|
119
|
+
* as overrides when available.
|
|
120
|
+
*
|
|
121
|
+
* @returns A list of Server objects
|
|
122
|
+
*/
|
|
123
|
+
resolveServers(): Server[] {
|
|
124
|
+
const { pollingTimeout, serverTimeout } = this.resolveTimeouts();
|
|
125
|
+
const urls = this.resolveUrls();
|
|
126
|
+
const serverConfigs = this.llamaSettings.servers ?? [];
|
|
127
|
+
|
|
128
|
+
return urls.map((url) => {
|
|
129
|
+
const config = serverConfigs.find((s) => s.url === url);
|
|
130
|
+
return new Server(
|
|
131
|
+
url,
|
|
132
|
+
config?.id,
|
|
133
|
+
config?.name,
|
|
134
|
+
serverTimeout,
|
|
135
|
+
pollingTimeout,
|
|
136
|
+
);
|
|
137
|
+
});
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Resolves API key for the provider ID using Pi's stored credentials.
|
|
142
|
+
*
|
|
143
|
+
* @returns The API key to use for the provider
|
|
144
|
+
*/
|
|
145
|
+
resolveApiKey(providerId: string): string {
|
|
146
|
+
const credential = readStoredCredential(providerId) as ApiKeyCredential;
|
|
147
|
+
return credential?.key ?? API_KEY_PLACEHOLDER;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Resolves the current thinking level from Pi.
|
|
152
|
+
*
|
|
153
|
+
* @returns The thinking level
|
|
154
|
+
*/
|
|
155
|
+
resolveThinkingLevel(): ModelThinkingLevel | undefined {
|
|
156
|
+
return this.settingsManager.getDefaultThinkingLevel();
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* Resolves the effective thinking budgets from settings.
|
|
161
|
+
*
|
|
162
|
+
* @returns An object with selected budgets for thinking levels
|
|
163
|
+
*/
|
|
164
|
+
resolveThinkingBudgets(): Record<ModelThinkingLevel, number> {
|
|
165
|
+
const settingsBudgets = this.settingsManager.getThinkingBudgets() ?? {};
|
|
166
|
+
return {
|
|
167
|
+
...THINKING_BUDGETS,
|
|
168
|
+
...settingsBudgets,
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Resolves whether the extension should react to model selection events.
|
|
174
|
+
*
|
|
175
|
+
* @returns `true` if the extension should load the model on model_select
|
|
176
|
+
*/
|
|
177
|
+
resolveReactToModelSelect(): boolean {
|
|
178
|
+
return this.llamaSettings.reactToModelSelect ?? REACT_TO_MODEL_SELECT;
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Resolves whether the extension should auto-load models on message.
|
|
183
|
+
*
|
|
184
|
+
* @returns `true` if the extension should auto-load models
|
|
185
|
+
*/
|
|
186
|
+
resolveAutoloadOnMessage(): boolean {
|
|
187
|
+
return this.llamaSettings.autoloadOnMessage ?? AUTOLOAD_ON_MESSAGE;
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
/**
|
|
191
|
+
* Resolves the timeout settings for polling and server checks.
|
|
192
|
+
*
|
|
193
|
+
* @returns Object with polling and server timeout values
|
|
194
|
+
*/
|
|
195
|
+
resolveTimeouts(): { pollingTimeout: number; serverTimeout: number } {
|
|
196
|
+
return {
|
|
197
|
+
pollingTimeout: this.llamaSettings.pollingTimeout ?? POLLING_TIMEOUT,
|
|
198
|
+
serverTimeout: this.llamaSettings.serverTimeout ?? SERVER_TIMEOUT,
|
|
199
|
+
};
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/**
|
|
204
|
+
* Shared singleton instance used across the extension.
|
|
205
|
+
*/
|
|
206
|
+
export const settings = new LlamaSettingsManager();
|