pi-llama-cpp 0.11.0 → 0.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -7,6 +7,7 @@ import {
7
7
  PROVIDER_PREFIX,
8
8
  SERVER_TIMEOUT,
9
9
  } from "../src/constants";
10
+ import type { ModelOverride } from "../src/interfaces/settings";
10
11
  import { settings } from "../src/managers/settings";
11
12
  import { Server } from "../src/server";
12
13
 
@@ -35,10 +36,12 @@ vi.mock("node:fs/promises", () => ({
35
36
  readFile: vi.fn(),
36
37
  writeFile: vi.fn(),
37
38
  rename: vi.fn(),
39
+ access: vi.fn(),
38
40
  }));
39
41
 
40
42
  // Import mocked modules
41
43
  import { getAgentDir } from "@earendil-works/pi-coding-agent";
44
+ import { access } from "node:fs/promises";
42
45
 
43
46
  describe("URL resolution fallback chain", () => {
44
47
  const mockGetAgentDir = vi.mocked(getAgentDir);
@@ -66,7 +69,7 @@ describe("URL resolution fallback chain", () => {
66
69
  // Ensure env var is not set (and not inherited from environment)
67
70
  delete process.env.LLAMA_SERVER_URL;
68
71
 
69
- const result = settings.resolveUrls();
72
+ const result = await settings.resolveUrls();
70
73
 
71
74
  expect(result).toEqual([LLAMA_SERVER_URL]);
72
75
  });
@@ -77,7 +80,7 @@ describe("URL resolution fallback chain", () => {
77
80
  });
78
81
  process.env.LLAMA_SERVER_URL = "http://env-url:8080";
79
82
 
80
- const result = settings.resolveUrls();
83
+ const result = await settings.resolveUrls();
81
84
 
82
85
  expect(result).toEqual(["http://env-url:8080"]);
83
86
  });
@@ -85,7 +88,7 @@ describe("URL resolution fallback chain", () => {
85
88
  it("should use env variable when no other config exists", async () => {
86
89
  process.env.LLAMA_SERVER_URL = "http://env-url:8080";
87
90
 
88
- const result = settings.resolveUrls();
91
+ const result = await settings.resolveUrls();
89
92
 
90
93
  expect(result).toEqual(["http://env-url:8080"]);
91
94
  });
@@ -95,7 +98,7 @@ describe("URL resolution fallback chain", () => {
95
98
  llamaServerUrl: "http://project:9999",
96
99
  });
97
100
 
98
- const result = settings.resolveUrls();
101
+ const result = await settings.resolveUrls();
99
102
 
100
103
  expect(result).toEqual(["http://project:9999"]);
101
104
  });
@@ -105,7 +108,7 @@ describe("URL resolution fallback chain", () => {
105
108
  llamaServerUrl: "http://global:8080",
106
109
  });
107
110
 
108
- const result = settings.resolveUrls();
111
+ const result = await settings.resolveUrls();
109
112
 
110
113
  expect(result).toEqual(["http://global:8080"]);
111
114
  });
@@ -113,7 +116,7 @@ describe("URL resolution fallback chain", () => {
113
116
  it("should strip trailing slashes from resolved URL", async () => {
114
117
  process.env.LLAMA_SERVER_URL = "http://localhost:8080/";
115
118
 
116
- const result = settings.resolveUrls();
119
+ const result = await settings.resolveUrls();
117
120
 
118
121
  expect(result).toEqual(["http://localhost:8080"]);
119
122
  });
@@ -121,8 +124,8 @@ describe("URL resolution fallback chain", () => {
121
124
  it("should cache the resolved URL on subsequent calls", async () => {
122
125
  process.env.LLAMA_SERVER_URL = "http://first:8080";
123
126
 
124
- const result1 = settings.resolveUrls();
125
- const result2 = settings.resolveUrls();
127
+ const result1 = await settings.resolveUrls();
128
+ const result2 = await settings.resolveUrls();
126
129
 
127
130
  expect(result1).toEqual(["http://first:8080"]);
128
131
  expect(result2).toEqual(["http://first:8080"]);
@@ -131,7 +134,7 @@ describe("URL resolution fallback chain", () => {
131
134
  it("should handle multiple URLs separated by semicolons", async () => {
132
135
  process.env.LLAMA_SERVER_URL = "http://first:8080;http://second:9090/";
133
136
 
134
- const result = settings.resolveUrls();
137
+ const result = await settings.resolveUrls();
135
138
 
136
139
  expect(result).toEqual(["http://first:8080", "http://second:9090"]);
137
140
  });
@@ -139,7 +142,7 @@ describe("URL resolution fallback chain", () => {
139
142
  it("should drop env URLs without an http(s) scheme, warn, and fall through", async () => {
140
143
  process.env.LLAMA_SERVER_URL = "127.0.0.1:8080";
141
144
 
142
- const result = settings.resolveUrls();
145
+ const result = await settings.resolveUrls();
143
146
 
144
147
  expect(result).toEqual([LLAMA_SERVER_URL]);
145
148
  expect(settings.takeWarnings()).toEqual([
@@ -155,7 +158,7 @@ describe("URL resolution fallback chain", () => {
155
158
  },
156
159
  });
157
160
 
158
- const result = settings.resolveUrls();
161
+ const result = await settings.resolveUrls();
159
162
 
160
163
  expect(result).toEqual(["http://good:8080"]);
161
164
  expect(settings.takeWarnings()).toEqual([
@@ -196,7 +199,7 @@ describe("llamaSettings.servers resolution", () => {
196
199
  },
197
200
  });
198
201
 
199
- const result = settings.resolveUrls();
202
+ const result = await settings.resolveUrls();
200
203
 
201
204
  expect(result).toEqual([
202
205
  "http://project-server:8080",
@@ -216,7 +219,7 @@ describe("llamaSettings.servers resolution", () => {
216
219
  },
217
220
  });
218
221
 
219
- const result = settings.resolveUrls();
222
+ const result = await settings.resolveUrls();
220
223
 
221
224
  expect(result).toEqual(["http://project:8080"]);
222
225
  });
@@ -228,7 +231,7 @@ describe("llamaSettings.servers resolution", () => {
228
231
  },
229
232
  });
230
233
 
231
- const result = settings.resolveUrls();
234
+ const result = await settings.resolveUrls();
232
235
 
233
236
  expect(result).toEqual(["http://global:8080"]);
234
237
  });
@@ -241,7 +244,7 @@ describe("llamaSettings.servers resolution", () => {
241
244
  },
242
245
  });
243
246
 
244
- const result = settings.resolveUrls();
247
+ const result = await settings.resolveUrls();
245
248
 
246
249
  expect(result).toEqual(["http://env:8080"]);
247
250
  });
@@ -253,7 +256,7 @@ describe("llamaSettings.servers resolution", () => {
253
256
  },
254
257
  });
255
258
 
256
- const result = settings.resolveUrls();
259
+ const result = await settings.resolveUrls();
257
260
 
258
261
  expect(result).toEqual(["http://server:9090"]);
259
262
  });
@@ -266,7 +269,7 @@ describe("llamaSettings.servers resolution", () => {
266
269
  },
267
270
  });
268
271
 
269
- const result = settings.resolveUrls();
272
+ const result = await settings.resolveUrls();
270
273
 
271
274
  expect(result).toEqual(["http://server:9090"]);
272
275
  });
@@ -279,7 +282,7 @@ describe("llamaSettings.servers resolution", () => {
279
282
  },
280
283
  });
281
284
 
282
- const result = settings.resolveUrls();
285
+ const result = await settings.resolveUrls();
283
286
 
284
287
  expect(result).toEqual(["http://legacy:8080"]);
285
288
  });
@@ -291,7 +294,7 @@ describe("llamaSettings.servers resolution", () => {
291
294
  },
292
295
  });
293
296
 
294
- const result = settings.resolveUrls();
297
+ const result = await settings.resolveUrls();
295
298
 
296
299
  expect(result).toEqual(["http://localhost:8080"]);
297
300
  });
@@ -438,7 +441,7 @@ describe("reactToModelSelect and autoloadOnMessage fallbacks", () => {
438
441
  it("should return true when reactToModelSelect is not set", async () => {
439
442
  const { settings } = await import("../src/managers/settings");
440
443
 
441
- const result = settings.resolveReactToModelSelect();
444
+ const result = await settings.resolveReactToModelSelect();
442
445
 
443
446
  expect(result).toBe(true);
444
447
  });
@@ -446,7 +449,7 @@ describe("reactToModelSelect and autoloadOnMessage fallbacks", () => {
446
449
  it("should return false when autoloadOnMessage is not set", async () => {
447
450
  const { settings } = await import("../src/managers/settings");
448
451
 
449
- const result = settings.resolveAutoloadOnMessage();
452
+ const result = await settings.resolveAutoloadOnMessage();
450
453
 
451
454
  expect(result).toBe(false);
452
455
  });
@@ -454,7 +457,7 @@ describe("reactToModelSelect and autoloadOnMessage fallbacks", () => {
454
457
  it("should return 'asc' when sortBy is not set", async () => {
455
458
  const { settings } = await import("../src/managers/settings");
456
459
 
457
- const result = settings.resolveSortBy();
460
+ const result = await settings.resolveSortBy();
458
461
 
459
462
  expect(result).toBe("asc");
460
463
  });
@@ -469,8 +472,8 @@ describe("reactToModelSelect and autoloadOnMessage fallbacks", () => {
469
472
 
470
473
  const { settings } = await import("../src/managers/settings");
471
474
 
472
- expect(settings.resolveReactToModelSelect()).toBe(false);
473
- expect(settings.resolveAutoloadOnMessage()).toBe(true);
475
+ expect(await settings.resolveReactToModelSelect()).toBe(false);
476
+ expect(await settings.resolveAutoloadOnMessage()).toBe(true);
474
477
  });
475
478
  });
476
479
 
@@ -495,7 +498,7 @@ describe("resolveServers", () => {
495
498
  mockGetGlobalSettings.mockReturnValue({});
496
499
  });
497
500
 
498
- it("should use llamaSettings.servers when configured", () => {
501
+ it("should use llamaSettings.servers when configured", async () => {
499
502
  mockGetProjectSettings.mockReturnValue({
500
503
  llamaSettings: {
501
504
  servers: [
@@ -504,7 +507,7 @@ describe("resolveServers", () => {
504
507
  },
505
508
  });
506
509
 
507
- const result = settings.resolveServers();
510
+ const result = await settings.resolveServers();
508
511
 
509
512
  expect(result).toHaveLength(1);
510
513
  expect(result[0].baseUrl).toBe("http://custom:8080");
@@ -514,20 +517,20 @@ describe("resolveServers", () => {
514
517
  it("should fall back to resolveUrls when servers is empty", async () => {
515
518
  process.env.LLAMA_SERVER_URL = "http://env-server:9090";
516
519
 
517
- const result = settings.resolveServers();
520
+ const result = await settings.resolveServers();
518
521
 
519
522
  expect(result).toHaveLength(1);
520
523
  expect(result[0].baseUrl).toBe("http://env-server:9090");
521
524
  });
522
525
 
523
- it("should fall back to default URL when no config exists", () => {
524
- const result = settings.resolveServers();
526
+ it("should fall back to default URL when no config exists", async () => {
527
+ const result = await settings.resolveServers();
525
528
 
526
529
  expect(result).toHaveLength(1);
527
530
  expect(result[0].baseUrl).toBe(LLAMA_SERVER_URL);
528
531
  });
529
532
 
530
- it("should apply id/name from llamaSettings.servers as overrides", () => {
533
+ it("should apply id/name from llamaSettings.servers as overrides", async () => {
531
534
  mockGetProjectSettings.mockReturnValue({
532
535
  llamaSettings: {
533
536
  servers: [
@@ -536,7 +539,7 @@ describe("resolveServers", () => {
536
539
  },
537
540
  });
538
541
 
539
- const result = settings.resolveServers();
542
+ const result = await settings.resolveServers();
540
543
 
541
544
  expect(result).toHaveLength(1);
542
545
  expect(result[0].baseUrl).toBe("http://127.0.0.1:8080");
@@ -544,7 +547,7 @@ describe("resolveServers", () => {
544
547
  expect(result[0].providerName).toBe(`Llama.cpp (Custom)`);
545
548
  });
546
549
 
547
- it("should handle multiple URLs with partial id/name overrides", () => {
550
+ it("should handle multiple URLs with partial id/name overrides", async () => {
548
551
  mockGetProjectSettings.mockReturnValue({
549
552
  llamaSettings: {
550
553
  servers: [{ url: "http://first:8080", id: "first-server" }],
@@ -552,7 +555,7 @@ describe("resolveServers", () => {
552
555
  });
553
556
  process.env.LLAMA_SERVER_URL = "http://first:8080;http://second:9090";
554
557
 
555
- const result = settings.resolveServers();
558
+ const result = await settings.resolveServers();
556
559
 
557
560
  expect(result).toHaveLength(2);
558
561
  expect(result[0].baseUrl).toBe("http://first:8080");
@@ -569,7 +572,7 @@ describe("resolveServers", () => {
569
572
  },
570
573
  });
571
574
 
572
- const result = settings.resolveServers();
575
+ const result = await settings.resolveServers();
573
576
 
574
577
  // env variable takes precedence via resolveUrls
575
578
  expect(result).toHaveLength(1);
@@ -585,7 +588,7 @@ describe("resolveTimeouts", () => {
585
588
  it("should return default timeouts when not configured", async () => {
586
589
  const { settings } = await import("../src/managers/settings");
587
590
 
588
- const result = settings.resolveTimeouts();
591
+ const result = await settings.resolveTimeouts();
589
592
 
590
593
  expect(result).toEqual({
591
594
  pollingTimeout: POLLING_TIMEOUT,
@@ -602,7 +605,7 @@ describe("resolveTimeouts", () => {
602
605
 
603
606
  const { settings } = await import("../src/managers/settings");
604
607
 
605
- const result = settings.resolveTimeouts();
608
+ const result = await settings.resolveTimeouts();
606
609
 
607
610
  expect(result.pollingTimeout).toBe(120000);
608
611
  expect(result.serverTimeout).toBe(SERVER_TIMEOUT);
@@ -617,7 +620,7 @@ describe("resolveTimeouts", () => {
617
620
 
618
621
  const { settings } = await import("../src/managers/settings");
619
622
 
620
- const result = settings.resolveTimeouts();
623
+ const result = await settings.resolveTimeouts();
621
624
 
622
625
  expect(result.pollingTimeout).toBe(POLLING_TIMEOUT);
623
626
  expect(result.serverTimeout).toBe(3000);
@@ -633,7 +636,7 @@ describe("resolveTimeouts", () => {
633
636
 
634
637
  const { settings } = await import("../src/managers/settings");
635
638
 
636
- const result = settings.resolveTimeouts();
639
+ const result = await settings.resolveTimeouts();
637
640
 
638
641
  expect(result).toEqual({
639
642
  pollingTimeout: 90000,
@@ -697,12 +700,20 @@ describe("setLlamaSetting", () => {
697
700
  const mockReadFile = vi.mocked(readFile);
698
701
  const mockWriteFile = vi.mocked(writeFile);
699
702
  const mockRename = vi.mocked(rename);
703
+ const mockAccess = vi.mocked(access);
700
704
 
701
- const SETTINGS_PATH = "/fake/agent/dir/settings.json";
705
+ const GLOBAL_SETTINGS_PATH = "/fake/agent/dir/settings.json";
706
+ const PROJECT_SETTINGS_PATH = "/fake/project/.pi/settings.json";
707
+ const FAKE_CWD = "/fake/project";
708
+
709
+ afterEach(() => {
710
+ vi.resetModules();
711
+ });
702
712
 
703
713
  beforeEach(() => {
704
714
  vi.clearAllMocks();
705
715
  mockGetAgentDir.mockReturnValue("/fake/agent/dir");
716
+ vi.spyOn(process, "cwd").mockReturnValue(FAKE_CWD);
706
717
  mockGetProjectSettings.mockReturnValue({});
707
718
  mockGetGlobalSettings.mockReturnValue({});
708
719
  mockReload.mockResolvedValue(undefined);
@@ -711,7 +722,72 @@ describe("setLlamaSetting", () => {
711
722
  mockRename.mockResolvedValue(undefined);
712
723
  });
713
724
 
725
+ it("should write to project settings when .pi/settings.json exists (auto scope)", async () => {
726
+ mockAccess.mockResolvedValue(undefined);
727
+ mockReadFile.mockResolvedValue("{}");
728
+
729
+ const { settings } = await import("../src/managers/settings");
730
+ await settings.setLlamaSetting("sortBy", "desc");
731
+
732
+ expect(mockWriteFile).toHaveBeenCalledWith(
733
+ `${PROJECT_SETTINGS_PATH}.tmp`,
734
+ expect.any(String),
735
+ "utf-8",
736
+ );
737
+ expect(mockRename).toHaveBeenCalledWith(
738
+ `${PROJECT_SETTINGS_PATH}.tmp`,
739
+ PROJECT_SETTINGS_PATH,
740
+ );
741
+ });
742
+
743
+ it("should write to global settings when .pi/settings.json does not exist (auto scope)", async () => {
744
+ mockAccess.mockRejectedValue(new Error("ENOENT"));
745
+ mockReadFile.mockResolvedValue("{}");
746
+
747
+ const { settings } = await import("../src/managers/settings");
748
+ await settings.setLlamaSetting("sortBy", "desc");
749
+
750
+ expect(mockWriteFile).toHaveBeenCalledWith(
751
+ `${GLOBAL_SETTINGS_PATH}.tmp`,
752
+ expect.any(String),
753
+ "utf-8",
754
+ );
755
+ expect(mockRename).toHaveBeenCalledWith(
756
+ `${GLOBAL_SETTINGS_PATH}.tmp`,
757
+ GLOBAL_SETTINGS_PATH,
758
+ );
759
+ });
760
+
761
+ it("should always write to global when scope is explicitly 'global'", async () => {
762
+ mockAccess.mockResolvedValue(undefined); // project exists but we override
763
+ mockReadFile.mockResolvedValue("{}");
764
+
765
+ const { settings } = await import("../src/managers/settings");
766
+ await settings.setLlamaSetting("sortBy", "desc", "global");
767
+
768
+ expect(mockWriteFile).toHaveBeenCalledWith(
769
+ `${GLOBAL_SETTINGS_PATH}.tmp`,
770
+ expect.any(String),
771
+ "utf-8",
772
+ );
773
+ });
774
+
775
+ it("should always write to project when scope is explicitly 'project'", async () => {
776
+ mockAccess.mockRejectedValue(new Error("ENOENT")); // project doesn't exist but we override
777
+ mockReadFile.mockResolvedValue("{}");
778
+
779
+ const { settings } = await import("../src/managers/settings");
780
+ await settings.setLlamaSetting("sortBy", "desc", "project");
781
+
782
+ expect(mockWriteFile).toHaveBeenCalledWith(
783
+ `${PROJECT_SETTINGS_PATH}.tmp`,
784
+ expect.any(String),
785
+ "utf-8",
786
+ );
787
+ });
788
+
714
789
  it("should write the merged llamaSettings key atomically and reload", async () => {
790
+ mockAccess.mockRejectedValue(new Error("ENOENT")); // no project settings
715
791
  mockReadFile.mockResolvedValue(
716
792
  JSON.stringify(
717
793
  { unrelated: true, llamaSettings: { reactToModelSelect: true } },
@@ -720,11 +796,12 @@ describe("setLlamaSetting", () => {
720
796
  ),
721
797
  );
722
798
 
799
+ const { settings } = await import("../src/managers/settings");
723
800
  await settings.setLlamaSetting("sortBy", "desc");
724
801
 
725
802
  expect(mockWriteFile).toHaveBeenCalledTimes(1);
726
803
  const [tmpPath, written, encoding] = mockWriteFile.mock.calls[0];
727
- expect(tmpPath).toBe(`${SETTINGS_PATH}.tmp`);
804
+ expect(tmpPath).toBe(`${GLOBAL_SETTINGS_PATH}.tmp`);
728
805
  expect(encoding).toBe("utf-8");
729
806
  const parsed = JSON.parse(written as string);
730
807
  expect(parsed).toEqual({
@@ -732,12 +809,16 @@ describe("setLlamaSetting", () => {
732
809
  llamaSettings: { reactToModelSelect: true, sortBy: "desc" },
733
810
  });
734
811
  expect(mockRename).toHaveBeenCalledWith(
735
- `${SETTINGS_PATH}.tmp`,
736
- SETTINGS_PATH,
812
+ `${GLOBAL_SETTINGS_PATH}.tmp`,
813
+ GLOBAL_SETTINGS_PATH,
737
814
  );
738
815
  expect(mockReload).toHaveBeenCalledTimes(1);
739
816
  });
740
817
 
818
+ afterEach(() => {
819
+ vi.resetModules();
820
+ });
821
+
741
822
  it("should reflect the new value in resolvers immediately after the write", async () => {
742
823
  mockSettingsManager.reload.mockImplementation(async () => {
743
824
  mockGetGlobalSettings.mockReturnValue({
@@ -745,14 +826,17 @@ describe("setLlamaSetting", () => {
745
826
  });
746
827
  });
747
828
 
829
+ const { settings } = await import("../src/managers/settings");
748
830
  await settings.setLlamaSetting("sortBy", "desc");
749
831
 
750
- expect(settings.resolveSortBy()).toBe("desc");
832
+ expect(await settings.resolveSortBy()).toBe("desc");
751
833
  });
752
834
 
753
835
  it("should reject and skip reload when the write fails", async () => {
836
+ mockAccess.mockRejectedValue(new Error("ENOENT"));
754
837
  mockWriteFile.mockRejectedValue(new Error("ENOSPC: simulated"));
755
838
 
839
+ const { settings } = await import("../src/managers/settings");
756
840
  await expect(settings.setLlamaSetting("sortBy", "desc")).rejects.toThrow(
757
841
  "ENOSPC",
758
842
  );
@@ -760,8 +844,10 @@ describe("setLlamaSetting", () => {
760
844
  });
761
845
 
762
846
  it("should reject and leave the file untouched when the JSON is invalid", async () => {
847
+ mockAccess.mockRejectedValue(new Error("ENOENT"));
763
848
  mockReadFile.mockResolvedValue("{ broken");
764
849
 
850
+ const { settings } = await import("../src/managers/settings");
765
851
  await expect(settings.setLlamaSetting("sortBy", "desc")).rejects.toThrow(
766
852
  /Cannot parse/,
767
853
  );
@@ -770,6 +856,8 @@ describe("setLlamaSetting", () => {
770
856
  });
771
857
 
772
858
  it("should persist booleans and numbers with type fidelity", async () => {
859
+ mockAccess.mockRejectedValue(new Error("ENOENT"));
860
+ const { settings } = await import("../src/managers/settings");
773
861
  await settings.setLlamaSetting("reactToModelSelect", false);
774
862
 
775
863
  const [, firstWrite] = mockWriteFile.mock.calls[0];
@@ -791,3 +879,361 @@ describe("setLlamaSetting", () => {
791
879
  expect(() => new LlamaSettingsManager()).not.toThrow();
792
880
  });
793
881
  });
882
+
883
+ describe("resolveServerOverrides", () => {
884
+ const mockGetAgentDir = vi.mocked(getAgentDir);
885
+ const mockGetProjectSettings = vi.mocked(
886
+ mockSettingsManager.getProjectSettings,
887
+ );
888
+ const mockGetGlobalSettings = vi.mocked(
889
+ mockSettingsManager.getGlobalSettings,
890
+ );
891
+
892
+ afterEach(() => {
893
+ vi.resetModules();
894
+ });
895
+
896
+ beforeEach(() => {
897
+ vi.clearAllMocks();
898
+ mockGetAgentDir.mockReturnValue("/fake/agent/dir");
899
+ mockGetProjectSettings.mockReturnValue({});
900
+ mockGetGlobalSettings.mockReturnValue({});
901
+ });
902
+
903
+ it("should return overrides for a server that has them configured", async () => {
904
+ mockGetProjectSettings.mockReturnValue({
905
+ llamaSettings: {
906
+ servers: [
907
+ {
908
+ url: "http://127.0.0.1:8080",
909
+ overrides: {
910
+ "llama-3-8b": { cost: { input: 0.2, output: 0.6 } },
911
+ "llama-3-70b": {
912
+ cost: {
913
+ input: 0.1,
914
+ output: 0.3,
915
+ cacheRead: 0.01,
916
+ cacheWrite: 0.02,
917
+ },
918
+ },
919
+ },
920
+ },
921
+ ],
922
+ },
923
+ });
924
+
925
+ const result = await settings.resolveServerOverrides(
926
+ "http://127.0.0.1:8080",
927
+ );
928
+
929
+ expect(result).toEqual({
930
+ "llama-3-8b": { cost: { input: 0.2, output: 0.6 } },
931
+ "llama-3-70b": {
932
+ cost: {
933
+ input: 0.1,
934
+ output: 0.3,
935
+ cacheRead: 0.01,
936
+ cacheWrite: 0.02,
937
+ },
938
+ },
939
+ });
940
+ });
941
+
942
+ it("should return empty object for a server without overrides", async () => {
943
+ mockGetProjectSettings.mockReturnValue({
944
+ llamaSettings: {
945
+ servers: [{ url: "http://127.0.0.1:8080" }],
946
+ },
947
+ });
948
+
949
+ const result = await settings.resolveServerOverrides(
950
+ "http://127.0.0.1:8080",
951
+ );
952
+
953
+ expect(result).toEqual({});
954
+ });
955
+
956
+ it("should return empty object when server URL is not in config", async () => {
957
+ mockGetProjectSettings.mockReturnValue({
958
+ llamaSettings: {
959
+ servers: [{ url: "http://127.0.0.1:9090" }],
960
+ },
961
+ });
962
+
963
+ const result = await settings.resolveServerOverrides(
964
+ "http://127.0.0.1:8080",
965
+ );
966
+
967
+ expect(result).toEqual({});
968
+ });
969
+
970
+ it("should use global settings when no project config exists", async () => {
971
+ mockGetGlobalSettings.mockReturnValue({
972
+ llamaSettings: {
973
+ servers: [
974
+ {
975
+ url: "http://global:8080",
976
+ overrides: { "model-a": { cost: { input: 0.5 } } },
977
+ },
978
+ ],
979
+ },
980
+ });
981
+
982
+ const result = await settings.resolveServerOverrides("http://global:8080");
983
+
984
+ expect(result).toEqual({ "model-a": { cost: { input: 0.5 } } });
985
+ });
986
+
987
+ it("should prioritize project overrides over global overrides", async () => {
988
+ mockGetProjectSettings.mockReturnValue({
989
+ llamaSettings: {
990
+ servers: [
991
+ {
992
+ url: "http://shared:8080",
993
+ overrides: { "model-b": { cost: { input: 0.1, output: 0.2 } } },
994
+ },
995
+ ],
996
+ },
997
+ });
998
+ mockGetGlobalSettings.mockReturnValue({
999
+ llamaSettings: {
1000
+ servers: [
1001
+ {
1002
+ url: "http://shared:8080",
1003
+ overrides: { "model-b": { cost: { input: 0.5, output: 0.5 } } },
1004
+ },
1005
+ ],
1006
+ },
1007
+ });
1008
+
1009
+ const result = await settings.resolveServerOverrides("http://shared:8080");
1010
+
1011
+ expect(result).toEqual({
1012
+ "model-b": { cost: { input: 0.1, output: 0.2 } },
1013
+ });
1014
+ });
1015
+
1016
+ it("should return empty object when servers list is empty", async () => {
1017
+ mockGetProjectSettings.mockReturnValue({
1018
+ llamaSettings: { servers: [] },
1019
+ });
1020
+
1021
+ const result = await settings.resolveServerOverrides(
1022
+ "http://127.0.0.1:8080",
1023
+ );
1024
+
1025
+ expect(result).toEqual({});
1026
+ });
1027
+
1028
+ it("should return empty object when llamaSettings is missing", async () => {
1029
+ mockGetProjectSettings.mockReturnValue({});
1030
+
1031
+ const result = await settings.resolveServerOverrides(
1032
+ "http://127.0.0.1:8080",
1033
+ );
1034
+
1035
+ expect(result).toEqual({});
1036
+ });
1037
+
1038
+ it("should support partial override objects", async () => {
1039
+ mockGetProjectSettings.mockReturnValue({
1040
+ llamaSettings: {
1041
+ servers: [
1042
+ {
1043
+ url: "http://127.0.0.1:8080",
1044
+ overrides: { "partial-model": { cost: { input: 0.1 } } },
1045
+ },
1046
+ ],
1047
+ },
1048
+ });
1049
+
1050
+ const result = await settings.resolveServerOverrides(
1051
+ "http://127.0.0.1:8080",
1052
+ );
1053
+
1054
+ expect(result).toEqual({ "partial-model": { cost: { input: 0.1 } } });
1055
+ });
1056
+ });
1057
+
1058
+ describe("Server with overrides", () => {
1059
+ it("should store and expose resolved overrides", () => {
1060
+ const server = new Server(settings, {
1061
+ baseUrl: "http://127.0.0.1:8080",
1062
+ overrides: {
1063
+ "model-a": { cost: { input: 0.2, output: 0.6 } },
1064
+ "model-b": { cost: { input: 0.1, output: 0.3, cacheRead: 0.01 } },
1065
+ },
1066
+ });
1067
+
1068
+ expect(server.getOverrides()).toEqual({
1069
+ "model-a": { cost: { input: 0.2, output: 0.6 } },
1070
+ "model-b": { cost: { input: 0.1, output: 0.3, cacheRead: 0.01 } },
1071
+ });
1072
+ });
1073
+
1074
+ it("should return empty object when no overrides are provided", () => {
1075
+ const server = new Server(settings, {
1076
+ baseUrl: "http://127.0.0.1:8080",
1077
+ });
1078
+
1079
+ expect(server.getOverrides()).toEqual({});
1080
+ });
1081
+ });
1082
+
1083
+ describe("resolveServers passes overrides", () => {
1084
+ const mockGetAgentDir = vi.mocked(getAgentDir);
1085
+ const mockGetProjectSettings = vi.mocked(
1086
+ mockSettingsManager.getProjectSettings,
1087
+ );
1088
+ const mockGetGlobalSettings = vi.mocked(
1089
+ mockSettingsManager.getGlobalSettings,
1090
+ );
1091
+
1092
+ afterEach(() => {
1093
+ vi.resetModules();
1094
+ });
1095
+
1096
+ beforeEach(() => {
1097
+ vi.clearAllMocks();
1098
+ mockGetAgentDir.mockReturnValue("/fake/agent/dir");
1099
+ mockGetProjectSettings.mockReturnValue({});
1100
+ mockGetGlobalSettings.mockReturnValue({});
1101
+ });
1102
+
1103
+ it("should pass resolved overrides to Server instances", async () => {
1104
+ mockGetProjectSettings.mockReturnValue({
1105
+ llamaSettings: {
1106
+ servers: [
1107
+ {
1108
+ url: "http://overrides-server:8080",
1109
+ overrides: { "model-x": { cost: { input: 0.5, output: 1.0 } } },
1110
+ },
1111
+ {
1112
+ url: "http://no-overrides-server:9090",
1113
+ },
1114
+ ],
1115
+ },
1116
+ });
1117
+
1118
+ const result = await settings.resolveServers();
1119
+
1120
+ expect(result).toHaveLength(2);
1121
+ expect(result[0].getOverrides()).toEqual({
1122
+ "model-x": { cost: { input: 0.5, output: 1.0 } },
1123
+ });
1124
+ expect(result[1].getOverrides()).toEqual({});
1125
+ });
1126
+ });
1127
+
1128
+ describe("Server.findOverrideForModel", () => {
1129
+ function createServer(overrides: Record<string, ModelOverride>): Server {
1130
+ return new Server(settings as any, {
1131
+ baseUrl: "http://127.0.0.1:8080",
1132
+ overrides,
1133
+ });
1134
+ }
1135
+
1136
+ it("should return undefined when overrides is empty", () => {
1137
+ const server = createServer({});
1138
+ expect(server.findOverrideForModel("llama-3-8b")).toBeUndefined();
1139
+ });
1140
+
1141
+ it("should return undefined when no key matches", () => {
1142
+ const server = createServer({
1143
+ mistral: { cost: { input: 0.1 } },
1144
+ "gpt-4": { cost: { input: 0.3 } },
1145
+ });
1146
+ expect(server.findOverrideForModel("llama-3-8b")).toBeUndefined();
1147
+ });
1148
+
1149
+ it("should match exact ID", () => {
1150
+ const server = createServer({
1151
+ "llama-3-8b": { cost: { input: 0.2, output: 0.6 } },
1152
+ });
1153
+ expect(server.findOverrideForModel("llama-3-8b")).toEqual({
1154
+ cost: { input: 0.2, output: 0.6 },
1155
+ });
1156
+ });
1157
+
1158
+ it("should match prefix", () => {
1159
+ const server = createServer({
1160
+ llama: { cost: { input: 0.01, output: 0.02 } },
1161
+ });
1162
+ expect(server.findOverrideForModel("llama-3-8b")).toEqual({
1163
+ cost: { input: 0.01, output: 0.02 },
1164
+ });
1165
+ });
1166
+
1167
+ it("should prefer longest match (most specific)", () => {
1168
+ const server = createServer({
1169
+ llama: { cost: { input: 0.01, output: 0.02 } },
1170
+ "llama-3": { cost: { input: 0.05, output: 0.1 } },
1171
+ "llama-3-8b": { cost: { input: 0.2, output: 0.6 } },
1172
+ });
1173
+ expect(server.findOverrideForModel("llama-3-8b")).toEqual({
1174
+ cost: { input: 0.2, output: 0.6 },
1175
+ });
1176
+ });
1177
+
1178
+ it("should match the second-longest when exact match is absent", () => {
1179
+ const server = createServer({
1180
+ llama: { cost: { input: 0.01, output: 0.02 } },
1181
+ "llama-3": { cost: { input: 0.05, output: 0.1 } },
1182
+ "llama-3-8b": { cost: { input: 0.2, output: 0.6 } },
1183
+ });
1184
+ expect(server.findOverrideForModel("llama-3-70b")).toEqual({
1185
+ cost: { input: 0.05, output: 0.1 },
1186
+ });
1187
+ });
1188
+
1189
+ it("should skip empty keys", () => {
1190
+ const server = createServer({
1191
+ "": { cost: { input: 0.001 } },
1192
+ llama: { cost: { input: 0.01 } },
1193
+ });
1194
+ expect(server.findOverrideForModel("llama-3-8b")).toEqual({
1195
+ cost: { input: 0.01 },
1196
+ });
1197
+ });
1198
+
1199
+ it("should not match when model ID is shorter than key", () => {
1200
+ const server = createServer({
1201
+ "llama-3-8b": { cost: { input: 0.2 } },
1202
+ });
1203
+ expect(server.findOverrideForModel("llama")).toBeUndefined();
1204
+ });
1205
+
1206
+ it("should handle single matching key", () => {
1207
+ const server = createServer({
1208
+ qwen: { cost: { input: 0.1, output: 0.3 } },
1209
+ });
1210
+ expect(server.findOverrideForModel("qwen-3-8b")).toEqual({
1211
+ cost: { input: 0.1, output: 0.3 },
1212
+ });
1213
+ });
1214
+
1215
+ it("should handle overlapping but non-prefix matches", () => {
1216
+ const server = createServer({
1217
+ model: { cost: { input: 0.1 } },
1218
+ "model-a": { cost: { input: 0.2 } },
1219
+ });
1220
+ // "model" matches "model-a" and "model-b"
1221
+ // "model-a" matches only "model-a"
1222
+ expect(server.findOverrideForModel("model-a")).toEqual({
1223
+ cost: { input: 0.2 },
1224
+ });
1225
+ expect(server.findOverrideForModel("model-b")).toEqual({
1226
+ cost: { input: 0.1 },
1227
+ });
1228
+ });
1229
+
1230
+ it("should return overrides with capabilities and reasoning", () => {
1231
+ const server = createServer({
1232
+ qwen: { capabilities: ["text", "image"], reasoning: false },
1233
+ });
1234
+ expect(server.findOverrideForModel("qwen-3-8b")).toEqual({
1235
+ capabilities: ["text", "image"],
1236
+ reasoning: false,
1237
+ });
1238
+ });
1239
+ });