adaptive-memory-multi-model-router 2.13.18 → 2.13.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dockerignore +82 -0
- package/.env.example +303 -0
- package/.github/ISSUE_TEMPLATE/bug_report.md +83 -12
- package/.github/ISSUE_TEMPLATE/config.yml +12 -6
- package/.github/ISSUE_TEMPLATE/feature_request.md +61 -10
- package/.github/PULL_REQUEST_TEMPLATE.md +53 -26
- package/.github/dependabot.yml +9 -0
- package/.github/workflows/codeql.yml +38 -0
- package/.github/workflows/npm-publish.yml +20 -0
- package/.github/workflows/stale.yml +56 -0
- package/ARCHITECTURE.md +346 -0
- package/AUDIT_REPORT.md +28 -0
- package/CHANGELOG.md +386 -22
- package/CONTRIBUTORS.md +20 -0
- package/Dockerfile +53 -0
- package/Dockerfile.proxy +33 -0
- package/PR_STATUS_REPORT.md +148 -0
- package/README.md +22 -0
- package/RUNKIT.md +83 -0
- package/_schema.html +61 -15
- package/articles/AI_AGENT_LLM_ROUTING.md +150 -0
- package/articles/FROM_ZERO_TO_10K.md +107 -0
- package/articles/LLM_BENCHMARK_DEEP_DIVE.md +153 -0
- package/articles/TWEETS_10K_DOWNLOADS.md +47 -0
- package/articles/TWEETS_BENCHMARK_FIRST.md +46 -0
- package/articles/TWEETS_MCP_PLAY.md +51 -0
- package/articles/TWEETS_SEQUENTIAL_BROKEN.md +49 -0
- package/articles/TWEETS_WHY_BUILD.md +54 -0
- package/benchmark-results.json +26 -45
- package/cli/a3m +840 -0
- package/demo/package.json +13 -0
- package/demo/public/index.html +762 -0
- package/demo/server.js +405 -0
- package/dist/cli.js +4 -0
- package/docker-compose.yml +74 -0
- package/docs/.nojekyll +0 -0
- package/docs/BENCHMARK.md +96 -22
- package/docs/_config.yml +49 -0
- package/docs/api.html +513 -0
- package/docs/benchmark.html +387 -0
- package/docs/cli-cheatsheet.md +339 -0
- package/docs/comparison.md +108 -0
- package/docs/curl-examples.md +247 -0
- package/docs/index.html +390 -99
- package/docs/openapi.yaml +1318 -0
- package/docs/quick-start.html +366 -0
- package/docs/robots.txt +1 -1
- package/docs/sitemap.xml +23 -5
- package/docs/styles.css +682 -0
- package/examples/README.md +61 -0
- package/examples/a3m-sdk.js +124 -0
- package/examples/basic-route.js +54 -0
- package/examples/chat-loop.js +202 -0
- package/examples/classify-then-route.js +102 -0
- package/examples/cost-compare.js +120 -0
- package/examples/ensemble.js +160 -0
- package/integrations/langchain/README.md +216 -0
- package/integrations/langchain/a3m_langchain.ts +1360 -0
- package/integrations/langchain/example.ts +287 -0
- package/integrations/vercel-ai-sdk/README.md +49 -0
- package/integrations/vercel-ai-sdk/a3m_provider.ts +78 -0
- package/integrations/vercel-ai-sdk/example.ts +25 -0
- package/llms-full.txt +43 -0
- package/llms.txt +9 -0
- package/mcp-server/README.md +188 -0
- package/mcp-server/package.json +29 -0
- package/mcp-server/src/index.ts +744 -0
- package/mcp-server/tsconfig.json +19 -0
- package/package.json +3 -3
- package/proxy/README.md +227 -0
- package/proxy/package-lock.json +831 -0
- package/proxy/package.json +17 -0
- package/proxy/rate-limit.js +145 -0
- package/proxy/rate-limit.test.js +311 -0
- package/proxy/server.js +970 -0
- package/scripts/banner.js +29 -0
- package/scripts/compare-providers.sh +230 -0
- package/scripts/cross_post.py +443 -0
- package/scripts/publish_fcc.py +106 -0
- package/scripts/push-to-gitee.sh +52 -0
- package/src/tui/dashboard.ts +13 -0
- package/tests/__mocks__/tokenUtils.ts +22 -0
- package/tests/memory/episodicMemory.test.ts +227 -0
- package/tests/package-lock.json +1628 -0
- package/tests/package.json +18 -0
- package/tests/routing/ensembleVoting.test.ts +236 -0
- package/tests/routing/providerRetry.test.ts +360 -0
- package/tests/routing/queryTypePresets.test.ts +206 -0
- package/tests/tsconfig.json +21 -0
- package/tests/vitest.config.ts +18 -0
- package/.env +0 -2
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "a3m-router-tests",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"private": true,
|
|
5
|
+
"type": "module",
|
|
6
|
+
"scripts": {
|
|
7
|
+
"test": "vitest run",
|
|
8
|
+
"test:watch": "vitest",
|
|
9
|
+
"test:coverage": "vitest run --coverage"
|
|
10
|
+
},
|
|
11
|
+
"devDependencies": {
|
|
12
|
+
"typescript": "^5.8.0",
|
|
13
|
+
"vitest": "^3.1.0"
|
|
14
|
+
},
|
|
15
|
+
"dependencies": {
|
|
16
|
+
"nanoid": "^5.0.0"
|
|
17
|
+
}
|
|
18
|
+
}
|
|
@@ -0,0 +1,236 @@
|
|
|
1
|
+
import { describe, it, expect, vi } from 'vitest';
|
|
2
|
+
import {
|
|
3
|
+
executeEnsemble,
|
|
4
|
+
mergeComplementary,
|
|
5
|
+
recordFeedback,
|
|
6
|
+
} from '../../tmlpd-pi-extension/src/routing/ensembleVoting';
|
|
7
|
+
|
|
8
|
+
describe('executeEnsemble', () => {
|
|
9
|
+
const defaultExecutors: Record<string, (q: string, s: string, c: string) => Promise<string | null>> = {
|
|
10
|
+
groq: vi.fn().mockResolvedValue(
|
|
11
|
+
'Simple answer with no details.'
|
|
12
|
+
),
|
|
13
|
+
nvidia: vi.fn().mockResolvedValue(
|
|
14
|
+
'Detailed answer including 42 numerical references. The API endpoint app.ts handles requests. ' +
|
|
15
|
+
'* Point one\n* Point two\n* Point three\n* Point four\n* Point five\n' +
|
|
16
|
+
'The system uses Docker, Redis, and GCS for infrastructure. npm install is required.'
|
|
17
|
+
),
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
it('scores detailed responses higher than short ones', async () => {
|
|
21
|
+
const result = await executeEnsemble(
|
|
22
|
+
'test query',
|
|
23
|
+
'system prompt',
|
|
24
|
+
'',
|
|
25
|
+
defaultExecutors,
|
|
26
|
+
{ providers: ['groq', 'nvidia'] }
|
|
27
|
+
);
|
|
28
|
+
|
|
29
|
+
expect(result.scores['nvidia']).toBeGreaterThan(result.scores['groq']);
|
|
30
|
+
expect(result.winner).toBe('nvidia');
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
it('selects the provider with the highest score as winner', async () => {
|
|
34
|
+
const executors = {
|
|
35
|
+
low: vi.fn().mockResolvedValue('Hi'),
|
|
36
|
+
mid: vi.fn().mockResolvedValue('A moderate answer with some text.'),
|
|
37
|
+
high: vi.fn().mockResolvedValue(
|
|
38
|
+
'Excellent comprehensive response. Contains 3 key points. The API integration uses app.ts. ' +
|
|
39
|
+
'* Point A\n* Point B\n* Point C\n* Point D\n* Point E'
|
|
40
|
+
),
|
|
41
|
+
};
|
|
42
|
+
|
|
43
|
+
const result = await executeEnsemble(
|
|
44
|
+
'query',
|
|
45
|
+
'sys',
|
|
46
|
+
'',
|
|
47
|
+
executors,
|
|
48
|
+
{ providers: ['low', 'mid', 'high'] }
|
|
49
|
+
);
|
|
50
|
+
|
|
51
|
+
expect(result.winner).toBe('high');
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
it('handles providers returning null (errors)', async () => {
|
|
55
|
+
const executors = {
|
|
56
|
+
good: vi.fn().mockResolvedValue('Valid answer here.'),
|
|
57
|
+
bad: vi.fn().mockRejectedValue(new Error('API failure')),
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
const result = await executeEnsemble(
|
|
61
|
+
'query',
|
|
62
|
+
'sys',
|
|
63
|
+
'',
|
|
64
|
+
executors,
|
|
65
|
+
{ providers: ['good', 'bad'] }
|
|
66
|
+
);
|
|
67
|
+
|
|
68
|
+
expect(result.allResults['good']).toBe('Valid answer here.');
|
|
69
|
+
expect(result.allResults['bad']).toBeNull();
|
|
70
|
+
expect(result.scores['bad']).toBe(0);
|
|
71
|
+
expect(result.winner).toBe('good');
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
it('handles all providers failing', async () => {
|
|
75
|
+
const executors = {
|
|
76
|
+
a: vi.fn().mockRejectedValue(new Error('fail')),
|
|
77
|
+
b: vi.fn().mockRejectedValue(new Error('fail')),
|
|
78
|
+
};
|
|
79
|
+
|
|
80
|
+
const result = await executeEnsemble(
|
|
81
|
+
'query',
|
|
82
|
+
'sys',
|
|
83
|
+
'',
|
|
84
|
+
executors,
|
|
85
|
+
{ providers: ['a', 'b'] }
|
|
86
|
+
);
|
|
87
|
+
|
|
88
|
+
expect(result.best).toBe('');
|
|
89
|
+
expect(Object.values(result.scores).every(s => s === 0)).toBe(true);
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
it('applies length penalty for very short responses', async () => {
|
|
93
|
+
const executors = {
|
|
94
|
+
short: vi.fn().mockResolvedValue('Hi'),
|
|
95
|
+
normal: vi.fn().mockResolvedValue('A normal length answer with several words in it.'),
|
|
96
|
+
};
|
|
97
|
+
|
|
98
|
+
const result = await executeEnsemble(
|
|
99
|
+
'query',
|
|
100
|
+
'sys',
|
|
101
|
+
'',
|
|
102
|
+
executors,
|
|
103
|
+
{ providers: ['short', 'normal'] }
|
|
104
|
+
);
|
|
105
|
+
|
|
106
|
+
expect(result.scores['short']).toBeLessThan(result.scores['normal']);
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
it('applies specificity bonus for responses with numbers', async () => {
|
|
110
|
+
const executors = {
|
|
111
|
+
vague: vi.fn().mockResolvedValue('The system works well and is quite good.'),
|
|
112
|
+
specific: vi.fn().mockResolvedValue('The system processes 42 requests per second with 99.9% uptime.'),
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
const result = await executeEnsemble(
|
|
116
|
+
'query',
|
|
117
|
+
'sys',
|
|
118
|
+
'',
|
|
119
|
+
executors,
|
|
120
|
+
{ providers: ['vague', 'specific'] }
|
|
121
|
+
);
|
|
122
|
+
|
|
123
|
+
expect(result.scores['specific']).toBeGreaterThan(result.scores['vague']);
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
it('applies structure bonus for multi-line responses', async () => {
|
|
127
|
+
const executors = {
|
|
128
|
+
oneLine: vi.fn().mockResolvedValue('Just a single line answer here.'),
|
|
129
|
+
multiLine: vi.fn().mockResolvedValue('Line one\nLine two\nLine three\nLine four\nLine five'),
|
|
130
|
+
};
|
|
131
|
+
|
|
132
|
+
const result = await executeEnsemble(
|
|
133
|
+
'query',
|
|
134
|
+
'sys',
|
|
135
|
+
'',
|
|
136
|
+
executors,
|
|
137
|
+
{ providers: ['oneLine', 'multiLine'] }
|
|
138
|
+
);
|
|
139
|
+
|
|
140
|
+
expect(result.scores['multiLine']).toBeGreaterThan(result.scores['oneLine']);
|
|
141
|
+
});
|
|
142
|
+
|
|
143
|
+
it('includes timing information in result', async () => {
|
|
144
|
+
const executors = {
|
|
145
|
+
fast: vi.fn().mockImplementation(
|
|
146
|
+
() => new Promise(r => setTimeout(() => r('Quick answer.'), 5))
|
|
147
|
+
),
|
|
148
|
+
};
|
|
149
|
+
|
|
150
|
+
const result = await executeEnsemble(
|
|
151
|
+
'query',
|
|
152
|
+
'sys',
|
|
153
|
+
'',
|
|
154
|
+
executors,
|
|
155
|
+
{ providers: ['fast'] }
|
|
156
|
+
);
|
|
157
|
+
|
|
158
|
+
expect(result.timing.totalMs).toBeGreaterThanOrEqual(0);
|
|
159
|
+
expect(result.timing.perProvider['fast']).toBeGreaterThanOrEqual(0);
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
it('produces reasoning string explaining winner selection', async () => {
|
|
163
|
+
const executors = {
|
|
164
|
+
a: vi.fn().mockResolvedValue('Answer A with some detail.'),
|
|
165
|
+
b: vi.fn().mockResolvedValue('Answer B with more specific 42 details and technical API references.'),
|
|
166
|
+
};
|
|
167
|
+
|
|
168
|
+
const result = await executeEnsemble(
|
|
169
|
+
'query',
|
|
170
|
+
'sys',
|
|
171
|
+
'',
|
|
172
|
+
executors,
|
|
173
|
+
{ providers: ['a', 'b'] }
|
|
174
|
+
);
|
|
175
|
+
|
|
176
|
+
expect(result.reasoning).toBeTruthy();
|
|
177
|
+
expect(result.reasoning).toContain(result.winner);
|
|
178
|
+
});
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
describe('mergeComplementary', () => {
|
|
182
|
+
it('merges multiple results into sections', () => {
|
|
183
|
+
const merged = mergeComplementary(['Answer one', 'Answer two']);
|
|
184
|
+
expect(merged).toContain('### Provider 1');
|
|
185
|
+
expect(merged).toContain('### Provider 2');
|
|
186
|
+
expect(merged).toContain('Answer one');
|
|
187
|
+
expect(merged).toContain('Answer two');
|
|
188
|
+
expect(merged).toContain('---');
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
it('filters out empty results', () => {
|
|
192
|
+
const merged = mergeComplementary(['Valid', '', 'Also valid']);
|
|
193
|
+
expect(merged).toContain('### Provider 1');
|
|
194
|
+
expect(merged).toContain('### Provider 2');
|
|
195
|
+
expect(merged).not.toContain('### Provider 3');
|
|
196
|
+
});
|
|
197
|
+
|
|
198
|
+
it('truncates at maxLength', () => {
|
|
199
|
+
const long = 'A'.repeat(3000);
|
|
200
|
+
const merged = mergeComplementary([long, long], 500);
|
|
201
|
+
expect(merged.length).toBeLessThanOrEqual(500);
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
it('returns empty string for all-empty input', () => {
|
|
205
|
+
expect(mergeComplementary([])).toBe('');
|
|
206
|
+
expect(mergeComplementary(['', '', null as unknown as string])).toBe('');
|
|
207
|
+
});
|
|
208
|
+
});
|
|
209
|
+
|
|
210
|
+
describe('recordFeedback', () => {
|
|
211
|
+
it('increments good count for helpful feedback', () => {
|
|
212
|
+
const history: Record<string, { good: number; bad: number }> = {};
|
|
213
|
+
const updated = recordFeedback('groq', true, history);
|
|
214
|
+
expect(updated['groq'].good).toBe(1);
|
|
215
|
+
expect(updated['groq'].bad).toBe(0);
|
|
216
|
+
});
|
|
217
|
+
|
|
218
|
+
it('increments bad count for unhelpful feedback', () => {
|
|
219
|
+
const history: Record<string, { good: number; bad: number }> = { groq: { good: 1, bad: 0 } };
|
|
220
|
+
const updated = recordFeedback('groq', false, history);
|
|
221
|
+
expect(updated['groq'].good).toBe(1);
|
|
222
|
+
expect(updated['groq'].bad).toBe(1);
|
|
223
|
+
});
|
|
224
|
+
|
|
225
|
+
it('initializes history entry if missing', () => {
|
|
226
|
+
const history: Record<string, { good: number; bad: number }> = {};
|
|
227
|
+
const updated = recordFeedback('new-provider', true, history);
|
|
228
|
+
expect(updated['new-provider']).toEqual({ good: 1, bad: 0 });
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
it('returns the same history object (mutates in place)', () => {
|
|
232
|
+
const history: Record<string, { good: number; bad: number }> = {};
|
|
233
|
+
const updated = recordFeedback('groq', true, history);
|
|
234
|
+
expect(updated).toBe(history);
|
|
235
|
+
});
|
|
236
|
+
});
|
|
@@ -0,0 +1,360 @@
|
|
|
1
|
+
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
|
2
|
+
|
|
3
|
+
// Mock tokenUtils BEFORE importing providerRetry
|
|
4
|
+
vi.mock('../../src/utils/tokenUtils', () => ({
|
|
5
|
+
countTokens: (text: string) => {
|
|
6
|
+
if (!text || text.length === 0) return 0;
|
|
7
|
+
return Math.ceil(text.trim().split(/\s+/).length * 1.3);
|
|
8
|
+
},
|
|
9
|
+
estimateTokens: (text: string) => {
|
|
10
|
+
if (!text || text.length === 0) return 0;
|
|
11
|
+
return Math.ceil(text.trim().split(/\s+/).length * 1.3);
|
|
12
|
+
},
|
|
13
|
+
}));
|
|
14
|
+
|
|
15
|
+
import {
|
|
16
|
+
ProviderRetryHandler,
|
|
17
|
+
createRetryHandler,
|
|
18
|
+
DEFAULT_RETRY_CONFIG,
|
|
19
|
+
DEFAULT_PROVIDER_CONFIG,
|
|
20
|
+
PROVIDER_CONTEXT_LIMITS,
|
|
21
|
+
RetryConfig,
|
|
22
|
+
ProviderRetryConfig,
|
|
23
|
+
} from '../../src/routing/providerRetry';
|
|
24
|
+
|
|
25
|
+
// ============================================================
|
|
26
|
+
// HELPERS
|
|
27
|
+
// ============================================================
|
|
28
|
+
|
|
29
|
+
function expectInRange(actual: number, min: number, max: number, label: string): void {
|
|
30
|
+
expect(actual >= min && actual <= max).toBe(true);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
// ============================================================
|
|
34
|
+
// SUITE
|
|
35
|
+
// ============================================================
|
|
36
|
+
|
|
37
|
+
describe('ProviderRetryHandler', () => {
|
|
38
|
+
let handler: ProviderRetryHandler;
|
|
39
|
+
|
|
40
|
+
beforeEach(() => {
|
|
41
|
+
handler = new ProviderRetryHandler();
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
describe('constructor', () => {
|
|
45
|
+
it('initializes with default configs', () => {
|
|
46
|
+
expect(handler).toBeInstanceOf(ProviderRetryHandler);
|
|
47
|
+
});
|
|
48
|
+
|
|
49
|
+
it('initializes with custom provider configs', () => {
|
|
50
|
+
const custom: ProviderRetryConfig = {
|
|
51
|
+
customProvider: {
|
|
52
|
+
timeout: 5000,
|
|
53
|
+
retry: { maxRetries: 2, initialDelayMs: 500, maxDelayMs: 10000, backoffMultiplier: 2 },
|
|
54
|
+
rateLimitRetries: 3,
|
|
55
|
+
},
|
|
56
|
+
};
|
|
57
|
+
const customHandler = new ProviderRetryHandler(custom);
|
|
58
|
+
const cfg = customHandler.getConfig('customProvider');
|
|
59
|
+
expect(cfg.timeout).toBe(5000);
|
|
60
|
+
});
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
describe('getConfig', () => {
|
|
64
|
+
it('returns deepseek config with correct values', () => {
|
|
65
|
+
const cfg = handler.getConfig('deepseek');
|
|
66
|
+
expect(cfg.timeout).toBe(30000);
|
|
67
|
+
expect(cfg.retry.maxRetries).toBe(5);
|
|
68
|
+
expect(cfg.rateLimitRetries).toBe(3);
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
it('returns groq config with short timeout', () => {
|
|
72
|
+
const cfg = handler.getConfig('groq');
|
|
73
|
+
expect(cfg.timeout).toBe(10000);
|
|
74
|
+
expect(cfg.retry.maxRetries).toBe(2);
|
|
75
|
+
expect(cfg.rateLimitRetries).toBe(1);
|
|
76
|
+
});
|
|
77
|
+
|
|
78
|
+
it('falls back to default for unknown providers', () => {
|
|
79
|
+
const cfg = handler.getConfig('nonexistent');
|
|
80
|
+
expect(cfg.timeout).toBe(15000);
|
|
81
|
+
expect(cfg.retry.maxRetries).toBe(3);
|
|
82
|
+
});
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
describe('configureProvider', () => {
|
|
86
|
+
it('adds a new custom provider', () => {
|
|
87
|
+
handler.configureProvider('my-provider', {
|
|
88
|
+
timeout: 9999,
|
|
89
|
+
retry: { maxRetries: 10, initialDelayMs: 100, maxDelayMs: 5000, backoffMultiplier: 1.5 },
|
|
90
|
+
rateLimitRetries: 5,
|
|
91
|
+
});
|
|
92
|
+
const cfg = handler.getConfig('my-provider');
|
|
93
|
+
expect(cfg.timeout).toBe(9999);
|
|
94
|
+
expect(cfg.retry.maxRetries).toBe(10);
|
|
95
|
+
expect(cfg.retry.initialDelayMs).toBe(100);
|
|
96
|
+
expect(cfg.rateLimitRetries).toBe(5);
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
it('overrides existing provider partially', () => {
|
|
100
|
+
handler.configureProvider('groq', { timeout: 50000 });
|
|
101
|
+
const cfg = handler.getConfig('groq');
|
|
102
|
+
expect(cfg.timeout).toBe(50000);
|
|
103
|
+
// Other values unchanged
|
|
104
|
+
expect(cfg.retry.maxRetries).toBe(2);
|
|
105
|
+
expect(cfg.rateLimitRetries).toBe(1);
|
|
106
|
+
});
|
|
107
|
+
});
|
|
108
|
+
|
|
109
|
+
describe('isRetryableError', () => {
|
|
110
|
+
it('returns true for common transient errors', () => {
|
|
111
|
+
expect(handler.isRetryableError({ code: 'ECONNRESET' })).toBe(true);
|
|
112
|
+
expect(handler.isRetryableError({ code: 'ETIMEDOUT' })).toBe(true);
|
|
113
|
+
expect(handler.isRetryableError({ code: 'ECONNREFUSED' })).toBe(true);
|
|
114
|
+
expect(handler.isRetryableError({ code: 'EAI_AGAIN' })).toBe(true);
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
it('returns true for 5xx status codes', () => {
|
|
118
|
+
expect(handler.isRetryableError({ status: 500 })).toBe(true);
|
|
119
|
+
expect(handler.isRetryableError({ status: 502 })).toBe(true);
|
|
120
|
+
expect(handler.isRetryableError({ status: 503 })).toBe(true);
|
|
121
|
+
expect(handler.isRetryableError({ status: 504 })).toBe(true);
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
it('returns true for 429 rate limit', () => {
|
|
125
|
+
expect(handler.isRetryableError({ status: 429 })).toBe(true);
|
|
126
|
+
expect(handler.isRetryableError({ statusCode: 429 })).toBe(true);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
it('returns false for 4xx client errors', () => {
|
|
130
|
+
expect(handler.isRetryableError({ status: 400 })).toBe(false);
|
|
131
|
+
expect(handler.isRetryableError({ status: 404 })).toBe(false);
|
|
132
|
+
});
|
|
133
|
+
|
|
134
|
+
it('returns false for permanent provider state errors', () => {
|
|
135
|
+
expect(handler.isRetryableError({ status: 401 })).toBe(false);
|
|
136
|
+
expect(handler.isRetryableError({ status: 403 })).toBe(false);
|
|
137
|
+
expect(handler.isRetryableError({ message: 'insufficient balance' })).toBe(false);
|
|
138
|
+
expect(handler.isRetryableError({ message: 'invalid API key' })).toBe(false);
|
|
139
|
+
expect(handler.isRetryableError({ message: 'quota exhausted' })).toBe(false);
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
it('returns false for null/undefined error', () => {
|
|
143
|
+
expect(handler.isRetryableError(null)).toBe(false);
|
|
144
|
+
expect(handler.isRetryableError(undefined)).toBe(false);
|
|
145
|
+
});
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
describe('isRateLimitError', () => {
|
|
149
|
+
it('detects 429 in status field', () => {
|
|
150
|
+
expect(handler.isRateLimitError({ status: 429 })).toBe(true);
|
|
151
|
+
});
|
|
152
|
+
|
|
153
|
+
it('detects 429 in statusCode field', () => {
|
|
154
|
+
expect(handler.isRateLimitError({ statusCode: 429 })).toBe(true);
|
|
155
|
+
});
|
|
156
|
+
|
|
157
|
+
it('returns false for non-429 errors', () => {
|
|
158
|
+
expect(handler.isRateLimitError({ status: 200 })).toBe(false);
|
|
159
|
+
expect(handler.isRateLimitError({ status: 500 })).toBe(false);
|
|
160
|
+
expect(handler.isRateLimitError({})).toBe(false);
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
describe('calculateBackoffDelay', () => {
|
|
165
|
+
const defaultConfig: RetryConfig = {
|
|
166
|
+
maxRetries: 3,
|
|
167
|
+
initialDelayMs: 1000,
|
|
168
|
+
maxDelayMs: 30000,
|
|
169
|
+
backoffMultiplier: 2,
|
|
170
|
+
retryableErrors: ['ECONNRESET'],
|
|
171
|
+
};
|
|
172
|
+
|
|
173
|
+
it('returns delay in range [0.5*base, base] for attempt 0', () => {
|
|
174
|
+
const delay = handler.calculateBackoffDelay(0, defaultConfig);
|
|
175
|
+
// base = 1000, jitter = [500, 1000]
|
|
176
|
+
expectInRange(delay, 500, 1000, 'Attempt 0 delay');
|
|
177
|
+
});
|
|
178
|
+
|
|
179
|
+
it('returns delay in range [0.5*base, base] for attempt 1', () => {
|
|
180
|
+
const delay = handler.calculateBackoffDelay(1, defaultConfig);
|
|
181
|
+
// base = 1000 * 2^1 = 2000, jitter = [1000, 2000]
|
|
182
|
+
expectInRange(delay, 1000, 2000, 'Attempt 1 delay');
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
it('returns delay in range [0.5*base, base] for attempt 2', () => {
|
|
186
|
+
const delay = handler.calculateBackoffDelay(2, defaultConfig);
|
|
187
|
+
// base = 1000 * 2^2 = 4000, jitter = [2000, 4000]
|
|
188
|
+
expectInRange(delay, 2000, 4000, 'Attempt 2 delay');
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
it('caps delay at maxDelayMs', () => {
|
|
192
|
+
const cappedConfig: RetryConfig = { ...defaultConfig, maxDelayMs: 5000 };
|
|
193
|
+
const delay = handler.calculateBackoffDelay(10, cappedConfig);
|
|
194
|
+
expect(delay).toBeLessThanOrEqual(5000);
|
|
195
|
+
});
|
|
196
|
+
|
|
197
|
+
it('respects Retry-After header for 429 errors', () => {
|
|
198
|
+
const rateLimitError = {
|
|
199
|
+
status: 429,
|
|
200
|
+
headers: { 'retry-after': '5' },
|
|
201
|
+
};
|
|
202
|
+
const delay = handler.calculateBackoffDelay(0, defaultConfig, rateLimitError);
|
|
203
|
+
// Retry-After: 5 seconds = 5000ms, with some tolerance
|
|
204
|
+
expectInRange(delay, 4500, 5500, 'Retry-After delay');
|
|
205
|
+
});
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
describe('validateContextWindow', () => {
|
|
209
|
+
it('returns valid for short prompts', () => {
|
|
210
|
+
const result = handler.validateContextWindow('openai', 'Short prompt');
|
|
211
|
+
expect(result.valid).toBe(true);
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
it('returns invalid for long prompts on small-context providers', () => {
|
|
215
|
+
const longText = Array(10000).join('word '); // ~9 chars * 10000 = ~90K chars
|
|
216
|
+
const result = handler.validateContextWindow('cerebras', longText);
|
|
217
|
+
expect(result.valid).toBe(false);
|
|
218
|
+
expect(result.reason).toBeTruthy();
|
|
219
|
+
expect(result.suggestedProvider).toBeTruthy();
|
|
220
|
+
});
|
|
221
|
+
|
|
222
|
+
it('suggests a provider with larger context when validation fails', () => {
|
|
223
|
+
const longText = Array(10000).join('word ');
|
|
224
|
+
const result = handler.validateContextWindow('cerebras', longText);
|
|
225
|
+
expect(result.suggestedProvider).toBeTruthy();
|
|
226
|
+
const suggestedLimit = PROVIDER_CONTEXT_LIMITS[result.suggestedProvider!] || 0;
|
|
227
|
+
const cerebrasLimit = PROVIDER_CONTEXT_LIMITS['cerebras'] || 0;
|
|
228
|
+
expect(suggestedLimit).toBeGreaterThan(cerebrasLimit);
|
|
229
|
+
});
|
|
230
|
+
|
|
231
|
+
it('returns valid for large prompts on large-context providers', () => {
|
|
232
|
+
const longText = Array(10000).join('word ');
|
|
233
|
+
const result = handler.validateContextWindow('minimax', longText);
|
|
234
|
+
expect(result.valid).toBe(true);
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
it('accepts explicit token count parameter', () => {
|
|
238
|
+
const result = handler.validateContextWindow('groq', 'test', 100);
|
|
239
|
+
expect(result.valid).toBe(true);
|
|
240
|
+
});
|
|
241
|
+
});
|
|
242
|
+
|
|
243
|
+
describe('stats tracking', () => {
|
|
244
|
+
it('starts with zeroed stats', () => {
|
|
245
|
+
const stats = handler.getStats('openai');
|
|
246
|
+
expect(stats.totalRequests).toBe(0);
|
|
247
|
+
expect(stats.successfulRequests).toBe(0);
|
|
248
|
+
expect(stats.failedRequests).toBe(0);
|
|
249
|
+
expect(stats.totalRetries).toBe(0);
|
|
250
|
+
});
|
|
251
|
+
|
|
252
|
+
it('getAllStats returns all providers', () => {
|
|
253
|
+
const allStats = handler.getAllStats();
|
|
254
|
+
expect(allStats['openai']).toBeDefined();
|
|
255
|
+
expect(allStats['deepseek']).toBeDefined();
|
|
256
|
+
expect(allStats['groq']).toBeDefined();
|
|
257
|
+
expect(Object.keys(allStats).length).toBeGreaterThan(5);
|
|
258
|
+
});
|
|
259
|
+
|
|
260
|
+
it('resetStats clears single provider', () => {
|
|
261
|
+
handler.resetStats('openai');
|
|
262
|
+
const stats = handler.getStats('openai');
|
|
263
|
+
expect(stats.totalRequests).toBe(0);
|
|
264
|
+
});
|
|
265
|
+
|
|
266
|
+
it('resetStats with no arg clears all', () => {
|
|
267
|
+
handler.resetStats();
|
|
268
|
+
const allStats = handler.getAllStats();
|
|
269
|
+
for (const stats of Object.values(allStats)) {
|
|
270
|
+
expect(stats.totalRequests).toBe(0);
|
|
271
|
+
}
|
|
272
|
+
});
|
|
273
|
+
});
|
|
274
|
+
|
|
275
|
+
describe('executeWithRetry', () => {
|
|
276
|
+
it('succeeds on first attempt', async () => {
|
|
277
|
+
const fn = vi.fn().mockResolvedValue('success');
|
|
278
|
+
const result = await handler.executeWithRetry('groq', fn);
|
|
279
|
+
expect(result).toBe('success');
|
|
280
|
+
expect(fn).toHaveBeenCalledTimes(1);
|
|
281
|
+
});
|
|
282
|
+
|
|
283
|
+
it('retries on transient errors then succeeds', async () => {
|
|
284
|
+
const fn = vi.fn()
|
|
285
|
+
.mockRejectedValueOnce({ code: 'ECONNRESET', message: 'Connection reset' })
|
|
286
|
+
.mockRejectedValueOnce({ code: 'ECONNRESET', message: 'Connection reset' })
|
|
287
|
+
.mockResolvedValue('success after retry');
|
|
288
|
+
|
|
289
|
+
const result = await handler.executeWithRetry('groq', fn, {
|
|
290
|
+
onRetry: vi.fn(),
|
|
291
|
+
});
|
|
292
|
+
expect(result).toBe('success after retry');
|
|
293
|
+
expect(fn).toHaveBeenCalledTimes(3);
|
|
294
|
+
});
|
|
295
|
+
|
|
296
|
+
it('fails after exhausting retries', async () => {
|
|
297
|
+
const fn = vi.fn().mockRejectedValue({ code: 'ECONNRESET', message: 'Persistent failure' });
|
|
298
|
+
|
|
299
|
+
await expect(
|
|
300
|
+
handler.executeWithRetry('groq', fn)
|
|
301
|
+
).rejects.toThrow();
|
|
302
|
+
expect(fn).toHaveBeenCalledTimes(3); // initial + 2 retries (groq maxRetries=2)
|
|
303
|
+
});
|
|
304
|
+
|
|
305
|
+
it('does not retry on non-retryable errors', async () => {
|
|
306
|
+
const fn = vi.fn().mockRejectedValue({ status: 401, message: 'Unauthorized' });
|
|
307
|
+
|
|
308
|
+
await expect(
|
|
309
|
+
handler.executeWithRetry('groq', fn)
|
|
310
|
+
).rejects.toThrow();
|
|
311
|
+
expect(fn).toHaveBeenCalledTimes(1);
|
|
312
|
+
});
|
|
313
|
+
|
|
314
|
+
it('calls onRetry callback on each retry', async () => {
|
|
315
|
+
const fn = vi.fn()
|
|
316
|
+
.mockRejectedValueOnce({ code: 'ECONNRESET' })
|
|
317
|
+
.mockResolvedValue('ok');
|
|
318
|
+
const onRetry = vi.fn();
|
|
319
|
+
|
|
320
|
+
await handler.executeWithRetry('groq', fn, { onRetry });
|
|
321
|
+
expect(onRetry).toHaveBeenCalledTimes(1);
|
|
322
|
+
expect(onRetry).toHaveBeenCalledWith(1, { code: 'ECONNRESET' }, expect.any(Number));
|
|
323
|
+
});
|
|
324
|
+
|
|
325
|
+
it('respects custom timeout', async () => {
|
|
326
|
+
const slowFn = vi.fn().mockImplementation(
|
|
327
|
+
() => new Promise(r => setTimeout(r, 100))
|
|
328
|
+
);
|
|
329
|
+
|
|
330
|
+
// Short timeout should reject
|
|
331
|
+
await expect(
|
|
332
|
+
handler.executeWithRetry('groq', slowFn, { timeout: 5 })
|
|
333
|
+
).rejects.toThrow(/timed out/i);
|
|
334
|
+
});
|
|
335
|
+
});
|
|
336
|
+
|
|
337
|
+
describe('PROVIDER_CONTEXT_LIMITS', () => {
|
|
338
|
+
it('contains expected providers', () => {
|
|
339
|
+
expect(PROVIDER_CONTEXT_LIMITS['openai']).toBe(128000);
|
|
340
|
+
expect(PROVIDER_CONTEXT_LIMITS['anthropic']).toBe(200000);
|
|
341
|
+
expect(PROVIDER_CONTEXT_LIMITS['minimax']).toBe(1000000);
|
|
342
|
+
expect(PROVIDER_CONTEXT_LIMITS['groq']).toBe(32000);
|
|
343
|
+
expect(PROVIDER_CONTEXT_LIMITS['default']).toBe(8192);
|
|
344
|
+
});
|
|
345
|
+
});
|
|
346
|
+
|
|
347
|
+
describe('createRetryHandler', () => {
|
|
348
|
+
it('creates handler with custom configs', () => {
|
|
349
|
+
const custom = createRetryHandler({
|
|
350
|
+
testProv: {
|
|
351
|
+
timeout: 1000,
|
|
352
|
+
retry: { maxRetries: 1, initialDelayMs: 100, maxDelayMs: 5000, backoffMultiplier: 2 },
|
|
353
|
+
rateLimitRetries: 1,
|
|
354
|
+
},
|
|
355
|
+
});
|
|
356
|
+
const cfg = custom.getConfig('testProv');
|
|
357
|
+
expect(cfg.timeout).toBe(1000);
|
|
358
|
+
});
|
|
359
|
+
});
|
|
360
|
+
});
|