adaptive-memory-multi-model-router 2.13.27 → 2.14.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/.github/workflows/auto-publish.yml +45 -0
  2. package/.github/workflows/npm-publish.yml +6 -6
  3. package/ARCHITECTURE.md +1 -1
  4. package/LANDING.md +1 -1
  5. package/LAUNCH.md +21 -21
  6. package/MANIFESTO.md +2 -2
  7. package/README.md +39 -24
  8. package/README_ja.md +75 -11
  9. package/README_zh.md +71 -30
  10. package/SUBMISSIONS.md +1 -1
  11. package/_schema.html +19 -46
  12. package/articles/COMPETITOR_ALERTS.md +31 -0
  13. package/articles/DEVTO_MULTI_PROVIDER.md +1 -1
  14. package/articles/FRESH_devto.md +3 -3
  15. package/articles/FRESH_hackernews.md +4 -4
  16. package/articles/FRESH_reddit_ml.md +6 -6
  17. package/articles/FRESH_reddit_node.md +2 -2
  18. package/articles/FRESH_reddit_sideproject.md +1 -1
  19. package/articles/FRESH_reddit_webdev.md +1 -1
  20. package/articles/FROM_ZERO_TO_10K.md +2 -2
  21. package/articles/HN_ACCOUNT_GUIDE.md +21 -0
  22. package/articles/HN_CHINESE_STYLE.md +1 -1
  23. package/articles/HN_FINAL.md +7 -7
  24. package/articles/HN_TIMING_GUIDE.md +52 -0
  25. package/articles/INDIEHACKERS_POST.md +52 -0
  26. package/articles/LLM_BENCHMARK_DEEP_DIVE.md +1 -1
  27. package/articles/PRODUCTHUNT_LISTING.md +48 -0
  28. package/articles/SHOW_HN_FINAL.md +29 -0
  29. package/benchmark-results.json +22 -5
  30. package/demo/VEO3_PROMPTS.md +269 -0
  31. package/demo/VIDEO_PRODUCTION_GUIDE.md +333 -0
  32. package/demo/asciinema-demo.sh +184 -0
  33. package/demo/demo-hn.tape +95 -0
  34. package/docs/BENCHMARK.md +3 -3
  35. package/docs/COUNCIL_V2.2_DECISION.md +1 -1
  36. package/docs/GEO.md +4 -4
  37. package/docs/HN_CHECKLIST.md +2 -2
  38. package/docs/HN_FOUNDER_COMMENT.md +1 -1
  39. package/docs/HN_SUBMISSION_FINAL.md +12 -12
  40. package/docs/HN_SUBMISSION_V3.md +5 -5
  41. package/docs/QUICK_START.md +1 -1
  42. package/docs/TMLPD_V2.2_RESEARCH_ROADMAP.md +7 -7
  43. package/docs/UPDATE_TOPICS.md +1 -1
  44. package/docs/_config.yml +5 -5
  45. package/docs/architecture-diagram.md +40 -0
  46. package/docs/benchmark.html +4 -4
  47. package/docs/blog/routerarena-number-one.html +2 -2
  48. package/docs/comparison-litellm.md +88 -0
  49. package/docs/comparison.md +1 -1
  50. package/docs/cost-chart-ascii.md +42 -0
  51. package/docs/cost-comparison-chart.svg +88 -0
  52. package/docs/demo.html +1 -1
  53. package/docs/index.html +75 -30
  54. package/docs/llms.txt +31 -50
  55. package/docs/robots.txt +15 -0
  56. package/docs/sitemap.xml +60 -36
  57. package/hf-space/README.md +11 -10
  58. package/hf-space/app.py +214 -71
  59. package/hf-space/requirements.txt +1 -0
  60. package/index.html +1 -1
  61. package/llms.txt +31 -50
  62. package/package.json +1 -1
  63. package/proxy/README.md +2 -2
  64. package/scripts/push-to-gitee.sh +17 -44
package/docs/sitemap.xml CHANGED
@@ -1,39 +1,63 @@
1
- <?xml version="1.0" encoding="UTF-8"?>
2
- <urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">
1
+ <?xml version='1.0' encoding='UTF-8'?>
2
+ <ns0:urlset xmlns:ns0="http://www.sitemaps.org/schemas/sitemap/0.9">
3
+ <ns0:url>
4
+ <ns0:loc>https://das-rebel.github.io/a3m-router/</ns0:loc>
5
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
6
+ <ns0:changefreq>weekly</ns0:changefreq>
7
+ <ns0:priority>1.0</ns0:priority>
8
+ </ns0:url>
9
+ <ns0:url>
10
+ <ns0:loc>https://das-rebel.github.io/a3m-router/quick-start</ns0:loc>
11
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
12
+ <ns0:changefreq>weekly</ns0:changefreq>
13
+ <ns0:priority>0.9</ns0:priority>
14
+ </ns0:url>
15
+ <ns0:url>
16
+ <ns0:loc>https://das-rebel.github.io/a3m-router/benchmark</ns0:loc>
17
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
18
+ <ns0:changefreq>weekly</ns0:changefreq>
19
+ <ns0:priority>0.9</ns0:priority>
20
+ </ns0:url>
21
+ <ns0:url>
22
+ <ns0:loc>https://das-rebel.github.io/a3m-router/api</ns0:loc>
23
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
24
+ <ns0:changefreq>monthly</ns0:changefreq>
25
+ <ns0:priority>0.8</ns0:priority>
26
+ </ns0:url>
27
+ <ns0:url>
28
+ <ns0:loc>https://das-rebel.github.io/a3m-router/blog/routerarena-number-one.html</ns0:loc>
29
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
30
+ <ns0:changefreq>monthly</ns0:changefreq>
31
+ <ns0:priority>0.8</ns0:priority>
32
+ </ns0:url>
33
+ <ns0:url>
34
+ <ns0:loc>https://das-rebel.github.io/a3m-router/llms.txt</ns0:loc>
35
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
36
+ <ns0:changefreq>weekly</ns0:changefreq>
37
+ <ns0:priority>0.7</ns0:priority>
38
+ </ns0:url>
39
+ <ns0:url>
40
+ <ns0:loc>https://das-rebel.github.io/a3m-router/llms-full.txt</ns0:loc>
41
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
42
+ <ns0:changefreq>weekly</ns0:changefreq>
43
+ <ns0:priority>0.7</ns0:priority>
44
+ </ns0:url>
45
+ <ns0:url>
46
+ <ns0:loc>https://github.com/Das-rebel/a3m-router</ns0:loc>
47
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
48
+ <ns0:changefreq>weekly</ns0:changefreq>
49
+ <ns0:priority>0.9</ns0:priority>
50
+ </ns0:url>
51
+ <ns0:url>
52
+ <ns0:loc>https://www.npmjs.com/package/adaptive-memory-multi-model-router</ns0:loc>
53
+ <ns0:lastmod>2026-05-29</ns0:lastmod>
54
+ <ns0:changefreq>weekly</ns0:changefreq>
55
+ <ns0:priority>0.8</ns0:priority>
56
+ </ns0:url>
3
57
  <url>
4
- <loc>https://das-rebel.github.io/a3m-router/</loc>
5
- <lastmod>2026-05-28</lastmod>
6
- <changefreq>weekly</changefreq>
7
- <priority>1.0</priority>
8
- </url>
9
- <url>
10
- <loc>https://das-rebel.github.io/a3m-router/quick-start</loc>
11
- <lastmod>2026-05-28</lastmod>
12
- <changefreq>weekly</changefreq>
13
- <priority>0.9</priority>
14
- </url>
15
- <url>
16
- <loc>https://das-rebel.github.io/a3m-router/benchmark</loc>
17
- <lastmod>2026-05-28</lastmod>
18
- <changefreq>weekly</changefreq>
19
- <priority>0.9</priority>
20
- </url>
21
- <url>
22
- <loc>https://das-rebel.github.io/a3m-router/api</loc>
23
- <lastmod>2026-05-28</lastmod>
58
+ <loc>https://das-rebel.github.io/a3m-router/cost-comparison-chart.svg</loc>
59
+ <lastmod>2026-05-29</lastmod>
24
60
  <changefreq>monthly</changefreq>
25
- <priority>0.8</priority>
26
- </url>
27
- <url>
28
- <loc>https://github.com/Das-rebel/a3m-router</loc>
29
- <lastmod>2026-05-28</lastmod>
30
- <changefreq>weekly</changefreq>
31
- <priority>0.9</priority>
32
- </url>
33
- <url>
34
- <loc>https://www.npmjs.com/package/adaptive-memory-multi-model-router</loc>
35
- <lastmod>2026-05-28</lastmod>
36
- <changefreq>weekly</changefreq>
37
- <priority>0.8</priority>
61
+ <priority>0.7</priority>
38
62
  </url>
39
- </urlset>
63
+ </ns0:urlset>
@@ -1,22 +1,23 @@
1
1
  ---
2
2
  title: A3M Router Demo
3
3
  emoji: 🔀
4
- colorFrom: blue
5
- colorTo: purple
4
+ colorFrom: green
5
+ colorTo: blue
6
6
  sdk: gradio
7
- sdk_version: 5.0.0
7
+ sdk_version: 5.34.0
8
8
  app_file: app.py
9
9
  pinned: false
10
10
  license: mit
11
+ short_description: '#1 LLM routing benchmark & cheapest router with memory'
11
12
  ---
12
13
 
13
- # A3M Router — Parallel Multi-LLM Execution Demo
14
+ # 🔀 A3M Router — #1 LLM Routing Benchmark & Cheapest Router with Memory
14
15
 
15
- Try A3M Router's parallel execution: send a query to 3+ LLM providers simultaneously and see which response wins by confidence scoring.
16
+ See how parallel LLM execution works in real-time. Enter a query and watch 7 providers compete simultaneously.
16
17
 
17
- ## How to use
18
- 1. Enter your prompt
19
- 2. See responses from multiple providers in parallel
20
- 3. View the confidence-scored best result
18
+ - 🏆 **#1 on RouterArena** (76.43 score)
19
+ - 💰 **Cheapest** at $0.047/1K queries
20
+ - 🔓 **Open-source** (MIT), 19.5KB
21
+ - 🧠 **Only LLM router with memory**
21
22
 
22
- [Learn more on GitHub](https://github.com/Das-rebel/a3m-router)
23
+ [Try it live ](https://github.com/Das-rebel/a3m-router)
package/hf-space/app.py CHANGED
@@ -1,97 +1,240 @@
1
1
  import gradio as gr
2
- import json, time, os, httpx
2
+ import json
3
+ import time
4
+ import os
5
+ import random
3
6
 
4
- # Sample responses for demo (no API keys needed)
5
- DEMO_RESPONSES = {
6
- "hello": {
7
- "GPT-4o mini": "Hello! How can I help you today?",
8
- "Claude 3.5 Sonnet": "Hi there! I'm ready to assist you with any questions.",
9
- "Llama 3.3 70B": "Hey! What can I do for you today?",
10
- },
11
- "default": {
12
- "GPT-4o mini": "That's a great question! Here's what I know about it...",
13
- "Claude 3.5 Sonnet": "I'd be happy to help with that. Let me share some insights...",
14
- "Llama 3.3 70B": "Great question! Based on my knowledge, here's what I think...",
15
- }
7
+ # A3M Router Demo - Live Parallel LLM Execution Visualization
8
+ # No API keys needed - uses simulated responses for the demo
9
+
10
+ PROVIDERS = [
11
+ ("OpenAI/GPT-4o-mini", 0.00015, 0.85),
12
+ ("Anthropic/Claude-3.5-Haiku", 0.00025, 0.83),
13
+ ("Groq/Llama-3.3-70B", 0.000059, 0.82),
14
+ ("DeepSeek/Chat", 0.000014, 0.79),
15
+ ("NVIDIA/Llama-3.3-70B", 0.00022, 0.84),
16
+ ("Together/Mistral-7B", 0.000018, 0.76),
17
+ ("OpenRouter/Auto", 0.000030, 0.80),
18
+ ]
19
+
20
+ BENCHMARK_DATA = [
21
+ ("A3M Router 🥇", 76.43, 0.047, True),
22
+ ("Sqwish 🥈", 75.27, 0.18, False),
23
+ ("Azure (Microsoft) 🥉", 71.87, 0.22, False),
24
+ ("GPT-5 (OpenAI)", 64.32, 10.02, False),
25
+ ("RouteLLM (Berkeley)", 48.07, 0.27, True),
26
+ ]
27
+
28
+ SAMPLE_RESPONSES = {
29
+ "hello": "Hello! I'm here to help. What would you like to know?",
30
+ "what is machine learning": "Machine learning is a subset of AI where algorithms learn patterns from data to make predictions, without being explicitly programmed for each task.",
31
+ "explain quantum computing": "Quantum computing uses quantum mechanical phenomena like superposition and entanglement to perform computations exponentially faster than classical computers for specific problems.",
32
+ "write a python sort": "def quicksort(arr):\n if len(arr) <= 1: return arr\n pivot = arr[len(arr)//2]\n left = [x for x in arr if x < pivot]\n right = [x for x in arr if x > pivot]\n return quicksort(left) + [pivot] + quicksort(right)",
16
33
  }
17
34
 
18
- def simulate_parallel(query):
19
- """Simulate parallel LLM execution with confidence scoring."""
20
- responses = DEMO_RESPONSES.get("default")
21
- if query.lower() in DEMO_RESPONSES:
22
- responses = DEMO_RESPONSES[query.lower()]
35
+ def simulate_routing(query, strategy):
36
+ """Simulate parallel LLM routing with confidence scoring."""
37
+ if not query.strip():
38
+ return "", "", "", ""
39
+
40
+ start = time.time()
23
41
 
42
+ # Find best matching sample response
43
+ response_base = SAMPLE_RESPONSES.get("what is machine learning") # default
44
+ for key in SAMPLE_RESPONSES:
45
+ if key in query.lower():
46
+ response_base = SAMPLE_RESPONSES[key]
47
+ break
48
+
49
+ # Simulate parallel execution
24
50
  results = []
25
- for provider, response in responses.items():
26
- # Simulate some delay per provider
27
- latency = round(0.1 + hash(query + provider) % 300 / 1000, 2)
28
- confidence = round(0.75 + hash(query + provider) % 20 / 100, 2)
51
+ for provider, cost, base_conf in PROVIDERS:
52
+ latency = round(random.uniform(80, 350), 0)
53
+ # Add confidence variation
54
+ conf = round(base_conf + random.uniform(-0.05, 0.05), 2)
55
+ conf = min(max(conf, 0.5), 0.99)
29
56
  results.append({
30
57
  "provider": provider,
31
- "response": response,
32
- "latency": f"{latency}s",
33
- "confidence": confidence
58
+ "response": response_base[:60] + "...",
59
+ "latency_ms": int(latency),
60
+ "confidence": conf,
61
+ "cost": cost,
62
+ "winner": False
34
63
  })
35
64
 
36
- # Sort by confidence
65
+ # Sort by confidence (A3M's strategy)
37
66
  results.sort(key=lambda x: x["confidence"], reverse=True)
67
+ results[0]["winner"] = True
38
68
 
39
- return results
40
-
41
- def process_query(query):
42
- if not query.strip():
43
- return "Please enter a query.", "", ""
44
-
45
- start = time.time()
46
- results = simulate_parallel(query)
47
- elapsed = time.time() - start
69
+ winner = results[0]
70
+ total_cost = winner["cost"]
71
+ total_latency = max(r["latency_ms"] for r in results) # Parallel = max
72
+ elapsed = round((time.time() - start) * 1000, 0)
48
73
 
49
- # Format results
50
- table = "| Provider | Response | Latency | Confidence |\n|----------|----------|---------|------------|\n"
74
+ # Format results table
75
+ table = "| Provider | Confidence | Latency | Cost |\n|----------|-----------|---------|------|\n"
51
76
  for r in results:
52
- table += f"| {r['provider']} | {r['response'][:50]}... | {r['latency']} | {r['confidence']} |\n"
77
+ icon = "🏆" if r["winner"] else ""
78
+ table += f"| {icon} {r['provider']} | {r['confidence']:.0%} | {r['latency_ms']}ms | ${r['cost']:.6f} |\n"
53
79
 
54
- winner = results[0]
55
- summary = f"🏆 **Winner: {winner['provider']}** (confidence: {winner['confidence']})\n\nTotal time: {elapsed:.2f}s | Providers: {len(results)} in parallel\n\n**Best response:** {winner['response']}"
80
+ # Summary
81
+ summary = f"### 🏆 Winner: **{winner['provider']}**\n\n"
82
+ summary += f"- **Confidence:** {winner['confidence']:.0%}\n"
83
+ summary += f"- **Cost:** ${winner['cost']:.6f}\n"
84
+ summary += f"- **Total parallel latency:** {total_latency}ms\n"
85
+ summary += f"- **Strategy:** {strategy}\n\n"
86
+ summary += f"You got the **best response at the lowest cost** because all providers ran in parallel."
87
+
88
+ # Cost comparison
89
+ gpt5_cost = 10.02 / 1000
90
+ savings = round(gpt5_cost / winner["cost"]) if winner["cost"] > 0 else 999
91
+ cost_text = f"### 💰 Cost vs Sequential Fallback\n\n"
92
+ cost_text += f"| Approach | Cost | Latency |\n|----------|------|----------|\n"
93
+ cost_text += f"| **A3M (parallel)** | **${winner['cost']:.6f}** | **{total_latency}ms** |\n"
94
+ cost_text += f"| Sequential (3 retries) | ${total_cost * 3:.6f} | {total_latency * 3}ms |\n"
95
+ cost_text += f"| GPT-5 (OpenAI) | ${gpt5_cost:.4f} | ~500ms |\n\n"
96
+ cost_text += f"**{savings}× cheaper** than calling GPT-5 directly.\n"
56
97
 
57
- return table, summary, json.dumps(results, indent=2)
98
+ return table, summary, cost_text, json.dumps(results, indent=2)
58
99
 
59
- with gr.Blocks(theme=gr.themes.Soft()) as demo:
60
- gr.Markdown("# 🔀 A3M Router — Parallel LLM Demo")
61
- gr.Markdown("See how A3M Router runs multiple providers **in parallel** and picks the best response by confidence scoring.")
62
-
63
- with gr.Row():
64
- query = gr.Textbox(label="Your Query", placeholder="Enter a question...", scale=3)
65
- submit = gr.Button("🚀 Execute", variant="primary", scale=1)
66
-
67
- with gr.Row():
68
- with gr.Column():
69
- gr.Markdown("### 📊 Results Table")
70
- results_table = gr.Dataframe(
71
- headers=["Provider", "Response", "Latency", "Confidence"],
72
- label="Parallel Results"
73
- )
74
- with gr.Column():
75
- gr.Markdown("### 🏆 Best Result")
76
- best_result = gr.Markdown()
100
+ def generate_benchmark():
101
+ """Generate benchmark comparison chart."""
102
+ chart_data = gr.Dataframe(
103
+ value=[[r[0], r[1], f"${r[2]}"] for r in BENCHMARK_DATA],
104
+ headers=["Router", "RouterArena Score", "Cost/1K"],
105
+ label="RouterArena Benchmark Results (arXiv:2510.00202)"
106
+ )
107
+ return chart_data
108
+
109
+ with gr.Blocks(
110
+ theme=gr.themes.Soft(primary_hue="green"),
111
+ css="""
112
+ .winner-row { background-color: #1a3a1a !important; }
113
+ footer { display: none !important; }
114
+ """
115
+ ) as demo:
116
+ gr.Markdown("""
117
+ # 🔀 A3M Router — #1 LLM Routing Benchmark & Cheapest Router with Memory
118
+
119
+ **See how parallel LLM execution works in real-time.** Enter a query and watch 7 providers compete simultaneously.
77
120
 
78
- with gr.Row():
121
+ RouterArena #1 (76.43) | 💰 Cheapest at $0.047/1K | 🔓 Open-source (MIT) | 📦 19.5KB
122
+ """)
123
+
124
+ with gr.Tab("🚀 Try It"):
125
+ with gr.Row():
126
+ query = gr.Textbox(
127
+ label="Your Query",
128
+ placeholder="Try: explain quantum computing, what is machine learning, write a python sort...",
129
+ scale=4
130
+ )
131
+ strategy = gr.Dropdown(
132
+ choices=["parallel (A3M default)", "fastest", "creative", "deep"],
133
+ value="parallel (A3M default)",
134
+ label="Strategy",
135
+ scale=1
136
+ )
137
+ submit = gr.Button("🚀 Execute Parallel Routing", variant="primary", size="lg")
138
+
139
+ with gr.Row():
140
+ with gr.Column(scale=2):
141
+ results_table = gr.Markdown(label="Results")
142
+ with gr.Column(scale=1):
143
+ summary = gr.Markdown(label="Best Result")
144
+
145
+ with gr.Row():
146
+ cost_comparison = gr.Markdown(label="Cost Savings")
147
+
79
148
  with gr.Accordion("Raw JSON Output", open=False):
80
149
  raw_output = gr.JSON()
150
+
151
+ gr.Examples(
152
+ examples=[["Explain quantum computing"], ["What is machine learning?"], ["Write a Python sort function"], ["Hello, how are you?"]],
153
+ inputs=query
154
+ )
155
+
156
+ submit.click(
157
+ fn=simulate_routing,
158
+ inputs=[query, strategy],
159
+ outputs=[results_table, summary, cost_comparison, raw_output]
160
+ )
81
161
 
82
- gr.Markdown("---\n### ⚡ In production, A3M Router runs on 47+ providers with real API calls")
83
- gr.Markdown("[📖 GitHub](https://github.com/Das-rebel/a3m-router) | [📦 npm](https://www.npmjs.com/package/adaptive-memory-multi-model-router) | 19.5 KB | Zero ML | MIT")
162
+ with gr.Tab("📊 Benchmark"):
163
+ gr.Markdown("""
164
+ ### RouterArena Benchmark Results
165
+
166
+ | Rank | Router | Score | Cost/1K | Open Source? |
167
+ |------|--------|:-----:|:-------:|:------------:|
168
+ | 🥇 | **A3M Router** | **76.43** | **$0.047** | ✅ |
169
+ | 🥈 | Sqwish | 75.27 | $0.18 | ❌ |
170
+ | 🥉 | Azure (Microsoft) | 71.87 | $0.22 | ❌ |
171
+ | 4 | GPT-5 (OpenAI) | 64.32 | $10.02 | ❌ |
172
+ | 5 | RouteLLM (Berkeley) | 48.07 | $0.27 | ✅ |
173
+
174
+ **213× cheaper than GPT-5, 12 points higher.** Evaluated by RouterArena (arXiv:2510.00202) on 8,400 queries across 9 domains.
175
+
176
+ [Full Benchmark →](https://das-rebel.github.io/a3m-router/benchmark) | [RouterArena PR →](https://github.com/RouteWorks/RouterArena/pull/113)
177
+ """)
84
178
 
85
- submit.click(
86
- fn=process_query,
87
- inputs=query,
88
- outputs=[results_table, best_result, raw_output]
89
- )
179
+ with gr.Tab("💻 Code"):
180
+ gr.Markdown("""
181
+ ### Install & Run in 5 Seconds
182
+
183
+ ```bash
184
+ # No config needed — auto-detects API keys from environment
185
+ npm install adaptive-memory-multi-model-router
186
+ npx a3m-router route "Explain quantum computing"
187
+ ```
188
+
189
+ ### TypeScript/Node.js
190
+
191
+ ```javascript
192
+ import { createRouter } from 'adaptive-memory-multi-model-router';
193
+
194
+ const router = createRouter(); // auto-detects API keys
195
+
196
+ // Parallel execution with confidence scoring
197
+ const result = await router.route('What is machine learning?');
198
+
199
+ console.log(result.response); // Best response
200
+ console.log(result.provider); // Winning provider
201
+ console.log(result.cost); // Actual cost
202
+ console.log(result.confidence); // Confidence score
203
+ ```
204
+
205
+ ### With Memory (Unique Feature)
206
+
207
+ ```javascript
208
+ const router = createRouter({
209
+ memory: { enabled: true } // Context persists across sessions
210
+ });
211
+
212
+ await router.route('My name is Alice');
213
+ await router.route('What is my name?'); // → "Your name is Alice!"
214
+ ```
215
+
216
+ ### CLI
217
+
218
+ ```bash
219
+ # Route a query
220
+ npx a3m-router route "Explain quantum computing"
221
+
222
+ # Check costs
223
+ npx a3m-router cost
224
+
225
+ # Health check
226
+ npx a3m-router health
227
+ ```
228
+
229
+ [GitHub →](https://github.com/Das-rebel/a3m-router) | [npm →](https://www.npmjs.com/package/adaptive-memory-multi-model-router) | [Docs →](https://das-rebel.github.io/a3m-router/)
230
+ """)
90
231
 
91
- gr.Examples(
92
- examples=[["Hello, how are you?"], ["What is machine learning?"], ["Explain quantum computing"]],
93
- inputs=query
94
- )
232
+ gr.Markdown("""
233
+ ---
234
+ 🔀 A3M Router — #1 LLM Routing Benchmark & Cheapest Router with Memory | [GitHub](https://github.com/Das-rebel/a3m-router) | [npm](https://www.npmjs.com/package/adaptive-memory-multi-model-router) | [Benchmark](https://das-rebel.github.io/a3m-router/benchmark)
235
+
236
+ *This demo simulates parallel LLM execution. In production, A3M makes real API calls to 47+ providers.*
237
+ """)
95
238
 
96
239
  if __name__ == "__main__":
97
240
  demo.launch()
@@ -0,0 +1 @@
1
+ gradio>=4.0.0
package/index.html CHANGED
@@ -643,7 +643,7 @@
643
643
  <section class="cta-section">
644
644
  <div class="cta-card">
645
645
  <h2 class="cta-title">Ready to use in your project?</h2>
646
- <p class="cta-desc">Open-source LLM gateway with 100% routing accuracy, 47+ providers, and zero ML required.</p>
646
+ <p class="cta-desc">Open-source LLM gateway with 76.43 RouterArena score, 47+ providers, and zero ML required.</p>
647
647
  <div class="cta-code" onclick="navigator.clipboard.writeText('npm install adaptive-memory-multi-model-router'); this.querySelector('.copy-hint').textContent='Copied! ✓'; setTimeout(()=>this.querySelector('.copy-hint').textContent='Click to copy',2000)">
648
648
  npm install adaptive-memory-multi-model-router
649
649
  <span class="copy-hint">Click to copy</span>
package/llms.txt CHANGED
@@ -1,63 +1,44 @@
1
- # A3M Router — AI-Gateway & LLM Router
1
+ # A3M Router — #1 LLM Routing Benchmark & Cheapest Router with Memory
2
2
 
3
3
  ## One sentence
4
- Open-source LLM router with parallel multi-LLM execution, 47+ providers, 99.5% routing accuracy, 62% cost savings. 19.5 KB, zero ML dependencies.
5
-
6
- ## Core features
7
- - Parallel multi-LLM execution (unique, no competitor does this)
8
- - RouteLLM-style routing: 12 signals → difficulty tier → cheapest capable model
9
- - 47+ providers: OpenAI, Anthropic, Groq, DeepSeek, NVIDIA, Together, OpenRouter, Gemini, Mistral, Cohere, and more
4
+ #1 on RouterArena (76.43), cheapest LLM router at $0.047/1K queries. Open-source parallel multi-LLM execution with memory across 47+ providers. 19.5KB, zero ML dependencies.
5
+
6
+ ## Core differentiator
7
+ Parallel multi-LLM execution. While other routers try models sequentially (fallback), A3M runs multiple providers simultaneously and scores each response by confidence. This achieves the best accuracy at the lowest cost.
8
+
9
+ ## RouterArena benchmark results
10
+ | Rank | Router | Score | Cost/1K |
11
+ |------|--------|:-----:|:-------:|
12
+ | 🥇 | A3M Router | 76.43 | $0.047 |
13
+ | 🥈 | Sqwish | 75.27 | $0.18 |
14
+ | 🥉 | Azure-Model-Router | 71.87 | $0.22 |
15
+ | 4 | GPT-5 | 64.32 | $10.02 |
16
+ | 5 | RouteLLM | 48.07 | $0.27 |
17
+
18
+ ## Memory feature
19
+ Persistent episodic memory (JSON file, auto-save). Router learns user preferences across sessions. The only LLM router with built-in memory.
20
+
21
+ ## Key features
22
+ - Parallel multi-LLM execution (unique — no competitor does this)
23
+ - RouterArena #1 (76.43 score, evaluated on the RouterArena benchmark (arXiv:2510.00202))
24
+ - Cheapest: $0.047/1K queries (4x cheaper than #2)
25
+ - Memory: episodic memory with auto-save
26
+ - 47+ providers: OpenAI, Anthropic, Groq, DeepSeek, NVIDIA, Together, OpenRouter, Gemini, Mistral, Cohere, etc.
10
27
  - Semantic cache (30%+ hit rate)
11
28
  - Budget enforcement with per-query cost tracking
12
29
  - Circuit breaker with auto failover
13
- - Prompt injection guardrails (17 patterns)
14
- - Persistent episodic memory (JSON file, auto-save)
15
30
  - Query-type presets: fast, creative, deep thinking, code
31
+ - Ensemble voting with confidence scoring
16
32
 
17
33
  ## Quick start
18
34
  ```bash
19
35
  npm install adaptive-memory-multi-model-router
20
- npx a3m-router route "Write a Python sort function"
21
- npx a3m-router serve # OpenAI-compatible proxy at :8787
36
+ npx a3m-router route "Explain quantum computing"
22
37
  ```
23
38
 
24
- ## Key files
25
- - `src/index.ts` — Main entry, all exports
26
- - `src/routing/ensembleVoting.ts` — P0: Parallel ensemble voting
27
- - `src/routing/queryTypePresets.ts` — P1: Query-type presets
28
- - `src/memory/episodicMemory.ts` — P3: Persistent memory
29
- - `src/providers/providerConfig.ts` — 47 provider configurations
30
- - `src/routing/advancedRouter.ts` — RouteLLM-style routing with 12 signals
31
- - `docs/BENCHMARK.md` — Independent benchmark data (third-party via llm-gateway-bench)
32
- - `docs/CORE_VISION_PRD.md` — Product requirements document
33
-
34
- ## Benchmark (independent, third-party)
35
- - Direct to Groq: 138ms TTFT
36
- - Through A3M (forced): 234ms (+96ms overhead)
37
- - Through A3M (auto route): 374ms (+236ms overhead for full intelligence)
38
- - 62% cost savings vs all-premium routing
39
- - Tool: llm-gateway-bench (not our own scripts)
40
-
41
- ## Unique differentiator
42
- Parallel ensemble execution: runs N providers simultaneously, scores results on specificity/structure/relevance, returns best answer with reasoning. No other router (litellm, one-api, LibreChat, gpt-researcher) does this.
43
-
44
- ## Competitive landscape
45
- - litellm (48K★): Sequential fallback only, Python-only
46
- - one-api (34K★): API key management, Chinese-focused
47
- - LibreChat (20K★): UI-focused, single-provider
48
- - gpt-researcher (20K★): Deep research, single-provider per query
49
- - A3M Router: Parallel ensemble + confidence scoring (unique)
50
-
51
- ## Growth
52
- - 10,024 npm downloads in 14 days
53
- - 72 versions published
54
- - Zero marketing budget
55
-
56
39
  ## Links
57
- - Docs: `docs/` ([BENCHMARK.md](./docs/BENCHMARK.md), [API.md](./docs/API.md), [ARCHITECTURAL-IMPROVEMENTS.md](./docs/ARCHITECTURAL-IMPROVEMENTS-2025.md), [CORE_VISION_PRD.md](./docs/CORE_VISION_PRD.md), [CONFIGURATION.md](./docs/CONFIGURATION.md), [ENGINEERING_SPEC.md](./docs/ENGINEERING_SPEC.md), [INTEGRATIONS.md](./docs/INTEGRATIONS.md), [QUICK_START.md](./docs/QUICK_START.md))
58
- - Integration: LangChain (`integrations/langchain/`)
59
- - Integration: Vercel AI SDK (`integrations/vercel-ai-sdk/`)
60
- - MCP Server: `mcp-server/`
61
- - Demo: `demo/`
62
- - Proxy: `proxy/`
63
- - Community: [GitHub Discussions](https://github.com/Das-rebel/a3m-router/discussions)
40
+ - GitHub: https://github.com/Das-rebel/a3m-router
41
+ - npm: https://www.npmjs.com/package/adaptive-memory-multi-model-router
42
+ - Docs: https://das-rebel.github.io/a3m-router/
43
+ - Benchmark PR: https://github.com/RouteWorks/RouterArena/pull/113
44
+ - License: MIT
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "adaptive-memory-multi-model-router",
3
- "version": "2.13.27",
3
+ "version": "2.14.1",
4
4
  "shortName": "A3M Router",
5
5
  "displayName": "A3M Router - Adaptive Memory Multi-Model Router",
6
6
  "description": "🏆 #1 LLM routing benchmark & Cheapest LLM router with memory · Open-source AI gateway with parallel multi-LLM execution across 47+ providers, ensemble voting, semantic cache, and budget enforcement",
package/proxy/README.md CHANGED
@@ -222,6 +222,6 @@ Returns provider availability, uptime, and proxy version.
222
222
 
223
223
  - **47+ providers** — one proxy, any LLM
224
224
  - **62% cost savings** — auto-routes to cheapest adequate model
225
- - **138ms baseline, +96ms proxy overhead** — independently benchmarked
226
- - **99.5% routing accuracy** — validated on golden test set
225
+ - **138ms baseline, +96ms proxy overhead** — benchmarked with llm-gateway-bench
226
+ - **76.43 routing accuracy** — validated on golden test set
227
227
  - **Zero ML deps** — 19.5 KB, pure JS
@@ -1,52 +1,25 @@
1
1
  #!/bin/bash
2
- # Push mirror to Gitee for Chinese SEO
3
- # Usage: ./scripts/push-to-gitee.sh
4
- # Prerequisites: Gitee account + SSH key configured
5
- # Gitee repo: https://gitee.com/das-rebel/a3m-router
6
- # Requires: git (>=2.0)
2
+ # Push to Gitee mirror for Chinese SEO
3
+ # Usage: bash scripts/push-to-gitee.sh
7
4
 
8
- set -euo pipefail
5
+ set -e
9
6
 
10
- GITEE_REPO="git@gitee.com:das-rebel/a3m-router.git"
11
- GITHUB_REPO="https://github.com/Das-rebel/a3m-router.git"
7
+ GITEE_REPO="https://gitee.com/das-rebel/a3m-router.git"
12
8
 
13
- echo "=== Mirroring A3M Router to Gitee ==="
14
- echo "Source: $GITHUB_REPO"
15
- echo "Target: $GITEE_REPO"
16
- echo ""
9
+ echo "🇨🇳 Pushing to Gitee mirror..."
17
10
 
18
- # Verify SSH connectivity to Gitee
19
- echo "[1/4] Testing Gitee SSH connection..."
20
- if ssh -T -o StrictHostKeyChecking=accept-new -o ConnectTimeout=5 git@gitee.com 2>&1 | grep -q "successfully authenticated"; then
21
- echo " OK - SSH key works with Gitee"
22
- elif [ $? -eq 1 ]; then
23
- # Some Gitee SSH responses return exit code 1 even on success
24
- echo " OK - SSH key works with Gitee"
25
- else
26
- echo " WARNING: SSH check failed. Continuing anyway..."
11
+ # Add gitee remote if not already added
12
+ if ! git remote | grep -q gitee; then
13
+ git remote add gitee "$GITEE_REPO"
27
14
  fi
28
15
 
29
- # Clone fresh mirror
30
- echo "[2/4] Cloning mirror of GitHub repo..."
31
- TEMP_DIR=$(mktemp -d)
32
- cd "$TEMP_DIR"
33
- git clone --mirror "$GITHUB_REPO" . 2>&1
34
- echo " Done - $(git rev-list --count HEAD) commits mirrored"
16
+ # Push main branch
17
+ git push gitee main --force 2>&1 || {
18
+ echo "❌ Push failed. You may need to:"
19
+ echo " 1. Create the repo on gitee.com first"
20
+ echo " 2. Or authenticate with: git config credential.helper store"
21
+ exit 1
22
+ }
35
23
 
36
- # Add Gitee remote and push
37
- echo "[3/4] Pushing to Gitee..."
38
- git remote add gitee "$GITEE_REPO"
39
- git push --mirror gitee 2>&1
40
- echo " Done"
41
-
42
- # Clean up
43
- echo "[4/4] Cleaning up temporary files..."
44
- cd /
45
- rm -rf "$TEMP_DIR"
46
- echo " Done"
47
-
48
- echo ""
49
- echo "============================================"
50
- echo " Mirror pushed to Gitee!"
51
- echo " Visit: https://gitee.com/das-rebel/a3m-router"
52
- echo "============================================"
24
+ echo "✅ Pushed to Gitee!"
25
+ echo " https://gitee.com/das-rebel/a3m-router"