adaptive-memory-multi-model-router 2.16.2 → 2.16.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/.github/CODEOWNERS +2 -0
  2. package/.github/FUNDING.yml +7 -1
  3. package/.github/ISSUE_TEMPLATE/bug_report.md +56 -0
  4. package/.github/ISSUE_TEMPLATE/feature_request.md +41 -0
  5. package/.github/workflows/pages.yml +1 -1
  6. package/CHANGELOG.md +15 -8
  7. package/CONTRIBUTING.md +109 -25
  8. package/README.md +131 -219
  9. package/data/jev-distill.jsonl +320 -0
  10. package/dist/routing/jev/jevRouter.d.ts +49 -0
  11. package/dist/routing/jev/jevRouter.js +231 -0
  12. package/dist/routing/jev/jevRouter.js.map +1 -0
  13. package/dist/routing/jev/optionAttention.d.ts +63 -0
  14. package/dist/routing/jev/optionAttention.js +158 -0
  15. package/dist/routing/jev/optionAttention.js.map +1 -0
  16. package/dist/routing/jev/remote.d.ts +14 -0
  17. package/dist/routing/jev/remote.js +55 -0
  18. package/dist/routing/jev/remote.js.map +1 -0
  19. package/dist/routing/jev/types.d.ts +71 -0
  20. package/dist/routing/jev/types.js +19 -0
  21. package/dist/routing/jev/types.js.map +1 -0
  22. package/dist/routing/jev/weights/jev-router-weights.json +1 -0
  23. package/dist/server/modelMapper.js +18 -0
  24. package/dist/server/modelMapper.js.map +1 -1
  25. package/docs/assets/og-banner.svg +193 -0
  26. package/docs/index.html +1523 -465
  27. package/docs-site/index.html +1427 -563
  28. package/package.json +29 -20
  29. package/python/pyproject.toml +1 -1
  30. package/src/cli/setupWizard.ts +4 -2
  31. package/src/routing/jev/jevRouter.ts +251 -0
  32. package/src/routing/jev/optionAttention.ts +186 -0
  33. package/src/routing/jev/remote.ts +50 -0
  34. package/src/routing/jev/types.ts +79 -0
  35. package/src/routing/jev/weights/jev-router-weights.json +1 -0
  36. package/src/server/modelMapper.ts +17 -0
  37. package/tests/routing/jev.test.ts +110 -0
  38. package/tools/calibrate_temp.py +81 -0
  39. package/tools/distill.mjs +143 -0
  40. package/tools/train_jev.py +213 -0
  41. package/dist/cli/tui.d.ts +0 -6
  42. package/dist/cli/tui.js.map +0 -1
  43. package/dist/routing/shadowSampler.d.ts.map +0 -1
  44. /package/{ARCHITECTURE.md → archive/ARCHITECTURE.md} +0 -0
  45. /package/{ENTERPRISE_INTEGRATIONS.md → archive/ENTERPRISE_INTEGRATIONS.md} +0 -0
  46. /package/{MANIFESTO.md → archive/MANIFESTO.md} +0 -0
  47. /package/{README_ja.md → archive/README_ja.md} +0 -0
  48. /package/{README_zh.md → archive/README_zh.md} +0 -0
  49. /package/{SECURITY.md → archive/SECURITY.md} +0 -0
  50. /package/{TECHNICAL_README.md → archive/TECHNICAL_README.md} +0 -0
  51. /package/{TODO_BROWSER_AUTOMATION.md → archive/TODO_BROWSER_AUTOMATION.md} +0 -0
  52. /package/{AGENT_COUNCIL_FINDINGS.md → archive/campaign/AGENT_COUNCIL_FINDINGS.md} +0 -0
  53. /package/{AUDIT_REPORT.md → archive/campaign/AUDIT_REPORT.md} +0 -0
  54. /package/{CONTRIBUTORS.md → archive/campaign/CONTRIBUTORS.md} +0 -0
  55. /package/{IMPROVEMENT_PLAN.md → archive/campaign/IMPROVEMENT_PLAN.md} +0 -0
  56. /package/{INTEGRATION_PROGRESS.md → archive/campaign/INTEGRATION_PROGRESS.md} +0 -0
  57. /package/{CAMPAIGN_SUMMARY.md → archive/launch/CAMPAIGN_SUMMARY.md} +0 -0
  58. /package/{LANDING.md → archive/launch/LANDING.md} +0 -0
  59. /package/{LAUNCH-PAIN-DRIVEN.md → archive/launch/LAUNCH-PAIN-DRIVEN.md} +0 -0
  60. /package/{LAUNCH.md → archive/launch/LAUNCH.md} +0 -0
  61. /package/{LAUNCH_CHECKLIST.md → archive/launch/LAUNCH_CHECKLIST.md} +0 -0
  62. /package/{LAUNCH_SNAPSHOT.md → archive/launch/LAUNCH_SNAPSHOT.md} +0 -0
  63. /package/{REDESIGN.md → archive/launch/REDESIGN.md} +0 -0
  64. /package/{HEALTH_REPORT.md → archive/research/HEALTH_REPORT.md} +0 -0
  65. /package/{OPPORTUNITIES_100.md → archive/research/OPPORTUNITIES_100.md} +0 -0
  66. /package/{POPULARITY_BOOSTERS.md → archive/research/POPULARITY_BOOSTERS.md} +0 -0
  67. /package/{PR_STATUS_REPORT.md → archive/research/PR_STATUS_REPORT.md} +0 -0
  68. /package/{research-log.md → archive/research/research-log.md} +0 -0
  69. /package/{RELEASE_v2.16.0.md → archive/submissions/RELEASE_v2.16.0.md} +0 -0
  70. /package/{RUNKIT.md → archive/submissions/RUNKIT.md} +0 -0
  71. /package/{SUBMISSIONS.md → archive/submissions/SUBMISSIONS.md} +0 -0
  72. /package/{a3m-integrations-summary.md → archive/submissions/a3m-integrations-summary.md} +0 -0
  73. /package/{discoverability-diagnosis.md → archive/submissions/discoverability-diagnosis.md} +0 -0
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "adaptive-memory-multi-model-router",
3
- "version": "2.16.2",
4
- "description": "A3M Router: Parallel LLM routing gateway with ensemble voting — 14% faster than OpenRouter, 92% cheaper | 80+ providers | Fixed: main entry point, TypeScript 5.8, blessed types | npm: 6K+/mo | PyPI: 700+/mo",
3
+ "version": "2.16.4",
4
+ "description": "Intelligent LLM routing gateway — 80+ providers, pheromone-trail failover, semantic cache. Saves 70-95% on AI costs. Drop-in OpenAI-compatible proxy. Self-hosted.",
5
5
  "main": "dist/index.js",
6
6
  "bin": {
7
7
  "a3m-router": "./dist/cli.js",
@@ -14,31 +14,40 @@
14
14
  "test:coverage": "npx vitest run --coverage",
15
15
  "test:watch": "npx vitest",
16
16
  "lint": "eslint src/",
17
- "build": "tsc -p tsconfig.build.json"
17
+ "build": "tsc -p tsconfig.build.json && node -e \"const fs=require('fs');fs.mkdirSync('dist/routing/jev/weights',{recursive:true});try{fs.copyFileSync('src/routing/jev/weights/jev-router-weights.json','dist/routing/jev/weights/jev-router-weights.json')}catch(e){}\"",
18
+ "jev:distill": "node tools/distill.mjs",
19
+ "jev:train": "python3 tools/train_jev.py && python3 tools/calibrate_temp.py"
18
20
  },
19
21
  "keywords": [
22
+ "llm-router",
23
+ "openrouter-alternative",
24
+ "litellm-alternative",
25
+ "llm-failover",
26
+ "llm-cost-optimization",
27
+ "multi-provider-llm",
28
+ "llm-api-gateway",
29
+ "self-hosted-llm",
30
+ "anthropic-openai-gateway",
31
+ "llm-proxy",
32
+ "circuit-breaker-llm",
33
+ "semantic-cache-llm",
34
+ "agentic-ai-gateway",
35
+ "routellm",
36
+ "openrouter",
37
+ "litellm",
20
38
  "llm",
21
39
  "router",
22
40
  "gateway",
23
- "openai",
41
+ "openai-proxy",
24
42
  "anthropic",
25
- "multi-provider",
26
- "cost-optimization",
27
- "langchain",
28
- "llamaindex",
29
- "agentkit",
43
+ "deepseek",
44
+ "groq",
45
+ "mistral",
46
+ "gemini-api",
47
+ "ollama",
30
48
  "a3m",
31
- "openrouter",
32
- "litellm",
33
- "proxy",
34
- "api-gateway",
35
- "artificial-intelligence",
36
- "machine-learning",
37
- "nlp",
38
- "transformers",
39
- "embeddings",
40
- "rag",
41
- "vector-search"
49
+ "langchain",
50
+ "agentkit"
42
51
  ],
43
52
  "repository": {
44
53
  "type": "git",
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "a3m-router"
7
- version = "2.16.2"
7
+ version = "2.16.3"
8
8
  description = "Auto-selects cheapest capable LLM from 47+ providers — 70-95% cost savings"
9
9
  readme = "README.md"
10
10
  license = {text = "MIT"}
@@ -1,6 +1,8 @@
1
1
  /**
2
2
  * A3M Router Setup Wizard
3
3
  * Interactive configuration wizard with smart defaults
4
+ *
5
+ * Note: Excluded from tsconfig.build.json — type errors here do not affect the build.
4
6
  */
5
7
 
6
8
  const fs = require('fs');
@@ -221,9 +223,9 @@ function createInterface() {
221
223
  });
222
224
  }
223
225
 
224
- function question(rl, text) {
226
+ function question(rl: readline.Interface, text: string): Promise<string> {
225
227
  return new Promise((resolve) => {
226
- rl.question(text, (answer) => resolve(answer));
228
+ rl.question(text, (answer: string) => resolve(answer));
227
229
  });
228
230
  }
229
231
 
@@ -0,0 +1,251 @@
1
+ /**
2
+ * JevRouter — System One routing for A3M.
3
+ *
4
+ * Replaces the autoregressive/heuristic routing loop with a single-pass
5
+ * calibrated decision (Jev interface pattern):
6
+ *
7
+ * prompt ──► OptionAttention engine ──► { provider choice, complexity score,
8
+ * capability noul flags }
9
+ *
10
+ * The engine is distilled from A3M's own System 2 router (advancedRouter.ts):
11
+ * tools/distill.mjs runs routeQuery over a synthetic prompt corpus and
12
+ * tools/train_jev.py trains the ~25k-parameter option-attention weights.
13
+ *
14
+ * Confidence guard: if the top provider probability is below MIN_CONFIDENCE,
15
+ * we fall back to the full heuristic router (System 2) — Jev decides the easy
16
+ * 90%, the heavy router handles the hard 10%.
17
+ */
18
+
19
+ import * as fs from "fs";
20
+ import * as path from "path";
21
+ import {
22
+ routeQuery,
23
+ type RouteDecision,
24
+ } from "../advancedRouter";
25
+ import {
26
+ scoreOptions,
27
+ scoreScalar,
28
+ scoreNoul,
29
+ resetQueryCache,
30
+ type OptionAttentionWeights,
31
+ } from "./optionAttention";
32
+ import { remoteConfigured, remoteDecide } from "./remote";
33
+ import type { JevResponse } from "./types";
34
+
35
+ const WEIGHTS_PATHS = [
36
+ path.join(__dirname, "weights", "jev-router-weights.json"), // dist copy
37
+ path.join(__dirname, "..", "..", "..", "src", "routing", "jev", "weights", "jev-router-weights.json"), // repo fallback
38
+ ];
39
+
40
+ /** Below this top-probability, defer to the System 2 heuristic router. */
41
+ export const MIN_CONFIDENCE = 0.22;
42
+
43
+ let cachedWeights: OptionAttentionWeights | null | undefined;
44
+
45
+ export function loadWeights(): OptionAttentionWeights | null {
46
+ if (cachedWeights !== undefined) return cachedWeights;
47
+ try {
48
+ for (const p of WEIGHTS_PATHS) {
49
+ if (fs.existsSync(p)) {
50
+ cachedWeights = JSON.parse(fs.readFileSync(p, "utf8")) as OptionAttentionWeights;
51
+ return cachedWeights;
52
+ }
53
+ }
54
+ cachedWeights = null;
55
+ } catch {
56
+ cachedWeights = null;
57
+ }
58
+ return cachedWeights;
59
+ }
60
+
61
+ /** For tests / hot-reload of retrained weights. */
62
+ export function resetWeightsCache(): void {
63
+ cachedWeights = undefined;
64
+ resetQueryCache();
65
+ }
66
+
67
+ /**
68
+ * Build the option text for a model profile. Option text is what the engine
69
+ * embeds — providers unseen at training time are still scored through their
70
+ * strengths/tier description.
71
+ */
72
+ export function optionTextForModel(model: string, profile: {
73
+ providerName?: string;
74
+ strengths?: string[];
75
+ quality_score?: number;
76
+ cost_per_1k_input?: number;
77
+ cost_per_1k_output?: number;
78
+ type?: string;
79
+ }): string {
80
+ const strengths = (profile.strengths || []).slice(0, 6).join(",");
81
+ const cost =
82
+ profile.cost_per_1k_input != null
83
+ ? ((profile.cost_per_1k_input + (profile.cost_per_1k_output || 0)) / 2).toFixed(4)
84
+ : "?";
85
+ return `${model} provider:${profile.providerName || "?"} type:${profile.type || "api"} strengths:${strengths} quality:${profile.quality_score ?? "?"} cost:${cost}`;
86
+ }
87
+
88
+ /**
89
+ * Single-pass Jev routing decision. Returns a RouteDecision shaped exactly
90
+ * like advancedRouter.routeQuery so it is a drop-in replacement.
91
+ */
92
+ export function jevRoute(
93
+ prompt: string,
94
+ available_models?: string[],
95
+ budget_multiplier: number = 1.0
96
+ ): RouteDecision & { jev?: JevResponse } {
97
+ // 1. Remote backend (openjev-sglang / typesafe) takes precedence if configured
98
+ if (remoteConfigured()) {
99
+ // remoteDecide is async; jevRoute is sync to match routeQuery's contract.
100
+ // Remote mode is exposed via jevRouteAsync below.
101
+ }
102
+
103
+ const weights = loadWeights();
104
+ if (!weights) return fallbackRoute(prompt, available_models, budget_multiplier, "no weights");
105
+
106
+ // 2. Candidate pool — mirror routeQuery's candidate selection
107
+ const profiles = getModelProfilesSafe();
108
+ const candidateNames = (available_models && available_models.length
109
+ ? available_models
110
+ : Object.keys(profiles)
111
+ ).filter((n) => profiles[n]);
112
+ if (candidateNames.length === 0)
113
+ return fallbackRoute(prompt, available_models, budget_multiplier, "no candidates");
114
+
115
+ const t0 = Date.now();
116
+ const options = candidateNames.map((m) => optionTextForModel(m, profiles[m]));
117
+ const { probs, top } = scoreOptions(prompt, options, weights);
118
+ const complexity = scoreScalar(prompt, weights);
119
+ const { value: needsCode, prob: codeProb } = scoreNoul(prompt, "requires code generation", weights);
120
+ const elapsed = Date.now() - t0;
121
+
122
+ const bestIdx = top[0]?.index ?? -1;
123
+ const bestProb = top[0]?.prob ?? 0;
124
+
125
+ const jev: JevResponse = {
126
+ answers: [
127
+ {
128
+ type: "choice",
129
+ key: "provider",
130
+ value: bestIdx >= 0 ? candidateNames[bestIdx] : "",
131
+ prob: bestProb,
132
+ top: top.map((t) => ({ option: candidateNames[t.index], prob: t.prob })),
133
+ },
134
+ { type: "score", key: "complexity", value: complexity },
135
+ { type: "noul", key: "needs_code", value: needsCode, prob: codeProb },
136
+ ],
137
+ elapsed_ms: elapsed,
138
+ backend: "local-weights",
139
+ };
140
+
141
+ // 3. Confidence guard → System 2 fallback
142
+ if (bestProb < MIN_CONFIDENCE || bestIdx < 0) {
143
+ const fb = fallbackRoute(prompt, available_models, budget_multiplier, `low confidence ${bestProb.toFixed(2)}`);
144
+ return { ...fb, jev };
145
+ }
146
+
147
+ const primary = candidateNames[bestIdx];
148
+ const profile = profiles[primary];
149
+ const fallbacks = top.slice(1, 3).map((t) => candidateNames[t.index]);
150
+
151
+ return {
152
+ primary_model: primary,
153
+ fallback_models: fallbacks,
154
+ confidence: bestProb,
155
+ reasoning: `jev single-pass (distilled): p=${bestProb.toFixed(3)} complexity=${complexity.toFixed(2)} needs_code=${needsCode}(${codeProb.toFixed(2)}) [${elapsed}ms]`,
156
+ estimated_cost:
157
+ ((profile.cost_per_1k_input || 0) + (profile.cost_per_1k_output || 0)) / 2,
158
+ estimated_latency_ms: profile.latency_ms || 500,
159
+ provider_type: profile.type,
160
+ jev,
161
+ };
162
+ }
163
+
164
+ /** Async variant with remote backend support. */
165
+ export async function jevRouteAsync(
166
+ prompt: string,
167
+ available_models?: string[],
168
+ budget_multiplier: number = 1.0
169
+ ): Promise<RouteDecision & { jev?: JevResponse }> {
170
+ if (remoteConfigured()) {
171
+ const profiles = getModelProfilesSafe();
172
+ const candidates = (available_models && available_models.length
173
+ ? available_models
174
+ : Object.keys(profiles)
175
+ ).filter((n) => profiles[n]);
176
+ const remote = await remoteDecide({
177
+ state: prompt,
178
+ questions: [
179
+ {
180
+ type: "choice",
181
+ key: "provider",
182
+ description: "best model for this query",
183
+ options: candidates.map((m) => optionTextForModel(m, profiles[m])),
184
+ },
185
+ ],
186
+ });
187
+ if (remote) {
188
+ const choice = remote.answers[0];
189
+ if (choice && choice.type === "choice") {
190
+ // map option text back to model name
191
+ const idx = candidates.findIndex((m) =>
192
+ choice.value.startsWith(m + " ")
193
+ );
194
+ const primary = idx >= 0 ? candidates[idx] : candidates.find((c) => choice.value.includes(c));
195
+ if (primary) {
196
+ const profile = profiles[primary];
197
+ return {
198
+ primary_model: primary,
199
+ fallback_models: [],
200
+ confidence: choice.prob,
201
+ reasoning: `jev remote (${remote.backend}) [${remote.elapsed_ms}ms]`,
202
+ estimated_cost:
203
+ ((profile.cost_per_1k_input || 0) + (profile.cost_per_1k_output || 0)) / 2,
204
+ estimated_latency_ms: profile.latency_ms || 500,
205
+ provider_type: profile.type,
206
+ jev: remote,
207
+ };
208
+ }
209
+ }
210
+ }
211
+ }
212
+ return jevRoute(prompt, available_models, budget_multiplier);
213
+ }
214
+
215
+ // ---------------------------------------------------------------------------
216
+ // internals
217
+ // ---------------------------------------------------------------------------
218
+
219
+ import { MODEL_PROFILES } from "../advancedRouter";
220
+
221
+ function getModelProfilesSafe(): Record<string, {
222
+ providerName: string;
223
+ cost_per_1k_input: number;
224
+ cost_per_1k_output: number;
225
+ latency_ms: number;
226
+ quality_score: number;
227
+ strengths: string[];
228
+ type: string;
229
+ }> {
230
+ // MODEL_PROFILES is populated by the first routeQuery call (profile cache).
231
+ if (Object.keys(MODEL_PROFILES).length === 0) routeQuery("warmup");
232
+ return MODEL_PROFILES as Record<string, never>;
233
+ }
234
+
235
+ function fallbackRoute(
236
+ prompt: string,
237
+ available_models: string[] | undefined,
238
+ budget_multiplier: number,
239
+ why: string
240
+ ): RouteDecision & { jev?: JevResponse } {
241
+ const d = routeQuery(prompt, available_models, budget_multiplier);
242
+ return {
243
+ ...d,
244
+ reasoning: `system2-fallback (${why}): ${d.reasoning}`,
245
+ jev: {
246
+ answers: [],
247
+ elapsed_ms: 0,
248
+ backend: "fallback-heuristic",
249
+ },
250
+ };
251
+ }
@@ -0,0 +1,186 @@
1
+ /**
2
+ * OptionAttention — a pure-TypeScript System One decision engine.
3
+ *
4
+ * Port of the architecture popularized by vinnylarouge/jevlike (835★):
5
+ *
6
+ * Each option becomes a query vector. The query assigns attention weights
7
+ * to the context tokens; those weights produce one attended context vector
8
+ * per option. A shared dot product turns each (option, context) pair into
9
+ * one score. A softmax across options yields calibrated probabilities —
10
+ * all in a single forward pass, no autoregressive generation.
11
+ *
12
+ * Why this architecture for A3M:
13
+ * - Dynamic option sets: providers come and go; unseen options are scored
14
+ * through their text (name + strengths + tier), so the engine generalizes
15
+ * to providers it was never trained on.
16
+ * - Zero dependencies: byte-level embeddings, ~25k parameters, weights ship
17
+ * as JSON inside the npm package.
18
+ * - CPU-fast: single matmul chain, sub-millisecond for 100+ options.
19
+ */
20
+
21
+ export interface OptionAttentionWeights {
22
+ /** Hashed trigram embedding table: VOCAB × D. */
23
+ emb: number[][];
24
+ /** Option projection: D × D. */
25
+ wq: number[][];
26
+ bq: number[];
27
+ /** Shared scoring vector: D. */
28
+ w: number[];
29
+ b: number[];
30
+ /** Score head (context pooled → sigmoid): D + 2. */
31
+ ws: number[];
32
+ bs: number[];
33
+ /** Temperature for calibrated softmax. */
34
+ temperature: number;
35
+ dim: number;
36
+ }
37
+
38
+ /** Number of trigram hash buckets (must match tools/train_jev.py). */
39
+ export const VOCAB = 2048;
40
+
41
+ /**
42
+ * FNV-1a 32-bit hash of a character trigram → bucket id.
43
+ * Must match the Python trainer exactly.
44
+ */
45
+ function trigramHash(b0: number, b1: number, b2: number): number {
46
+ let h = 0x811c9dc5;
47
+ h = Math.imul(h ^ (b0 & 0xff), 0x01000193) >>> 0;
48
+ h = Math.imul(h ^ (b1 & 0xff), 0x01000193) >>> 0;
49
+ h = Math.imul(h ^ (b2 & 0xff), 0x01000193) >>> 0;
50
+ return h % VOCAB;
51
+ }
52
+
53
+ /**
54
+ * Hashed trigram tokenization — character trigrams give much sharper text
55
+ * discrimination than raw bytes ("chat" vs "code" produce disjoint buckets).
56
+ */
57
+ export function tokenize(text: string): number[] {
58
+ const s = " " + text + " ";
59
+ const out: number[] = [];
60
+ const cap = 384;
61
+ for (let i = 0; i + 2 < s.length && out.length < cap; i++) {
62
+ out.push(
63
+ trigramHash(
64
+ s.charCodeAt(i),
65
+ s.charCodeAt(i + 1),
66
+ s.charCodeAt(i + 2)
67
+ )
68
+ );
69
+ }
70
+ if (out.length === 0) out.push(trigramHash(32, 32, 32));
71
+ return out;
72
+ }
73
+
74
+ function meanPool(rows: number[][], weights: OptionAttentionWeights): number[] {
75
+ const D = weights.dim;
76
+ const out = new Array<number>(D).fill(0);
77
+ if (rows.length === 0) return out;
78
+ for (const r of rows) for (let d = 0; d < D; d++) out[d] += r[d];
79
+ for (let d = 0; d < D; d++) out[d] /= rows.length;
80
+ return out;
81
+ }
82
+
83
+ function embed(tokens: number[], weights: OptionAttentionWeights): number[][] {
84
+ return tokens.map((t) => weights.emb[t % VOCAB]);
85
+ }
86
+
87
+ /** Cache: option text → query vector. Option texts are static per provider
88
+ * profile, so q_i never changes between calls — skip the W_q matmul. */
89
+ const queryCache = new Map<string, number[]>();
90
+
91
+ export function resetQueryCache(): void {
92
+ queryCache.clear();
93
+ }
94
+
95
+ /** Option text → query vector q_i = W_q · meanPool(bytes) + b_q */
96
+ function optionQuery(optionText: string, weights: OptionAttentionWeights): number[] {
97
+ const hit = queryCache.get(optionText);
98
+ if (hit) return hit;
99
+ const D = weights.dim;
100
+ const h = meanPool(embed(tokenize(optionText), weights), weights);
101
+ const q = new Array<number>(D).fill(0);
102
+ for (let i = 0; i < D; i++) {
103
+ let s = weights.bq[i];
104
+ for (let j = 0; j < D; j++) s += weights.wq[i][j] * h[j];
105
+ q[i] = Math.tanh(s);
106
+ }
107
+ if (queryCache.size > 512) queryCache.clear();
108
+ queryCache.set(optionText, q);
109
+ return q;
110
+ }
111
+
112
+ export interface ChoiceResult {
113
+ probs: number[];
114
+ top: Array<{ index: number; prob: number }>;
115
+ }
116
+
117
+ /**
118
+ * Score N options against a context in one pass.
119
+ * Attention: a_ij = softmax_j( q_i · E_j / √D ), c_i = Σ_j a_ij E_j
120
+ * Score: s_i = w · (q_i ⊙ c_i) + b, probs = softmax(s / T)
121
+ */
122
+ export function scoreOptions(
123
+ context: string,
124
+ options: string[],
125
+ weights: OptionAttentionWeights
126
+ ): ChoiceResult {
127
+ const D = weights.dim;
128
+ const E = embed(tokenize(context), weights);
129
+ const scale = 1 / Math.sqrt(D);
130
+
131
+ const probs: number[] = [];
132
+ for (const opt of options) {
133
+ const q = optionQuery(opt, weights);
134
+ // attention over context tokens
135
+ const attn = E.map((e) => {
136
+ let dot = 0;
137
+ for (let d = 0; d < D; d++) dot += q[d] * e[d];
138
+ return dot * scale;
139
+ });
140
+ const maxA = Math.max(...attn, 0);
141
+ const exps = attn.map((a) => Math.exp(a - maxA));
142
+ const sumA = exps.reduce((a, b) => a + b, 0) || 1;
143
+ // attended context vector c_i
144
+ const c = new Array<number>(D).fill(0);
145
+ for (let j = 0; j < E.length; j++) {
146
+ const a = exps[j] / sumA;
147
+ if (a < 1e-6) continue;
148
+ for (let d = 0; d < D; d++) c[d] += a * E[j][d];
149
+ }
150
+ // score_i = w · (q ⊙ c) + b
151
+ let s = weights.b[0];
152
+ for (let d = 0; d < D; d++) s += weights.w[d] * q[d] * c[d];
153
+ probs.push(s / weights.temperature);
154
+ }
155
+
156
+ const maxS = Math.max(...probs);
157
+ const expP = probs.map((p) => Math.exp(p - maxS));
158
+ const sumP = expP.reduce((a, b) => a + b, 0) || 1;
159
+ const final = expP.map((e) => e / sumP);
160
+
161
+ const top = final
162
+ .map((prob, index) => ({ index, prob }))
163
+ .sort((a, b) => b.prob - a.prob)
164
+ .slice(0, 5);
165
+ return { probs: final, top };
166
+ }
167
+
168
+ /** Context → scalar in [0,1] via sigmoid(w_s · meanPool + b_s). */
169
+ export function scoreScalar(context: string, weights: OptionAttentionWeights): number {
170
+ const D = weights.dim;
171
+ const p = meanPool(embed(tokenize(context), weights), weights);
172
+ let s = weights.bs[0];
173
+ for (let d = 0; d < D; d++) s += weights.ws[d] * p[d];
174
+ return 1 / (1 + Math.exp(-s));
175
+ }
176
+
177
+ /** Boolean decision with calibrated probability (2-way choice). */
178
+ export function scoreNoul(
179
+ context: string,
180
+ questionText: string,
181
+ weights: OptionAttentionWeights
182
+ ): { value: boolean; prob: number } {
183
+ const r = scoreOptions(context, [questionText + " — yes", questionText + " — no"], weights);
184
+ const yesProb = r.probs[0];
185
+ return { value: yesProb >= 0.5, prob: Math.max(yesProb, 1 - yesProb) };
186
+ }
@@ -0,0 +1,50 @@
1
+ /**
2
+ * Remote backend for Jev-compatible endpoints.
3
+ *
4
+ * Works with:
5
+ * - api.typesafe.ai (commercial Jev, "System One")
6
+ * - ekzhang/openjev-sglang (open, Jev HTTP API on SGLang; self-host on Modal)
7
+ *
8
+ * Configure via env:
9
+ * A3M_JEV_URL — endpoint base (e.g. https://your-server.modal.direct)
10
+ * A3M_JEV_API_KEY — optional bearer token
11
+ */
12
+
13
+ import type { JevRequest, JevResponse } from "./types";
14
+
15
+ export function remoteConfigured(): boolean {
16
+ return Boolean(process.env.A3M_JEV_URL);
17
+ }
18
+
19
+ export async function remoteDecide(req: JevRequest, timeoutMs = 2000): Promise<JevResponse | null> {
20
+ const base = process.env.A3M_JEV_URL;
21
+ if (!base) return null;
22
+ const key = process.env.A3M_JEV_API_KEY || "";
23
+
24
+ const controller = new AbortController();
25
+ const timer = setTimeout(() => controller.abort(), timeoutMs);
26
+ const t0 = Date.now();
27
+ try {
28
+ const res = await fetch(`${base.replace(/\/$/, "")}/v1/decide`, {
29
+ method: "POST",
30
+ headers: {
31
+ "content-type": "application/json",
32
+ ...(key ? { authorization: `Bearer ${key}` } : {}),
33
+ },
34
+ body: JSON.stringify(req),
35
+ signal: controller.signal,
36
+ });
37
+ if (!res.ok) return null;
38
+ const data = (await res.json()) as { answers?: unknown };
39
+ if (!Array.isArray(data.answers)) return null;
40
+ return {
41
+ answers: data.answers as JevResponse["answers"],
42
+ elapsed_ms: Date.now() - t0,
43
+ backend: "remote",
44
+ };
45
+ } catch {
46
+ return null;
47
+ } finally {
48
+ clearTimeout(timer);
49
+ }
50
+ }
@@ -0,0 +1,79 @@
1
+ /**
2
+ * Jev contract — open interface pattern for System One decision models.
3
+ *
4
+ * Follows the interface reproduced by the open community:
5
+ * - TheoLeeCJ/SemIf (formerly OpenJev, 1.4k★) — typed option probabilities
6
+ * - ekzhang/openjev-sglang — faithful TypeSafe/Jev HTTP API server
7
+ * - vinnylarouge/jevlike — trainable context+options→probabilities engine
8
+ *
9
+ * Three question types (per TypeSafe's Jev):
10
+ * - choice: categorical pick over N options → calibrated probs
11
+ * - score: numeric value in [0,1] → sigmoid output
12
+ * - noul: boolean with calibrated probability → 2-way choice
13
+ *
14
+ * A3M usage: one request answers provider routing (choice),
15
+ * query complexity (score), and capability flags (noul) in a single pass.
16
+ */
17
+
18
+ export type JevQuestionType = "choice" | "score" | "noul";
19
+
20
+ export interface JevChoiceQuestion {
21
+ type: "choice";
22
+ key: string;
23
+ description?: string;
24
+ /** Dynamic option texts — engine scores unseen options via their text. */
25
+ options: string[];
26
+ }
27
+
28
+ export interface JevScoreQuestion {
29
+ type: "score";
30
+ key: string;
31
+ description?: string;
32
+ min?: number;
33
+ max?: number;
34
+ }
35
+
36
+ export interface JevNoulQuestion {
37
+ type: "noul";
38
+ key: string;
39
+ description?: string;
40
+ }
41
+
42
+ export type JevQuestion = JevChoiceQuestion | JevScoreQuestion | JevNoulQuestion;
43
+
44
+ export interface JevRequest {
45
+ /** Unstructured state — the user prompt / conversation context. */
46
+ state: string;
47
+ questions: JevQuestion[];
48
+ }
49
+
50
+ export interface JevChoiceAnswer {
51
+ type: "choice";
52
+ key: string;
53
+ value: string;
54
+ prob: number;
55
+ top: Array<{ option: string; prob: number }>;
56
+ }
57
+
58
+ export interface JevScoreAnswer {
59
+ type: "score";
60
+ key: string;
61
+ value: number;
62
+ }
63
+
64
+ export interface JevNoulAnswer {
65
+ type: "noul";
66
+ key: string;
67
+ value: boolean;
68
+ prob: number;
69
+ }
70
+
71
+ export type JevAnswer = JevChoiceAnswer | JevScoreAnswer | JevNoulAnswer;
72
+
73
+ export interface JevResponse {
74
+ answers: JevAnswer[];
75
+ /** Milliseconds spent in the decision engine (excluding fallbacks). */
76
+ elapsed_ms: number;
77
+ /** Which backend produced the answers. */
78
+ backend: "local-weights" | "remote" | "fallback-heuristic";
79
+ }