model-router-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- app/analytics/service.py +119 -0
- app/analyzer/analyzer.py +67 -0
- app/analyzer/heuristics.py +192 -0
- app/api/routes.py +589 -0
- app/budgets/manager.py +39 -0
- app/cli/main.py +287 -0
- app/config/settings.py +43 -0
- app/experiments/service.py +85 -0
- app/fallback/handler.py +105 -0
- app/models/schemas.py +127 -0
- app/observability/events.py +43 -0
- app/providers/base.py +46 -0
- app/providers/external_providers.py +321 -0
- app/providers/mock_provider.py +108 -0
- app/providers/ollama_provider.py +141 -0
- app/providers/registry.py +35 -0
- app/router/engine.py +150 -0
- app/router/rules_engine.py +73 -0
- app/router/scoring.py +154 -0
- app/static/assets/index-CQFztymk.js +63 -0
- app/static/assets/index-DWa3sE4Y.css +2 -0
- app/static/favicon.png +0 -0
- app/static/favicon.svg +1 -0
- app/static/icons.svg +24 -0
- app/static/index.html +17 -0
- app/static/logo.png +0 -0
- app/storage/database.py +366 -0
- app/storage/models.py +202 -0
- model_router_cli-1.0.0.dist-info/METADATA +343 -0
- model_router_cli-1.0.0.dist-info/RECORD +38 -0
- model_router_cli-1.0.0.dist-info/WHEEL +5 -0
- model_router_cli-1.0.0.dist-info/entry_points.txt +2 -0
- model_router_cli-1.0.0.dist-info/licenses/LICENSE +22 -0
- model_router_cli-1.0.0.dist-info/top_level.txt +2 -0
- tests/test_analyzer.py +41 -0
- tests/test_e2e.py +127 -0
- tests/test_providers.py +21 -0
- tests/test_router.py +78 -0
app/router/engine.py
ADDED
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
import uuid
|
|
2
|
+
import datetime
|
|
3
|
+
from typing import List, Dict, Any, Optional
|
|
4
|
+
from app.models.schemas import (
|
|
5
|
+
ModelMetadata,
|
|
6
|
+
RequestAnalysis,
|
|
7
|
+
RoutingDecision,
|
|
8
|
+
CandidateScore,
|
|
9
|
+
PriorityLevel,
|
|
10
|
+
)
|
|
11
|
+
from app.router.scoring import filter_candidate_models, compute_candidate_score
|
|
12
|
+
from app.router.rules_engine import evaluate_routing_rules
|
|
13
|
+
from app.storage.models import RoutingRuleRecord
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def generate_routing_explanation(
|
|
17
|
+
selected_model: ModelMetadata,
|
|
18
|
+
analysis: RequestAnalysis,
|
|
19
|
+
policy_name: str,
|
|
20
|
+
rule_name: Optional[str] = None,
|
|
21
|
+
) -> List[str]:
|
|
22
|
+
reasons = []
|
|
23
|
+
|
|
24
|
+
if rule_name:
|
|
25
|
+
reasons.append(f"Matching rule applied: \"{rule_name}\"")
|
|
26
|
+
|
|
27
|
+
# Task explanation
|
|
28
|
+
reasons.append(f"Task classified as {analysis.task_type.value} with {analysis.complexity_label.value} complexity ({analysis.complexity:.2f})")
|
|
29
|
+
|
|
30
|
+
# Reasoning / Code requirements
|
|
31
|
+
if analysis.reasoning_required:
|
|
32
|
+
reasons.append("Multi-step reasoning & deep inference required for problem domain")
|
|
33
|
+
if analysis.coding_required:
|
|
34
|
+
reasons.append("Syntactical & algorithmic coding capabilities required")
|
|
35
|
+
|
|
36
|
+
# Model specific match
|
|
37
|
+
if selected_model.tier.value == "POWER":
|
|
38
|
+
reasons.append(f"High reasoning score ({selected_model.quality_score * 100:.0f}%) and deep context ({selected_model.context_window:,} tokens)")
|
|
39
|
+
elif selected_model.tier.value == "FAST":
|
|
40
|
+
reasons.append(f"Low latency priority and high speed score ({selected_model.speed_score * 100:.0f}%)")
|
|
41
|
+
else:
|
|
42
|
+
reasons.append("Balanced quality and cost profile selected for optimal efficiency")
|
|
43
|
+
|
|
44
|
+
# Cost / local factor
|
|
45
|
+
if selected_model.cost_per_input_token == 0.0:
|
|
46
|
+
reasons.append("Local zero-cost execution prioritized")
|
|
47
|
+
elif analysis.cost_sensitivity == PriorityLevel.HIGH:
|
|
48
|
+
reasons.append("Cost sensitivity is HIGH — model offers lowest token pricing")
|
|
49
|
+
|
|
50
|
+
# Policy
|
|
51
|
+
reasons.append(f"Evaluated under '{policy_name}' policy weights")
|
|
52
|
+
|
|
53
|
+
return reasons
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def route_request(
|
|
57
|
+
analysis: RequestAnalysis,
|
|
58
|
+
available_models: List[ModelMetadata],
|
|
59
|
+
policy_weights: Dict[str, float],
|
|
60
|
+
policy_name: str = "balanced",
|
|
61
|
+
rules: Optional[List[RoutingRuleRecord]] = None,
|
|
62
|
+
budget_percent: float = 0.0,
|
|
63
|
+
request_id: Optional[str] = None,
|
|
64
|
+
) -> RoutingDecision:
|
|
65
|
+
req_id = request_id or str(uuid.uuid4())
|
|
66
|
+
decision_id = f"dec_{uuid.uuid4().hex[:12]}"
|
|
67
|
+
now_iso = datetime.datetime.utcnow().isoformat()
|
|
68
|
+
|
|
69
|
+
# 1. Rule evaluation
|
|
70
|
+
rule_action, rule_target, rule_name = None, None, None
|
|
71
|
+
if rules:
|
|
72
|
+
rule_action, rule_target, rule_name = evaluate_routing_rules(rules, analysis, budget_percent)
|
|
73
|
+
|
|
74
|
+
# 2. Hard filter candidates
|
|
75
|
+
eligible_models, rejected_dict = filter_candidate_models(available_models, analysis)
|
|
76
|
+
|
|
77
|
+
if not eligible_models:
|
|
78
|
+
# Fallback to any active model if everything was filtered
|
|
79
|
+
eligible_models = [m for m in available_models if m.is_active] or available_models
|
|
80
|
+
|
|
81
|
+
# Handle Rule Action: ROUTE_TO specific model
|
|
82
|
+
if rule_action == "ROUTE_TO" and rule_target:
|
|
83
|
+
exact_match = next((m for m in eligible_models if m.id == rule_target), None)
|
|
84
|
+
if exact_match:
|
|
85
|
+
scores = [compute_candidate_score(m, analysis, policy_weights) for m in eligible_models]
|
|
86
|
+
score_map = {s.model_id: s for s in scores}
|
|
87
|
+
selected_score = score_map.get(exact_match.id) or compute_candidate_score(exact_match, analysis, policy_weights)
|
|
88
|
+
|
|
89
|
+
reasons = generate_routing_explanation(exact_match, analysis, policy_name, rule_name)
|
|
90
|
+
return RoutingDecision(
|
|
91
|
+
decision_id=decision_id,
|
|
92
|
+
request_id=req_id,
|
|
93
|
+
selected_model=exact_match.id,
|
|
94
|
+
selected_model_name=exact_match.name,
|
|
95
|
+
provider=exact_match.provider,
|
|
96
|
+
tier=exact_match.tier.value if hasattr(exact_match.tier, "value") else str(exact_match.tier),
|
|
97
|
+
confidence=0.98,
|
|
98
|
+
policy_used=policy_name,
|
|
99
|
+
reasons=reasons,
|
|
100
|
+
candidate_scores=scores,
|
|
101
|
+
rejected_candidates=rejected_dict,
|
|
102
|
+
estimated_cost_usd=selected_score.estimated_cost_usd,
|
|
103
|
+
estimated_latency_ms=selected_score.estimated_latency_ms,
|
|
104
|
+
rule_applied=rule_name,
|
|
105
|
+
timestamp=now_iso,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
# Handle Rule Action: FORCE_TIER
|
|
109
|
+
if rule_action == "FORCE_TIER" and rule_target:
|
|
110
|
+
tier_filtered = [m for m in eligible_models if m.tier.value.upper() == rule_target.upper()]
|
|
111
|
+
if tier_filtered:
|
|
112
|
+
eligible_models = tier_filtered
|
|
113
|
+
|
|
114
|
+
# 3. Score all eligible models
|
|
115
|
+
candidate_scores: List[CandidateScore] = []
|
|
116
|
+
for model in eligible_models:
|
|
117
|
+
sc = compute_candidate_score(model, analysis, policy_weights)
|
|
118
|
+
candidate_scores.append(sc)
|
|
119
|
+
|
|
120
|
+
# Sort descending by overall_score
|
|
121
|
+
candidate_scores.sort(key=lambda s: s.overall_score, reverse=True)
|
|
122
|
+
best_candidate = candidate_scores[0]
|
|
123
|
+
selected_model = next(m for m in eligible_models if m.id == best_candidate.model_id)
|
|
124
|
+
|
|
125
|
+
# Confidence calculation: score difference from second candidate
|
|
126
|
+
if len(candidate_scores) > 1:
|
|
127
|
+
gap = best_candidate.overall_score - candidate_scores[1].overall_score
|
|
128
|
+
confidence = min(0.99, max(0.70, round(0.80 + (gap / 100.0), 2)))
|
|
129
|
+
else:
|
|
130
|
+
confidence = 0.95
|
|
131
|
+
|
|
132
|
+
reasons = generate_routing_explanation(selected_model, analysis, policy_name, rule_name)
|
|
133
|
+
|
|
134
|
+
return RoutingDecision(
|
|
135
|
+
decision_id=decision_id,
|
|
136
|
+
request_id=req_id,
|
|
137
|
+
selected_model=selected_model.id,
|
|
138
|
+
selected_model_name=selected_model.name,
|
|
139
|
+
provider=selected_model.provider,
|
|
140
|
+
tier=selected_model.tier.value if hasattr(selected_model.tier, "value") else str(selected_model.tier),
|
|
141
|
+
confidence=confidence,
|
|
142
|
+
policy_used=policy_name,
|
|
143
|
+
reasons=reasons,
|
|
144
|
+
candidate_scores=candidate_scores,
|
|
145
|
+
rejected_candidates=rejected_dict,
|
|
146
|
+
estimated_cost_usd=best_candidate.estimated_cost_usd,
|
|
147
|
+
estimated_latency_ms=best_candidate.estimated_latency_ms,
|
|
148
|
+
rule_applied=rule_name,
|
|
149
|
+
timestamp=now_iso,
|
|
150
|
+
)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
from typing import List, Optional, Tuple
|
|
2
|
+
from app.storage.models import RoutingRuleRecord
|
|
3
|
+
from app.models.schemas import RequestAnalysis
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def evaluate_routing_rules(
|
|
7
|
+
rules: List[RoutingRuleRecord],
|
|
8
|
+
analysis: RequestAnalysis,
|
|
9
|
+
current_budget_percent: float = 0.0,
|
|
10
|
+
) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
|
11
|
+
"""
|
|
12
|
+
Evaluates enabled routing rules in order of priority (highest priority number first).
|
|
13
|
+
Returns: (action_type, action_target, matched_rule_name) or (None, None, None)
|
|
14
|
+
"""
|
|
15
|
+
sorted_rules = sorted([r for r in rules if r.is_enabled], key=lambda x: x.priority, reverse=True)
|
|
16
|
+
|
|
17
|
+
for rule in sorted_rules:
|
|
18
|
+
field = rule.condition_field.lower()
|
|
19
|
+
op = rule.condition_operator
|
|
20
|
+
target_val = rule.condition_value
|
|
21
|
+
|
|
22
|
+
matched = False
|
|
23
|
+
|
|
24
|
+
if field == "task_type":
|
|
25
|
+
val = analysis.task_type.value
|
|
26
|
+
if op == "==":
|
|
27
|
+
matched = (val == target_val)
|
|
28
|
+
elif op == "!=":
|
|
29
|
+
matched = (val != target_val)
|
|
30
|
+
elif op == "contains":
|
|
31
|
+
matched = (target_val in val)
|
|
32
|
+
|
|
33
|
+
elif field == "complexity":
|
|
34
|
+
val = analysis.complexity
|
|
35
|
+
try:
|
|
36
|
+
target_num = float(target_val)
|
|
37
|
+
if op == ">=":
|
|
38
|
+
matched = (val >= target_num)
|
|
39
|
+
elif op == ">":
|
|
40
|
+
matched = (val > target_num)
|
|
41
|
+
elif op == "<=":
|
|
42
|
+
matched = (val <= target_num)
|
|
43
|
+
elif op == "<":
|
|
44
|
+
matched = (val < target_num)
|
|
45
|
+
elif op == "==":
|
|
46
|
+
matched = (abs(val - target_num) < 0.01)
|
|
47
|
+
except ValueError:
|
|
48
|
+
pass
|
|
49
|
+
|
|
50
|
+
elif field in ("budget_percent", "monthly_budget_percent"):
|
|
51
|
+
try:
|
|
52
|
+
target_num = float(target_val)
|
|
53
|
+
if op == ">=":
|
|
54
|
+
matched = (current_budget_percent >= target_num)
|
|
55
|
+
elif op == ">":
|
|
56
|
+
matched = (current_budget_percent > target_num)
|
|
57
|
+
except ValueError:
|
|
58
|
+
pass
|
|
59
|
+
|
|
60
|
+
elif field == "context_size":
|
|
61
|
+
try:
|
|
62
|
+
target_num = int(target_val)
|
|
63
|
+
if op == ">=":
|
|
64
|
+
matched = (analysis.context_size >= target_num)
|
|
65
|
+
elif op == ">":
|
|
66
|
+
matched = (analysis.context_size > target_num)
|
|
67
|
+
except ValueError:
|
|
68
|
+
pass
|
|
69
|
+
|
|
70
|
+
if matched:
|
|
71
|
+
return rule.action_type, rule.action_target, rule.name
|
|
72
|
+
|
|
73
|
+
return None, None, None
|
app/router/scoring.py
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
from typing import List, Dict, Tuple, Optional
|
|
2
|
+
from app.models.schemas import ModelMetadata, RequestAnalysis, CandidateScore, PriorityLevel
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
def filter_candidate_models(
|
|
6
|
+
models: List[ModelMetadata],
|
|
7
|
+
analysis: RequestAnalysis,
|
|
8
|
+
) -> Tuple[List[ModelMetadata], Dict[str, str]]:
|
|
9
|
+
"""
|
|
10
|
+
Hard-filtering step: eliminates models that cannot satisfy non-negotiable requirements.
|
|
11
|
+
Returns: (eligible_models, rejected_candidates_with_reasons)
|
|
12
|
+
"""
|
|
13
|
+
eligible: List[ModelMetadata] = []
|
|
14
|
+
rejected: Dict[str, str] = {}
|
|
15
|
+
|
|
16
|
+
for m in models:
|
|
17
|
+
if not m.is_active:
|
|
18
|
+
rejected[m.id] = "Model is currently disabled/inactive."
|
|
19
|
+
continue
|
|
20
|
+
|
|
21
|
+
# 1. Context Window check
|
|
22
|
+
if analysis.context_size > m.context_window:
|
|
23
|
+
rejected[m.id] = (
|
|
24
|
+
f"Context window ({m.context_window} tokens) is smaller than required input ({analysis.context_size} tokens)."
|
|
25
|
+
)
|
|
26
|
+
continue
|
|
27
|
+
|
|
28
|
+
# 2. Vision requirement check
|
|
29
|
+
if analysis.vision_required and not m.supports_vision:
|
|
30
|
+
rejected[m.id] = "Task requires vision / multimodal capabilities which this model does not support."
|
|
31
|
+
continue
|
|
32
|
+
|
|
33
|
+
# 3. Coding capability check (for high complexity code tasks)
|
|
34
|
+
if analysis.coding_required and analysis.complexity >= 0.70 and not m.supports_coding:
|
|
35
|
+
rejected[m.id] = "Task requires specialized coding capabilities for complex logic."
|
|
36
|
+
continue
|
|
37
|
+
|
|
38
|
+
# 4. Deep reasoning capability check
|
|
39
|
+
if analysis.reasoning_required and analysis.complexity >= 0.85 and not m.supports_reasoning:
|
|
40
|
+
rejected[m.id] = "Task requires advanced reasoning capabilities for multi-step problem solving."
|
|
41
|
+
continue
|
|
42
|
+
|
|
43
|
+
eligible.append(m)
|
|
44
|
+
|
|
45
|
+
# If all filtered out (edge case), return active models to avoid total failure
|
|
46
|
+
if not eligible and models:
|
|
47
|
+
for m in models:
|
|
48
|
+
if m.is_active:
|
|
49
|
+
eligible.append(m)
|
|
50
|
+
rejected.clear()
|
|
51
|
+
|
|
52
|
+
return eligible, rejected
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def compute_candidate_score(
|
|
56
|
+
model: ModelMetadata,
|
|
57
|
+
analysis: RequestAnalysis,
|
|
58
|
+
weights: Dict[str, float],
|
|
59
|
+
) -> CandidateScore:
|
|
60
|
+
"""
|
|
61
|
+
Multi-criteria configurable scoring formula.
|
|
62
|
+
Quality, Speed, CostEfficiency, CapabilityMatch, Reliability.
|
|
63
|
+
Normalized score: 0 to 100.
|
|
64
|
+
"""
|
|
65
|
+
# 1. Quality Component (0.0 to 1.0)
|
|
66
|
+
quality_comp = model.quality_score
|
|
67
|
+
if analysis.quality_requirement == PriorityLevel.HIGH:
|
|
68
|
+
# Boost premium models for high quality requirements
|
|
69
|
+
if model.tier.value == "POWER":
|
|
70
|
+
quality_comp = min(1.0, quality_comp * 1.15)
|
|
71
|
+
elif analysis.quality_requirement == PriorityLevel.LOW:
|
|
72
|
+
quality_comp = model.quality_score * 0.9
|
|
73
|
+
|
|
74
|
+
# 2. Speed Component (0.0 to 1.0)
|
|
75
|
+
speed_comp = model.speed_score
|
|
76
|
+
if analysis.latency_priority == PriorityLevel.HIGH:
|
|
77
|
+
if model.tier.value == "FAST":
|
|
78
|
+
speed_comp = min(1.0, speed_comp * 1.20)
|
|
79
|
+
elif analysis.latency_priority == PriorityLevel.LOW:
|
|
80
|
+
speed_comp = model.speed_score * 0.85
|
|
81
|
+
|
|
82
|
+
# 3. Cost Efficiency Component (0.0 to 1.0)
|
|
83
|
+
# Zero-cost local models get 1.0, cheaper cloud models get high score
|
|
84
|
+
combined_token_cost = (model.cost_per_input_token * 1000) + (model.cost_per_output_token * 1000)
|
|
85
|
+
if combined_token_cost == 0.0:
|
|
86
|
+
cost_efficiency = 1.0
|
|
87
|
+
else:
|
|
88
|
+
# Scale: $0.0001 -> 0.95, $0.01 -> 0.50, $0.05 -> 0.10
|
|
89
|
+
cost_efficiency = max(0.05, min(0.99, 1.0 / (1.0 + (combined_token_cost * 100.0))))
|
|
90
|
+
|
|
91
|
+
if analysis.cost_sensitivity == PriorityLevel.HIGH:
|
|
92
|
+
cost_efficiency = min(1.0, cost_efficiency * 1.25)
|
|
93
|
+
|
|
94
|
+
# 4. Capability Match Component (0.0 to 1.0)
|
|
95
|
+
cap_points = 0.0
|
|
96
|
+
total_caps = 0.0
|
|
97
|
+
|
|
98
|
+
if analysis.coding_required:
|
|
99
|
+
total_caps += 1.0
|
|
100
|
+
if model.supports_coding:
|
|
101
|
+
cap_points += 1.0
|
|
102
|
+
|
|
103
|
+
if analysis.reasoning_required:
|
|
104
|
+
total_caps += 1.0
|
|
105
|
+
if model.supports_reasoning:
|
|
106
|
+
cap_points += 1.0
|
|
107
|
+
|
|
108
|
+
if analysis.vision_required:
|
|
109
|
+
total_caps += 1.0
|
|
110
|
+
if model.supports_vision:
|
|
111
|
+
cap_points += 1.0
|
|
112
|
+
|
|
113
|
+
capability_comp = (cap_points / total_caps) if total_caps > 0 else 0.90
|
|
114
|
+
|
|
115
|
+
# 5. Reliability Component (0.0 to 1.0)
|
|
116
|
+
reliability_comp = model.reliability_score
|
|
117
|
+
|
|
118
|
+
# Weighted Sum
|
|
119
|
+
w_q = weights.get("quality_weight", 0.35)
|
|
120
|
+
w_c = weights.get("cost_weight", 0.25)
|
|
121
|
+
w_s = weights.get("speed_weight", 0.20)
|
|
122
|
+
w_cap = weights.get("capability_weight", 0.15)
|
|
123
|
+
w_r = weights.get("reliability_weight", 0.05)
|
|
124
|
+
total_w = w_q + w_c + w_s + w_cap + w_r
|
|
125
|
+
|
|
126
|
+
raw_score = (
|
|
127
|
+
(quality_comp * w_q)
|
|
128
|
+
+ (cost_efficiency * w_c)
|
|
129
|
+
+ (speed_comp * w_s)
|
|
130
|
+
+ (capability_comp * w_cap)
|
|
131
|
+
+ (reliability_comp * w_r)
|
|
132
|
+
) / (total_w if total_w > 0 else 1.0)
|
|
133
|
+
|
|
134
|
+
overall_score = round(raw_score * 100.0, 2)
|
|
135
|
+
|
|
136
|
+
# Estimated metrics
|
|
137
|
+
est_latency = 120.0 if model.tier.value == "FAST" else (350.0 if model.tier.value == "BALANCED" else 750.0)
|
|
138
|
+
est_cost = (analysis.context_size * model.cost_per_input_token) + (250 * model.cost_per_output_token)
|
|
139
|
+
|
|
140
|
+
return CandidateScore(
|
|
141
|
+
model_id=model.id,
|
|
142
|
+
model_name=model.name,
|
|
143
|
+
provider=model.provider,
|
|
144
|
+
tier=model.tier.value if hasattr(model.tier, "value") else str(model.tier),
|
|
145
|
+
overall_score=overall_score,
|
|
146
|
+
quality_component=round(quality_comp * 100.0, 1),
|
|
147
|
+
cost_component=round(cost_efficiency * 100.0, 1),
|
|
148
|
+
speed_component=round(speed_comp * 100.0, 1),
|
|
149
|
+
capability_component=round(capability_comp * 100.0, 1),
|
|
150
|
+
reliability_component=round(reliability_comp * 100.0, 1),
|
|
151
|
+
estimated_latency_ms=est_latency,
|
|
152
|
+
estimated_cost_usd=round(est_cost, 6),
|
|
153
|
+
eligible=True,
|
|
154
|
+
)
|