model-router-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
app/router/engine.py ADDED
@@ -0,0 +1,150 @@
1
+ import uuid
2
+ import datetime
3
+ from typing import List, Dict, Any, Optional
4
+ from app.models.schemas import (
5
+ ModelMetadata,
6
+ RequestAnalysis,
7
+ RoutingDecision,
8
+ CandidateScore,
9
+ PriorityLevel,
10
+ )
11
+ from app.router.scoring import filter_candidate_models, compute_candidate_score
12
+ from app.router.rules_engine import evaluate_routing_rules
13
+ from app.storage.models import RoutingRuleRecord
14
+
15
+
16
+ def generate_routing_explanation(
17
+ selected_model: ModelMetadata,
18
+ analysis: RequestAnalysis,
19
+ policy_name: str,
20
+ rule_name: Optional[str] = None,
21
+ ) -> List[str]:
22
+ reasons = []
23
+
24
+ if rule_name:
25
+ reasons.append(f"Matching rule applied: \"{rule_name}\"")
26
+
27
+ # Task explanation
28
+ reasons.append(f"Task classified as {analysis.task_type.value} with {analysis.complexity_label.value} complexity ({analysis.complexity:.2f})")
29
+
30
+ # Reasoning / Code requirements
31
+ if analysis.reasoning_required:
32
+ reasons.append("Multi-step reasoning & deep inference required for problem domain")
33
+ if analysis.coding_required:
34
+ reasons.append("Syntactical & algorithmic coding capabilities required")
35
+
36
+ # Model specific match
37
+ if selected_model.tier.value == "POWER":
38
+ reasons.append(f"High reasoning score ({selected_model.quality_score * 100:.0f}%) and deep context ({selected_model.context_window:,} tokens)")
39
+ elif selected_model.tier.value == "FAST":
40
+ reasons.append(f"Low latency priority and high speed score ({selected_model.speed_score * 100:.0f}%)")
41
+ else:
42
+ reasons.append("Balanced quality and cost profile selected for optimal efficiency")
43
+
44
+ # Cost / local factor
45
+ if selected_model.cost_per_input_token == 0.0:
46
+ reasons.append("Local zero-cost execution prioritized")
47
+ elif analysis.cost_sensitivity == PriorityLevel.HIGH:
48
+ reasons.append("Cost sensitivity is HIGH — model offers lowest token pricing")
49
+
50
+ # Policy
51
+ reasons.append(f"Evaluated under '{policy_name}' policy weights")
52
+
53
+ return reasons
54
+
55
+
56
+ def route_request(
57
+ analysis: RequestAnalysis,
58
+ available_models: List[ModelMetadata],
59
+ policy_weights: Dict[str, float],
60
+ policy_name: str = "balanced",
61
+ rules: Optional[List[RoutingRuleRecord]] = None,
62
+ budget_percent: float = 0.0,
63
+ request_id: Optional[str] = None,
64
+ ) -> RoutingDecision:
65
+ req_id = request_id or str(uuid.uuid4())
66
+ decision_id = f"dec_{uuid.uuid4().hex[:12]}"
67
+ now_iso = datetime.datetime.utcnow().isoformat()
68
+
69
+ # 1. Rule evaluation
70
+ rule_action, rule_target, rule_name = None, None, None
71
+ if rules:
72
+ rule_action, rule_target, rule_name = evaluate_routing_rules(rules, analysis, budget_percent)
73
+
74
+ # 2. Hard filter candidates
75
+ eligible_models, rejected_dict = filter_candidate_models(available_models, analysis)
76
+
77
+ if not eligible_models:
78
+ # Fallback to any active model if everything was filtered
79
+ eligible_models = [m for m in available_models if m.is_active] or available_models
80
+
81
+ # Handle Rule Action: ROUTE_TO specific model
82
+ if rule_action == "ROUTE_TO" and rule_target:
83
+ exact_match = next((m for m in eligible_models if m.id == rule_target), None)
84
+ if exact_match:
85
+ scores = [compute_candidate_score(m, analysis, policy_weights) for m in eligible_models]
86
+ score_map = {s.model_id: s for s in scores}
87
+ selected_score = score_map.get(exact_match.id) or compute_candidate_score(exact_match, analysis, policy_weights)
88
+
89
+ reasons = generate_routing_explanation(exact_match, analysis, policy_name, rule_name)
90
+ return RoutingDecision(
91
+ decision_id=decision_id,
92
+ request_id=req_id,
93
+ selected_model=exact_match.id,
94
+ selected_model_name=exact_match.name,
95
+ provider=exact_match.provider,
96
+ tier=exact_match.tier.value if hasattr(exact_match.tier, "value") else str(exact_match.tier),
97
+ confidence=0.98,
98
+ policy_used=policy_name,
99
+ reasons=reasons,
100
+ candidate_scores=scores,
101
+ rejected_candidates=rejected_dict,
102
+ estimated_cost_usd=selected_score.estimated_cost_usd,
103
+ estimated_latency_ms=selected_score.estimated_latency_ms,
104
+ rule_applied=rule_name,
105
+ timestamp=now_iso,
106
+ )
107
+
108
+ # Handle Rule Action: FORCE_TIER
109
+ if rule_action == "FORCE_TIER" and rule_target:
110
+ tier_filtered = [m for m in eligible_models if m.tier.value.upper() == rule_target.upper()]
111
+ if tier_filtered:
112
+ eligible_models = tier_filtered
113
+
114
+ # 3. Score all eligible models
115
+ candidate_scores: List[CandidateScore] = []
116
+ for model in eligible_models:
117
+ sc = compute_candidate_score(model, analysis, policy_weights)
118
+ candidate_scores.append(sc)
119
+
120
+ # Sort descending by overall_score
121
+ candidate_scores.sort(key=lambda s: s.overall_score, reverse=True)
122
+ best_candidate = candidate_scores[0]
123
+ selected_model = next(m for m in eligible_models if m.id == best_candidate.model_id)
124
+
125
+ # Confidence calculation: score difference from second candidate
126
+ if len(candidate_scores) > 1:
127
+ gap = best_candidate.overall_score - candidate_scores[1].overall_score
128
+ confidence = min(0.99, max(0.70, round(0.80 + (gap / 100.0), 2)))
129
+ else:
130
+ confidence = 0.95
131
+
132
+ reasons = generate_routing_explanation(selected_model, analysis, policy_name, rule_name)
133
+
134
+ return RoutingDecision(
135
+ decision_id=decision_id,
136
+ request_id=req_id,
137
+ selected_model=selected_model.id,
138
+ selected_model_name=selected_model.name,
139
+ provider=selected_model.provider,
140
+ tier=selected_model.tier.value if hasattr(selected_model.tier, "value") else str(selected_model.tier),
141
+ confidence=confidence,
142
+ policy_used=policy_name,
143
+ reasons=reasons,
144
+ candidate_scores=candidate_scores,
145
+ rejected_candidates=rejected_dict,
146
+ estimated_cost_usd=best_candidate.estimated_cost_usd,
147
+ estimated_latency_ms=best_candidate.estimated_latency_ms,
148
+ rule_applied=rule_name,
149
+ timestamp=now_iso,
150
+ )
@@ -0,0 +1,73 @@
1
+ from typing import List, Optional, Tuple
2
+ from app.storage.models import RoutingRuleRecord
3
+ from app.models.schemas import RequestAnalysis
4
+
5
+
6
+ def evaluate_routing_rules(
7
+ rules: List[RoutingRuleRecord],
8
+ analysis: RequestAnalysis,
9
+ current_budget_percent: float = 0.0,
10
+ ) -> Tuple[Optional[str], Optional[str], Optional[str]]:
11
+ """
12
+ Evaluates enabled routing rules in order of priority (highest priority number first).
13
+ Returns: (action_type, action_target, matched_rule_name) or (None, None, None)
14
+ """
15
+ sorted_rules = sorted([r for r in rules if r.is_enabled], key=lambda x: x.priority, reverse=True)
16
+
17
+ for rule in sorted_rules:
18
+ field = rule.condition_field.lower()
19
+ op = rule.condition_operator
20
+ target_val = rule.condition_value
21
+
22
+ matched = False
23
+
24
+ if field == "task_type":
25
+ val = analysis.task_type.value
26
+ if op == "==":
27
+ matched = (val == target_val)
28
+ elif op == "!=":
29
+ matched = (val != target_val)
30
+ elif op == "contains":
31
+ matched = (target_val in val)
32
+
33
+ elif field == "complexity":
34
+ val = analysis.complexity
35
+ try:
36
+ target_num = float(target_val)
37
+ if op == ">=":
38
+ matched = (val >= target_num)
39
+ elif op == ">":
40
+ matched = (val > target_num)
41
+ elif op == "<=":
42
+ matched = (val <= target_num)
43
+ elif op == "<":
44
+ matched = (val < target_num)
45
+ elif op == "==":
46
+ matched = (abs(val - target_num) < 0.01)
47
+ except ValueError:
48
+ pass
49
+
50
+ elif field in ("budget_percent", "monthly_budget_percent"):
51
+ try:
52
+ target_num = float(target_val)
53
+ if op == ">=":
54
+ matched = (current_budget_percent >= target_num)
55
+ elif op == ">":
56
+ matched = (current_budget_percent > target_num)
57
+ except ValueError:
58
+ pass
59
+
60
+ elif field == "context_size":
61
+ try:
62
+ target_num = int(target_val)
63
+ if op == ">=":
64
+ matched = (analysis.context_size >= target_num)
65
+ elif op == ">":
66
+ matched = (analysis.context_size > target_num)
67
+ except ValueError:
68
+ pass
69
+
70
+ if matched:
71
+ return rule.action_type, rule.action_target, rule.name
72
+
73
+ return None, None, None
app/router/scoring.py ADDED
@@ -0,0 +1,154 @@
1
+ from typing import List, Dict, Tuple, Optional
2
+ from app.models.schemas import ModelMetadata, RequestAnalysis, CandidateScore, PriorityLevel
3
+
4
+
5
+ def filter_candidate_models(
6
+ models: List[ModelMetadata],
7
+ analysis: RequestAnalysis,
8
+ ) -> Tuple[List[ModelMetadata], Dict[str, str]]:
9
+ """
10
+ Hard-filtering step: eliminates models that cannot satisfy non-negotiable requirements.
11
+ Returns: (eligible_models, rejected_candidates_with_reasons)
12
+ """
13
+ eligible: List[ModelMetadata] = []
14
+ rejected: Dict[str, str] = {}
15
+
16
+ for m in models:
17
+ if not m.is_active:
18
+ rejected[m.id] = "Model is currently disabled/inactive."
19
+ continue
20
+
21
+ # 1. Context Window check
22
+ if analysis.context_size > m.context_window:
23
+ rejected[m.id] = (
24
+ f"Context window ({m.context_window} tokens) is smaller than required input ({analysis.context_size} tokens)."
25
+ )
26
+ continue
27
+
28
+ # 2. Vision requirement check
29
+ if analysis.vision_required and not m.supports_vision:
30
+ rejected[m.id] = "Task requires vision / multimodal capabilities which this model does not support."
31
+ continue
32
+
33
+ # 3. Coding capability check (for high complexity code tasks)
34
+ if analysis.coding_required and analysis.complexity >= 0.70 and not m.supports_coding:
35
+ rejected[m.id] = "Task requires specialized coding capabilities for complex logic."
36
+ continue
37
+
38
+ # 4. Deep reasoning capability check
39
+ if analysis.reasoning_required and analysis.complexity >= 0.85 and not m.supports_reasoning:
40
+ rejected[m.id] = "Task requires advanced reasoning capabilities for multi-step problem solving."
41
+ continue
42
+
43
+ eligible.append(m)
44
+
45
+ # If all filtered out (edge case), return active models to avoid total failure
46
+ if not eligible and models:
47
+ for m in models:
48
+ if m.is_active:
49
+ eligible.append(m)
50
+ rejected.clear()
51
+
52
+ return eligible, rejected
53
+
54
+
55
+ def compute_candidate_score(
56
+ model: ModelMetadata,
57
+ analysis: RequestAnalysis,
58
+ weights: Dict[str, float],
59
+ ) -> CandidateScore:
60
+ """
61
+ Multi-criteria configurable scoring formula.
62
+ Quality, Speed, CostEfficiency, CapabilityMatch, Reliability.
63
+ Normalized score: 0 to 100.
64
+ """
65
+ # 1. Quality Component (0.0 to 1.0)
66
+ quality_comp = model.quality_score
67
+ if analysis.quality_requirement == PriorityLevel.HIGH:
68
+ # Boost premium models for high quality requirements
69
+ if model.tier.value == "POWER":
70
+ quality_comp = min(1.0, quality_comp * 1.15)
71
+ elif analysis.quality_requirement == PriorityLevel.LOW:
72
+ quality_comp = model.quality_score * 0.9
73
+
74
+ # 2. Speed Component (0.0 to 1.0)
75
+ speed_comp = model.speed_score
76
+ if analysis.latency_priority == PriorityLevel.HIGH:
77
+ if model.tier.value == "FAST":
78
+ speed_comp = min(1.0, speed_comp * 1.20)
79
+ elif analysis.latency_priority == PriorityLevel.LOW:
80
+ speed_comp = model.speed_score * 0.85
81
+
82
+ # 3. Cost Efficiency Component (0.0 to 1.0)
83
+ # Zero-cost local models get 1.0, cheaper cloud models get high score
84
+ combined_token_cost = (model.cost_per_input_token * 1000) + (model.cost_per_output_token * 1000)
85
+ if combined_token_cost == 0.0:
86
+ cost_efficiency = 1.0
87
+ else:
88
+ # Scale: $0.0001 -> 0.95, $0.01 -> 0.50, $0.05 -> 0.10
89
+ cost_efficiency = max(0.05, min(0.99, 1.0 / (1.0 + (combined_token_cost * 100.0))))
90
+
91
+ if analysis.cost_sensitivity == PriorityLevel.HIGH:
92
+ cost_efficiency = min(1.0, cost_efficiency * 1.25)
93
+
94
+ # 4. Capability Match Component (0.0 to 1.0)
95
+ cap_points = 0.0
96
+ total_caps = 0.0
97
+
98
+ if analysis.coding_required:
99
+ total_caps += 1.0
100
+ if model.supports_coding:
101
+ cap_points += 1.0
102
+
103
+ if analysis.reasoning_required:
104
+ total_caps += 1.0
105
+ if model.supports_reasoning:
106
+ cap_points += 1.0
107
+
108
+ if analysis.vision_required:
109
+ total_caps += 1.0
110
+ if model.supports_vision:
111
+ cap_points += 1.0
112
+
113
+ capability_comp = (cap_points / total_caps) if total_caps > 0 else 0.90
114
+
115
+ # 5. Reliability Component (0.0 to 1.0)
116
+ reliability_comp = model.reliability_score
117
+
118
+ # Weighted Sum
119
+ w_q = weights.get("quality_weight", 0.35)
120
+ w_c = weights.get("cost_weight", 0.25)
121
+ w_s = weights.get("speed_weight", 0.20)
122
+ w_cap = weights.get("capability_weight", 0.15)
123
+ w_r = weights.get("reliability_weight", 0.05)
124
+ total_w = w_q + w_c + w_s + w_cap + w_r
125
+
126
+ raw_score = (
127
+ (quality_comp * w_q)
128
+ + (cost_efficiency * w_c)
129
+ + (speed_comp * w_s)
130
+ + (capability_comp * w_cap)
131
+ + (reliability_comp * w_r)
132
+ ) / (total_w if total_w > 0 else 1.0)
133
+
134
+ overall_score = round(raw_score * 100.0, 2)
135
+
136
+ # Estimated metrics
137
+ est_latency = 120.0 if model.tier.value == "FAST" else (350.0 if model.tier.value == "BALANCED" else 750.0)
138
+ est_cost = (analysis.context_size * model.cost_per_input_token) + (250 * model.cost_per_output_token)
139
+
140
+ return CandidateScore(
141
+ model_id=model.id,
142
+ model_name=model.name,
143
+ provider=model.provider,
144
+ tier=model.tier.value if hasattr(model.tier, "value") else str(model.tier),
145
+ overall_score=overall_score,
146
+ quality_component=round(quality_comp * 100.0, 1),
147
+ cost_component=round(cost_efficiency * 100.0, 1),
148
+ speed_component=round(speed_comp * 100.0, 1),
149
+ capability_component=round(capability_comp * 100.0, 1),
150
+ reliability_component=round(reliability_comp * 100.0, 1),
151
+ estimated_latency_ms=est_latency,
152
+ estimated_cost_usd=round(est_cost, 6),
153
+ eligible=True,
154
+ )