durallm 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. durallm/__init__.py +265 -0
  2. durallm/_env.py +13 -0
  3. durallm/agent/__init__.py +27 -0
  4. durallm/agent/context.py +274 -0
  5. durallm/agent/failover_plan.py +34 -0
  6. durallm/agent/idempotency.py +220 -0
  7. durallm/agent/state.py +124 -0
  8. durallm/agent/tool_validation.py +272 -0
  9. durallm/breaker/__init__.py +31 -0
  10. durallm/breaker/circuit_breaker.py +302 -0
  11. durallm/breaker/metrics.py +119 -0
  12. durallm/breaker/registry.py +53 -0
  13. durallm/breaker/state.py +28 -0
  14. durallm/capability/__init__.py +14 -0
  15. durallm/capability/profile.py +146 -0
  16. durallm/capability/registry.py +121 -0
  17. durallm/classifier.py +421 -0
  18. durallm/config.py +92 -0
  19. durallm/continuation/__init__.py +22 -0
  20. durallm/continuation/models.py +110 -0
  21. durallm/continuation/sqlite.py +268 -0
  22. durallm/continuation/store.py +236 -0
  23. durallm/demo.py +198 -0
  24. durallm/discovery.py +254 -0
  25. durallm/errors.py +155 -0
  26. durallm/execution/__init__.py +19 -0
  27. durallm/execution/deadline.py +76 -0
  28. durallm/execution/executor.py +885 -0
  29. durallm/execution/ledger.py +82 -0
  30. durallm/execution/policy.py +48 -0
  31. durallm/gateway.py +163 -0
  32. durallm/health/__init__.py +13 -0
  33. durallm/health/telemetry.py +205 -0
  34. durallm/mcp/__init__.py +13 -0
  35. durallm/mcp/proxy.py +303 -0
  36. durallm/models.py +120 -0
  37. durallm/observability/logger.py +91 -0
  38. durallm/pools.py +306 -0
  39. durallm/protocol/__init__.py +43 -0
  40. durallm/protocol/anthropic.py +268 -0
  41. durallm/protocol/gemini.py +268 -0
  42. durallm/protocol/ir.py +100 -0
  43. durallm/protocol/openai.py +307 -0
  44. durallm/providers/__init__.py +35 -0
  45. durallm/providers/adapters.py +547 -0
  46. durallm/providers/base.py +193 -0
  47. durallm/proxy.py +942 -0
  48. durallm/pruner.py +137 -0
  49. durallm/router.py +341 -0
  50. durallm/routing/__init__.py +28 -0
  51. durallm/routing/budget.py +71 -0
  52. durallm/routing/cache.py +113 -0
  53. durallm/routing/decision.py +79 -0
  54. durallm/routing/keys.py +138 -0
  55. durallm/routing/quality.py +103 -0
  56. durallm/routing/requirements.py +156 -0
  57. durallm/routing/resources.py +63 -0
  58. durallm/routing/router.py +332 -0
  59. durallm/routing/scorer.py +118 -0
  60. durallm/routing/tokenizer.py +73 -0
  61. durallm/security/defense.py +84 -0
  62. durallm/storage/__init__.py +16 -0
  63. durallm/storage/contracts.py +120 -0
  64. durallm/storage/sqlite.py +552 -0
  65. durallm/storage/tool_ledger.py +142 -0
  66. durallm/streaming/__init__.py +19 -0
  67. durallm/streaming/modes.py +190 -0
  68. durallm/streaming/parser.py +190 -0
  69. durallm/translators.py +341 -0
  70. durallm/validation/response.py +105 -0
  71. durallm-0.2.0.dist-info/METADATA +368 -0
  72. durallm-0.2.0.dist-info/RECORD +77 -0
  73. durallm-0.2.0.dist-info/WHEEL +5 -0
  74. durallm-0.2.0.dist-info/entry_points.txt +5 -0
  75. durallm-0.2.0.dist-info/licenses/LICENSE +21 -0
  76. durallm-0.2.0.dist-info/top_level.txt +2 -0
  77. llm_circuit_breaker/__init__.py +39 -0
durallm/__init__.py ADDED
@@ -0,0 +1,265 @@
1
+ """⚡ LLM Circuit Breaker
2
+
3
+ Self-Hostable Agent Resilience Gateway with Capability-Aware Routing,
4
+ Semantic Failover, and Zero-Dependency Autonomous Recovery.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ # V2 Core Exports
10
+ from durallm.agent import (
11
+ AgentState,
12
+ ContextBudget,
13
+ ContextManager,
14
+ StateSnapshot,
15
+ ToolCallResult,
16
+ ToolCallValidator,
17
+ ToolValidationReport,
18
+ extract_diagnostic_summary,
19
+ )
20
+ from durallm.breaker import (
21
+ DEFAULT_BREAKER_REGISTRY,
22
+ CircuitBreaker,
23
+ CircuitBreakerConfig,
24
+ CircuitBreakerRegistry,
25
+ CircuitBreakerState,
26
+ StateTransitionEvent,
27
+ )
28
+ from durallm.capability import (
29
+ DEFAULT_CAPABILITY_REGISTRY,
30
+ CapabilityRegistry,
31
+ Endpoint,
32
+ ModelProfile,
33
+ )
34
+ from durallm.classifier import (
35
+ ClassifiedError,
36
+ FailoverReason,
37
+ FailureCategory,
38
+ FailureClassification,
39
+ classify_api_error,
40
+ classify_failure,
41
+ parse_retry_after,
42
+ )
43
+ from durallm.config import GatewayConfig
44
+ from durallm.continuation import (
45
+ ACP_VERSION,
46
+ Checkpoint,
47
+ ContinuationEvent,
48
+ ContinuationRequest,
49
+ ContinuationStore,
50
+ ContinuationTurn,
51
+ InMemoryContinuationStore,
52
+ )
53
+ from durallm.discovery import (
54
+ discover_free_models,
55
+ is_model_free,
56
+ supports_tool_calling,
57
+ )
58
+ from durallm.errors import (
59
+ BreakerOpenError,
60
+ CircuitBreakerError,
61
+ ContextOverflowError,
62
+ ContinuationProtocolError,
63
+ CycleDetectedError,
64
+ DeadlineExceededError,
65
+ GatewayError,
66
+ IndeterminateToolOperationError,
67
+ NoHealthyRouteError,
68
+ ProbeAdmissionDeniedError,
69
+ ToolOperationProtocolError,
70
+ UnsafeToolCallError,
71
+ )
72
+ from durallm.execution import (
73
+ AttemptLedger,
74
+ Deadline,
75
+ ExecutionPolicy,
76
+ FallbackPolicy,
77
+ GatewayExecutor,
78
+ RetryPolicy,
79
+ )
80
+ from durallm.health import (
81
+ DEFAULT_HEALTH_STORE,
82
+ EndpointHealthSnapshot,
83
+ HealthTelemetryStore,
84
+ )
85
+ from durallm.mcp import (
86
+ DEFAULT_MCP_PROXY,
87
+ MCPProxy,
88
+ MCPToolDefinition,
89
+ )
90
+ from durallm.models import AttemptRecord
91
+ from durallm.pools import (
92
+ POOL_MANAGER,
93
+ IsolatedPoolManager,
94
+ RouteDefinition,
95
+ )
96
+ from durallm.protocol import (
97
+ NormalizedMessage,
98
+ NormalizedRequest,
99
+ NormalizedResponse,
100
+ NormalizedToolCall,
101
+ NormalizedToolDefinition,
102
+ NormalizedToolResult,
103
+ anthropic_request_to_ir,
104
+ gemini_response_to_ir,
105
+ ir_to_anthropic_request,
106
+ ir_to_anthropic_response,
107
+ ir_to_gemini_request,
108
+ ir_to_openai_request,
109
+ ir_to_openai_response,
110
+ openai_request_to_ir,
111
+ openai_response_to_ir,
112
+ )
113
+ from durallm.proxy import (
114
+ CircuitBreakerGatewayHandler,
115
+ create_proxy_app,
116
+ start_proxy_server,
117
+ )
118
+ from durallm.pruner import (
119
+ estimate_tokens,
120
+ prune_anthropic_request,
121
+ prune_openai_request,
122
+ )
123
+ from durallm.router import UniversalFailoverRouter
124
+ from durallm.routing import (
125
+ CandidateEvaluation,
126
+ CapabilityRouter,
127
+ RequirementVector,
128
+ RoutingDecision,
129
+ RoutingScorer,
130
+ )
131
+ from durallm.streaming import (
132
+ MidStreamFailurePolicy,
133
+ StreamingMetrics,
134
+ StreamingMode,
135
+ interruption_sse,
136
+ synthesize_anthropic_sse,
137
+ synthesize_openai_sse,
138
+ )
139
+ from durallm.translators import (
140
+ anthropic_to_openai_request,
141
+ clean_gemini_schema,
142
+ convert_gemini_to_openai_response,
143
+ convert_openai_to_gemini_payload,
144
+ openai_to_anthropic_response,
145
+ repair_json_string,
146
+ )
147
+
148
+ __version__ = "0.2.0"
149
+
150
+ __all__ = [
151
+ # Breaker & Registry
152
+ "CircuitBreaker",
153
+ "CircuitBreakerConfig",
154
+ "CircuitBreakerState",
155
+ "CircuitBreakerRegistry",
156
+ "DEFAULT_BREAKER_REGISTRY",
157
+ "StateTransitionEvent",
158
+ # Capability & Routing
159
+ "ModelProfile",
160
+ "Endpoint",
161
+ "CapabilityRegistry",
162
+ "DEFAULT_CAPABILITY_REGISTRY",
163
+ "CapabilityRouter",
164
+ "RequirementVector",
165
+ "RoutingDecision",
166
+ "CandidateEvaluation",
167
+ "RoutingScorer",
168
+ # Execution & Deadlines
169
+ "GatewayExecutor",
170
+ "ExecutionPolicy",
171
+ "RetryPolicy",
172
+ "FallbackPolicy",
173
+ "Deadline",
174
+ "AttemptLedger",
175
+ "AttemptRecord",
176
+ # Agent & Tool Safety
177
+ "AgentState",
178
+ "StateSnapshot",
179
+ "ToolCallValidator",
180
+ "ToolCallResult",
181
+ "ToolValidationReport",
182
+ "ContextManager",
183
+ "ContextBudget",
184
+ "extract_diagnostic_summary",
185
+ "MCPProxy",
186
+ "DEFAULT_MCP_PROXY",
187
+ "MCPToolDefinition",
188
+ # Agent Continuation Protocol
189
+ "ACP_VERSION",
190
+ "Checkpoint",
191
+ "ContinuationEvent",
192
+ "ContinuationRequest",
193
+ "ContinuationStore",
194
+ "ContinuationTurn",
195
+ "InMemoryContinuationStore",
196
+ # Protocol IR
197
+ "NormalizedRequest",
198
+ "NormalizedResponse",
199
+ "NormalizedMessage",
200
+ "NormalizedToolCall",
201
+ "NormalizedToolDefinition",
202
+ "NormalizedToolResult",
203
+ "anthropic_request_to_ir",
204
+ "ir_to_anthropic_request",
205
+ "ir_to_anthropic_response",
206
+ "openai_request_to_ir",
207
+ "ir_to_openai_request",
208
+ "openai_response_to_ir",
209
+ "ir_to_openai_response",
210
+ "ir_to_gemini_request",
211
+ "gemini_response_to_ir",
212
+ # Streaming & Health
213
+ "StreamingMode",
214
+ "MidStreamFailurePolicy",
215
+ "StreamingMetrics",
216
+ "interruption_sse",
217
+ "synthesize_anthropic_sse",
218
+ "synthesize_openai_sse",
219
+ "HealthTelemetryStore",
220
+ "EndpointHealthSnapshot",
221
+ "DEFAULT_HEALTH_STORE",
222
+ # Classifier & Taxonomy
223
+ "FailureCategory",
224
+ "FailoverReason",
225
+ "FailureClassification",
226
+ "classify_failure",
227
+ "parse_retry_after",
228
+ "classify_api_error",
229
+ "ClassifiedError",
230
+ # Errors
231
+ "GatewayError",
232
+ "CircuitBreakerError",
233
+ "BreakerOpenError",
234
+ "ProbeAdmissionDeniedError",
235
+ "DeadlineExceededError",
236
+ "NoHealthyRouteError",
237
+ "UnsafeToolCallError",
238
+ "ContinuationProtocolError",
239
+ "IndeterminateToolOperationError",
240
+ "ToolOperationProtocolError",
241
+ "ContextOverflowError",
242
+ "CycleDetectedError",
243
+ # Configuration
244
+ "GatewayConfig",
245
+ # V1 Compatibility
246
+ "UniversalFailoverRouter",
247
+ "POOL_MANAGER",
248
+ "IsolatedPoolManager",
249
+ "RouteDefinition",
250
+ "prune_anthropic_request",
251
+ "prune_openai_request",
252
+ "estimate_tokens",
253
+ "anthropic_to_openai_request",
254
+ "openai_to_anthropic_response",
255
+ "clean_gemini_schema",
256
+ "convert_openai_to_gemini_payload",
257
+ "convert_gemini_to_openai_response",
258
+ "repair_json_string",
259
+ "discover_free_models",
260
+ "is_model_free",
261
+ "supports_tool_calling",
262
+ "CircuitBreakerGatewayHandler",
263
+ "start_proxy_server",
264
+ "create_proxy_app",
265
+ ]
durallm/_env.py ADDED
@@ -0,0 +1,13 @@
1
+ """Opt-in environment flags shared by the V1 and V3 planes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+
7
+ # Loopback / RFC1918 upstreams (Ollama, LM Studio, a LAN proxy) are blocked unless this is set.
8
+ ALLOW_LOCAL_UPSTREAM_ENV = "LLM_BREAKER_ALLOW_LOCAL_UPSTREAM"
9
+
10
+
11
+ def env_flag(name: str) -> bool:
12
+ """True when an opt-in environment variable is set to 1/true/yes/on."""
13
+ return os.getenv(name, "").strip().lower() in ("1", "true", "yes", "on")
@@ -0,0 +1,27 @@
1
+ """Agent Semantic Resilience Subsystem."""
2
+
3
+ from durallm.agent.context import (
4
+ ContextBudget,
5
+ ContextManager,
6
+ estimate_tokens,
7
+ extract_diagnostic_summary,
8
+ )
9
+ from durallm.agent.state import AgentState, StateSnapshot
10
+ from durallm.agent.tool_validation import (
11
+ ToolCallResult,
12
+ ToolCallValidator,
13
+ ToolValidationReport,
14
+ )
15
+
16
+ __all__ = [
17
+ "AgentState",
18
+ "StateSnapshot",
19
+ "ToolCallValidator",
20
+ "ToolCallResult",
21
+ "ToolValidationReport",
22
+ "ContextManager",
23
+ "ContextBudget",
24
+ "estimate_tokens",
25
+ "extract_diagnostic_summary",
26
+ ]
27
+
@@ -0,0 +1,274 @@
1
+ """Budget-Aware Context Manager and Structured Semantic Compactor."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import copy
6
+ import json
7
+ import logging
8
+ import re
9
+ from dataclasses import dataclass
10
+ from typing import Any, Tuple
11
+
12
+ from durallm.errors import ContextOverflowError
13
+ from durallm.protocol.ir import (
14
+ NormalizedRequest,
15
+ )
16
+
17
+ logger = logging.getLogger("durallm.agent.context")
18
+
19
+
20
+ def estimate_tokens(payload: Any) -> int:
21
+ """Safe, conservative token estimation (~3.8 characters per token)."""
22
+ if isinstance(payload, str):
23
+ text_len = len(payload)
24
+ elif isinstance(payload, NormalizedRequest):
25
+ parts = []
26
+ if payload.system_instruction:
27
+ parts.append(payload.system_instruction)
28
+ for m in payload.messages:
29
+ if m.content:
30
+ parts.append(m.content)
31
+ if m.reasoning_content:
32
+ parts.append(m.reasoning_content)
33
+ for tc in m.tool_calls:
34
+ parts.append(tc.raw_arguments or json.dumps(tc.arguments))
35
+ for tr in m.tool_results:
36
+ parts.append(tr.content)
37
+ text_len = sum(len(p) for p in parts)
38
+ else:
39
+ try:
40
+ text_len = len(json.dumps(payload, ensure_ascii=False))
41
+ except Exception:
42
+ text_len = len(str(payload))
43
+ return max(1, (text_len + 3) // 4)
44
+
45
+
46
+ DIAGNOSTIC_PATTERNS = re.compile(
47
+ r"("
48
+ r"(?:error|fatal|fail(?:ed|ure)?|exception|critical|traceback|panic)\b"
49
+ r"|(?:exit(?:\s+code)?|returncode)\s*[:=]?\s*\d+"
50
+ r"|(?:AssertionError|TypeError|ValueError|KeyError|IndexError|AttributeError|SyntaxError|NameError)"
51
+ r"|(?:\bFAILED\b|\bFAIL\b|\b=== FAILURES ===\b|\b=== ERRORS ===\b)"
52
+ r"|error\[E\d+\]"
53
+ r"|error TS\d+"
54
+ r"|(?:[\w\.\-/]+\.\w+):(\d+)(?::(\d+))?:\s*(?:fatal )?(?:error|warning)"
55
+ r"|File \"[^\"]+\", line \d+"
56
+ r"|diff --git"
57
+ r"|@@ -[0-9,]+ \+[0-9,]+ @@"
58
+ r")",
59
+ re.IGNORECASE,
60
+ )
61
+
62
+
63
+ def extract_diagnostic_summary(raw_content: str, max_chars: int = 600) -> str:
64
+ """
65
+ Extract structured diagnostic information from compiler outputs, test traces,
66
+ and tool outputs rather than blind text slicing.
67
+ Preserves:
68
+ - Compiler error sites (file:line:col: error)
69
+ - Test failure assertions (FAILED test_..., AssertionError)
70
+ - Python/runtime tracebacks
71
+ - Exit codes and execution summaries
72
+ """
73
+ if len(raw_content) <= max_chars:
74
+ return raw_content
75
+
76
+ # 1. Attempt JSON structured extraction
77
+ try:
78
+ data = json.loads(raw_content)
79
+ if isinstance(data, dict):
80
+ extracted = {}
81
+ for k in [
82
+ "status", "exit_code", "returncode", "error", "errors", "message",
83
+ "path", "file", "id", "count", "stderr", "stdout",
84
+ ]:
85
+ if k in data:
86
+ extracted[k] = data[k]
87
+ if extracted:
88
+ formatted = (
89
+ f"[Structured Tool Output Summary (by Circuit Breaker)]:\n"
90
+ f"{json.dumps(extracted, ensure_ascii=False, indent=2)}\n"
91
+ f"... (remaining payload truncated to preserve context budget)"
92
+ )
93
+ if len(formatted) <= max_chars:
94
+ return formatted
95
+ return formatted[:max_chars] + "\n... [truncated]"
96
+ except Exception:
97
+ pass
98
+
99
+ # 2. Text / Log file extraction: hunt for diagnostic lines
100
+ lines = raw_content.splitlines()
101
+ if len(lines) <= 4:
102
+ half = max(50, (max_chars - 60) // 2)
103
+ return (
104
+ "[Historical Tool Output compacted by Circuit Breaker to fit target budget]\n"
105
+ + raw_content[:half]
106
+ + "\n... [truncated] ...\n"
107
+ + raw_content[-half:]
108
+ )
109
+
110
+ diagnostic_lines: list[str] = []
111
+ seen = set()
112
+ for ln in lines:
113
+ cleaned = ln.strip()
114
+ if not cleaned:
115
+ continue
116
+ # Skip repetitive progress lines
117
+ if re.match(r"^[\.FEsxX]+\s+\[\s*\d+%\]", cleaned) or re.match(r"^\.+$", cleaned):
118
+ continue
119
+ if DIAGNOSTIC_PATTERNS.search(cleaned):
120
+ if cleaned not in seen:
121
+ seen.add(cleaned)
122
+ diagnostic_lines.append(cleaned)
123
+
124
+ header_lines = [ln.strip() for ln in lines[:3] if ln.strip()]
125
+ tail_lines = [ln.strip() for ln in lines[-3:] if ln.strip()]
126
+
127
+ parts = [
128
+ "[Historical Tool Output compacted by Circuit Breaker to fit target budget]",
129
+ f"--- HEAD ({len(lines)} total lines) ---",
130
+ "\n".join(header_lines),
131
+ ]
132
+
133
+ if diagnostic_lines:
134
+ max_diag = max(2, min(len(diagnostic_lines), 8))
135
+ parts.extend([
136
+ f"--- EXTRACTED DIAGNOSTICS & ERRORS ({len(diagnostic_lines)} findings) ---",
137
+ "\n".join(diagnostic_lines[:max_diag]),
138
+ ])
139
+
140
+ parts.extend([
141
+ "--- TAIL ---",
142
+ "\n".join(tail_lines),
143
+ ])
144
+
145
+ summary = "\n".join(parts)
146
+ if len(summary) > max_chars:
147
+ summary = summary[:max_chars] + "\n... [truncated]"
148
+ return summary
149
+
150
+
151
+ def extract_structured_tool_summary(raw_content: str, max_chars: int = 500) -> str:
152
+ """Backward-compatible wrapper for extract_diagnostic_summary."""
153
+ return extract_diagnostic_summary(raw_content, max_chars=max_chars)
154
+
155
+
156
+ @dataclass
157
+ class ContextBudget:
158
+ """Model context window and reserved output token budget."""
159
+ model_context_window: int = 65536
160
+ desired_output_tokens: int = 4096
161
+ safety_margin_tokens: int = 2048
162
+
163
+ @property
164
+ def available_input_budget(self) -> int:
165
+ """Remaining tokens available for input prompt history."""
166
+ return max(512, self.model_context_window - self.desired_output_tokens - self.safety_margin_tokens)
167
+
168
+
169
+ class ContextManager:
170
+ """
171
+ Manages request context size, enforcing explicit token budgets and hierarchical compaction.
172
+ Compaction hierarchy:
173
+ 1. System instructions (never dropped)
174
+ 2. Root user objective (first user prompt, never dropped)
175
+ 3. Active constraints (never dropped)
176
+ 4. Diagnostic compaction of large compiler/test traces in tool results and messages
177
+ 5. Recent execution turns (latest preserve_tail_turns intact)
178
+ 6. Evict oldest intermediate pairs between root objective and recent tail turns.
179
+ """
180
+
181
+ def __init__(self, preserve_tail_turns: int = 6):
182
+ self.preserve_tail_turns = preserve_tail_turns
183
+
184
+ def compact(
185
+ self,
186
+ request: NormalizedRequest,
187
+ budget: ContextBudget,
188
+ ) -> Tuple[NormalizedRequest, bool]:
189
+ """
190
+ Compact request to fit strictly within the target model's available input budget.
191
+ Returns (compacted_request, was_compacted).
192
+ Raises ContextOverflowError when the protected content (system instruction, root
193
+ objective, preserved tail turns) still exceeds the budget after every compaction phase.
194
+ """
195
+ current_tokens = estimate_tokens(request)
196
+ target_tokens = budget.available_input_budget
197
+
198
+ if current_tokens <= target_tokens:
199
+ return request, False
200
+
201
+ logger.info(
202
+ "Request size (%d tokens) exceeds available budget (%d tokens). Initiating hierarchical compaction.",
203
+ current_tokens, target_tokens
204
+ )
205
+
206
+ compacted = copy.deepcopy(request)
207
+ messages = compacted.messages
208
+ first_user_idx = next((idx for idx, message in enumerate(messages) if message.role == "user"), None)
209
+ protected_prefix_count = (first_user_idx + 1) if first_user_idx is not None else 1
210
+
211
+ # Phase 1: In-place diagnostic compaction of oversized tool results and logs
212
+ for idx, m in enumerate(messages):
213
+ for tr in m.tool_results:
214
+ if len(tr.content) > 400:
215
+ tr.content = extract_diagnostic_summary(tr.content, max_chars=400)
216
+ # Compact message content if it contains huge logs (e.g. OpenCode compiler/test outputs)
217
+ # Do not drop root objective or final instruction entirely; compact huge diagnostic bodies
218
+ is_root_user = (idx == first_user_idx)
219
+ is_final_msg = (idx == len(messages) - 1)
220
+ if m.content and len(m.content) > 1000:
221
+ if not is_root_user and not is_final_msg:
222
+ m.content = extract_diagnostic_summary(m.content, max_chars=600)
223
+ elif is_final_msg and not is_root_user:
224
+ # Final instruction has huge attached log; compact the log while keeping tail instructions
225
+ m.content = extract_diagnostic_summary(m.content, max_chars=1200)
226
+
227
+ if estimate_tokens(compacted) <= target_tokens:
228
+ return compacted, True
229
+
230
+ # Phase 2: If message turns exceed tail window, evict intermediate turns
231
+ cutoff_idx = len(messages) - self.preserve_tail_turns
232
+ for idx in range(protected_prefix_count, max(protected_prefix_count, cutoff_idx)):
233
+ m = messages[idx]
234
+ for tr in m.tool_results:
235
+ if len(tr.content) > 300:
236
+ tr.content = extract_diagnostic_summary(tr.content, max_chars=300)
237
+ if m.content and len(m.content) > 600 and m.role == "assistant":
238
+ m.content = (
239
+ m.content[:300]
240
+ + "\n... [Prior assistant reasoning compacted by Circuit Breaker] ...\n"
241
+ + m.content[-300:]
242
+ )
243
+
244
+ if estimate_tokens(compacted) <= target_tokens:
245
+ return compacted, True
246
+
247
+ start_evict_idx = protected_prefix_count
248
+ while len(compacted.messages) > (self.preserve_tail_turns + protected_prefix_count):
249
+ if estimate_tokens(compacted) <= target_tokens:
250
+ break
251
+ compacted.messages.pop(start_evict_idx)
252
+
253
+ if estimate_tokens(compacted) <= target_tokens:
254
+ return compacted, True
255
+
256
+ # Phase 3: Aggressive compaction of tail turns if still over budget
257
+ for idx, m in enumerate(compacted.messages):
258
+ if idx == first_user_idx:
259
+ continue
260
+ for tr in m.tool_results:
261
+ if len(tr.content) > 250:
262
+ tr.content = extract_diagnostic_summary(tr.content, max_chars=250)
263
+ if m.content and len(m.content) > 500:
264
+ m.content = extract_diagnostic_summary(m.content, max_chars=400)
265
+
266
+ self._require_fit(compacted, target_tokens)
267
+ return compacted, True
268
+
269
+ @staticmethod
270
+ def _require_fit(request: NormalizedRequest, target_tokens: int) -> None:
271
+ remaining = estimate_tokens(request)
272
+ if remaining > target_tokens:
273
+ raise ContextOverflowError(required_tokens=remaining, available_budget=target_tokens)
274
+
@@ -0,0 +1,34 @@
1
+ """Explainable Semantic Failover Plan."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ import uuid
7
+ from dataclasses import asdict, dataclass, field
8
+ from typing import Any, Dict, List, Optional
9
+
10
+
11
+ @dataclass
12
+ class FailoverPlan:
13
+ """
14
+ Explicit, auditable plan created when transferring execution from one provider/model to another.
15
+ Coordinates state preservation, context budgeting, protocol translation, and tool safety.
16
+ """
17
+ plan_id: str = field(default_factory=lambda: f"fplan_{uuid.uuid4().hex[:8]}")
18
+ request_id: str = ""
19
+ source_endpoint: Optional[str] = None
20
+ target_endpoint: str = ""
21
+ failover_reason: str = ""
22
+ state_snapshot_id: Optional[str] = None
23
+ required_transformations: List[str] = field(default_factory=list)
24
+ context_tokens_before: int = 0
25
+ context_tokens_after: int = 0
26
+ context_compaction_applied: bool = False
27
+ tool_transformations: List[str] = field(default_factory=list)
28
+ risk_flags: List[str] = field(default_factory=list)
29
+ remaining_deadline_ms: float = 0.0
30
+ remaining_cost_budget_usd: Optional[float] = None
31
+ created_at_monotonic: float = field(default_factory=time.monotonic)
32
+
33
+ def to_dict(self) -> Dict[str, Any]:
34
+ return asdict(self)