durallm 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- durallm/__init__.py +265 -0
- durallm/_env.py +13 -0
- durallm/agent/__init__.py +27 -0
- durallm/agent/context.py +274 -0
- durallm/agent/failover_plan.py +34 -0
- durallm/agent/idempotency.py +220 -0
- durallm/agent/state.py +124 -0
- durallm/agent/tool_validation.py +272 -0
- durallm/breaker/__init__.py +31 -0
- durallm/breaker/circuit_breaker.py +302 -0
- durallm/breaker/metrics.py +119 -0
- durallm/breaker/registry.py +53 -0
- durallm/breaker/state.py +28 -0
- durallm/capability/__init__.py +14 -0
- durallm/capability/profile.py +146 -0
- durallm/capability/registry.py +121 -0
- durallm/classifier.py +421 -0
- durallm/config.py +92 -0
- durallm/continuation/__init__.py +22 -0
- durallm/continuation/models.py +110 -0
- durallm/continuation/sqlite.py +268 -0
- durallm/continuation/store.py +236 -0
- durallm/demo.py +198 -0
- durallm/discovery.py +254 -0
- durallm/errors.py +155 -0
- durallm/execution/__init__.py +19 -0
- durallm/execution/deadline.py +76 -0
- durallm/execution/executor.py +885 -0
- durallm/execution/ledger.py +82 -0
- durallm/execution/policy.py +48 -0
- durallm/gateway.py +163 -0
- durallm/health/__init__.py +13 -0
- durallm/health/telemetry.py +205 -0
- durallm/mcp/__init__.py +13 -0
- durallm/mcp/proxy.py +303 -0
- durallm/models.py +120 -0
- durallm/observability/logger.py +91 -0
- durallm/pools.py +306 -0
- durallm/protocol/__init__.py +43 -0
- durallm/protocol/anthropic.py +268 -0
- durallm/protocol/gemini.py +268 -0
- durallm/protocol/ir.py +100 -0
- durallm/protocol/openai.py +307 -0
- durallm/providers/__init__.py +35 -0
- durallm/providers/adapters.py +547 -0
- durallm/providers/base.py +193 -0
- durallm/proxy.py +942 -0
- durallm/pruner.py +137 -0
- durallm/router.py +341 -0
- durallm/routing/__init__.py +28 -0
- durallm/routing/budget.py +71 -0
- durallm/routing/cache.py +113 -0
- durallm/routing/decision.py +79 -0
- durallm/routing/keys.py +138 -0
- durallm/routing/quality.py +103 -0
- durallm/routing/requirements.py +156 -0
- durallm/routing/resources.py +63 -0
- durallm/routing/router.py +332 -0
- durallm/routing/scorer.py +118 -0
- durallm/routing/tokenizer.py +73 -0
- durallm/security/defense.py +84 -0
- durallm/storage/__init__.py +16 -0
- durallm/storage/contracts.py +120 -0
- durallm/storage/sqlite.py +552 -0
- durallm/storage/tool_ledger.py +142 -0
- durallm/streaming/__init__.py +19 -0
- durallm/streaming/modes.py +190 -0
- durallm/streaming/parser.py +190 -0
- durallm/translators.py +341 -0
- durallm/validation/response.py +105 -0
- durallm-0.2.0.dist-info/METADATA +368 -0
- durallm-0.2.0.dist-info/RECORD +77 -0
- durallm-0.2.0.dist-info/WHEEL +5 -0
- durallm-0.2.0.dist-info/entry_points.txt +5 -0
- durallm-0.2.0.dist-info/licenses/LICENSE +21 -0
- durallm-0.2.0.dist-info/top_level.txt +2 -0
- llm_circuit_breaker/__init__.py +39 -0
durallm/__init__.py
ADDED
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
"""⚡ LLM Circuit Breaker
|
|
2
|
+
|
|
3
|
+
Self-Hostable Agent Resilience Gateway with Capability-Aware Routing,
|
|
4
|
+
Semantic Failover, and Zero-Dependency Autonomous Recovery.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
# V2 Core Exports
|
|
10
|
+
from durallm.agent import (
|
|
11
|
+
AgentState,
|
|
12
|
+
ContextBudget,
|
|
13
|
+
ContextManager,
|
|
14
|
+
StateSnapshot,
|
|
15
|
+
ToolCallResult,
|
|
16
|
+
ToolCallValidator,
|
|
17
|
+
ToolValidationReport,
|
|
18
|
+
extract_diagnostic_summary,
|
|
19
|
+
)
|
|
20
|
+
from durallm.breaker import (
|
|
21
|
+
DEFAULT_BREAKER_REGISTRY,
|
|
22
|
+
CircuitBreaker,
|
|
23
|
+
CircuitBreakerConfig,
|
|
24
|
+
CircuitBreakerRegistry,
|
|
25
|
+
CircuitBreakerState,
|
|
26
|
+
StateTransitionEvent,
|
|
27
|
+
)
|
|
28
|
+
from durallm.capability import (
|
|
29
|
+
DEFAULT_CAPABILITY_REGISTRY,
|
|
30
|
+
CapabilityRegistry,
|
|
31
|
+
Endpoint,
|
|
32
|
+
ModelProfile,
|
|
33
|
+
)
|
|
34
|
+
from durallm.classifier import (
|
|
35
|
+
ClassifiedError,
|
|
36
|
+
FailoverReason,
|
|
37
|
+
FailureCategory,
|
|
38
|
+
FailureClassification,
|
|
39
|
+
classify_api_error,
|
|
40
|
+
classify_failure,
|
|
41
|
+
parse_retry_after,
|
|
42
|
+
)
|
|
43
|
+
from durallm.config import GatewayConfig
|
|
44
|
+
from durallm.continuation import (
|
|
45
|
+
ACP_VERSION,
|
|
46
|
+
Checkpoint,
|
|
47
|
+
ContinuationEvent,
|
|
48
|
+
ContinuationRequest,
|
|
49
|
+
ContinuationStore,
|
|
50
|
+
ContinuationTurn,
|
|
51
|
+
InMemoryContinuationStore,
|
|
52
|
+
)
|
|
53
|
+
from durallm.discovery import (
|
|
54
|
+
discover_free_models,
|
|
55
|
+
is_model_free,
|
|
56
|
+
supports_tool_calling,
|
|
57
|
+
)
|
|
58
|
+
from durallm.errors import (
|
|
59
|
+
BreakerOpenError,
|
|
60
|
+
CircuitBreakerError,
|
|
61
|
+
ContextOverflowError,
|
|
62
|
+
ContinuationProtocolError,
|
|
63
|
+
CycleDetectedError,
|
|
64
|
+
DeadlineExceededError,
|
|
65
|
+
GatewayError,
|
|
66
|
+
IndeterminateToolOperationError,
|
|
67
|
+
NoHealthyRouteError,
|
|
68
|
+
ProbeAdmissionDeniedError,
|
|
69
|
+
ToolOperationProtocolError,
|
|
70
|
+
UnsafeToolCallError,
|
|
71
|
+
)
|
|
72
|
+
from durallm.execution import (
|
|
73
|
+
AttemptLedger,
|
|
74
|
+
Deadline,
|
|
75
|
+
ExecutionPolicy,
|
|
76
|
+
FallbackPolicy,
|
|
77
|
+
GatewayExecutor,
|
|
78
|
+
RetryPolicy,
|
|
79
|
+
)
|
|
80
|
+
from durallm.health import (
|
|
81
|
+
DEFAULT_HEALTH_STORE,
|
|
82
|
+
EndpointHealthSnapshot,
|
|
83
|
+
HealthTelemetryStore,
|
|
84
|
+
)
|
|
85
|
+
from durallm.mcp import (
|
|
86
|
+
DEFAULT_MCP_PROXY,
|
|
87
|
+
MCPProxy,
|
|
88
|
+
MCPToolDefinition,
|
|
89
|
+
)
|
|
90
|
+
from durallm.models import AttemptRecord
|
|
91
|
+
from durallm.pools import (
|
|
92
|
+
POOL_MANAGER,
|
|
93
|
+
IsolatedPoolManager,
|
|
94
|
+
RouteDefinition,
|
|
95
|
+
)
|
|
96
|
+
from durallm.protocol import (
|
|
97
|
+
NormalizedMessage,
|
|
98
|
+
NormalizedRequest,
|
|
99
|
+
NormalizedResponse,
|
|
100
|
+
NormalizedToolCall,
|
|
101
|
+
NormalizedToolDefinition,
|
|
102
|
+
NormalizedToolResult,
|
|
103
|
+
anthropic_request_to_ir,
|
|
104
|
+
gemini_response_to_ir,
|
|
105
|
+
ir_to_anthropic_request,
|
|
106
|
+
ir_to_anthropic_response,
|
|
107
|
+
ir_to_gemini_request,
|
|
108
|
+
ir_to_openai_request,
|
|
109
|
+
ir_to_openai_response,
|
|
110
|
+
openai_request_to_ir,
|
|
111
|
+
openai_response_to_ir,
|
|
112
|
+
)
|
|
113
|
+
from durallm.proxy import (
|
|
114
|
+
CircuitBreakerGatewayHandler,
|
|
115
|
+
create_proxy_app,
|
|
116
|
+
start_proxy_server,
|
|
117
|
+
)
|
|
118
|
+
from durallm.pruner import (
|
|
119
|
+
estimate_tokens,
|
|
120
|
+
prune_anthropic_request,
|
|
121
|
+
prune_openai_request,
|
|
122
|
+
)
|
|
123
|
+
from durallm.router import UniversalFailoverRouter
|
|
124
|
+
from durallm.routing import (
|
|
125
|
+
CandidateEvaluation,
|
|
126
|
+
CapabilityRouter,
|
|
127
|
+
RequirementVector,
|
|
128
|
+
RoutingDecision,
|
|
129
|
+
RoutingScorer,
|
|
130
|
+
)
|
|
131
|
+
from durallm.streaming import (
|
|
132
|
+
MidStreamFailurePolicy,
|
|
133
|
+
StreamingMetrics,
|
|
134
|
+
StreamingMode,
|
|
135
|
+
interruption_sse,
|
|
136
|
+
synthesize_anthropic_sse,
|
|
137
|
+
synthesize_openai_sse,
|
|
138
|
+
)
|
|
139
|
+
from durallm.translators import (
|
|
140
|
+
anthropic_to_openai_request,
|
|
141
|
+
clean_gemini_schema,
|
|
142
|
+
convert_gemini_to_openai_response,
|
|
143
|
+
convert_openai_to_gemini_payload,
|
|
144
|
+
openai_to_anthropic_response,
|
|
145
|
+
repair_json_string,
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
__version__ = "0.2.0"
|
|
149
|
+
|
|
150
|
+
__all__ = [
|
|
151
|
+
# Breaker & Registry
|
|
152
|
+
"CircuitBreaker",
|
|
153
|
+
"CircuitBreakerConfig",
|
|
154
|
+
"CircuitBreakerState",
|
|
155
|
+
"CircuitBreakerRegistry",
|
|
156
|
+
"DEFAULT_BREAKER_REGISTRY",
|
|
157
|
+
"StateTransitionEvent",
|
|
158
|
+
# Capability & Routing
|
|
159
|
+
"ModelProfile",
|
|
160
|
+
"Endpoint",
|
|
161
|
+
"CapabilityRegistry",
|
|
162
|
+
"DEFAULT_CAPABILITY_REGISTRY",
|
|
163
|
+
"CapabilityRouter",
|
|
164
|
+
"RequirementVector",
|
|
165
|
+
"RoutingDecision",
|
|
166
|
+
"CandidateEvaluation",
|
|
167
|
+
"RoutingScorer",
|
|
168
|
+
# Execution & Deadlines
|
|
169
|
+
"GatewayExecutor",
|
|
170
|
+
"ExecutionPolicy",
|
|
171
|
+
"RetryPolicy",
|
|
172
|
+
"FallbackPolicy",
|
|
173
|
+
"Deadline",
|
|
174
|
+
"AttemptLedger",
|
|
175
|
+
"AttemptRecord",
|
|
176
|
+
# Agent & Tool Safety
|
|
177
|
+
"AgentState",
|
|
178
|
+
"StateSnapshot",
|
|
179
|
+
"ToolCallValidator",
|
|
180
|
+
"ToolCallResult",
|
|
181
|
+
"ToolValidationReport",
|
|
182
|
+
"ContextManager",
|
|
183
|
+
"ContextBudget",
|
|
184
|
+
"extract_diagnostic_summary",
|
|
185
|
+
"MCPProxy",
|
|
186
|
+
"DEFAULT_MCP_PROXY",
|
|
187
|
+
"MCPToolDefinition",
|
|
188
|
+
# Agent Continuation Protocol
|
|
189
|
+
"ACP_VERSION",
|
|
190
|
+
"Checkpoint",
|
|
191
|
+
"ContinuationEvent",
|
|
192
|
+
"ContinuationRequest",
|
|
193
|
+
"ContinuationStore",
|
|
194
|
+
"ContinuationTurn",
|
|
195
|
+
"InMemoryContinuationStore",
|
|
196
|
+
# Protocol IR
|
|
197
|
+
"NormalizedRequest",
|
|
198
|
+
"NormalizedResponse",
|
|
199
|
+
"NormalizedMessage",
|
|
200
|
+
"NormalizedToolCall",
|
|
201
|
+
"NormalizedToolDefinition",
|
|
202
|
+
"NormalizedToolResult",
|
|
203
|
+
"anthropic_request_to_ir",
|
|
204
|
+
"ir_to_anthropic_request",
|
|
205
|
+
"ir_to_anthropic_response",
|
|
206
|
+
"openai_request_to_ir",
|
|
207
|
+
"ir_to_openai_request",
|
|
208
|
+
"openai_response_to_ir",
|
|
209
|
+
"ir_to_openai_response",
|
|
210
|
+
"ir_to_gemini_request",
|
|
211
|
+
"gemini_response_to_ir",
|
|
212
|
+
# Streaming & Health
|
|
213
|
+
"StreamingMode",
|
|
214
|
+
"MidStreamFailurePolicy",
|
|
215
|
+
"StreamingMetrics",
|
|
216
|
+
"interruption_sse",
|
|
217
|
+
"synthesize_anthropic_sse",
|
|
218
|
+
"synthesize_openai_sse",
|
|
219
|
+
"HealthTelemetryStore",
|
|
220
|
+
"EndpointHealthSnapshot",
|
|
221
|
+
"DEFAULT_HEALTH_STORE",
|
|
222
|
+
# Classifier & Taxonomy
|
|
223
|
+
"FailureCategory",
|
|
224
|
+
"FailoverReason",
|
|
225
|
+
"FailureClassification",
|
|
226
|
+
"classify_failure",
|
|
227
|
+
"parse_retry_after",
|
|
228
|
+
"classify_api_error",
|
|
229
|
+
"ClassifiedError",
|
|
230
|
+
# Errors
|
|
231
|
+
"GatewayError",
|
|
232
|
+
"CircuitBreakerError",
|
|
233
|
+
"BreakerOpenError",
|
|
234
|
+
"ProbeAdmissionDeniedError",
|
|
235
|
+
"DeadlineExceededError",
|
|
236
|
+
"NoHealthyRouteError",
|
|
237
|
+
"UnsafeToolCallError",
|
|
238
|
+
"ContinuationProtocolError",
|
|
239
|
+
"IndeterminateToolOperationError",
|
|
240
|
+
"ToolOperationProtocolError",
|
|
241
|
+
"ContextOverflowError",
|
|
242
|
+
"CycleDetectedError",
|
|
243
|
+
# Configuration
|
|
244
|
+
"GatewayConfig",
|
|
245
|
+
# V1 Compatibility
|
|
246
|
+
"UniversalFailoverRouter",
|
|
247
|
+
"POOL_MANAGER",
|
|
248
|
+
"IsolatedPoolManager",
|
|
249
|
+
"RouteDefinition",
|
|
250
|
+
"prune_anthropic_request",
|
|
251
|
+
"prune_openai_request",
|
|
252
|
+
"estimate_tokens",
|
|
253
|
+
"anthropic_to_openai_request",
|
|
254
|
+
"openai_to_anthropic_response",
|
|
255
|
+
"clean_gemini_schema",
|
|
256
|
+
"convert_openai_to_gemini_payload",
|
|
257
|
+
"convert_gemini_to_openai_response",
|
|
258
|
+
"repair_json_string",
|
|
259
|
+
"discover_free_models",
|
|
260
|
+
"is_model_free",
|
|
261
|
+
"supports_tool_calling",
|
|
262
|
+
"CircuitBreakerGatewayHandler",
|
|
263
|
+
"start_proxy_server",
|
|
264
|
+
"create_proxy_app",
|
|
265
|
+
]
|
durallm/_env.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Opt-in environment flags shared by the V1 and V3 planes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
|
|
7
|
+
# Loopback / RFC1918 upstreams (Ollama, LM Studio, a LAN proxy) are blocked unless this is set.
|
|
8
|
+
ALLOW_LOCAL_UPSTREAM_ENV = "LLM_BREAKER_ALLOW_LOCAL_UPSTREAM"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def env_flag(name: str) -> bool:
|
|
12
|
+
"""True when an opt-in environment variable is set to 1/true/yes/on."""
|
|
13
|
+
return os.getenv(name, "").strip().lower() in ("1", "true", "yes", "on")
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Agent Semantic Resilience Subsystem."""
|
|
2
|
+
|
|
3
|
+
from durallm.agent.context import (
|
|
4
|
+
ContextBudget,
|
|
5
|
+
ContextManager,
|
|
6
|
+
estimate_tokens,
|
|
7
|
+
extract_diagnostic_summary,
|
|
8
|
+
)
|
|
9
|
+
from durallm.agent.state import AgentState, StateSnapshot
|
|
10
|
+
from durallm.agent.tool_validation import (
|
|
11
|
+
ToolCallResult,
|
|
12
|
+
ToolCallValidator,
|
|
13
|
+
ToolValidationReport,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"AgentState",
|
|
18
|
+
"StateSnapshot",
|
|
19
|
+
"ToolCallValidator",
|
|
20
|
+
"ToolCallResult",
|
|
21
|
+
"ToolValidationReport",
|
|
22
|
+
"ContextManager",
|
|
23
|
+
"ContextBudget",
|
|
24
|
+
"estimate_tokens",
|
|
25
|
+
"extract_diagnostic_summary",
|
|
26
|
+
]
|
|
27
|
+
|
durallm/agent/context.py
ADDED
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
"""Budget-Aware Context Manager and Structured Semantic Compactor."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import copy
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from typing import Any, Tuple
|
|
11
|
+
|
|
12
|
+
from durallm.errors import ContextOverflowError
|
|
13
|
+
from durallm.protocol.ir import (
|
|
14
|
+
NormalizedRequest,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger("durallm.agent.context")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def estimate_tokens(payload: Any) -> int:
|
|
21
|
+
"""Safe, conservative token estimation (~3.8 characters per token)."""
|
|
22
|
+
if isinstance(payload, str):
|
|
23
|
+
text_len = len(payload)
|
|
24
|
+
elif isinstance(payload, NormalizedRequest):
|
|
25
|
+
parts = []
|
|
26
|
+
if payload.system_instruction:
|
|
27
|
+
parts.append(payload.system_instruction)
|
|
28
|
+
for m in payload.messages:
|
|
29
|
+
if m.content:
|
|
30
|
+
parts.append(m.content)
|
|
31
|
+
if m.reasoning_content:
|
|
32
|
+
parts.append(m.reasoning_content)
|
|
33
|
+
for tc in m.tool_calls:
|
|
34
|
+
parts.append(tc.raw_arguments or json.dumps(tc.arguments))
|
|
35
|
+
for tr in m.tool_results:
|
|
36
|
+
parts.append(tr.content)
|
|
37
|
+
text_len = sum(len(p) for p in parts)
|
|
38
|
+
else:
|
|
39
|
+
try:
|
|
40
|
+
text_len = len(json.dumps(payload, ensure_ascii=False))
|
|
41
|
+
except Exception:
|
|
42
|
+
text_len = len(str(payload))
|
|
43
|
+
return max(1, (text_len + 3) // 4)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
DIAGNOSTIC_PATTERNS = re.compile(
|
|
47
|
+
r"("
|
|
48
|
+
r"(?:error|fatal|fail(?:ed|ure)?|exception|critical|traceback|panic)\b"
|
|
49
|
+
r"|(?:exit(?:\s+code)?|returncode)\s*[:=]?\s*\d+"
|
|
50
|
+
r"|(?:AssertionError|TypeError|ValueError|KeyError|IndexError|AttributeError|SyntaxError|NameError)"
|
|
51
|
+
r"|(?:\bFAILED\b|\bFAIL\b|\b=== FAILURES ===\b|\b=== ERRORS ===\b)"
|
|
52
|
+
r"|error\[E\d+\]"
|
|
53
|
+
r"|error TS\d+"
|
|
54
|
+
r"|(?:[\w\.\-/]+\.\w+):(\d+)(?::(\d+))?:\s*(?:fatal )?(?:error|warning)"
|
|
55
|
+
r"|File \"[^\"]+\", line \d+"
|
|
56
|
+
r"|diff --git"
|
|
57
|
+
r"|@@ -[0-9,]+ \+[0-9,]+ @@"
|
|
58
|
+
r")",
|
|
59
|
+
re.IGNORECASE,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def extract_diagnostic_summary(raw_content: str, max_chars: int = 600) -> str:
|
|
64
|
+
"""
|
|
65
|
+
Extract structured diagnostic information from compiler outputs, test traces,
|
|
66
|
+
and tool outputs rather than blind text slicing.
|
|
67
|
+
Preserves:
|
|
68
|
+
- Compiler error sites (file:line:col: error)
|
|
69
|
+
- Test failure assertions (FAILED test_..., AssertionError)
|
|
70
|
+
- Python/runtime tracebacks
|
|
71
|
+
- Exit codes and execution summaries
|
|
72
|
+
"""
|
|
73
|
+
if len(raw_content) <= max_chars:
|
|
74
|
+
return raw_content
|
|
75
|
+
|
|
76
|
+
# 1. Attempt JSON structured extraction
|
|
77
|
+
try:
|
|
78
|
+
data = json.loads(raw_content)
|
|
79
|
+
if isinstance(data, dict):
|
|
80
|
+
extracted = {}
|
|
81
|
+
for k in [
|
|
82
|
+
"status", "exit_code", "returncode", "error", "errors", "message",
|
|
83
|
+
"path", "file", "id", "count", "stderr", "stdout",
|
|
84
|
+
]:
|
|
85
|
+
if k in data:
|
|
86
|
+
extracted[k] = data[k]
|
|
87
|
+
if extracted:
|
|
88
|
+
formatted = (
|
|
89
|
+
f"[Structured Tool Output Summary (by Circuit Breaker)]:\n"
|
|
90
|
+
f"{json.dumps(extracted, ensure_ascii=False, indent=2)}\n"
|
|
91
|
+
f"... (remaining payload truncated to preserve context budget)"
|
|
92
|
+
)
|
|
93
|
+
if len(formatted) <= max_chars:
|
|
94
|
+
return formatted
|
|
95
|
+
return formatted[:max_chars] + "\n... [truncated]"
|
|
96
|
+
except Exception:
|
|
97
|
+
pass
|
|
98
|
+
|
|
99
|
+
# 2. Text / Log file extraction: hunt for diagnostic lines
|
|
100
|
+
lines = raw_content.splitlines()
|
|
101
|
+
if len(lines) <= 4:
|
|
102
|
+
half = max(50, (max_chars - 60) // 2)
|
|
103
|
+
return (
|
|
104
|
+
"[Historical Tool Output compacted by Circuit Breaker to fit target budget]\n"
|
|
105
|
+
+ raw_content[:half]
|
|
106
|
+
+ "\n... [truncated] ...\n"
|
|
107
|
+
+ raw_content[-half:]
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
diagnostic_lines: list[str] = []
|
|
111
|
+
seen = set()
|
|
112
|
+
for ln in lines:
|
|
113
|
+
cleaned = ln.strip()
|
|
114
|
+
if not cleaned:
|
|
115
|
+
continue
|
|
116
|
+
# Skip repetitive progress lines
|
|
117
|
+
if re.match(r"^[\.FEsxX]+\s+\[\s*\d+%\]", cleaned) or re.match(r"^\.+$", cleaned):
|
|
118
|
+
continue
|
|
119
|
+
if DIAGNOSTIC_PATTERNS.search(cleaned):
|
|
120
|
+
if cleaned not in seen:
|
|
121
|
+
seen.add(cleaned)
|
|
122
|
+
diagnostic_lines.append(cleaned)
|
|
123
|
+
|
|
124
|
+
header_lines = [ln.strip() for ln in lines[:3] if ln.strip()]
|
|
125
|
+
tail_lines = [ln.strip() for ln in lines[-3:] if ln.strip()]
|
|
126
|
+
|
|
127
|
+
parts = [
|
|
128
|
+
"[Historical Tool Output compacted by Circuit Breaker to fit target budget]",
|
|
129
|
+
f"--- HEAD ({len(lines)} total lines) ---",
|
|
130
|
+
"\n".join(header_lines),
|
|
131
|
+
]
|
|
132
|
+
|
|
133
|
+
if diagnostic_lines:
|
|
134
|
+
max_diag = max(2, min(len(diagnostic_lines), 8))
|
|
135
|
+
parts.extend([
|
|
136
|
+
f"--- EXTRACTED DIAGNOSTICS & ERRORS ({len(diagnostic_lines)} findings) ---",
|
|
137
|
+
"\n".join(diagnostic_lines[:max_diag]),
|
|
138
|
+
])
|
|
139
|
+
|
|
140
|
+
parts.extend([
|
|
141
|
+
"--- TAIL ---",
|
|
142
|
+
"\n".join(tail_lines),
|
|
143
|
+
])
|
|
144
|
+
|
|
145
|
+
summary = "\n".join(parts)
|
|
146
|
+
if len(summary) > max_chars:
|
|
147
|
+
summary = summary[:max_chars] + "\n... [truncated]"
|
|
148
|
+
return summary
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def extract_structured_tool_summary(raw_content: str, max_chars: int = 500) -> str:
|
|
152
|
+
"""Backward-compatible wrapper for extract_diagnostic_summary."""
|
|
153
|
+
return extract_diagnostic_summary(raw_content, max_chars=max_chars)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class ContextBudget:
|
|
158
|
+
"""Model context window and reserved output token budget."""
|
|
159
|
+
model_context_window: int = 65536
|
|
160
|
+
desired_output_tokens: int = 4096
|
|
161
|
+
safety_margin_tokens: int = 2048
|
|
162
|
+
|
|
163
|
+
@property
|
|
164
|
+
def available_input_budget(self) -> int:
|
|
165
|
+
"""Remaining tokens available for input prompt history."""
|
|
166
|
+
return max(512, self.model_context_window - self.desired_output_tokens - self.safety_margin_tokens)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
class ContextManager:
|
|
170
|
+
"""
|
|
171
|
+
Manages request context size, enforcing explicit token budgets and hierarchical compaction.
|
|
172
|
+
Compaction hierarchy:
|
|
173
|
+
1. System instructions (never dropped)
|
|
174
|
+
2. Root user objective (first user prompt, never dropped)
|
|
175
|
+
3. Active constraints (never dropped)
|
|
176
|
+
4. Diagnostic compaction of large compiler/test traces in tool results and messages
|
|
177
|
+
5. Recent execution turns (latest preserve_tail_turns intact)
|
|
178
|
+
6. Evict oldest intermediate pairs between root objective and recent tail turns.
|
|
179
|
+
"""
|
|
180
|
+
|
|
181
|
+
def __init__(self, preserve_tail_turns: int = 6):
|
|
182
|
+
self.preserve_tail_turns = preserve_tail_turns
|
|
183
|
+
|
|
184
|
+
def compact(
|
|
185
|
+
self,
|
|
186
|
+
request: NormalizedRequest,
|
|
187
|
+
budget: ContextBudget,
|
|
188
|
+
) -> Tuple[NormalizedRequest, bool]:
|
|
189
|
+
"""
|
|
190
|
+
Compact request to fit strictly within the target model's available input budget.
|
|
191
|
+
Returns (compacted_request, was_compacted).
|
|
192
|
+
Raises ContextOverflowError when the protected content (system instruction, root
|
|
193
|
+
objective, preserved tail turns) still exceeds the budget after every compaction phase.
|
|
194
|
+
"""
|
|
195
|
+
current_tokens = estimate_tokens(request)
|
|
196
|
+
target_tokens = budget.available_input_budget
|
|
197
|
+
|
|
198
|
+
if current_tokens <= target_tokens:
|
|
199
|
+
return request, False
|
|
200
|
+
|
|
201
|
+
logger.info(
|
|
202
|
+
"Request size (%d tokens) exceeds available budget (%d tokens). Initiating hierarchical compaction.",
|
|
203
|
+
current_tokens, target_tokens
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
compacted = copy.deepcopy(request)
|
|
207
|
+
messages = compacted.messages
|
|
208
|
+
first_user_idx = next((idx for idx, message in enumerate(messages) if message.role == "user"), None)
|
|
209
|
+
protected_prefix_count = (first_user_idx + 1) if first_user_idx is not None else 1
|
|
210
|
+
|
|
211
|
+
# Phase 1: In-place diagnostic compaction of oversized tool results and logs
|
|
212
|
+
for idx, m in enumerate(messages):
|
|
213
|
+
for tr in m.tool_results:
|
|
214
|
+
if len(tr.content) > 400:
|
|
215
|
+
tr.content = extract_diagnostic_summary(tr.content, max_chars=400)
|
|
216
|
+
# Compact message content if it contains huge logs (e.g. OpenCode compiler/test outputs)
|
|
217
|
+
# Do not drop root objective or final instruction entirely; compact huge diagnostic bodies
|
|
218
|
+
is_root_user = (idx == first_user_idx)
|
|
219
|
+
is_final_msg = (idx == len(messages) - 1)
|
|
220
|
+
if m.content and len(m.content) > 1000:
|
|
221
|
+
if not is_root_user and not is_final_msg:
|
|
222
|
+
m.content = extract_diagnostic_summary(m.content, max_chars=600)
|
|
223
|
+
elif is_final_msg and not is_root_user:
|
|
224
|
+
# Final instruction has huge attached log; compact the log while keeping tail instructions
|
|
225
|
+
m.content = extract_diagnostic_summary(m.content, max_chars=1200)
|
|
226
|
+
|
|
227
|
+
if estimate_tokens(compacted) <= target_tokens:
|
|
228
|
+
return compacted, True
|
|
229
|
+
|
|
230
|
+
# Phase 2: If message turns exceed tail window, evict intermediate turns
|
|
231
|
+
cutoff_idx = len(messages) - self.preserve_tail_turns
|
|
232
|
+
for idx in range(protected_prefix_count, max(protected_prefix_count, cutoff_idx)):
|
|
233
|
+
m = messages[idx]
|
|
234
|
+
for tr in m.tool_results:
|
|
235
|
+
if len(tr.content) > 300:
|
|
236
|
+
tr.content = extract_diagnostic_summary(tr.content, max_chars=300)
|
|
237
|
+
if m.content and len(m.content) > 600 and m.role == "assistant":
|
|
238
|
+
m.content = (
|
|
239
|
+
m.content[:300]
|
|
240
|
+
+ "\n... [Prior assistant reasoning compacted by Circuit Breaker] ...\n"
|
|
241
|
+
+ m.content[-300:]
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
if estimate_tokens(compacted) <= target_tokens:
|
|
245
|
+
return compacted, True
|
|
246
|
+
|
|
247
|
+
start_evict_idx = protected_prefix_count
|
|
248
|
+
while len(compacted.messages) > (self.preserve_tail_turns + protected_prefix_count):
|
|
249
|
+
if estimate_tokens(compacted) <= target_tokens:
|
|
250
|
+
break
|
|
251
|
+
compacted.messages.pop(start_evict_idx)
|
|
252
|
+
|
|
253
|
+
if estimate_tokens(compacted) <= target_tokens:
|
|
254
|
+
return compacted, True
|
|
255
|
+
|
|
256
|
+
# Phase 3: Aggressive compaction of tail turns if still over budget
|
|
257
|
+
for idx, m in enumerate(compacted.messages):
|
|
258
|
+
if idx == first_user_idx:
|
|
259
|
+
continue
|
|
260
|
+
for tr in m.tool_results:
|
|
261
|
+
if len(tr.content) > 250:
|
|
262
|
+
tr.content = extract_diagnostic_summary(tr.content, max_chars=250)
|
|
263
|
+
if m.content and len(m.content) > 500:
|
|
264
|
+
m.content = extract_diagnostic_summary(m.content, max_chars=400)
|
|
265
|
+
|
|
266
|
+
self._require_fit(compacted, target_tokens)
|
|
267
|
+
return compacted, True
|
|
268
|
+
|
|
269
|
+
@staticmethod
|
|
270
|
+
def _require_fit(request: NormalizedRequest, target_tokens: int) -> None:
|
|
271
|
+
remaining = estimate_tokens(request)
|
|
272
|
+
if remaining > target_tokens:
|
|
273
|
+
raise ContextOverflowError(required_tokens=remaining, available_budget=target_tokens)
|
|
274
|
+
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Explainable Semantic Failover Plan."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
import uuid
|
|
7
|
+
from dataclasses import asdict, dataclass, field
|
|
8
|
+
from typing import Any, Dict, List, Optional
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class FailoverPlan:
|
|
13
|
+
"""
|
|
14
|
+
Explicit, auditable plan created when transferring execution from one provider/model to another.
|
|
15
|
+
Coordinates state preservation, context budgeting, protocol translation, and tool safety.
|
|
16
|
+
"""
|
|
17
|
+
plan_id: str = field(default_factory=lambda: f"fplan_{uuid.uuid4().hex[:8]}")
|
|
18
|
+
request_id: str = ""
|
|
19
|
+
source_endpoint: Optional[str] = None
|
|
20
|
+
target_endpoint: str = ""
|
|
21
|
+
failover_reason: str = ""
|
|
22
|
+
state_snapshot_id: Optional[str] = None
|
|
23
|
+
required_transformations: List[str] = field(default_factory=list)
|
|
24
|
+
context_tokens_before: int = 0
|
|
25
|
+
context_tokens_after: int = 0
|
|
26
|
+
context_compaction_applied: bool = False
|
|
27
|
+
tool_transformations: List[str] = field(default_factory=list)
|
|
28
|
+
risk_flags: List[str] = field(default_factory=list)
|
|
29
|
+
remaining_deadline_ms: float = 0.0
|
|
30
|
+
remaining_cost_budget_usd: Optional[float] = None
|
|
31
|
+
created_at_monotonic: float = field(default_factory=time.monotonic)
|
|
32
|
+
|
|
33
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
34
|
+
return asdict(self)
|