evolvingmachines-evolve 0.0.55.dev1355__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bridge/__init__.py +5 -0
- bridge/dist/bridge.bundle.cjs +1275 -0
- evolve/__init__.py +819 -0
- evolve/_http.py +71 -0
- evolve/agent.py +896 -0
- evolve/bridge.py +509 -0
- evolve/browser_credentials.py +265 -0
- evolve/browser_profiles.py +95 -0
- evolve/config.py +600 -0
- evolve/hosted.py +8958 -0
- evolve/integrations.py +173 -0
- evolve/managed_secrets.py +175 -0
- evolve/pipeline/__init__.py +59 -0
- evolve/pipeline/pipeline.py +512 -0
- evolve/pipeline/types.py +286 -0
- evolve/prompts/__init__.py +132 -0
- evolve/prompts/agent_md/judge.md +30 -0
- evolve/prompts/agent_md/reduce.md +7 -0
- evolve/prompts/agent_md/verify.md +33 -0
- evolve/prompts/user/judge.md +1 -0
- evolve/prompts/user/retry_feedback.md +9 -0
- evolve/prompts/user/verify.md +1 -0
- evolve/py.typed +0 -0
- evolve/results.py +315 -0
- evolve/retry.py +133 -0
- evolve/schema.py +107 -0
- evolve/sessions_client.py +167 -0
- evolve/storage_client.py +178 -0
- evolve/swarm/__init__.py +75 -0
- evolve/swarm/results.py +140 -0
- evolve/swarm/swarm.py +2116 -0
- evolve/swarm/types.py +241 -0
- evolve/utils.py +227 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/METADATA +52 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/RECORD +38 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/WHEEL +5 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/licenses/LICENSE +201 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/top_level.txt +2 -0
evolve/results.py
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""Result types for Evolve SDK."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
import math
|
|
5
|
+
from typing import Any, Dict, List, Literal, Optional, Union
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class CheckpointInfo:
|
|
10
|
+
"""Checkpoint metadata.
|
|
11
|
+
|
|
12
|
+
Matches TypeScript SDK's CheckpointInfo for exact parity.
|
|
13
|
+
Evidence: sdk-ts/src/types.ts lines 613-634
|
|
14
|
+
|
|
15
|
+
Attributes:
|
|
16
|
+
id: Checkpoint ID — pass as `from_checkpoint` to restore
|
|
17
|
+
hash: SHA-256 of tar.gz — integrity verification
|
|
18
|
+
tag: Session tag at checkpoint time — lineage tracking
|
|
19
|
+
timestamp: ISO 8601 timestamp
|
|
20
|
+
size_bytes: Archive size in bytes
|
|
21
|
+
agent_type: Agent type that produced this checkpoint
|
|
22
|
+
model: Model that produced this checkpoint
|
|
23
|
+
workspace_mode: Workspace mode used when checkpoint was created
|
|
24
|
+
parent_id: Parent checkpoint ID — lineage tracking
|
|
25
|
+
comment: User-provided label for this checkpoint
|
|
26
|
+
"""
|
|
27
|
+
id: str
|
|
28
|
+
hash: str
|
|
29
|
+
tag: str
|
|
30
|
+
timestamp: str
|
|
31
|
+
size_bytes: Optional[int] = None
|
|
32
|
+
agent_type: Optional[str] = None
|
|
33
|
+
model: Optional[str] = None
|
|
34
|
+
workspace_mode: Optional[str] = None
|
|
35
|
+
parent_id: Optional[str] = None
|
|
36
|
+
comment: Optional[str] = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class AgentResponse:
|
|
41
|
+
"""Response from agent execution.
|
|
42
|
+
|
|
43
|
+
Matches TypeScript SDK's AgentResponse for exact parity.
|
|
44
|
+
|
|
45
|
+
Attributes:
|
|
46
|
+
sandbox_id: Sandbox ID
|
|
47
|
+
session_id: Dashboard session ID for trace/replay APIs, when known
|
|
48
|
+
browser: Managed browser runtime info, when a remote browser is configured
|
|
49
|
+
run_id: Run ID for spend/cost attribution (present for run(), None for execute_command())
|
|
50
|
+
exit_code: Command exit code
|
|
51
|
+
stdout: Standard output
|
|
52
|
+
stderr: Standard error
|
|
53
|
+
checkpoint: Checkpoint info if storage configured and run succeeded
|
|
54
|
+
"""
|
|
55
|
+
sandbox_id: str
|
|
56
|
+
exit_code: int
|
|
57
|
+
stdout: str
|
|
58
|
+
stderr: str
|
|
59
|
+
session_id: Optional[str] = None
|
|
60
|
+
browser: Optional[Dict[str, str]] = None
|
|
61
|
+
run_id: Optional[str] = None
|
|
62
|
+
checkpoint: Optional[CheckpointInfo] = None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# Backward compatibility alias
|
|
66
|
+
ExecuteResult = AgentResponse
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class SessionStatus:
|
|
71
|
+
"""Runtime status snapshot for sandbox and agent."""
|
|
72
|
+
sandbox_id: Optional[str]
|
|
73
|
+
sandbox: str
|
|
74
|
+
agent: str
|
|
75
|
+
active_process_id: Optional[str]
|
|
76
|
+
has_run: bool
|
|
77
|
+
timestamp: str
|
|
78
|
+
browser: Optional[Dict[str, str]] = None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass
|
|
82
|
+
class OutputResult:
|
|
83
|
+
"""Result from get_output_files() with optional schema validation.
|
|
84
|
+
|
|
85
|
+
Matches TypeScript SDK's OutputResult<T> for exact parity.
|
|
86
|
+
Evidence: sdk-ts/src/types.ts lines 258-268
|
|
87
|
+
|
|
88
|
+
Attributes:
|
|
89
|
+
files: Output files from output/ folder
|
|
90
|
+
data: Parsed and validated result.json data (None if no schema or validation failed)
|
|
91
|
+
error: Validation or parse error message, if any
|
|
92
|
+
raw_data: Raw result.json string when parse or validation failed (for debugging)
|
|
93
|
+
"""
|
|
94
|
+
files: Dict[str, Union[str, bytes]] = field(default_factory=dict)
|
|
95
|
+
data: Optional[Any] = None
|
|
96
|
+
error: Optional[str] = None
|
|
97
|
+
raw_data: Optional[str] = None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass
|
|
101
|
+
class RunCost:
|
|
102
|
+
"""Cost breakdown for a single run() invocation.
|
|
103
|
+
|
|
104
|
+
Matches TypeScript SDK's RunCost for exact parity.
|
|
105
|
+
|
|
106
|
+
Attributes:
|
|
107
|
+
run_id: Run ID matching AgentResponse.run_id
|
|
108
|
+
index: 1-based chronological position in session
|
|
109
|
+
cost: Total cost in USD as billed to your Evolve account
|
|
110
|
+
tokens: Token counts {'prompt': N, 'completion': N, 'cached': N}.
|
|
111
|
+
``prompt`` INCLUDES the cached share; ``cached`` is that share,
|
|
112
|
+
absent on servers predating the field (never 0-by-default).
|
|
113
|
+
model: Model used (e.g., 'claude-opus-4-8')
|
|
114
|
+
requests: Number of LLM API requests in this run
|
|
115
|
+
as_of: ISO timestamp when this data was fetched
|
|
116
|
+
is_complete: False if recent LLM calls may still be batching (~60s delay)
|
|
117
|
+
truncated: True if spend log pagination was capped
|
|
118
|
+
"""
|
|
119
|
+
run_id: str
|
|
120
|
+
index: int
|
|
121
|
+
cost: float
|
|
122
|
+
tokens: Dict[str, int]
|
|
123
|
+
model: str
|
|
124
|
+
requests: int
|
|
125
|
+
as_of: str
|
|
126
|
+
is_complete: bool
|
|
127
|
+
truncated: bool
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass
|
|
131
|
+
class SessionCost:
|
|
132
|
+
"""Cost breakdown for an entire agent session (all runs).
|
|
133
|
+
|
|
134
|
+
Matches TypeScript SDK's SessionCost for exact parity.
|
|
135
|
+
|
|
136
|
+
Attributes:
|
|
137
|
+
session_tag: Session tag matching get_session_tag()
|
|
138
|
+
total_cost: Total cost across all runs in USD
|
|
139
|
+
total_tokens: Aggregate token counts
|
|
140
|
+
{'prompt': N, 'completion': N, 'cached': N}. ``prompt`` INCLUDES
|
|
141
|
+
the cached share; ``cached`` is absent on servers predating it.
|
|
142
|
+
runs: Per-run breakdown, chronological order
|
|
143
|
+
as_of: ISO timestamp when this data was fetched
|
|
144
|
+
is_complete: False if session is still active or recently ended
|
|
145
|
+
truncated: True if spend log pagination was capped
|
|
146
|
+
"""
|
|
147
|
+
session_tag: str
|
|
148
|
+
total_cost: float
|
|
149
|
+
total_tokens: Dict[str, int]
|
|
150
|
+
runs: List[RunCost]
|
|
151
|
+
as_of: str
|
|
152
|
+
is_complete: bool
|
|
153
|
+
truncated: bool
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class UsageReading:
|
|
158
|
+
"""THE ONE-HOME USAGE READING — "what has this run's meter said so far".
|
|
159
|
+
|
|
160
|
+
Money and tokens come from the SAME gateway spend-log records, so the two
|
|
161
|
+
can never describe different sets of requests. Served under the one key
|
|
162
|
+
``usage``, with these exact keys, by the trial surfaces and the
|
|
163
|
+
managed-agents session surfaces alike — a renderer that reads one reads
|
|
164
|
+
the other unchanged. While the run is alive the platform's own poll
|
|
165
|
+
raises the numbers, so a polling reader sees them tick; once settled, the
|
|
166
|
+
settled figures replace the live ones under the same keys. The whole
|
|
167
|
+
object is None when the meter has never answered — never a fabricated
|
|
168
|
+
zero.
|
|
169
|
+
|
|
170
|
+
Attributes:
|
|
171
|
+
provisional: True while every number is a LOWER BOUND that can still
|
|
172
|
+
grow — the run is alive, or its settled lane is not yet
|
|
173
|
+
confirmed. False = settled; the reading will not move again.
|
|
174
|
+
spent_usd: Metered model spend so far, USD. None = the money was
|
|
175
|
+
never measured (a trial's ``spend_source`` lane ``assumed_cap``;
|
|
176
|
+
the token fields beside it may still carry real readings).
|
|
177
|
+
input_tokens: Prompt tokens so far, INCLUDING the cached share and
|
|
178
|
+
the cache-write share.
|
|
179
|
+
cached_input_tokens: The cached share of ``input_tokens`` (read from
|
|
180
|
+
the provider's prompt cache).
|
|
181
|
+
cache_write_tokens: The share of ``input_tokens`` WRITTEN to the
|
|
182
|
+
provider's prompt cache. Anthropic bills it at a premium above the
|
|
183
|
+
plain input price, so it is the fourth count ``spent_usd`` needs
|
|
184
|
+
to be reproducible from the tokens; providers without a
|
|
185
|
+
cache-write price report 0. None when the meter never answered —
|
|
186
|
+
and on a run settled before the platform recorded this share (an
|
|
187
|
+
older server omits the key), where the three counts beside it stay
|
|
188
|
+
real: None is never a fabricated 0.
|
|
189
|
+
output_tokens: Completion tokens so far.
|
|
190
|
+
as_of: When this reading was taken — show its age, never the figure
|
|
191
|
+
alone.
|
|
192
|
+
"""
|
|
193
|
+
provisional: bool
|
|
194
|
+
spent_usd: Optional[float]
|
|
195
|
+
input_tokens: Optional[int]
|
|
196
|
+
cached_input_tokens: Optional[int]
|
|
197
|
+
cache_write_tokens: Optional[int]
|
|
198
|
+
output_tokens: Optional[int]
|
|
199
|
+
as_of: Optional[str]
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _usage_reading_from_data(data: Any) -> Optional[UsageReading]:
|
|
203
|
+
"""The wire's usage reading, defensively — the same one rule as the
|
|
204
|
+
TypeScript SDK's ``mapUsageReading``: anything malformed answers None
|
|
205
|
+
(which already means "the meter never answered"), each numeric field is
|
|
206
|
+
taken only as a real number, and ``provisional`` must be a real bool —
|
|
207
|
+
without it the reading has no statement to make."""
|
|
208
|
+
if not isinstance(data, dict):
|
|
209
|
+
return None
|
|
210
|
+
provisional = data.get('provisional')
|
|
211
|
+
if not isinstance(provisional, bool):
|
|
212
|
+
return None
|
|
213
|
+
|
|
214
|
+
def _num(value: Any) -> Optional[float]:
|
|
215
|
+
# bool is an int subclass; a stray True must never become money. And a
|
|
216
|
+
# non-finite float (json.loads admits NaN/Infinity) is not a reading —
|
|
217
|
+
# refuse it like the TS side's Number.isFinite does.
|
|
218
|
+
if not isinstance(value, (int, float)) or isinstance(value, bool):
|
|
219
|
+
return None
|
|
220
|
+
return value if math.isfinite(value) else None
|
|
221
|
+
|
|
222
|
+
as_of = data.get('as_of')
|
|
223
|
+
return UsageReading(
|
|
224
|
+
provisional=provisional,
|
|
225
|
+
spent_usd=_num(data.get('spent_usd')),
|
|
226
|
+
input_tokens=_num(data.get('input_tokens')),
|
|
227
|
+
cached_input_tokens=_num(data.get('cached_input_tokens')),
|
|
228
|
+
cache_write_tokens=_num(data.get('cache_write_tokens')),
|
|
229
|
+
output_tokens=_num(data.get('output_tokens')),
|
|
230
|
+
as_of=as_of if isinstance(as_of, str) else None,
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
SessionEvent = Dict[str, Any]
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
@dataclass
|
|
238
|
+
class SessionInfo:
|
|
239
|
+
"""Historical session metadata from the standalone sessions() client.
|
|
240
|
+
|
|
241
|
+
Matches the TypeScript sessions() surface, with snake_case field names for
|
|
242
|
+
Python transport ergonomics.
|
|
243
|
+
"""
|
|
244
|
+
id: str
|
|
245
|
+
tag: str
|
|
246
|
+
agent: str
|
|
247
|
+
model: Optional[str]
|
|
248
|
+
provider: str
|
|
249
|
+
sandbox_id: Optional[str]
|
|
250
|
+
state: Literal['live', 'ended']
|
|
251
|
+
runtime_status: Literal['alive', 'dead', 'unknown']
|
|
252
|
+
cost: Optional[float]
|
|
253
|
+
created_at: str
|
|
254
|
+
ended_at: Optional[str]
|
|
255
|
+
step_count: int
|
|
256
|
+
tool_stats: Optional[Dict[str, int]]
|
|
257
|
+
#: The one-home usage reading — the SAME object, same keys, a trial
|
|
258
|
+
#: serves (see :class:`UsageReading`). None = the meter never answered
|
|
259
|
+
#: (and on servers predating the field).
|
|
260
|
+
usage: Optional[UsageReading] = None
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
@dataclass
|
|
264
|
+
class SessionPage:
|
|
265
|
+
"""Paginated session list response from the standalone sessions() client."""
|
|
266
|
+
items: List[SessionInfo]
|
|
267
|
+
next_cursor: Optional[str]
|
|
268
|
+
has_more: bool
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
@dataclass
|
|
272
|
+
class SessionTranscript:
|
|
273
|
+
"""One read of a session's transcript feed — the TypeScript SDK's
|
|
274
|
+
``SessionTranscript``: the parsed events after ``since`` plus the facts
|
|
275
|
+
the feed serves around them. There is no server-side paging: one read
|
|
276
|
+
answers everything after ``since``, and ``total`` counts ALL stored
|
|
277
|
+
events, so the next delta read passes ``since=total``.
|
|
278
|
+
"""
|
|
279
|
+
#: The session as the feed served it — the same shape ``get()`` returns,
|
|
280
|
+
#: so ``usage`` / ``cost`` is the run's total.
|
|
281
|
+
session: SessionInfo
|
|
282
|
+
#: The events after ``since``: what ``events()`` returns alone.
|
|
283
|
+
events: List[SessionEvent]
|
|
284
|
+
#: ALL stored events, independent of ``since``.
|
|
285
|
+
total: int
|
|
286
|
+
#: THE GATEWAY METER's per-call lines for this session (the spec's
|
|
287
|
+
#: GatewayUsageEvent, its own camelCase keys — ``call['update']['usage']``
|
|
288
|
+
#: carries ``promptTokens``, ``completionTokens``, ``cachedTokens`` and
|
|
289
|
+
#: ``costUsd``), in time order: one model call as the LiteLLM gateway
|
|
290
|
+
#: priced it, the same line a trial's trace carries in its gateway band.
|
|
291
|
+
#: Served whole on every read and beside ``events``, never inside them: a
|
|
292
|
+
#: session's ``since`` is an event COUNT, so a call line in the list would
|
|
293
|
+
#: corrupt every delta poller's cursor. The ONLY per-call tokens and money
|
|
294
|
+
#: a client may show (a harness's own ``usage`` line stays a raw record).
|
|
295
|
+
gateway_calls: List[Dict[str, Any]]
|
|
296
|
+
#: The server's write instant of each event's row, one per entry of
|
|
297
|
+
#: ``events``, index-aligned (the contract's ``SessionTranscript.storedAt``):
|
|
298
|
+
#: present on every row-served page (an empty page carries an empty list),
|
|
299
|
+
#: ``None`` when the transcript was served from its file, where no write
|
|
300
|
+
#: instant exists. It places the gateway meter's calls under the harness's
|
|
301
|
+
#: steps for harnesses whose lines carry no clock of their own (codex,
|
|
302
|
+
#: kimi, qwen); a reader that does not place calls needs nothing from it.
|
|
303
|
+
stored_at: Optional[List[str]] = None
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
@dataclass
|
|
307
|
+
class BrowserReplay:
|
|
308
|
+
"""Browser replay metadata and Dashboard-owned access URLs."""
|
|
309
|
+
session_id: str
|
|
310
|
+
status: Literal['ready']
|
|
311
|
+
replay_url: str
|
|
312
|
+
download_url: str
|
|
313
|
+
suggested_start_seconds: Optional[float] = None
|
|
314
|
+
size_bytes: Optional[int] = None
|
|
315
|
+
ready_at: Optional[str] = None
|
evolve/retry.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Retry Utility for Swarm operations.
|
|
2
|
+
|
|
3
|
+
Generic retry with exponential backoff.
|
|
4
|
+
Works with any result type that has a status field.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import asyncio
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Any, Awaitable, Callable, Optional, TypeVar
|
|
10
|
+
|
|
11
|
+
# =============================================================================
|
|
12
|
+
# CONSTANTS
|
|
13
|
+
# =============================================================================
|
|
14
|
+
|
|
15
|
+
DEFAULT_MAX_ATTEMPTS = 3
|
|
16
|
+
DEFAULT_BACKOFF_MS = 1000
|
|
17
|
+
DEFAULT_BACKOFF_MULTIPLIER = 2.0
|
|
18
|
+
|
|
19
|
+
# =============================================================================
|
|
20
|
+
# TYPES
|
|
21
|
+
# =============================================================================
|
|
22
|
+
|
|
23
|
+
# TypeVar for result types (SwarmResult, ReduceResult, etc.)
|
|
24
|
+
# Results must have a `status` field for default retry behavior.
|
|
25
|
+
TResult = TypeVar('TResult')
|
|
26
|
+
|
|
27
|
+
# Callback type for item retry events (must be defined before RetryConfig)
|
|
28
|
+
OnItemRetryCallback = Callable[[int, int, str], None] # (item_index, attempt, error)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _get_field(obj: Any, field: str, default: Any = None) -> Any:
|
|
32
|
+
"""Get field from dict or object (duck typing helper)."""
|
|
33
|
+
if isinstance(obj, dict):
|
|
34
|
+
return obj.get(field, default)
|
|
35
|
+
return getattr(obj, field, default)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class RetryConfig:
|
|
40
|
+
"""Per-item retry configuration.
|
|
41
|
+
|
|
42
|
+
Example:
|
|
43
|
+
# Basic retry on error
|
|
44
|
+
RetryConfig(max_attempts=3)
|
|
45
|
+
|
|
46
|
+
# With exponential backoff
|
|
47
|
+
RetryConfig(max_attempts=3, backoff_ms=1000, backoff_multiplier=2)
|
|
48
|
+
|
|
49
|
+
# Custom retry condition
|
|
50
|
+
RetryConfig(max_attempts=3, retry_on=lambda r: r.status == "error" or "timeout" in (r.error or ""))
|
|
51
|
+
|
|
52
|
+
# With callback
|
|
53
|
+
RetryConfig(max_attempts=3, on_item_retry=lambda i, a, e: print(f"Item {i} retry {a}: {e}"))
|
|
54
|
+
|
|
55
|
+
Args:
|
|
56
|
+
max_attempts: Maximum retry attempts (default: 3)
|
|
57
|
+
backoff_ms: Initial backoff in ms (default: 1000)
|
|
58
|
+
backoff_multiplier: Exponential backoff multiplier (default: 2)
|
|
59
|
+
retry_on: Custom retry condition (default: status == "error")
|
|
60
|
+
on_item_retry: Callback when retry occurs (item_index, attempt, error)
|
|
61
|
+
"""
|
|
62
|
+
max_attempts: int = DEFAULT_MAX_ATTEMPTS
|
|
63
|
+
backoff_ms: int = DEFAULT_BACKOFF_MS
|
|
64
|
+
backoff_multiplier: float = DEFAULT_BACKOFF_MULTIPLIER
|
|
65
|
+
retry_on: Optional[Callable[[Any], bool]] = None
|
|
66
|
+
on_item_retry: Optional[OnItemRetryCallback] = None
|
|
67
|
+
|
|
68
|
+
def should_retry(self, result: Any) -> bool:
|
|
69
|
+
"""Check if result should be retried."""
|
|
70
|
+
if self.retry_on is not None:
|
|
71
|
+
return self.retry_on(result)
|
|
72
|
+
# Default: retry on error status
|
|
73
|
+
return _get_field(result, 'status') == "error"
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# =============================================================================
|
|
77
|
+
# RETRY LOGIC
|
|
78
|
+
# =============================================================================
|
|
79
|
+
|
|
80
|
+
async def execute_with_retry(
|
|
81
|
+
fn: Callable[[int], Awaitable[TResult]],
|
|
82
|
+
config: Optional[RetryConfig] = None,
|
|
83
|
+
item_index: int = 0,
|
|
84
|
+
) -> TResult:
|
|
85
|
+
"""Execute a function with retry and exponential backoff.
|
|
86
|
+
|
|
87
|
+
Works with any result type that has a `status` field (SwarmResult, ReduceResult, etc.).
|
|
88
|
+
|
|
89
|
+
Args:
|
|
90
|
+
fn: Async function that receives attempt number (1-based) and returns a result
|
|
91
|
+
config: Retry configuration (optional, uses defaults if not provided)
|
|
92
|
+
item_index: Item index for callback (default: 0)
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
Result from the function
|
|
96
|
+
|
|
97
|
+
Example:
|
|
98
|
+
result = await execute_with_retry(
|
|
99
|
+
lambda attempt: self._execute_map_item(item, prompt, index, operation_id, params, timeout, attempt),
|
|
100
|
+
RetryConfig(max_attempts=3, backoff_ms=1000),
|
|
101
|
+
item_index=index,
|
|
102
|
+
)
|
|
103
|
+
"""
|
|
104
|
+
resolved = config or RetryConfig()
|
|
105
|
+
|
|
106
|
+
last_result: Optional[TResult] = None
|
|
107
|
+
attempts = 0
|
|
108
|
+
backoff = resolved.backoff_ms
|
|
109
|
+
|
|
110
|
+
while attempts < resolved.max_attempts:
|
|
111
|
+
attempts += 1
|
|
112
|
+
last_result = await fn(attempts)
|
|
113
|
+
|
|
114
|
+
# Check if we should retry
|
|
115
|
+
if not resolved.should_retry(last_result):
|
|
116
|
+
return last_result
|
|
117
|
+
|
|
118
|
+
# Don't retry if we've exhausted attempts
|
|
119
|
+
if attempts >= resolved.max_attempts:
|
|
120
|
+
break
|
|
121
|
+
|
|
122
|
+
# Notify of retry via callback in config
|
|
123
|
+
if resolved.on_item_retry is not None:
|
|
124
|
+
error = _get_field(last_result, 'error') or "Unknown error"
|
|
125
|
+
resolved.on_item_retry(item_index, attempts, error)
|
|
126
|
+
|
|
127
|
+
# Wait before retrying (convert ms to seconds)
|
|
128
|
+
await asyncio.sleep(backoff / 1000)
|
|
129
|
+
backoff = backoff * resolved.backoff_multiplier
|
|
130
|
+
|
|
131
|
+
# Return last result
|
|
132
|
+
assert last_result is not None
|
|
133
|
+
return last_result
|
evolve/schema.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Schema utilities for Pydantic models, dataclasses, and JSON Schema.
|
|
2
|
+
|
|
3
|
+
Provides unified detection, conversion, and validation for schema types.
|
|
4
|
+
Uses Pydantic's TypeAdapter for dataclass support.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import dataclasses
|
|
8
|
+
from typing import Any, Dict, Optional
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
# =============================================================================
|
|
12
|
+
# DETECTION
|
|
13
|
+
# =============================================================================
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def is_pydantic_model(obj: Any) -> bool:
|
|
17
|
+
"""Check if object is a Pydantic model class."""
|
|
18
|
+
try:
|
|
19
|
+
from pydantic import BaseModel
|
|
20
|
+
return isinstance(obj, type) and issubclass(obj, BaseModel)
|
|
21
|
+
except ImportError:
|
|
22
|
+
return False
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def is_dataclass(obj: Any) -> bool:
|
|
26
|
+
"""Check if object is a dataclass."""
|
|
27
|
+
return dataclasses.is_dataclass(obj) and isinstance(obj, type)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def is_json_schema(obj: Any) -> bool:
|
|
31
|
+
"""Check if object is a JSON Schema dict."""
|
|
32
|
+
return isinstance(obj, dict)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# =============================================================================
|
|
36
|
+
# CONVERSION
|
|
37
|
+
# =============================================================================
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def to_json_schema(schema: Any) -> Optional[Dict[str, Any]]:
|
|
41
|
+
"""Convert a schema to JSON Schema format.
|
|
42
|
+
|
|
43
|
+
Supports:
|
|
44
|
+
- Dict (JSON Schema) - passed through
|
|
45
|
+
- Pydantic models - uses model_json_schema()
|
|
46
|
+
- Dataclasses - uses Pydantic TypeAdapter
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
schema: Pydantic model, dataclass, or JSON Schema dict
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
JSON Schema dict, or None if schema is None
|
|
53
|
+
"""
|
|
54
|
+
if schema is None:
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
if is_json_schema(schema):
|
|
58
|
+
return schema
|
|
59
|
+
|
|
60
|
+
if is_pydantic_model(schema):
|
|
61
|
+
return schema.model_json_schema()
|
|
62
|
+
|
|
63
|
+
if is_dataclass(schema):
|
|
64
|
+
from pydantic import TypeAdapter
|
|
65
|
+
return TypeAdapter(schema).json_schema()
|
|
66
|
+
|
|
67
|
+
raise TypeError(
|
|
68
|
+
f"Schema must be a Pydantic model, dataclass, or dict. "
|
|
69
|
+
f"Got {type(schema).__name__}"
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# =============================================================================
|
|
74
|
+
# VALIDATION
|
|
75
|
+
# =============================================================================
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def validate_and_parse(
|
|
79
|
+
raw_json: str,
|
|
80
|
+
schema: Any,
|
|
81
|
+
strict: bool = False,
|
|
82
|
+
) -> Any:
|
|
83
|
+
"""Validate JSON string and parse into schema type.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
raw_json: Raw JSON string to validate
|
|
87
|
+
schema: Pydantic model or dataclass (returns instance)
|
|
88
|
+
strict: Use strict validation mode
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
Model/dataclass instance, or None for dict schemas
|
|
92
|
+
|
|
93
|
+
Raises:
|
|
94
|
+
ValidationError: If validation fails
|
|
95
|
+
"""
|
|
96
|
+
if schema is None or is_json_schema(schema):
|
|
97
|
+
# JSON Schema validation is handled by TS SDK
|
|
98
|
+
return None
|
|
99
|
+
|
|
100
|
+
if is_pydantic_model(schema):
|
|
101
|
+
return schema.model_validate_json(raw_json, strict=strict)
|
|
102
|
+
|
|
103
|
+
if is_dataclass(schema):
|
|
104
|
+
from pydantic import TypeAdapter
|
|
105
|
+
return TypeAdapter(schema).validate_json(raw_json, strict=strict)
|
|
106
|
+
|
|
107
|
+
return None
|