evolvingmachines-evolve 0.0.55.dev1355__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evolve/results.py ADDED
@@ -0,0 +1,315 @@
1
+ """Result types for Evolve SDK."""
2
+
3
+ from dataclasses import dataclass, field
4
+ import math
5
+ from typing import Any, Dict, List, Literal, Optional, Union
6
+
7
+
8
+ @dataclass
9
+ class CheckpointInfo:
10
+ """Checkpoint metadata.
11
+
12
+ Matches TypeScript SDK's CheckpointInfo for exact parity.
13
+ Evidence: sdk-ts/src/types.ts lines 613-634
14
+
15
+ Attributes:
16
+ id: Checkpoint ID — pass as `from_checkpoint` to restore
17
+ hash: SHA-256 of tar.gz — integrity verification
18
+ tag: Session tag at checkpoint time — lineage tracking
19
+ timestamp: ISO 8601 timestamp
20
+ size_bytes: Archive size in bytes
21
+ agent_type: Agent type that produced this checkpoint
22
+ model: Model that produced this checkpoint
23
+ workspace_mode: Workspace mode used when checkpoint was created
24
+ parent_id: Parent checkpoint ID — lineage tracking
25
+ comment: User-provided label for this checkpoint
26
+ """
27
+ id: str
28
+ hash: str
29
+ tag: str
30
+ timestamp: str
31
+ size_bytes: Optional[int] = None
32
+ agent_type: Optional[str] = None
33
+ model: Optional[str] = None
34
+ workspace_mode: Optional[str] = None
35
+ parent_id: Optional[str] = None
36
+ comment: Optional[str] = None
37
+
38
+
39
+ @dataclass
40
+ class AgentResponse:
41
+ """Response from agent execution.
42
+
43
+ Matches TypeScript SDK's AgentResponse for exact parity.
44
+
45
+ Attributes:
46
+ sandbox_id: Sandbox ID
47
+ session_id: Dashboard session ID for trace/replay APIs, when known
48
+ browser: Managed browser runtime info, when a remote browser is configured
49
+ run_id: Run ID for spend/cost attribution (present for run(), None for execute_command())
50
+ exit_code: Command exit code
51
+ stdout: Standard output
52
+ stderr: Standard error
53
+ checkpoint: Checkpoint info if storage configured and run succeeded
54
+ """
55
+ sandbox_id: str
56
+ exit_code: int
57
+ stdout: str
58
+ stderr: str
59
+ session_id: Optional[str] = None
60
+ browser: Optional[Dict[str, str]] = None
61
+ run_id: Optional[str] = None
62
+ checkpoint: Optional[CheckpointInfo] = None
63
+
64
+
65
+ # Backward compatibility alias
66
+ ExecuteResult = AgentResponse
67
+
68
+
69
+ @dataclass
70
+ class SessionStatus:
71
+ """Runtime status snapshot for sandbox and agent."""
72
+ sandbox_id: Optional[str]
73
+ sandbox: str
74
+ agent: str
75
+ active_process_id: Optional[str]
76
+ has_run: bool
77
+ timestamp: str
78
+ browser: Optional[Dict[str, str]] = None
79
+
80
+
81
+ @dataclass
82
+ class OutputResult:
83
+ """Result from get_output_files() with optional schema validation.
84
+
85
+ Matches TypeScript SDK's OutputResult<T> for exact parity.
86
+ Evidence: sdk-ts/src/types.ts lines 258-268
87
+
88
+ Attributes:
89
+ files: Output files from output/ folder
90
+ data: Parsed and validated result.json data (None if no schema or validation failed)
91
+ error: Validation or parse error message, if any
92
+ raw_data: Raw result.json string when parse or validation failed (for debugging)
93
+ """
94
+ files: Dict[str, Union[str, bytes]] = field(default_factory=dict)
95
+ data: Optional[Any] = None
96
+ error: Optional[str] = None
97
+ raw_data: Optional[str] = None
98
+
99
+
100
+ @dataclass
101
+ class RunCost:
102
+ """Cost breakdown for a single run() invocation.
103
+
104
+ Matches TypeScript SDK's RunCost for exact parity.
105
+
106
+ Attributes:
107
+ run_id: Run ID matching AgentResponse.run_id
108
+ index: 1-based chronological position in session
109
+ cost: Total cost in USD as billed to your Evolve account
110
+ tokens: Token counts {'prompt': N, 'completion': N, 'cached': N}.
111
+ ``prompt`` INCLUDES the cached share; ``cached`` is that share,
112
+ absent on servers predating the field (never 0-by-default).
113
+ model: Model used (e.g., 'claude-opus-4-8')
114
+ requests: Number of LLM API requests in this run
115
+ as_of: ISO timestamp when this data was fetched
116
+ is_complete: False if recent LLM calls may still be batching (~60s delay)
117
+ truncated: True if spend log pagination was capped
118
+ """
119
+ run_id: str
120
+ index: int
121
+ cost: float
122
+ tokens: Dict[str, int]
123
+ model: str
124
+ requests: int
125
+ as_of: str
126
+ is_complete: bool
127
+ truncated: bool
128
+
129
+
130
+ @dataclass
131
+ class SessionCost:
132
+ """Cost breakdown for an entire agent session (all runs).
133
+
134
+ Matches TypeScript SDK's SessionCost for exact parity.
135
+
136
+ Attributes:
137
+ session_tag: Session tag matching get_session_tag()
138
+ total_cost: Total cost across all runs in USD
139
+ total_tokens: Aggregate token counts
140
+ {'prompt': N, 'completion': N, 'cached': N}. ``prompt`` INCLUDES
141
+ the cached share; ``cached`` is absent on servers predating it.
142
+ runs: Per-run breakdown, chronological order
143
+ as_of: ISO timestamp when this data was fetched
144
+ is_complete: False if session is still active or recently ended
145
+ truncated: True if spend log pagination was capped
146
+ """
147
+ session_tag: str
148
+ total_cost: float
149
+ total_tokens: Dict[str, int]
150
+ runs: List[RunCost]
151
+ as_of: str
152
+ is_complete: bool
153
+ truncated: bool
154
+
155
+
156
+ @dataclass
157
+ class UsageReading:
158
+ """THE ONE-HOME USAGE READING — "what has this run's meter said so far".
159
+
160
+ Money and tokens come from the SAME gateway spend-log records, so the two
161
+ can never describe different sets of requests. Served under the one key
162
+ ``usage``, with these exact keys, by the trial surfaces and the
163
+ managed-agents session surfaces alike — a renderer that reads one reads
164
+ the other unchanged. While the run is alive the platform's own poll
165
+ raises the numbers, so a polling reader sees them tick; once settled, the
166
+ settled figures replace the live ones under the same keys. The whole
167
+ object is None when the meter has never answered — never a fabricated
168
+ zero.
169
+
170
+ Attributes:
171
+ provisional: True while every number is a LOWER BOUND that can still
172
+ grow — the run is alive, or its settled lane is not yet
173
+ confirmed. False = settled; the reading will not move again.
174
+ spent_usd: Metered model spend so far, USD. None = the money was
175
+ never measured (a trial's ``spend_source`` lane ``assumed_cap``;
176
+ the token fields beside it may still carry real readings).
177
+ input_tokens: Prompt tokens so far, INCLUDING the cached share and
178
+ the cache-write share.
179
+ cached_input_tokens: The cached share of ``input_tokens`` (read from
180
+ the provider's prompt cache).
181
+ cache_write_tokens: The share of ``input_tokens`` WRITTEN to the
182
+ provider's prompt cache. Anthropic bills it at a premium above the
183
+ plain input price, so it is the fourth count ``spent_usd`` needs
184
+ to be reproducible from the tokens; providers without a
185
+ cache-write price report 0. None when the meter never answered —
186
+ and on a run settled before the platform recorded this share (an
187
+ older server omits the key), where the three counts beside it stay
188
+ real: None is never a fabricated 0.
189
+ output_tokens: Completion tokens so far.
190
+ as_of: When this reading was taken — show its age, never the figure
191
+ alone.
192
+ """
193
+ provisional: bool
194
+ spent_usd: Optional[float]
195
+ input_tokens: Optional[int]
196
+ cached_input_tokens: Optional[int]
197
+ cache_write_tokens: Optional[int]
198
+ output_tokens: Optional[int]
199
+ as_of: Optional[str]
200
+
201
+
202
+ def _usage_reading_from_data(data: Any) -> Optional[UsageReading]:
203
+ """The wire's usage reading, defensively — the same one rule as the
204
+ TypeScript SDK's ``mapUsageReading``: anything malformed answers None
205
+ (which already means "the meter never answered"), each numeric field is
206
+ taken only as a real number, and ``provisional`` must be a real bool —
207
+ without it the reading has no statement to make."""
208
+ if not isinstance(data, dict):
209
+ return None
210
+ provisional = data.get('provisional')
211
+ if not isinstance(provisional, bool):
212
+ return None
213
+
214
+ def _num(value: Any) -> Optional[float]:
215
+ # bool is an int subclass; a stray True must never become money. And a
216
+ # non-finite float (json.loads admits NaN/Infinity) is not a reading —
217
+ # refuse it like the TS side's Number.isFinite does.
218
+ if not isinstance(value, (int, float)) or isinstance(value, bool):
219
+ return None
220
+ return value if math.isfinite(value) else None
221
+
222
+ as_of = data.get('as_of')
223
+ return UsageReading(
224
+ provisional=provisional,
225
+ spent_usd=_num(data.get('spent_usd')),
226
+ input_tokens=_num(data.get('input_tokens')),
227
+ cached_input_tokens=_num(data.get('cached_input_tokens')),
228
+ cache_write_tokens=_num(data.get('cache_write_tokens')),
229
+ output_tokens=_num(data.get('output_tokens')),
230
+ as_of=as_of if isinstance(as_of, str) else None,
231
+ )
232
+
233
+
234
+ SessionEvent = Dict[str, Any]
235
+
236
+
237
+ @dataclass
238
+ class SessionInfo:
239
+ """Historical session metadata from the standalone sessions() client.
240
+
241
+ Matches the TypeScript sessions() surface, with snake_case field names for
242
+ Python transport ergonomics.
243
+ """
244
+ id: str
245
+ tag: str
246
+ agent: str
247
+ model: Optional[str]
248
+ provider: str
249
+ sandbox_id: Optional[str]
250
+ state: Literal['live', 'ended']
251
+ runtime_status: Literal['alive', 'dead', 'unknown']
252
+ cost: Optional[float]
253
+ created_at: str
254
+ ended_at: Optional[str]
255
+ step_count: int
256
+ tool_stats: Optional[Dict[str, int]]
257
+ #: The one-home usage reading — the SAME object, same keys, a trial
258
+ #: serves (see :class:`UsageReading`). None = the meter never answered
259
+ #: (and on servers predating the field).
260
+ usage: Optional[UsageReading] = None
261
+
262
+
263
+ @dataclass
264
+ class SessionPage:
265
+ """Paginated session list response from the standalone sessions() client."""
266
+ items: List[SessionInfo]
267
+ next_cursor: Optional[str]
268
+ has_more: bool
269
+
270
+
271
+ @dataclass
272
+ class SessionTranscript:
273
+ """One read of a session's transcript feed — the TypeScript SDK's
274
+ ``SessionTranscript``: the parsed events after ``since`` plus the facts
275
+ the feed serves around them. There is no server-side paging: one read
276
+ answers everything after ``since``, and ``total`` counts ALL stored
277
+ events, so the next delta read passes ``since=total``.
278
+ """
279
+ #: The session as the feed served it — the same shape ``get()`` returns,
280
+ #: so ``usage`` / ``cost`` is the run's total.
281
+ session: SessionInfo
282
+ #: The events after ``since``: what ``events()`` returns alone.
283
+ events: List[SessionEvent]
284
+ #: ALL stored events, independent of ``since``.
285
+ total: int
286
+ #: THE GATEWAY METER's per-call lines for this session (the spec's
287
+ #: GatewayUsageEvent, its own camelCase keys — ``call['update']['usage']``
288
+ #: carries ``promptTokens``, ``completionTokens``, ``cachedTokens`` and
289
+ #: ``costUsd``), in time order: one model call as the LiteLLM gateway
290
+ #: priced it, the same line a trial's trace carries in its gateway band.
291
+ #: Served whole on every read and beside ``events``, never inside them: a
292
+ #: session's ``since`` is an event COUNT, so a call line in the list would
293
+ #: corrupt every delta poller's cursor. The ONLY per-call tokens and money
294
+ #: a client may show (a harness's own ``usage`` line stays a raw record).
295
+ gateway_calls: List[Dict[str, Any]]
296
+ #: The server's write instant of each event's row, one per entry of
297
+ #: ``events``, index-aligned (the contract's ``SessionTranscript.storedAt``):
298
+ #: present on every row-served page (an empty page carries an empty list),
299
+ #: ``None`` when the transcript was served from its file, where no write
300
+ #: instant exists. It places the gateway meter's calls under the harness's
301
+ #: steps for harnesses whose lines carry no clock of their own (codex,
302
+ #: kimi, qwen); a reader that does not place calls needs nothing from it.
303
+ stored_at: Optional[List[str]] = None
304
+
305
+
306
+ @dataclass
307
+ class BrowserReplay:
308
+ """Browser replay metadata and Dashboard-owned access URLs."""
309
+ session_id: str
310
+ status: Literal['ready']
311
+ replay_url: str
312
+ download_url: str
313
+ suggested_start_seconds: Optional[float] = None
314
+ size_bytes: Optional[int] = None
315
+ ready_at: Optional[str] = None
evolve/retry.py ADDED
@@ -0,0 +1,133 @@
1
+ """Retry Utility for Swarm operations.
2
+
3
+ Generic retry with exponential backoff.
4
+ Works with any result type that has a status field.
5
+ """
6
+
7
+ import asyncio
8
+ from dataclasses import dataclass
9
+ from typing import Any, Awaitable, Callable, Optional, TypeVar
10
+
11
+ # =============================================================================
12
+ # CONSTANTS
13
+ # =============================================================================
14
+
15
+ DEFAULT_MAX_ATTEMPTS = 3
16
+ DEFAULT_BACKOFF_MS = 1000
17
+ DEFAULT_BACKOFF_MULTIPLIER = 2.0
18
+
19
+ # =============================================================================
20
+ # TYPES
21
+ # =============================================================================
22
+
23
+ # TypeVar for result types (SwarmResult, ReduceResult, etc.)
24
+ # Results must have a `status` field for default retry behavior.
25
+ TResult = TypeVar('TResult')
26
+
27
+ # Callback type for item retry events (must be defined before RetryConfig)
28
+ OnItemRetryCallback = Callable[[int, int, str], None] # (item_index, attempt, error)
29
+
30
+
31
+ def _get_field(obj: Any, field: str, default: Any = None) -> Any:
32
+ """Get field from dict or object (duck typing helper)."""
33
+ if isinstance(obj, dict):
34
+ return obj.get(field, default)
35
+ return getattr(obj, field, default)
36
+
37
+
38
+ @dataclass
39
+ class RetryConfig:
40
+ """Per-item retry configuration.
41
+
42
+ Example:
43
+ # Basic retry on error
44
+ RetryConfig(max_attempts=3)
45
+
46
+ # With exponential backoff
47
+ RetryConfig(max_attempts=3, backoff_ms=1000, backoff_multiplier=2)
48
+
49
+ # Custom retry condition
50
+ RetryConfig(max_attempts=3, retry_on=lambda r: r.status == "error" or "timeout" in (r.error or ""))
51
+
52
+ # With callback
53
+ RetryConfig(max_attempts=3, on_item_retry=lambda i, a, e: print(f"Item {i} retry {a}: {e}"))
54
+
55
+ Args:
56
+ max_attempts: Maximum retry attempts (default: 3)
57
+ backoff_ms: Initial backoff in ms (default: 1000)
58
+ backoff_multiplier: Exponential backoff multiplier (default: 2)
59
+ retry_on: Custom retry condition (default: status == "error")
60
+ on_item_retry: Callback when retry occurs (item_index, attempt, error)
61
+ """
62
+ max_attempts: int = DEFAULT_MAX_ATTEMPTS
63
+ backoff_ms: int = DEFAULT_BACKOFF_MS
64
+ backoff_multiplier: float = DEFAULT_BACKOFF_MULTIPLIER
65
+ retry_on: Optional[Callable[[Any], bool]] = None
66
+ on_item_retry: Optional[OnItemRetryCallback] = None
67
+
68
+ def should_retry(self, result: Any) -> bool:
69
+ """Check if result should be retried."""
70
+ if self.retry_on is not None:
71
+ return self.retry_on(result)
72
+ # Default: retry on error status
73
+ return _get_field(result, 'status') == "error"
74
+
75
+
76
+ # =============================================================================
77
+ # RETRY LOGIC
78
+ # =============================================================================
79
+
80
+ async def execute_with_retry(
81
+ fn: Callable[[int], Awaitable[TResult]],
82
+ config: Optional[RetryConfig] = None,
83
+ item_index: int = 0,
84
+ ) -> TResult:
85
+ """Execute a function with retry and exponential backoff.
86
+
87
+ Works with any result type that has a `status` field (SwarmResult, ReduceResult, etc.).
88
+
89
+ Args:
90
+ fn: Async function that receives attempt number (1-based) and returns a result
91
+ config: Retry configuration (optional, uses defaults if not provided)
92
+ item_index: Item index for callback (default: 0)
93
+
94
+ Returns:
95
+ Result from the function
96
+
97
+ Example:
98
+ result = await execute_with_retry(
99
+ lambda attempt: self._execute_map_item(item, prompt, index, operation_id, params, timeout, attempt),
100
+ RetryConfig(max_attempts=3, backoff_ms=1000),
101
+ item_index=index,
102
+ )
103
+ """
104
+ resolved = config or RetryConfig()
105
+
106
+ last_result: Optional[TResult] = None
107
+ attempts = 0
108
+ backoff = resolved.backoff_ms
109
+
110
+ while attempts < resolved.max_attempts:
111
+ attempts += 1
112
+ last_result = await fn(attempts)
113
+
114
+ # Check if we should retry
115
+ if not resolved.should_retry(last_result):
116
+ return last_result
117
+
118
+ # Don't retry if we've exhausted attempts
119
+ if attempts >= resolved.max_attempts:
120
+ break
121
+
122
+ # Notify of retry via callback in config
123
+ if resolved.on_item_retry is not None:
124
+ error = _get_field(last_result, 'error') or "Unknown error"
125
+ resolved.on_item_retry(item_index, attempts, error)
126
+
127
+ # Wait before retrying (convert ms to seconds)
128
+ await asyncio.sleep(backoff / 1000)
129
+ backoff = backoff * resolved.backoff_multiplier
130
+
131
+ # Return last result
132
+ assert last_result is not None
133
+ return last_result
evolve/schema.py ADDED
@@ -0,0 +1,107 @@
1
+ """Schema utilities for Pydantic models, dataclasses, and JSON Schema.
2
+
3
+ Provides unified detection, conversion, and validation for schema types.
4
+ Uses Pydantic's TypeAdapter for dataclass support.
5
+ """
6
+
7
+ import dataclasses
8
+ from typing import Any, Dict, Optional
9
+
10
+
11
+ # =============================================================================
12
+ # DETECTION
13
+ # =============================================================================
14
+
15
+
16
+ def is_pydantic_model(obj: Any) -> bool:
17
+ """Check if object is a Pydantic model class."""
18
+ try:
19
+ from pydantic import BaseModel
20
+ return isinstance(obj, type) and issubclass(obj, BaseModel)
21
+ except ImportError:
22
+ return False
23
+
24
+
25
+ def is_dataclass(obj: Any) -> bool:
26
+ """Check if object is a dataclass."""
27
+ return dataclasses.is_dataclass(obj) and isinstance(obj, type)
28
+
29
+
30
+ def is_json_schema(obj: Any) -> bool:
31
+ """Check if object is a JSON Schema dict."""
32
+ return isinstance(obj, dict)
33
+
34
+
35
+ # =============================================================================
36
+ # CONVERSION
37
+ # =============================================================================
38
+
39
+
40
+ def to_json_schema(schema: Any) -> Optional[Dict[str, Any]]:
41
+ """Convert a schema to JSON Schema format.
42
+
43
+ Supports:
44
+ - Dict (JSON Schema) - passed through
45
+ - Pydantic models - uses model_json_schema()
46
+ - Dataclasses - uses Pydantic TypeAdapter
47
+
48
+ Args:
49
+ schema: Pydantic model, dataclass, or JSON Schema dict
50
+
51
+ Returns:
52
+ JSON Schema dict, or None if schema is None
53
+ """
54
+ if schema is None:
55
+ return None
56
+
57
+ if is_json_schema(schema):
58
+ return schema
59
+
60
+ if is_pydantic_model(schema):
61
+ return schema.model_json_schema()
62
+
63
+ if is_dataclass(schema):
64
+ from pydantic import TypeAdapter
65
+ return TypeAdapter(schema).json_schema()
66
+
67
+ raise TypeError(
68
+ f"Schema must be a Pydantic model, dataclass, or dict. "
69
+ f"Got {type(schema).__name__}"
70
+ )
71
+
72
+
73
+ # =============================================================================
74
+ # VALIDATION
75
+ # =============================================================================
76
+
77
+
78
+ def validate_and_parse(
79
+ raw_json: str,
80
+ schema: Any,
81
+ strict: bool = False,
82
+ ) -> Any:
83
+ """Validate JSON string and parse into schema type.
84
+
85
+ Args:
86
+ raw_json: Raw JSON string to validate
87
+ schema: Pydantic model or dataclass (returns instance)
88
+ strict: Use strict validation mode
89
+
90
+ Returns:
91
+ Model/dataclass instance, or None for dict schemas
92
+
93
+ Raises:
94
+ ValidationError: If validation fails
95
+ """
96
+ if schema is None or is_json_schema(schema):
97
+ # JSON Schema validation is handled by TS SDK
98
+ return None
99
+
100
+ if is_pydantic_model(schema):
101
+ return schema.model_validate_json(raw_json, strict=strict)
102
+
103
+ if is_dataclass(schema):
104
+ from pydantic import TypeAdapter
105
+ return TypeAdapter(schema).validate_json(raw_json, strict=strict)
106
+
107
+ return None