evolvingmachines-evolve 0.0.55.dev1355__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bridge/__init__.py +5 -0
- bridge/dist/bridge.bundle.cjs +1275 -0
- evolve/__init__.py +819 -0
- evolve/_http.py +71 -0
- evolve/agent.py +896 -0
- evolve/bridge.py +509 -0
- evolve/browser_credentials.py +265 -0
- evolve/browser_profiles.py +95 -0
- evolve/config.py +600 -0
- evolve/hosted.py +8958 -0
- evolve/integrations.py +173 -0
- evolve/managed_secrets.py +175 -0
- evolve/pipeline/__init__.py +59 -0
- evolve/pipeline/pipeline.py +512 -0
- evolve/pipeline/types.py +286 -0
- evolve/prompts/__init__.py +132 -0
- evolve/prompts/agent_md/judge.md +30 -0
- evolve/prompts/agent_md/reduce.md +7 -0
- evolve/prompts/agent_md/verify.md +33 -0
- evolve/prompts/user/judge.md +1 -0
- evolve/prompts/user/retry_feedback.md +9 -0
- evolve/prompts/user/verify.md +1 -0
- evolve/py.typed +0 -0
- evolve/results.py +315 -0
- evolve/retry.py +133 -0
- evolve/schema.py +107 -0
- evolve/sessions_client.py +167 -0
- evolve/storage_client.py +178 -0
- evolve/swarm/__init__.py +75 -0
- evolve/swarm/results.py +140 -0
- evolve/swarm/swarm.py +2116 -0
- evolve/swarm/types.py +241 -0
- evolve/utils.py +227 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/METADATA +52 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/RECORD +38 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/WHEEL +5 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/licenses/LICENSE +201 -0
- evolvingmachines_evolve-0.0.55.dev1355.dist-info/top_level.txt +2 -0
evolve/pipeline/types.py
ADDED
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Pipeline Types - Fluent API for chaining Swarm operations."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Any, Callable, Dict, Generic, List, Literal, Optional, TypeVar, Union
|
|
5
|
+
|
|
6
|
+
from ..swarm.types import (
|
|
7
|
+
BestOfConfig,
|
|
8
|
+
VerifyConfig,
|
|
9
|
+
SchemaType,
|
|
10
|
+
Prompt,
|
|
11
|
+
)
|
|
12
|
+
from ..swarm.results import SwarmResult, ReduceResult
|
|
13
|
+
from ..config import IntegrationsSetup
|
|
14
|
+
from ..retry import RetryConfig
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
T = TypeVar('T')
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
# =============================================================================
|
|
21
|
+
# EMIT OPTION (filter only)
|
|
22
|
+
# =============================================================================
|
|
23
|
+
|
|
24
|
+
EmitOption = Literal["success", "filtered", "all"]
|
|
25
|
+
"""What filter emits to the next step.
|
|
26
|
+
|
|
27
|
+
- "success": Items that passed condition (default)
|
|
28
|
+
- "filtered": Items that failed condition
|
|
29
|
+
- "all": Both success and filtered
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# =============================================================================
|
|
34
|
+
# STEP CONFIGURATIONS
|
|
35
|
+
# =============================================================================
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class MapConfig(Generic[T]):
|
|
39
|
+
"""Map step configuration."""
|
|
40
|
+
prompt: Prompt
|
|
41
|
+
"""Task prompt."""
|
|
42
|
+
name: Optional[str] = None
|
|
43
|
+
"""Step name for observability (appears in events)."""
|
|
44
|
+
system_prompt: Optional[str] = None
|
|
45
|
+
"""System prompt override."""
|
|
46
|
+
schema: Optional[SchemaType] = None
|
|
47
|
+
"""Schema for structured output."""
|
|
48
|
+
schema_options: Optional[Dict[str, Any]] = None
|
|
49
|
+
"""Validation options for JSON Schema."""
|
|
50
|
+
agent: Optional[Any] = None
|
|
51
|
+
"""Agent override."""
|
|
52
|
+
mcp_servers: Optional[Dict[str, Any]] = None
|
|
53
|
+
"""MCP servers override (replaces swarm default for this step)."""
|
|
54
|
+
skills: Optional[List[str]] = None
|
|
55
|
+
"""Skills override (replaces swarm default for this step)."""
|
|
56
|
+
integrations: Optional[IntegrationsSetup] = None
|
|
57
|
+
"""Integrations override (replaces swarm default for this step)."""
|
|
58
|
+
best_of: Optional[BestOfConfig] = None
|
|
59
|
+
"""BestOf configuration (mutually exclusive with verify)."""
|
|
60
|
+
verify: Optional[VerifyConfig] = None
|
|
61
|
+
"""Verify configuration (mutually exclusive with bestOf)."""
|
|
62
|
+
retry: Optional[RetryConfig] = None
|
|
63
|
+
"""Retry configuration."""
|
|
64
|
+
timeout_ms: Optional[int] = None
|
|
65
|
+
"""Timeout in ms."""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class FilterConfig(Generic[T]):
|
|
70
|
+
"""Filter step configuration."""
|
|
71
|
+
prompt: str
|
|
72
|
+
"""Evaluation prompt."""
|
|
73
|
+
schema: SchemaType
|
|
74
|
+
"""Schema for structured output (required)."""
|
|
75
|
+
condition: Callable[[Any], bool]
|
|
76
|
+
"""Condition function to determine pass/fail."""
|
|
77
|
+
name: Optional[str] = None
|
|
78
|
+
"""Step name for observability (appears in events)."""
|
|
79
|
+
system_prompt: Optional[str] = None
|
|
80
|
+
"""System prompt override."""
|
|
81
|
+
schema_options: Optional[Dict[str, Any]] = None
|
|
82
|
+
"""Validation options for JSON Schema."""
|
|
83
|
+
agent: Optional[Any] = None
|
|
84
|
+
"""Agent override."""
|
|
85
|
+
mcp_servers: Optional[Dict[str, Any]] = None
|
|
86
|
+
"""MCP servers override (replaces swarm default for this step)."""
|
|
87
|
+
skills: Optional[List[str]] = None
|
|
88
|
+
"""Skills override (replaces swarm default for this step)."""
|
|
89
|
+
integrations: Optional[IntegrationsSetup] = None
|
|
90
|
+
"""Integrations override (replaces swarm default for this step)."""
|
|
91
|
+
emit: EmitOption = "success"
|
|
92
|
+
"""What to emit to next step (default: "success")."""
|
|
93
|
+
verify: Optional[VerifyConfig] = None
|
|
94
|
+
"""Verify configuration."""
|
|
95
|
+
retry: Optional[RetryConfig] = None
|
|
96
|
+
"""Retry configuration."""
|
|
97
|
+
timeout_ms: Optional[int] = None
|
|
98
|
+
"""Timeout in ms."""
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class ReduceConfig(Generic[T]):
|
|
103
|
+
"""Reduce step configuration."""
|
|
104
|
+
prompt: str
|
|
105
|
+
"""Synthesis prompt."""
|
|
106
|
+
name: Optional[str] = None
|
|
107
|
+
"""Step name for observability (appears in events)."""
|
|
108
|
+
system_prompt: Optional[str] = None
|
|
109
|
+
"""System prompt override."""
|
|
110
|
+
schema: Optional[SchemaType] = None
|
|
111
|
+
"""Schema for structured output."""
|
|
112
|
+
schema_options: Optional[Dict[str, Any]] = None
|
|
113
|
+
"""Validation options for JSON Schema."""
|
|
114
|
+
agent: Optional[Any] = None
|
|
115
|
+
"""Agent override."""
|
|
116
|
+
mcp_servers: Optional[Dict[str, Any]] = None
|
|
117
|
+
"""MCP servers override (replaces swarm default for this step)."""
|
|
118
|
+
skills: Optional[List[str]] = None
|
|
119
|
+
"""Skills override (replaces swarm default for this step)."""
|
|
120
|
+
integrations: Optional[IntegrationsSetup] = None
|
|
121
|
+
"""Integrations override (replaces swarm default for this step)."""
|
|
122
|
+
verify: Optional[VerifyConfig] = None
|
|
123
|
+
"""Verify configuration."""
|
|
124
|
+
retry: Optional[RetryConfig] = None
|
|
125
|
+
"""Retry configuration."""
|
|
126
|
+
timeout_ms: Optional[int] = None
|
|
127
|
+
"""Timeout in ms."""
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# =============================================================================
|
|
131
|
+
# INTERNAL
|
|
132
|
+
# =============================================================================
|
|
133
|
+
|
|
134
|
+
StepType = Literal["map", "filter", "reduce"]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass
|
|
138
|
+
class Step:
|
|
139
|
+
"""Internal step representation."""
|
|
140
|
+
type: StepType
|
|
141
|
+
config: Union[MapConfig, FilterConfig, ReduceConfig]
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
# =============================================================================
|
|
145
|
+
# RESULTS
|
|
146
|
+
# =============================================================================
|
|
147
|
+
|
|
148
|
+
@dataclass
|
|
149
|
+
class StepResult(Generic[T]):
|
|
150
|
+
"""Result of a single pipeline step."""
|
|
151
|
+
type: StepType
|
|
152
|
+
index: int
|
|
153
|
+
duration_ms: int
|
|
154
|
+
results: Union[List[SwarmResult[T]], ReduceResult[T]]
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
@dataclass
|
|
158
|
+
class PipelineResult(Generic[T]):
|
|
159
|
+
"""Final result from pipeline execution."""
|
|
160
|
+
pipeline_run_id: str
|
|
161
|
+
"""Unique identifier for this pipeline run."""
|
|
162
|
+
steps: List[StepResult]
|
|
163
|
+
output: Union[List[SwarmResult[T]], ReduceResult[T]]
|
|
164
|
+
total_duration_ms: int
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# =============================================================================
|
|
168
|
+
# EVENTS
|
|
169
|
+
# =============================================================================
|
|
170
|
+
|
|
171
|
+
@dataclass
|
|
172
|
+
class StepEvent:
|
|
173
|
+
"""Step lifecycle event (base class)."""
|
|
174
|
+
type: StepType
|
|
175
|
+
index: int
|
|
176
|
+
name: Optional[str]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@dataclass
|
|
180
|
+
class StepStartEvent(StepEvent):
|
|
181
|
+
"""Emitted when step starts."""
|
|
182
|
+
item_count: int
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
@dataclass
|
|
186
|
+
class StepCompleteEvent(StepEvent):
|
|
187
|
+
"""Emitted when step completes."""
|
|
188
|
+
duration_ms: int
|
|
189
|
+
success_count: int
|
|
190
|
+
error_count: int
|
|
191
|
+
filtered_count: int
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
@dataclass
|
|
195
|
+
class StepErrorEvent(StepEvent):
|
|
196
|
+
"""Emitted when step errors."""
|
|
197
|
+
error: Exception
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
@dataclass
|
|
201
|
+
class ItemRetryEvent:
|
|
202
|
+
"""Emitted on item retry."""
|
|
203
|
+
step_index: int
|
|
204
|
+
step_name: Optional[str]
|
|
205
|
+
item_index: int
|
|
206
|
+
attempt: int
|
|
207
|
+
error: str
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
@dataclass
|
|
211
|
+
class WorkerCompleteEvent:
|
|
212
|
+
"""Emitted when verify worker completes."""
|
|
213
|
+
step_index: int
|
|
214
|
+
step_name: Optional[str]
|
|
215
|
+
item_index: int
|
|
216
|
+
attempt: int
|
|
217
|
+
status: Literal["success", "error"]
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@dataclass
|
|
221
|
+
class VerifierCompleteEvent:
|
|
222
|
+
"""Emitted when verifier completes."""
|
|
223
|
+
step_index: int
|
|
224
|
+
step_name: Optional[str]
|
|
225
|
+
item_index: int
|
|
226
|
+
attempt: int
|
|
227
|
+
passed: bool
|
|
228
|
+
feedback: Optional[str]
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
@dataclass
|
|
232
|
+
class CandidateCompleteEvent:
|
|
233
|
+
"""Emitted when bestOf candidate completes."""
|
|
234
|
+
step_index: int
|
|
235
|
+
step_name: Optional[str]
|
|
236
|
+
item_index: int
|
|
237
|
+
candidate_index: int
|
|
238
|
+
status: Literal["success", "error"]
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
@dataclass
|
|
242
|
+
class JudgeCompleteEvent:
|
|
243
|
+
"""Emitted when bestOf judge completes."""
|
|
244
|
+
step_index: int
|
|
245
|
+
step_name: Optional[str]
|
|
246
|
+
item_index: int
|
|
247
|
+
winner_index: int
|
|
248
|
+
reasoning: str
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
@dataclass
|
|
252
|
+
class PipelineEvents:
|
|
253
|
+
"""Event handlers."""
|
|
254
|
+
on_step_start: Optional[Callable[[StepStartEvent], None]] = None
|
|
255
|
+
on_step_complete: Optional[Callable[[StepCompleteEvent], None]] = None
|
|
256
|
+
on_step_error: Optional[Callable[[StepErrorEvent], None]] = None
|
|
257
|
+
on_item_retry: Optional[Callable[[ItemRetryEvent], None]] = None
|
|
258
|
+
on_worker_complete: Optional[Callable[[WorkerCompleteEvent], None]] = None
|
|
259
|
+
on_verifier_complete: Optional[Callable[[VerifierCompleteEvent], None]] = None
|
|
260
|
+
on_candidate_complete: Optional[Callable[[CandidateCompleteEvent], None]] = None
|
|
261
|
+
on_judge_complete: Optional[Callable[[JudgeCompleteEvent], None]] = None
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
# Event name mapping for chainable .on() style
|
|
265
|
+
PipelineEventMap = {
|
|
266
|
+
"step_start": "on_step_start",
|
|
267
|
+
"step_complete": "on_step_complete",
|
|
268
|
+
"step_error": "on_step_error",
|
|
269
|
+
"item_retry": "on_item_retry",
|
|
270
|
+
"worker_complete": "on_worker_complete",
|
|
271
|
+
"verifier_complete": "on_verifier_complete",
|
|
272
|
+
"candidate_complete": "on_candidate_complete",
|
|
273
|
+
"judge_complete": "on_judge_complete",
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
# Event names (for type hints)
|
|
277
|
+
EventName = Literal[
|
|
278
|
+
"step_start",
|
|
279
|
+
"step_complete",
|
|
280
|
+
"step_error",
|
|
281
|
+
"item_retry",
|
|
282
|
+
"worker_complete",
|
|
283
|
+
"verifier_complete",
|
|
284
|
+
"candidate_complete",
|
|
285
|
+
"judge_complete",
|
|
286
|
+
]
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Prompt templates for Swarm abstractions.
|
|
2
|
+
|
|
3
|
+
Prompts are stored as markdown files for easy editing.
|
|
4
|
+
They are loaded at import time using importlib.resources.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from importlib import resources
|
|
9
|
+
from typing import Dict, Union
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _load_prompt(subdir: str, filename: str) -> str:
|
|
13
|
+
"""Load a prompt template from a .md file in a subdirectory."""
|
|
14
|
+
return resources.files(__package__).joinpath(subdir, filename).read_text(encoding='utf-8')
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
# Load agent prompts (system prompts - goes into CLAUDE.md-like context)
|
|
18
|
+
JUDGE_PROMPT: str = _load_prompt('agent_md', 'judge.md')
|
|
19
|
+
VERIFY_PROMPT: str = _load_prompt('agent_md', 'verify.md')
|
|
20
|
+
REDUCE_PROMPT: str = _load_prompt('agent_md', 'reduce.md')
|
|
21
|
+
|
|
22
|
+
# Load user prompts (task prompts - passed to .run())
|
|
23
|
+
JUDGE_USER_PROMPT: str = _load_prompt('user', 'judge.md')
|
|
24
|
+
VERIFY_USER_PROMPT: str = _load_prompt('user', 'verify.md')
|
|
25
|
+
RETRY_FEEDBACK_PROMPT: str = _load_prompt('user', 'retry_feedback.md')
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def apply_template(template: str, variables: Dict[str, str]) -> str:
|
|
29
|
+
"""Apply template variables to a template string.
|
|
30
|
+
|
|
31
|
+
Replaces {{variable}} with the corresponding value from variables dict.
|
|
32
|
+
|
|
33
|
+
ONE pass over the template: substituted values are emitted verbatim and
|
|
34
|
+
never rescanned, so a value that itself contains ``{{...}}`` (user prompts,
|
|
35
|
+
verifier feedback, criteria) is not re-expanded by a later variable.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
template: Template string with {{variable}} placeholders
|
|
39
|
+
variables: Dict mapping variable names to values
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
Template with placeholders replaced
|
|
43
|
+
"""
|
|
44
|
+
return re.sub(
|
|
45
|
+
r'\{\{(\w+)\}\}',
|
|
46
|
+
lambda match: variables.get(match.group(1), match.group(0)),
|
|
47
|
+
template,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def build_file_tree(files: Dict[str, Union[str, bytes]]) -> str:
|
|
52
|
+
"""Build a formatted file tree string for judge context.
|
|
53
|
+
|
|
54
|
+
Generates an ASCII tree representation of the files structure,
|
|
55
|
+
with comments explaining the purpose of each section.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
files: Dict mapping file paths to content
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
Formatted file tree string
|
|
62
|
+
"""
|
|
63
|
+
if not files:
|
|
64
|
+
return "context/\n (empty)"
|
|
65
|
+
|
|
66
|
+
# Get unique top-level folders, sorted with worker_task first
|
|
67
|
+
folders = sorted(
|
|
68
|
+
set(path.split("/")[0] for path in files.keys()),
|
|
69
|
+
key=lambda f: ("" if f == "worker_task" else f)
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
if not folders:
|
|
73
|
+
return "context/\n (empty)"
|
|
74
|
+
|
|
75
|
+
# Check what exists in worker_task/
|
|
76
|
+
has_system_prompt = "worker_task/system_prompt.txt" in files
|
|
77
|
+
has_schema = "worker_task/schema.json" in files
|
|
78
|
+
has_input = any(p.startswith("worker_task/input/") for p in files.keys())
|
|
79
|
+
|
|
80
|
+
# Build entries with comments
|
|
81
|
+
entries: list[tuple[str, str]] = [] # (line, comment)
|
|
82
|
+
|
|
83
|
+
for i, folder in enumerate(folders):
|
|
84
|
+
is_last_folder = i == len(folders) - 1
|
|
85
|
+
folder_prefix = "└── " if is_last_folder else "├── "
|
|
86
|
+
child_indent = " " if is_last_folder else "│ "
|
|
87
|
+
|
|
88
|
+
if folder == "worker_task":
|
|
89
|
+
entries.append((f"{folder_prefix}{folder}/", "task given to workers"))
|
|
90
|
+
|
|
91
|
+
# Build worker_task children (only show what exists)
|
|
92
|
+
children: list[tuple[str, str]] = []
|
|
93
|
+
if has_system_prompt:
|
|
94
|
+
children.append(("system_prompt.txt", "worker system prompt"))
|
|
95
|
+
children.append(("user_prompt.txt", "worker task prompt"))
|
|
96
|
+
if has_schema:
|
|
97
|
+
children.append(("schema.json", "expected output schema"))
|
|
98
|
+
if has_input:
|
|
99
|
+
children.append(("input/", "worker input files"))
|
|
100
|
+
|
|
101
|
+
for j, (name, comment) in enumerate(children):
|
|
102
|
+
is_last_child = j == len(children) - 1
|
|
103
|
+
child_prefix = "└── " if is_last_child else "├── "
|
|
104
|
+
entries.append((f"{child_indent}{child_prefix}{name}", comment))
|
|
105
|
+
|
|
106
|
+
elif folder.startswith("candidate_"):
|
|
107
|
+
idx = folder.replace("candidate_", "")
|
|
108
|
+
entries.append((f"{folder_prefix}{folder}/", f"worker {idx} solution"))
|
|
109
|
+
elif folder == "worker_output":
|
|
110
|
+
entries.append((f"{folder_prefix}{folder}/", "output to verify"))
|
|
111
|
+
elif folder.startswith("item_"):
|
|
112
|
+
idx = folder.replace("item_", "")
|
|
113
|
+
entries.append((f"{folder_prefix}{folder}/", f"input {idx}"))
|
|
114
|
+
else:
|
|
115
|
+
entries.append((f"{folder_prefix}{folder}/", ""))
|
|
116
|
+
|
|
117
|
+
# Calculate max line width for alignment
|
|
118
|
+
max_width = max(len(line) for line, _ in entries)
|
|
119
|
+
|
|
120
|
+
# Build final output with aligned comments
|
|
121
|
+
lines = ["context/"]
|
|
122
|
+
for line, comment in entries:
|
|
123
|
+
if comment:
|
|
124
|
+
padding = " " * (max_width - len(line) + 3)
|
|
125
|
+
lines.append(f"{line}{padding}# {comment}")
|
|
126
|
+
else:
|
|
127
|
+
lines.append(line)
|
|
128
|
+
|
|
129
|
+
return "\n".join(lines)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
__all__ = ['JUDGE_PROMPT', 'JUDGE_USER_PROMPT', 'VERIFY_PROMPT', 'VERIFY_USER_PROMPT', 'REDUCE_PROMPT', 'RETRY_FEEDBACK_PROMPT', 'apply_template', 'build_file_tree']
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
### 1. YOUR ROLE: BEST OF N JUDGE
|
|
2
|
+
|
|
3
|
+
You are a judge. {{candidateCount}} AI workers attempted the same task independently. Your job is to analyze their solution attempts and pick the best one based on the evaluation criteria below.
|
|
4
|
+
|
|
5
|
+
### 2. CONTEXT STRUCTURE
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
{{fileTree}}
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
### 3. YOUR EVALUATION CRITERIA
|
|
12
|
+
|
|
13
|
+
You must judge their work based on:
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
{{criteria}}
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
### 4. YOUR PROCESS
|
|
20
|
+
|
|
21
|
+
1. Read `worker_task/` to understand the task:
|
|
22
|
+
- Review the worker system prompt and task prompt
|
|
23
|
+
- Check the expected output schema (if present)
|
|
24
|
+
- Examine the worker input files in `input/`
|
|
25
|
+
2. Carefully review EACH solution attempt in `candidate_i/`
|
|
26
|
+
3. Compare outputs against the evaluation criteria
|
|
27
|
+
4. Reason through your findings — perform all necessary evidence-based analyses and verifications before deciding
|
|
28
|
+
5. Pick the best candidate (0-indexed)
|
|
29
|
+
|
|
30
|
+
**IMPORTANT:** Be thorough. Do not skip steps. Your judgment must be evidence-based — cite specific files, outputs, or discrepancies to justify your decision.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
### 1. YOUR ROLE: OUTPUT VERIFIER
|
|
2
|
+
|
|
3
|
+
You are a quality verifier. An AI worker produced output for a task. Your job is to verify whether the output meets the specified quality criteria.
|
|
4
|
+
|
|
5
|
+
### 2. CONTEXT STRUCTURE
|
|
6
|
+
|
|
7
|
+
```
|
|
8
|
+
{{fileTree}}
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
### 3. VERIFICATION CRITERIA
|
|
12
|
+
|
|
13
|
+
The output must satisfy:
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
{{criteria}}
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
### 4. YOUR PROCESS
|
|
20
|
+
|
|
21
|
+
1. Read `worker_task/` to understand what was asked:
|
|
22
|
+
- Review the worker system prompt (if present)
|
|
23
|
+
- Review the task prompt
|
|
24
|
+
- Check the expected output schema (if present)
|
|
25
|
+
- Examine any input files in `input/`
|
|
26
|
+
2. Carefully review the worker's output in `worker_output/`
|
|
27
|
+
3. Evaluate against the verification criteria
|
|
28
|
+
4. Reason through your findings
|
|
29
|
+
5. Make your decision
|
|
30
|
+
|
|
31
|
+
**IMPORTANT:** Be thorough and fair. Cite specific evidence. If the output generally achieves the goal with minor issues, consider passing. Only fail if there are significant problems that violate the criteria.
|
|
32
|
+
|
|
33
|
+
If failing, provide specific, actionable feedback explaining what needs to be fixed.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
Evaluate the candidates and select the best one. You must save your decision to the file `output/result.json`.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
Verify the worker output against the criteria. You must save your decision to the file `output/result.json`.
|
evolve/py.typed
ADDED
|
File without changes
|