lemans 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +10 -0
- data/README.md +17 -2
- data/exe/lemans-remote +141 -3
- data/exe/lemans-viewer +962 -0
- data/lib/lemans/agent.rb +4 -1
- data/lib/lemans/agents/miniswen.rb +10 -5
- data/lib/lemans/agents/miniswen_installed.rb +16 -5
- data/lib/lemans/agents/nop.rb +1 -1
- data/lib/lemans/agents/oracle.rb +1 -1
- data/lib/lemans/agents.rb +4 -2
- data/lib/lemans/cli.rb +85 -31
- data/lib/lemans/environments/docker/proxy/Dockerfile +4 -0
- data/lib/lemans/environments/docker/proxy/README.md +49 -0
- data/lib/lemans/environments/docker/proxy/server.rb +79 -0
- data/lib/lemans/environments/docker.rb +79 -16
- data/lib/lemans/result.rb +52 -1
- data/lib/lemans/runner/task.rb +9 -5
- data/lib/lemans/runner.rb +45 -2
- data/lib/lemans/store.rb +6 -0
- data/lib/lemans/stores/fs.rb +16 -4
- data/lib/lemans/trial/patch.rb +21 -0
- data/lib/lemans/trial.rb +107 -39
- data/lib/lemans/version.rb +1 -1
- data/lib/lemans.rb +2 -0
- data/lib/miniswen/agent.rb +52 -16
- data/lib/miniswen/cli.rb +8 -3
- data/lib/miniswen/environment/docker.rb +1 -1
- data/lib/miniswen/jail.rb +1 -1
- data/lib/miniswen/local.rb +1 -1
- data/lib/miniswen/version.rb +1 -1
- metadata +48 -1
data/lib/lemans/result.rb
CHANGED
|
@@ -136,6 +136,12 @@ module Lemans
|
|
|
136
136
|
def as_json(**) = to_h
|
|
137
137
|
end
|
|
138
138
|
|
|
139
|
+
Restart = Data.define(:trial, :step, :mode) do
|
|
140
|
+
def initialize(trial:, step:, mode: nil) = super
|
|
141
|
+
|
|
142
|
+
def as_json(**) = to_h.compact
|
|
143
|
+
end
|
|
144
|
+
|
|
139
145
|
Step = Data.define(:outcome, :usage, :duration) do
|
|
140
146
|
def as_json(**) = { outcome: outcome.as_json, usage: usage&.as_json, duration: }.compact
|
|
141
147
|
|
|
@@ -150,7 +156,7 @@ module Lemans
|
|
|
150
156
|
attr_reader :id, :task, :agent, :model, :index,
|
|
151
157
|
:profile_digest, :task_digest, :revision
|
|
152
158
|
|
|
153
|
-
attr_accessor :tags, :metadata
|
|
159
|
+
attr_accessor :tags, :metadata, :restarted_from
|
|
154
160
|
|
|
155
161
|
attr_reader :phases, :steps
|
|
156
162
|
|
|
@@ -171,6 +177,7 @@ module Lemans
|
|
|
171
177
|
@metadata = {}
|
|
172
178
|
@phases = []
|
|
173
179
|
@steps = nil
|
|
180
|
+
@restarted_from = nil
|
|
174
181
|
|
|
175
182
|
@id = id || "#{task}__#{SecureRandom.alphanumeric(7)}"
|
|
176
183
|
@outcome = Outcome.new(:pending)
|
|
@@ -239,12 +246,55 @@ module Lemans
|
|
|
239
246
|
self
|
|
240
247
|
end
|
|
241
248
|
|
|
249
|
+
# A step is settled once the next one started: it passed its gate and left a savepoint.
|
|
250
|
+
def settled_steps
|
|
251
|
+
names = phases.map(&:name)
|
|
252
|
+
(1..steps.to_a.size).count { names.include?(:"agent.#{it + 1}") }
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
# The step whose verification ran last, nil when none did
|
|
256
|
+
def verified_step
|
|
257
|
+
name = phases.reverse.find { it.name.to_s.match?(/\Averifier(\.\d+)?\z/) }&.name or return nil
|
|
258
|
+
name.to_s[/\.(\d+)\z/, 1]&.to_i || steps&.size || 1
|
|
259
|
+
end
|
|
260
|
+
|
|
261
|
+
# Carries over the steps before `step`; a reverification carries `step` itself too.
|
|
262
|
+
def restart!(source, step:, mode: nil)
|
|
263
|
+
@restarted_from = Restart.new(trial: source.id, step:, mode: mode&.to_s)
|
|
264
|
+
carried = mode == :reverify ? step : step - 1
|
|
265
|
+
|
|
266
|
+
if source.steps
|
|
267
|
+
source.steps.first(carried).each { step_completed!(it.outcome, it.usage, duration: it.duration) }
|
|
268
|
+
elsif carried.positive?
|
|
269
|
+
# A single-step run graded before keeps its agent outcome; a failed grading lost it.
|
|
270
|
+
completed!(source.scored? ? source.outcome : Outcome.new(:completed), source.usage)
|
|
271
|
+
end
|
|
272
|
+
self
|
|
273
|
+
end
|
|
274
|
+
|
|
275
|
+
# Splices in the named phases of the source run, shifted so that its
|
|
276
|
+
# `through` phase ends now, and moves this run's environment setup back in
|
|
277
|
+
# front of them: the timeline reads as one run.
|
|
278
|
+
def adopt_phases!(source, names, through: names.last, now: Time.now.utc)
|
|
279
|
+
delta = now - source.phases.find { it.name == through }.finished_at
|
|
280
|
+
|
|
281
|
+
setup = phases.first
|
|
282
|
+
shift = source.phases.find { it.name == setup.name }.finished_at + delta - setup.finished_at
|
|
283
|
+
phases[0] = Phase.new(setup.name, started_at: setup.started_at + shift, finished_at: setup.finished_at + shift)
|
|
284
|
+
|
|
285
|
+
source.phases.select { names.include?(it.name) }.each do |phase|
|
|
286
|
+
phases << Phase.new(phase.name, started_at: phase.started_at + delta, finished_at: phase.finished_at + delta)
|
|
287
|
+
end
|
|
288
|
+
self
|
|
289
|
+
end
|
|
290
|
+
|
|
242
291
|
private def aggregate_usage = steps.filter_map(&:usage).reduce(:+)
|
|
243
292
|
|
|
244
293
|
def as_json(**)
|
|
245
294
|
{
|
|
246
295
|
trial: id, task:, agent:, model:, index:,
|
|
247
296
|
profile_digest:, task_digest:, revision: revision&.as_json,
|
|
297
|
+
restarted_from: restarted_from&.as_json,
|
|
248
298
|
lemans_version: VERSION,
|
|
249
299
|
tags:, metadata:, phases: phases.map(&:as_json),
|
|
250
300
|
steps: steps&.map(&:as_json),
|
|
@@ -264,6 +314,7 @@ module Lemans
|
|
|
264
314
|
)
|
|
265
315
|
result.tags = data[:tags] || []
|
|
266
316
|
result.metadata = data[:metadata] || {}
|
|
317
|
+
result.restarted_from = Restart.new(**data[:restarted_from]) if data[:restarted_from]
|
|
267
318
|
phases_from(data).each { result.phases << it }
|
|
268
319
|
|
|
269
320
|
# Steps first: the stored outcome/usage below override the aggregates.
|
data/lib/lemans/runner/task.rb
CHANGED
|
@@ -20,18 +20,22 @@ module Lemans
|
|
|
20
20
|
def_delegators :definition, :name, :config
|
|
21
21
|
def_delegators :result, :id
|
|
22
22
|
|
|
23
|
-
private attr_reader :definition, :store, :reporter
|
|
23
|
+
private attr_reader :definition, :store, :reporter, :restart_from, :restart_mode
|
|
24
24
|
|
|
25
|
-
def initialize(model, task_definition, index: 0, store: nil, reporter: nil)
|
|
25
|
+
def initialize(model, task_definition, index: 0, store: nil, reporter: nil, restart_from: nil, restart_mode: nil)
|
|
26
26
|
@model = model
|
|
27
27
|
@definition = task_definition
|
|
28
28
|
@index = index
|
|
29
29
|
@store = store
|
|
30
30
|
@reporter = reporter
|
|
31
|
+
@restart_from = restart_from
|
|
32
|
+
@restart_mode = restart_mode
|
|
31
33
|
@status = :pending
|
|
32
34
|
|
|
33
|
-
# prepare the result object: it's used by the actual execution down the stack
|
|
34
|
-
|
|
35
|
+
# prepare the result object: it's used by the actual execution down the stack;
|
|
36
|
+
# a restart keeps the agent of the run it restarts
|
|
37
|
+
@result = Result.from_task(definition, index:, model: model || config.models.first,
|
|
38
|
+
**({ agent: restart_from.agent } if restart_from))
|
|
35
39
|
end
|
|
36
40
|
|
|
37
41
|
def with_reporter(reporter)
|
|
@@ -55,7 +59,7 @@ module Lemans
|
|
|
55
59
|
private
|
|
56
60
|
|
|
57
61
|
def execute!
|
|
58
|
-
Trial.new(definition, model, store:, result:).run
|
|
62
|
+
Trial.new(definition, model, store:, result:, agent: restart_from&.agent, restart_from:, restart_mode:).run
|
|
59
63
|
end
|
|
60
64
|
end
|
|
61
65
|
end
|
data/lib/lemans/runner.rb
CHANGED
|
@@ -20,20 +20,26 @@ module Lemans
|
|
|
20
20
|
|
|
21
21
|
attr_reader :config, :tasks, :store, :reporter
|
|
22
22
|
|
|
23
|
-
private attr_reader :resuming, :executor
|
|
23
|
+
private attr_reader :resuming, :restarts, :restart_mode, :allow_scored, :executor
|
|
24
24
|
|
|
25
|
-
def initialize(config, tasks, store: nil, reporter: nil, executor: nil, resume: false
|
|
25
|
+
def initialize(config, tasks, store: nil, reporter: nil, executor: nil, resume: false,
|
|
26
|
+
restarts: [], restart_mode: nil, allow_scored: false)
|
|
26
27
|
@config = config
|
|
27
28
|
@tasks = tasks
|
|
28
29
|
@store = store
|
|
29
30
|
@reporter = reporter
|
|
30
31
|
@executor = executor || Executor.new(config.concurrency)
|
|
31
32
|
@resuming = resume
|
|
33
|
+
@restarts = restarts
|
|
34
|
+
@restart_mode = restart_mode
|
|
35
|
+
@allow_scored = allow_scored
|
|
32
36
|
end
|
|
33
37
|
|
|
34
38
|
def resuming? = @resuming
|
|
35
39
|
|
|
36
40
|
def attempts
|
|
41
|
+
return @attempts ||= restart_attempts if restarts.any?
|
|
42
|
+
|
|
37
43
|
@attempts ||= config.agent.models.flat_map do |model|
|
|
38
44
|
@tasks.flat_map do |task|
|
|
39
45
|
completed = resuming? ? completed_attempts(task, model) : 0
|
|
@@ -64,6 +70,43 @@ module Lemans
|
|
|
64
70
|
|
|
65
71
|
private
|
|
66
72
|
|
|
73
|
+
# Every run is checked before any starts: one refused run stops the batch.
|
|
74
|
+
def restart_attempts
|
|
75
|
+
refusals = []
|
|
76
|
+
attempts = restarts.filter_map do |restart|
|
|
77
|
+
restart_attempt(restart)
|
|
78
|
+
rescue ConfigError => e
|
|
79
|
+
refusals << e.message
|
|
80
|
+
nil
|
|
81
|
+
end
|
|
82
|
+
raise ConfigError, refusals.join("\n") if refusals.any?
|
|
83
|
+
|
|
84
|
+
attempts
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
# A run named for a restart is restarted on purpose: a changed task or
|
|
88
|
+
# bench is the operator's call, and a reverification may grade a scored run.
|
|
89
|
+
def restart_attempt(restart)
|
|
90
|
+
refuse = ->(reason) { raise ConfigError, "cannot restart #{restart.id}: #{reason}" }
|
|
91
|
+
task = tasks.find { it.name == restart.task }
|
|
92
|
+
refuse.("task #{restart.task} is not in the bench") unless task
|
|
93
|
+
refuse.("it is already scored (--allow-scored to restart it anyway)") if restart.scored? && restart_mode != :reverify && !allow_scored
|
|
94
|
+
|
|
95
|
+
case restart_mode
|
|
96
|
+
when :reverify
|
|
97
|
+
refuse.("it never reached a verification") unless restart.verified_step
|
|
98
|
+
when :recover
|
|
99
|
+
refuse.("#{restart.agent} cannot recover a session") unless Agents.lookup(restart.agent).recoverable?
|
|
100
|
+
|
|
101
|
+
history = task.multistep? ? "agent.result.#{restart.settled_steps + 1}.json" : "agent.result.json"
|
|
102
|
+
refuse.("it has no #{history} to recover from") unless store&.read_artifact(restart, history)
|
|
103
|
+
else
|
|
104
|
+
refuse.("no step was settled before it failed") if restart.settled_steps.zero?
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
Task.new(restart.model, task, store:, index: restart.index || 1, restart_from: restart, restart_mode:)
|
|
108
|
+
end
|
|
109
|
+
|
|
67
110
|
def completed_attempts(task, model)
|
|
68
111
|
completed_runs.count do |run|
|
|
69
112
|
run.task == task.name &&
|
data/lib/lemans/store.rb
CHANGED
|
@@ -46,5 +46,11 @@ module Lemans
|
|
|
46
46
|
def read_artifact(result, path)
|
|
47
47
|
raise NotImplementedError
|
|
48
48
|
end
|
|
49
|
+
|
|
50
|
+
# Returns the sorted paths of the artifacts the result stored (the result
|
|
51
|
+
# record is not one of them)
|
|
52
|
+
def artifacts(result)
|
|
53
|
+
raise NotImplementedError
|
|
54
|
+
end
|
|
49
55
|
end
|
|
50
56
|
end
|
data/lib/lemans/stores/fs.rb
CHANGED
|
@@ -52,8 +52,8 @@ module Lemans
|
|
|
52
52
|
end
|
|
53
53
|
|
|
54
54
|
def delete(result)
|
|
55
|
-
dir =
|
|
56
|
-
return unless dir
|
|
55
|
+
dir = result_dir(result)
|
|
56
|
+
return unless dir.directory?
|
|
57
57
|
|
|
58
58
|
FileUtils.remove_entry(dir.to_s)
|
|
59
59
|
prune_empty_parents(dir.parent)
|
|
@@ -90,6 +90,13 @@ module Lemans
|
|
|
90
90
|
file.read if file.file?
|
|
91
91
|
end
|
|
92
92
|
|
|
93
|
+
def artifacts(result)
|
|
94
|
+
dir = result_dir(result)
|
|
95
|
+
return [] unless dir.directory?
|
|
96
|
+
|
|
97
|
+
(dir.glob("**/*").select(&:file?).map { it.relative_path_from(dir).to_s } - [ FILENAME ]).sort
|
|
98
|
+
end
|
|
99
|
+
|
|
93
100
|
private
|
|
94
101
|
|
|
95
102
|
def filtered(text) = filterer ? filterer.filter(text) : text
|
|
@@ -132,9 +139,14 @@ module Lemans
|
|
|
132
139
|
tmp&.delete if tmp&.exist?
|
|
133
140
|
end
|
|
134
141
|
|
|
135
|
-
# result.json is stored at <root>/<model-short>/<result-id
|
|
142
|
+
# result.json is stored at <root>/<model-short>/<result-id>; a tree
|
|
143
|
+
# arranged by hand (runs grouped under batch directories) is searched
|
|
144
|
+
# for the id instead.
|
|
136
145
|
def result_dir(result)
|
|
137
|
-
root.join((result.model || result.agent).to_s.split("/").last.tr("#", "-"), result.id)
|
|
146
|
+
canonical = root.join((result.model || result.agent).to_s.split("/").last.to_s.tr("#", "-"), result.id)
|
|
147
|
+
return canonical if canonical.directory?
|
|
148
|
+
|
|
149
|
+
root.glob("**/#{result.id}").find(&:directory?) || canonical
|
|
138
150
|
end
|
|
139
151
|
end
|
|
140
152
|
end
|
data/lib/lemans/trial/patch.rb
CHANGED
|
@@ -61,6 +61,15 @@ module Lemans
|
|
|
61
61
|
raise InfrastructureError, "could not savepoint the tree the step left behind" unless savepoint
|
|
62
62
|
end
|
|
63
63
|
|
|
64
|
+
# Brings a fresh tree to where an earlier run left it: the settled steps'
|
|
65
|
+
# patches applied in order and marked as the savepoint, then the work of
|
|
66
|
+
# the step that run stopped in, on top of it, so collect! sees it as this step's.
|
|
67
|
+
def replay!(settled, unsettled = nil)
|
|
68
|
+
settled.each { apply!(it) }
|
|
69
|
+
savepoint!
|
|
70
|
+
apply!(unsettled) if unsettled
|
|
71
|
+
end
|
|
72
|
+
|
|
64
73
|
# Puts the workdir back to the savepoint exactly: the verifier restored
|
|
65
74
|
# the graded surfaces from the baseline and may have littered the tree,
|
|
66
75
|
# and the next step's agent must find neither.
|
|
@@ -75,6 +84,18 @@ module Lemans
|
|
|
75
84
|
|
|
76
85
|
private
|
|
77
86
|
|
|
87
|
+
def apply!(contents)
|
|
88
|
+
return if contents.empty?
|
|
89
|
+
|
|
90
|
+
Tempfile.create(%w[agent .patch]) do |file|
|
|
91
|
+
file.write(contents)
|
|
92
|
+
file.flush
|
|
93
|
+
environment.upload(file.path, REMOTE_PATCH)
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
environment.exec!("#{git} apply --binary --whitespace=nowarn #{REMOTE_PATCH} && rm -f #{REMOTE_PATCH}", timeout:)
|
|
97
|
+
end
|
|
98
|
+
|
|
78
99
|
def save_diff(result, store, from, to, destination)
|
|
79
100
|
diffed = environment.exec("#{git} diff --binary #{from} #{to} > #{REMOTE_PATCH}", timeout:)
|
|
80
101
|
return unless diffed.success?
|
data/lib/lemans/trial.rb
CHANGED
|
@@ -12,13 +12,15 @@ module Lemans
|
|
|
12
12
|
class Trial
|
|
13
13
|
attr_reader :task, :config, :model, :agent_name, :environment, :result
|
|
14
14
|
|
|
15
|
-
private attr_reader :agent, :store, :snapshot, :patch, :current_step_index
|
|
15
|
+
private attr_reader :agent, :store, :restart_from, :restart_mode, :snapshot, :patch, :current_step_index
|
|
16
16
|
|
|
17
|
-
def initialize(task, model = nil, result: nil, store: nil, agent: nil, environment: nil)
|
|
17
|
+
def initialize(task, model = nil, result: nil, store: nil, agent: nil, environment: nil, restart_from: nil, restart_mode: nil)
|
|
18
18
|
@task = task
|
|
19
19
|
@config = task.config
|
|
20
20
|
@model = model || config.models.first
|
|
21
21
|
@store = store
|
|
22
|
+
@restart_from = restart_from
|
|
23
|
+
@restart_mode = restart_mode
|
|
22
24
|
|
|
23
25
|
@agent = agent.is_a?(Agent) ? agent : Agents.build(agent || config.agent_name, profile: config.agent, model: @model)
|
|
24
26
|
@agent_name = @agent.name
|
|
@@ -47,9 +49,12 @@ module Lemans
|
|
|
47
49
|
@snapshot = nil
|
|
48
50
|
@patch = nil
|
|
49
51
|
@current_step_index = nil
|
|
52
|
+
@recovered_since = nil
|
|
50
53
|
end
|
|
51
54
|
|
|
52
55
|
def run
|
|
56
|
+
restart! if restart_from
|
|
57
|
+
|
|
53
58
|
phase(:environment_setup) do
|
|
54
59
|
environment.start
|
|
55
60
|
|
|
@@ -64,45 +69,55 @@ module Lemans
|
|
|
64
69
|
# Seal the git state to collect the agent's patch later
|
|
65
70
|
@patch = Patch.new(task, environment)
|
|
66
71
|
patch.seal!
|
|
72
|
+
patch.replay!(settled_patches, (restart_artifact("agent.patch") if restart_mode)) if restart_from
|
|
67
73
|
|
|
68
|
-
agent
|
|
74
|
+
# A reverification of the last step runs no agent
|
|
75
|
+
agent.install(task, environment) unless restart_mode == :reverify && restart_step == task.steps
|
|
69
76
|
|
|
70
77
|
environment.switch_network_policy!(config.agent.environment.network)
|
|
71
78
|
end
|
|
72
79
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
80
|
+
adopt_phases! if restart_from
|
|
81
|
+
|
|
82
|
+
pending = restart_mode
|
|
83
|
+
each_step(from: restart_from ? restart_step : 1) do |step_task|
|
|
84
|
+
mode, pending = pending, nil
|
|
85
|
+
|
|
86
|
+
unless mode == :reverify
|
|
87
|
+
history = restart_artifact("agent.result.json") if mode == :recover
|
|
88
|
+
response =
|
|
89
|
+
phase(:agent, started_at: (@recovered_since if mode == :recover)) do
|
|
90
|
+
agent.run(step_task, environment, history:)
|
|
91
|
+
rescue InfrastructureError, ::Miniswen::InfrastructureError => e
|
|
92
|
+
# Mark the failure here, where the agent phase is still known
|
|
93
|
+
result.failed!(:agent_error, e.message)
|
|
94
|
+
collect_patch!
|
|
95
|
+
raise
|
|
96
|
+
rescue ::Miniswen::AccountingError
|
|
97
|
+
# Classified by the outer rescue; the work is still on disk
|
|
98
|
+
collect_patch!
|
|
99
|
+
raise
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
# Whatever the agent brought back is evidence, a failed run's included
|
|
103
|
+
save_trajectory!(response.trajectory)
|
|
104
|
+
store&.save_artifact(result, response.raw_result, path: with_step_index("agent.result.json")) if response.raw_result
|
|
105
|
+
|
|
106
|
+
if response.error?
|
|
107
|
+
result.failed!(:agent_error, response.error)
|
|
84
108
|
collect_patch!
|
|
85
|
-
|
|
109
|
+
return result
|
|
86
110
|
end
|
|
87
111
|
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
result.failed!(:agent_error, response.error)
|
|
94
|
-
collect_patch!
|
|
95
|
-
return result
|
|
96
|
-
end
|
|
112
|
+
if task.multistep?
|
|
113
|
+
result.step_completed!(response.outcome, response.usage, duration: result.phases.last.duration)
|
|
114
|
+
else
|
|
115
|
+
result.completed!(response.outcome, response.usage)
|
|
116
|
+
end
|
|
97
117
|
|
|
98
|
-
|
|
99
|
-
result.step_completed!(response.outcome, response.usage, duration: result.phases.last.duration)
|
|
100
|
-
else
|
|
101
|
-
result.completed!(response.outcome, response.usage)
|
|
118
|
+
check_cost_limit!
|
|
102
119
|
end
|
|
103
120
|
|
|
104
|
-
check_cost_limit!
|
|
105
|
-
|
|
106
121
|
collect_patch!
|
|
107
122
|
if step_task.final_step?
|
|
108
123
|
patch.compile!(result, store) if task.multistep? && store
|
|
@@ -166,31 +181,84 @@ module Lemans
|
|
|
166
181
|
store.save_artifact(result, JSON.pretty_generate(trajectory.to_atif), path:)
|
|
167
182
|
end
|
|
168
183
|
|
|
169
|
-
def each_step
|
|
184
|
+
def each_step(from: 1)
|
|
170
185
|
return yield task unless task.multistep?
|
|
171
186
|
|
|
172
187
|
catch(:halt) do
|
|
173
|
-
|
|
174
|
-
|
|
188
|
+
from.upto(task.steps) do |index|
|
|
189
|
+
# The first step of a restart finds a fresh tree, already replayed
|
|
190
|
+
resume_agent! if index > from
|
|
175
191
|
@current_step_index = index
|
|
176
192
|
yield task.for_step(index)
|
|
177
193
|
end
|
|
178
194
|
end
|
|
179
195
|
end
|
|
180
196
|
|
|
197
|
+
# The step a restart begins at: the one a reverification grades again, or
|
|
198
|
+
# the first one not settled (run afresh, or recovered from its history).
|
|
199
|
+
def restart_step
|
|
200
|
+
@restart_step ||= restart_mode == :reverify ? restart_from.verified_step : restart_from.settled_steps + 1
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
# The new run carries the settled steps' evidence, and the agent's part of
|
|
204
|
+
# a step it grades again: the patch is collected anew.
|
|
205
|
+
def restart!
|
|
206
|
+
result.restart!(restart_from, step: restart_step, mode: restart_mode)
|
|
207
|
+
carried = restart_mode == :reverify ? %w[trajectory.json agent.result.json].map { restart_path(it) } : []
|
|
208
|
+
|
|
209
|
+
store.artifacts(restart_from).each do |path|
|
|
210
|
+
next unless path[%r{\.(\d+)(?:\.[^./]+)?\z}, 1].to_i.between?(1, restart_step - 1) || carried.include?(path)
|
|
211
|
+
|
|
212
|
+
store.save_artifact(result, store.read_artifact(restart_from, path), path:)
|
|
213
|
+
end
|
|
214
|
+
end
|
|
215
|
+
|
|
216
|
+
def settled_patches
|
|
217
|
+
(1...restart_step).map do |index|
|
|
218
|
+
store.read_artifact(restart_from, "agent.#{index}.patch") ||
|
|
219
|
+
raise(InfrastructureError, "#{restart_from.id} has no agent.#{index}.patch to replay")
|
|
220
|
+
end
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
def restart_artifact(name)
|
|
224
|
+
path = restart_path(name)
|
|
225
|
+
store.read_artifact(restart_from, path) || raise(InfrastructureError, "#{restart_from.id} has no #{path} to restart from")
|
|
226
|
+
end
|
|
227
|
+
|
|
228
|
+
def restart_path(name) = task.multistep? ? with_step_index(name, restart_step) : name
|
|
229
|
+
|
|
230
|
+
# A recovered step's agent phase reaches back over the part the source
|
|
231
|
+
# run already spent, so the step lasts as long as both parts together.
|
|
232
|
+
def adopt_phases!
|
|
233
|
+
names = restart_from.phases.map(&:name).select { it.to_s[/\.(\d+)\z/, 1].to_i.between?(1, restart_step - 1) }
|
|
234
|
+
agent_phase = restart_path("agent").to_sym
|
|
235
|
+
now = Time.now.utc
|
|
236
|
+
|
|
237
|
+
case restart_mode
|
|
238
|
+
when :reverify
|
|
239
|
+
result.adopt_phases!(restart_from, names + [ agent_phase ], now:)
|
|
240
|
+
when :recover
|
|
241
|
+
result.adopt_phases!(restart_from, names, through: agent_phase, now:)
|
|
242
|
+
partial = restart_from.phases.find { it.name == agent_phase }
|
|
243
|
+
@recovered_since = now - (partial.finished_at - partial.started_at)
|
|
244
|
+
else
|
|
245
|
+
result.adopt_phases!(restart_from, names, now:)
|
|
246
|
+
end
|
|
247
|
+
end
|
|
248
|
+
|
|
181
249
|
def resume_agent!
|
|
182
250
|
patch.restore!
|
|
183
251
|
environment.exec!("rm -rf #{Verifier::TESTS_DIR} #{Shellwords.escape(task.verifier.logs_dir)}")
|
|
184
252
|
environment.switch_network_policy!(config.agent.environment.network)
|
|
185
253
|
end
|
|
186
254
|
|
|
187
|
-
def with_step_index(path)
|
|
188
|
-
return path unless
|
|
255
|
+
def with_step_index(path, index = current_step_index)
|
|
256
|
+
return path unless index
|
|
189
257
|
|
|
190
258
|
*pre, last = path.to_s.split(".")
|
|
191
|
-
return "#{last}.#{
|
|
259
|
+
return "#{last}.#{index}" if pre.empty?
|
|
192
260
|
|
|
193
|
-
[ *pre,
|
|
261
|
+
[ *pre, index, last ].join(".")
|
|
194
262
|
end
|
|
195
263
|
|
|
196
264
|
def sandbox_ttl
|
|
@@ -209,8 +277,8 @@ module Lemans
|
|
|
209
277
|
)
|
|
210
278
|
end
|
|
211
279
|
|
|
212
|
-
def phase(name)
|
|
213
|
-
result.phase_started(with_step_index(name).to_sym)
|
|
280
|
+
def phase(name, started_at: nil)
|
|
281
|
+
result.phase_started(with_step_index(name).to_sym, started_at)
|
|
214
282
|
yield
|
|
215
283
|
ensure
|
|
216
284
|
result.phase_finished(with_step_index(name).to_sym)
|
data/lib/lemans/version.rb
CHANGED
data/lib/lemans.rb
CHANGED
|
@@ -11,6 +11,8 @@ loader.ignore("#{__dir__}/miniswen.rb", "#{__dir__}/miniswen")
|
|
|
11
11
|
loader.ignore("#{__dir__}/lemans/trial/verifier/assets")
|
|
12
12
|
# templates/ holds the files `lemans init` scaffolds, not Ruby the harness loads.
|
|
13
13
|
loader.ignore("#{__dir__}/lemans/cli/templates")
|
|
14
|
+
# proxy/ holds the image the Docker backend runs for allowlists, not Ruby the harness loads.
|
|
15
|
+
loader.ignore("#{__dir__}/lemans/environments/docker/proxy")
|
|
14
16
|
loader.setup
|
|
15
17
|
|
|
16
18
|
require "miniswen"
|
data/lib/miniswen/agent.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "json"
|
|
4
|
+
require "time"
|
|
4
5
|
require "miniswen/version"
|
|
5
6
|
require "miniswen/ruby_llm"
|
|
6
7
|
|
|
@@ -261,25 +262,17 @@ module Miniswen
|
|
|
261
262
|
@reporter = reporter
|
|
262
263
|
end
|
|
263
264
|
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
]
|
|
273
|
-
|
|
274
|
-
@steps = 0
|
|
275
|
-
@cost = 0.0
|
|
276
|
-
|
|
277
|
-
@totals = { input_tokens: 0, output_tokens: 0, cached_tokens: 0, thinking_tokens: 0 }
|
|
265
|
+
# With a `history` (the Result of an interrupted run), the session goes on
|
|
266
|
+
# where it stopped, under the budget the earlier part already spent.
|
|
267
|
+
def run(instruction = nil, history: nil)
|
|
268
|
+
if history
|
|
269
|
+
resume(history)
|
|
270
|
+
else
|
|
271
|
+
start(instruction)
|
|
272
|
+
end
|
|
278
273
|
|
|
279
|
-
@cost_known = true
|
|
280
274
|
@consecutive_format_errors = 0
|
|
281
275
|
@refused_turns = 0
|
|
282
|
-
@started_at = @clock.call
|
|
283
276
|
|
|
284
277
|
loop do
|
|
285
278
|
(status = limit_reached) and return finish(status)
|
|
@@ -329,6 +322,49 @@ module Miniswen
|
|
|
329
322
|
|
|
330
323
|
private
|
|
331
324
|
|
|
325
|
+
def start(instruction)
|
|
326
|
+
uname = execute("uname -srvm").output.to_s.strip
|
|
327
|
+
@messages = [
|
|
328
|
+
{ role: "system", content: SYSTEM_TEMPLATE },
|
|
329
|
+
{ role: "user", content: format(INSTANCE_TEMPLATE,
|
|
330
|
+
instruction: instruction,
|
|
331
|
+
system_information: uname,
|
|
332
|
+
macos_sed_note: uname.start_with?("Darwin") ? "\n#{MACOS_SED_NOTE}" : "") }
|
|
333
|
+
]
|
|
334
|
+
|
|
335
|
+
@steps = 0
|
|
336
|
+
@cost = 0.0
|
|
337
|
+
@cost_known = true
|
|
338
|
+
@totals = { input_tokens: 0, output_tokens: 0, cached_tokens: 0, thinking_tokens: 0 }
|
|
339
|
+
@started_at = @clock.call
|
|
340
|
+
end
|
|
341
|
+
|
|
342
|
+
def resume(history)
|
|
343
|
+
@messages = answered_messages(history.messages)
|
|
344
|
+
|
|
345
|
+
@steps = history.steps.to_i
|
|
346
|
+
@cost = history.cost_usd.to_f
|
|
347
|
+
@cost_known = !history.cost_usd.nil?
|
|
348
|
+
@totals = { input_tokens: history.input_tokens.to_i, output_tokens: history.output_tokens.to_i,
|
|
349
|
+
cached_tokens: history.cached_tokens.to_i, thinking_tokens: history.thinking_tokens.to_i }
|
|
350
|
+
@started_at = @clock.call - elapsed(history.messages)
|
|
351
|
+
end
|
|
352
|
+
|
|
353
|
+
# A turn whose calls did not all get a result cannot go back to a provider:
|
|
354
|
+
# it is dropped, and the model decides again from the tree it left.
|
|
355
|
+
def answered_messages(messages)
|
|
356
|
+
turn = messages.rindex { it[:role] == "assistant" && it[:tool_calls] }
|
|
357
|
+
return messages.dup unless turn
|
|
358
|
+
|
|
359
|
+
answered = messages[(turn + 1)..].filter_map { it[:tool_call_id] if it[:role] == "tool" }
|
|
360
|
+
messages[turn][:tool_calls].all? { answered.include?(it[:id]) } ? messages.dup : messages[0...turn]
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
def elapsed(messages)
|
|
364
|
+
times = messages.filter_map { Time.parse(it[:timestamp]) if it[:role] == "assistant" && it[:timestamp] }
|
|
365
|
+
times.empty? ? 0.0 : times.max - times.min
|
|
366
|
+
end
|
|
367
|
+
|
|
332
368
|
def execute(command)
|
|
333
369
|
environment.exec(command, timeout: exec_timeout, env: EXEC_ENV)
|
|
334
370
|
end
|
data/lib/miniswen/cli.rb
CHANGED
|
@@ -25,6 +25,7 @@ module Miniswen
|
|
|
25
25
|
@jail = false
|
|
26
26
|
@allowed_hosts = nil
|
|
27
27
|
@workdir = nil
|
|
28
|
+
@history = nil
|
|
28
29
|
end
|
|
29
30
|
|
|
30
31
|
def run
|
|
@@ -56,7 +57,7 @@ module Miniswen
|
|
|
56
57
|
agent = Agent.new(model:, reporter:, environment:, **options)
|
|
57
58
|
|
|
58
59
|
begin
|
|
59
|
-
result = agent.run(instruction)
|
|
60
|
+
result = agent.run(instruction, history: @history && Agent::Result.from_h(JSON.parse(@history)))
|
|
60
61
|
rescue StandardError => e
|
|
61
62
|
write_results(agent.partial_result(error_message(e)))
|
|
62
63
|
raise
|
|
@@ -105,7 +106,7 @@ module Miniswen
|
|
|
105
106
|
|
|
106
107
|
def parse_args!
|
|
107
108
|
parser = OptionParser.new do |opts|
|
|
108
|
-
opts.banner = "Usage: miniswen -m MODEL -p INSTRUCTION [...options]"
|
|
109
|
+
opts.banner = "Usage: miniswen -m MODEL (-p INSTRUCTION | --continue-from PATH) [...options]"
|
|
109
110
|
|
|
110
111
|
opts.on("-m MODEL", "--model=MODEL", String,
|
|
111
112
|
"LLM to use (litellm format, e.g.: openrouter/openai/gpt-5.6-luna") do |v|
|
|
@@ -116,6 +117,10 @@ module Miniswen
|
|
|
116
117
|
@instruction = File.file?(v) ? File.read(v) : v
|
|
117
118
|
end
|
|
118
119
|
|
|
120
|
+
opts.on("--continue-from=PATH", String, "Continue the session saved by --results-path at PATH (no -p needed)") do |v|
|
|
121
|
+
@history = File.read(v)
|
|
122
|
+
end
|
|
123
|
+
|
|
119
124
|
opts.on("--max-steps=STEPS", Integer, "Max steps count") do |v|
|
|
120
125
|
options[:max_steps] = v
|
|
121
126
|
end
|
|
@@ -197,7 +202,7 @@ module Miniswen
|
|
|
197
202
|
return if @refresh_registry
|
|
198
203
|
|
|
199
204
|
raise "Use -m to specify the model" unless @model
|
|
200
|
-
raise "Please, provide instructions via -p option" unless @instruction
|
|
205
|
+
raise "Please, provide instructions via -p option" unless @instruction || @history
|
|
201
206
|
end
|
|
202
207
|
end
|
|
203
208
|
end
|
|
@@ -20,7 +20,7 @@ module Miniswen
|
|
|
20
20
|
env&.each { |key, value| argv += [ "--env", "#{key}=#{value}" ] }
|
|
21
21
|
argv << id
|
|
22
22
|
argv += [ "timeout", timeout.ceil.to_s ] if timeout&.positive?
|
|
23
|
-
argv += [ "
|
|
23
|
+
argv += [ "bash", "-c", command ]
|
|
24
24
|
|
|
25
25
|
Open3.popen2e(*argv) do |stdin, pipe, wait|
|
|
26
26
|
stdin.close
|