lemans 1.3.5 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +21 -0
- data/README.md +17 -2
- data/exe/lemans-remote +141 -3
- data/exe/lemans-viewer +962 -0
- data/lib/lemans/agent.rb +4 -1
- data/lib/lemans/agents/miniswen.rb +10 -5
- data/lib/lemans/agents/miniswen_installed.rb +23 -5
- data/lib/lemans/agents/nop.rb +1 -1
- data/lib/lemans/agents/oracle.rb +1 -1
- data/lib/lemans/agents.rb +4 -2
- data/lib/lemans/cli/templates/bench/bench.yml +7 -0
- data/lib/lemans/cli.rb +85 -31
- data/lib/lemans/config/verifier.rb +10 -2
- data/lib/lemans/environments/docker/proxy/Dockerfile +4 -0
- data/lib/lemans/environments/docker/proxy/README.md +49 -0
- data/lib/lemans/environments/docker/proxy/server.rb +79 -0
- data/lib/lemans/environments/docker.rb +79 -16
- data/lib/lemans/result.rb +52 -1
- data/lib/lemans/runner/task.rb +9 -5
- data/lib/lemans/runner.rb +45 -2
- data/lib/lemans/store.rb +6 -0
- data/lib/lemans/stores/fs.rb +16 -4
- data/lib/lemans/trial/patch.rb +21 -0
- data/lib/lemans/trial/setup.rb +1 -0
- data/lib/lemans/trial.rb +109 -41
- data/lib/lemans/version.rb +1 -1
- data/lib/lemans.rb +2 -0
- data/lib/miniswen/agent.rb +65 -18
- data/lib/miniswen/cli.rb +20 -4
- data/lib/miniswen/environment/docker.rb +1 -1
- data/lib/miniswen/jail/proxy.rb +62 -0
- data/lib/miniswen/jail.rb +36 -9
- data/lib/miniswen/local.rb +6 -7
- data/lib/miniswen/testing.rb +7 -1
- data/lib/miniswen/version.rb +1 -1
- metadata +49 -1
data/lib/lemans/agent.rb
CHANGED
|
@@ -15,6 +15,9 @@ module Lemans
|
|
|
15
15
|
|
|
16
16
|
attr_reader :profile, :model
|
|
17
17
|
|
|
18
|
+
# Whether #run can go on from the history (raw result) of an interrupted run
|
|
19
|
+
def self.recoverable? = false
|
|
20
|
+
|
|
18
21
|
def initialize(profile:, model: nil)
|
|
19
22
|
@profile = profile
|
|
20
23
|
|
|
@@ -28,7 +31,7 @@ module Lemans
|
|
|
28
31
|
def install(_task, _environment) = nil
|
|
29
32
|
|
|
30
33
|
# Run the task.
|
|
31
|
-
def run(task, environment)
|
|
34
|
+
def run(task, environment, history: nil)
|
|
32
35
|
raise NotImplementedError
|
|
33
36
|
end
|
|
34
37
|
|
|
@@ -21,9 +21,13 @@ module Lemans
|
|
|
21
21
|
cost_limit: :cost_ceiling_reached
|
|
22
22
|
}.freeze
|
|
23
23
|
|
|
24
|
-
def
|
|
25
|
-
|
|
24
|
+
def self.recoverable? = true
|
|
25
|
+
|
|
26
|
+
def run(task, environment, history: nil)
|
|
27
|
+
history &&= ::Miniswen::Agent::Result.from_h(JSON.parse(history))
|
|
28
|
+
run_result = obtain_result(task, environment, history)
|
|
26
29
|
trajectory = trajectory_for(run_result)
|
|
30
|
+
raw_result = raw_result_for(run_result)
|
|
27
31
|
|
|
28
32
|
# A failed model call is still an answer: the trial saves the
|
|
29
33
|
# trajectory as evidence before failing.
|
|
@@ -39,16 +43,17 @@ module Lemans
|
|
|
39
43
|
|
|
40
44
|
private
|
|
41
45
|
|
|
42
|
-
def obtain_result(task, environment)
|
|
46
|
+
def obtain_result(task, environment, history)
|
|
43
47
|
agent = agent_for(environment)
|
|
44
48
|
begin
|
|
45
|
-
agent.run(task.instruction)
|
|
49
|
+
agent.run(task.instruction, history:)
|
|
46
50
|
rescue InfrastructureError, ::Miniswen::InfrastructureError => e
|
|
47
51
|
agent.partial_result(e.message)
|
|
48
52
|
end
|
|
49
53
|
end
|
|
50
54
|
|
|
51
|
-
|
|
55
|
+
# The run as miniswen records it, signatures included: what a recovery continues from
|
|
56
|
+
def raw_result_for(run_result) = JSON.generate(run_result.to_h)
|
|
52
57
|
|
|
53
58
|
def agent_for(environment)
|
|
54
59
|
raise ConfigError, "miniswen needs a model to drive" if model.to_s.empty?
|
|
@@ -13,6 +13,7 @@ module Lemans
|
|
|
13
13
|
class MiniswenInstalled < Miniswen
|
|
14
14
|
NAME = "miniswen-installed"
|
|
15
15
|
RESULTS_PATH = "/tmp/lemans-miniswen.result.json"
|
|
16
|
+
HISTORY_PATH = "/tmp/lemans-miniswen.history.json"
|
|
16
17
|
INSTALL_TIMEOUT_SEC = 300
|
|
17
18
|
# The CLI enforces max-time itself, but between steps only: a command
|
|
18
19
|
# started just before the deadline runs to its own exec timeout first,
|
|
@@ -31,8 +32,9 @@ module Lemans
|
|
|
31
32
|
|
|
32
33
|
# An in-sandbox run self-reports: everything but the verifier's reward
|
|
33
34
|
# comes from a file the sandbox wrote.
|
|
34
|
-
def obtain_result(task, environment)
|
|
35
|
-
|
|
35
|
+
def obtain_result(task, environment, history)
|
|
36
|
+
upload_history(environment, history) if history
|
|
37
|
+
run = environment.exec(command_for(task, history:), timeout: outer_timeout, env: provider_env(environment))
|
|
36
38
|
|
|
37
39
|
begin
|
|
38
40
|
Tempfile.create(%w[miniswen .result.json]) do |file|
|
|
@@ -47,7 +49,7 @@ module Lemans
|
|
|
47
49
|
end
|
|
48
50
|
end
|
|
49
51
|
|
|
50
|
-
|
|
52
|
+
def raw_result_for(_run_result) = @raw_result
|
|
51
53
|
|
|
52
54
|
def outer_timeout = profile.timeout + profile.exec_timeout + EXEC_SLACK_SEC
|
|
53
55
|
|
|
@@ -59,14 +61,30 @@ module Lemans
|
|
|
59
61
|
raise ConfigError, "miniswen-installed: #{e.message}"
|
|
60
62
|
end
|
|
61
63
|
|
|
62
|
-
def
|
|
64
|
+
def allowed_hosts
|
|
65
|
+
policy = profile.environment.network
|
|
66
|
+
policy.mode == "allowlist" ? policy.domains : []
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
def upload_history(environment, history)
|
|
70
|
+
Tempfile.create(%w[miniswen .history.json]) do |file|
|
|
71
|
+
file.write(JSON.generate(history.to_h))
|
|
72
|
+
file.flush
|
|
73
|
+
environment.upload(file.path, HISTORY_PATH)
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def command_for(task, history: nil)
|
|
63
78
|
argv = [ "miniswen", "-q", "--no-refresh-registry", "--jail",
|
|
64
|
-
"-m", model.to_s,
|
|
79
|
+
"-m", model.to_s,
|
|
80
|
+
*(history ? [ "--continue-from", HISTORY_PATH ] : [ "-p", task.instruction ]),
|
|
65
81
|
"--results-path", RESULTS_PATH,
|
|
66
82
|
"--max-steps", profile.step_limit, "--max-time", profile.timeout.to_i,
|
|
67
83
|
"--exec-timeout", profile.exec_timeout.to_i,
|
|
68
84
|
"--max-output-tokens", profile.max_output_tokens ]
|
|
69
85
|
argv += [ "--max-cost", profile.cost_limit.to_i ] if profile.cost_limit
|
|
86
|
+
argv += [ "--workdir", task.environment.workdir ]
|
|
87
|
+
argv += [ "--allow-hosts", allowed_hosts.join(",") ] if allowed_hosts.any?
|
|
70
88
|
argv.map { Shellwords.escape(it.to_s) }.join(" ")
|
|
71
89
|
end
|
|
72
90
|
end
|
data/lib/lemans/agents/nop.rb
CHANGED
data/lib/lemans/agents/oracle.rb
CHANGED
|
@@ -13,7 +13,7 @@ module Lemans
|
|
|
13
13
|
ENTRYPOINT = "solve.sh"
|
|
14
14
|
PATCH = "solution.patch"
|
|
15
15
|
|
|
16
|
-
def run(task, environment)
|
|
16
|
+
def run(task, environment, history: nil)
|
|
17
17
|
files = task.solution_files
|
|
18
18
|
if files.empty?
|
|
19
19
|
raise ConfigError, "#{task.name}: no solution/ to run — the oracle has nothing to prove" if
|
data/lib/lemans/agents.rb
CHANGED
|
@@ -11,11 +11,13 @@ module Lemans
|
|
|
11
11
|
"miniswen-installed" => "MiniswenInstalled"
|
|
12
12
|
}.freeze
|
|
13
13
|
|
|
14
|
-
def self.build(name, profile:, model: nil)
|
|
14
|
+
def self.build(name, profile:, model: nil) = lookup(name).new(profile: profile, model: model)
|
|
15
|
+
|
|
16
|
+
def self.lookup(name)
|
|
15
17
|
constant = REGISTRY[name] or
|
|
16
18
|
raise ConfigError, "unknown agent #{name.inspect} (known: #{REGISTRY.keys.join(", ")})"
|
|
17
19
|
|
|
18
|
-
const_get(constant)
|
|
20
|
+
const_get(constant)
|
|
19
21
|
end
|
|
20
22
|
end
|
|
21
23
|
end
|
|
@@ -74,6 +74,7 @@ agent:
|
|
|
74
74
|
|
|
75
75
|
# The sandbox network while the agent works: just enough to reach the model
|
|
76
76
|
# Use mode: none when using a local agent (miniswen)
|
|
77
|
+
# With mode allowlist, agent commands reach these hosts through the jail's proxy
|
|
77
78
|
environment:
|
|
78
79
|
network:
|
|
79
80
|
mode: allowlist
|
|
@@ -87,6 +88,12 @@ verifier:
|
|
|
87
88
|
# bare command list). Runs after the agent finished, before the tests:
|
|
88
89
|
# setup: [gem install debug]
|
|
89
90
|
|
|
91
|
+
# Grading runs sealed unless the task names hosts, say to install the gems a solution adds:
|
|
92
|
+
# environment:
|
|
93
|
+
# network:
|
|
94
|
+
# mode: allowlist
|
|
95
|
+
# hosts: [index.rubygems.org, rubygems.org]
|
|
96
|
+
|
|
90
97
|
# A command that must pass before the graded checks run, e.g. the app's own
|
|
91
98
|
# test suite:
|
|
92
99
|
# preverify: bin/rails test
|
data/lib/lemans/cli.rb
CHANGED
|
@@ -78,49 +78,80 @@ module Lemans
|
|
|
78
78
|
return
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
81
|
+
execute(runner, store, tasks)
|
|
82
|
+
rescue ConfigError => e
|
|
83
|
+
raise Thor::Error, "lemans: #{e.message}"
|
|
84
|
+
rescue Interrupt
|
|
85
|
+
say ""
|
|
86
|
+
exit 130
|
|
87
|
+
end
|
|
88
88
|
|
|
89
|
-
|
|
89
|
+
desc "restart RUN...", "Continue failed multistep runs from their last settled step in new runs"
|
|
90
|
+
long_desc <<~DESC
|
|
91
|
+
RUN is a run directory or a trial id; every run named restarts under the same options. The new run replays the settled steps' agent patches in a
|
|
92
|
+
fresh sandbox and starts at the next step; the failed run stays as it is. --recover also replays
|
|
93
|
+
the failed step's partial patch and lets the agent go on from its history; --reverify replays the
|
|
94
|
+
graded step's patch and runs its verification again.
|
|
95
|
+
DESC
|
|
96
|
+
option :bench, default: ".", desc: "Directory holding bench.yml"
|
|
97
|
+
option :runs_dir, default: "./runs", desc: "Directory holding the runs; the new runs go there too"
|
|
98
|
+
option :concurrency, type: :numeric, aliases: "-c", desc: "Restarts in flight at once (default: the bench's)"
|
|
99
|
+
option :backend, enum: Environments::BACKENDS.keys, desc: "Sandbox backend (default: daytona)"
|
|
100
|
+
option :max_output_tokens, type: :numeric, banner: "TOKENS",
|
|
101
|
+
desc: "Cap the agent's output per model call (default: the provider's)"
|
|
102
|
+
option :recover, type: :boolean, default: false,
|
|
103
|
+
desc: "Continue the failed step's agent session from its saved history"
|
|
104
|
+
option :reverify, type: :boolean, default: false,
|
|
105
|
+
desc: "Grade the last verified step again (with the current tests) and go on from there"
|
|
106
|
+
option :allow_scored, type: :boolean, default: false, desc: "Restart a scored run (--reverify always may)"
|
|
107
|
+
def restart(*runs)
|
|
108
|
+
raise Thor::Error, "lemans: name the run(s) to restart" if runs.empty?
|
|
109
|
+
raise Thor::Error, "lemans: --recover and --reverify exclude each other" if options[:recover] && options[:reverify]
|
|
90
110
|
|
|
91
|
-
|
|
111
|
+
Miniswen.refresh_registry!
|
|
92
112
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
113
|
+
store = Stores::FS.new(options[:runs_dir], filterer: SecretsFilter.default)
|
|
114
|
+
ids = runs.map { File.basename(it) }.uniq
|
|
115
|
+
found = store.fetch.select { ids.include?(it.id) }.to_h { [ it.id, it ] }
|
|
116
|
+
missing = ids - found.keys
|
|
117
|
+
raise Thor::Error, "lemans: no run #{missing.join(", ")} under #{options[:runs_dir]}" if missing.any?
|
|
96
118
|
|
|
97
|
-
|
|
98
|
-
|
|
119
|
+
sources = found.values_at(*ids)
|
|
120
|
+
|
|
121
|
+
# The board lays out every model the runs used, as wide as their highest attempt
|
|
122
|
+
config = Config.load_file(options[:bench])
|
|
123
|
+
config.load_options(**options.transform_keys(&:to_sym), model: sources.map(&:model).uniq,
|
|
124
|
+
attempts: sources.filter_map(&:index).max)
|
|
125
|
+
|
|
126
|
+
tasks = filter_tasks(config.tasks, name: sources.map(&:task).uniq)
|
|
127
|
+
|
|
128
|
+
mode = (:recover if options[:recover]) || (:reverify if options[:reverify])
|
|
129
|
+
runner = Runner.new(config, tasks, store:, restarts: sources, restart_mode: mode, allow_scored: options[:allow_scored])
|
|
130
|
+
|
|
131
|
+
execute(runner, store, tasks)
|
|
99
132
|
rescue ConfigError => e
|
|
100
133
|
raise Thor::Error, "lemans: #{e.message}"
|
|
101
134
|
rescue Interrupt
|
|
102
135
|
say ""
|
|
103
136
|
exit 130
|
|
104
|
-
ensure
|
|
105
|
-
reporter&.stop
|
|
106
137
|
end
|
|
107
138
|
|
|
108
|
-
desc "clobber", "Delete run results"
|
|
109
|
-
option :runs_dir, default: "./runs", desc: "Directory holding run directories"
|
|
139
|
+
desc "clobber [RUNS_DIR]", "Delete run results"
|
|
140
|
+
option :runs_dir, default: "./runs", desc: "Directory holding run directories (or pass it as RUNS_DIR)"
|
|
110
141
|
option :task, desc: "Only these tasks' runs", repeatable: true
|
|
111
142
|
option :ttl, desc: "Only runs older than this (10m, 2h, 1d)"
|
|
112
143
|
option :invalid, type: :boolean, default: false, desc: "Only runs that measured nothing (invalid or unreadable)"
|
|
113
144
|
option :force, type: :boolean, default: false, aliases: "-f", desc: "Delete without asking"
|
|
114
|
-
def clobber
|
|
115
|
-
store = Stores::FS.new(
|
|
145
|
+
def clobber(runs_dir = options[:runs_dir])
|
|
146
|
+
store = Stores::FS.new(runs_dir)
|
|
116
147
|
clobber = Clobber.new(store, tasks: options[:task], ttl: options[:ttl], invalid: options[:invalid])
|
|
117
148
|
|
|
118
149
|
doomed = clobber.matches
|
|
119
|
-
return say "lemans: nothing to clobber under #{
|
|
150
|
+
return say "lemans: nothing to clobber under #{runs_dir}" if doomed.empty?
|
|
120
151
|
|
|
121
152
|
unless options[:force]
|
|
122
153
|
doomed.each { say it.id }
|
|
123
|
-
return say "lemans: nothing deleted" unless yes?("Delete #{doomed.size} run(s) under #{
|
|
154
|
+
return say "lemans: nothing deleted" unless yes?("Delete #{doomed.size} run(s) under #{runs_dir}? [y/N]")
|
|
124
155
|
end
|
|
125
156
|
|
|
126
157
|
removed = clobber.execute!
|
|
@@ -129,15 +160,15 @@ module Lemans
|
|
|
129
160
|
raise Thor::Error, "lemans: #{e.message}"
|
|
130
161
|
end
|
|
131
162
|
|
|
132
|
-
desc "regrade", "Re-grade stored results from their checks.json after a verification_test.rb grading change"
|
|
163
|
+
desc "regrade [RUNS_DIR]", "Re-grade stored results from their checks.json after a verification_test.rb grading change"
|
|
133
164
|
option :bench, default: ".", desc: "Directory holding bench.yml"
|
|
134
165
|
option :task, desc: "Re-grade these tasks' runs", repeatable: true, required: true
|
|
135
|
-
option :runs_dir, default: "./runs", desc: "Directory holding run directories"
|
|
166
|
+
option :runs_dir, default: "./runs", desc: "Directory holding run directories (or pass it as RUNS_DIR)"
|
|
136
167
|
option :mapping, banner: "PATH",
|
|
137
168
|
desc: "Grade by this checks.json-shaped file (every check `fail` or `fail (allowed)`, plus `grading`) " \
|
|
138
169
|
"instead of reading verification_test.rb"
|
|
139
|
-
def regrade
|
|
140
|
-
store = Stores::FS.new(
|
|
170
|
+
def regrade(runs_dir = options[:runs_dir])
|
|
171
|
+
store = Stores::FS.new(runs_dir)
|
|
141
172
|
tasks = filter_tasks(Config.load_file(options[:bench]).tasks, name: options[:task])
|
|
142
173
|
raise Thor::Error, "lemans: --mapping re-grades one task at a time" if options[:mapping] && tasks.size > 1
|
|
143
174
|
|
|
@@ -153,14 +184,14 @@ module Lemans
|
|
|
153
184
|
end
|
|
154
185
|
|
|
155
186
|
say ""
|
|
156
|
-
say_status :report, "collecting results from #{
|
|
187
|
+
say_status :report, "collecting results from #{runs_dir}", :cyan
|
|
157
188
|
print_report Report.load(store, names: tasks.map(&:name))
|
|
158
189
|
rescue ConfigError => e
|
|
159
190
|
raise Thor::Error, "lemans: #{e.message}"
|
|
160
191
|
end
|
|
161
192
|
|
|
162
|
-
desc "report", "Summarize run results as a table or CSV"
|
|
163
|
-
option :runs_dir, default: "runs", desc: "Directory holding run directories"
|
|
193
|
+
desc "report [RUNS_DIR]", "Summarize run results as a table or CSV"
|
|
194
|
+
option :runs_dir, default: "runs", desc: "Directory holding run directories (or pass it as RUNS_DIR)"
|
|
164
195
|
option :tag, desc: "Only runs whose result carries this tag", repeatable: true
|
|
165
196
|
option :task, desc: "Only these tasks' runs", repeatable: true
|
|
166
197
|
option :metadata, banner: "KEY:VALUE", desc: "Only runs whose task metadata has this value (every pair must match)",
|
|
@@ -169,8 +200,8 @@ module Lemans
|
|
|
169
200
|
option :aggregate, aliases: "-A", banner: "COLUMNS", lazy_default: "task-model",
|
|
170
201
|
desc: "Group results by 1-3 dash-joined columns (task, agent, model)"
|
|
171
202
|
option :sort, aliases: "-S", banner: "COLUMN", desc: "Sort by a column"
|
|
172
|
-
def report
|
|
173
|
-
store = Stores::FS.new(
|
|
203
|
+
def report(runs_dir = options[:runs_dir])
|
|
204
|
+
store = Stores::FS.new(runs_dir)
|
|
174
205
|
results = Report.load(store, tags: options[:tag], names: options[:task],
|
|
175
206
|
metadata: Report.metadata_filter(options[:metadata]))
|
|
176
207
|
raise Thor::Error, "lemans: no matching results found" if results.empty?
|
|
@@ -184,6 +215,29 @@ module Lemans
|
|
|
184
215
|
|
|
185
216
|
private
|
|
186
217
|
|
|
218
|
+
def execute(runner, store, tasks)
|
|
219
|
+
reporter =
|
|
220
|
+
if interactive?
|
|
221
|
+
BoardReporter.new(tasks: tasks.map(&:name), models: runner.config.models,
|
|
222
|
+
attempts: runner.config.attempts, total: runner.attempts.size)
|
|
223
|
+
else
|
|
224
|
+
ProgressReporter.new(shell:, tasks: tasks.map(&:name))
|
|
225
|
+
end
|
|
226
|
+
|
|
227
|
+
reporter.start
|
|
228
|
+
|
|
229
|
+
summary = runner.run(reporter)
|
|
230
|
+
|
|
231
|
+
say ""
|
|
232
|
+
say_status :report, "collecting results from #{options[:runs_dir]}", :cyan
|
|
233
|
+
print_report Report.load(store)
|
|
234
|
+
|
|
235
|
+
exit 130 if summary.status == :interrupted
|
|
236
|
+
exit 1 if summary.status == :invalid
|
|
237
|
+
ensure
|
|
238
|
+
reporter&.stop
|
|
239
|
+
end
|
|
240
|
+
|
|
187
241
|
def filter_tasks(tasks, tags: nil, name: nil)
|
|
188
242
|
tasks = tasks.dup
|
|
189
243
|
|
|
@@ -26,6 +26,10 @@ module Lemans
|
|
|
26
26
|
conf.logs_dir = absolute_path!(data["logs_dir"]) if data["logs_dir"]
|
|
27
27
|
conf.verification = root.join(data["verification"]) if data["verification"]
|
|
28
28
|
|
|
29
|
+
if (network_data = data.dig("environment", "network"))
|
|
30
|
+
conf.environment = Environment.new(network: NetworkPolicy.from_config(network_data))
|
|
31
|
+
end
|
|
32
|
+
|
|
29
33
|
conf
|
|
30
34
|
end
|
|
31
35
|
|
|
@@ -42,7 +46,9 @@ module Lemans
|
|
|
42
46
|
end
|
|
43
47
|
end
|
|
44
48
|
|
|
45
|
-
attr_accessor :timeout, :setup, :command, :preverify, :restore_paths, :logs_dir, :verification
|
|
49
|
+
attr_accessor :timeout, :setup, :command, :preverify, :restore_paths, :logs_dir, :verification, :environment
|
|
50
|
+
|
|
51
|
+
Environment = Struct.new(:network, keyword_init: true)
|
|
46
52
|
|
|
47
53
|
def initialize
|
|
48
54
|
@timeout = 10 * 60
|
|
@@ -52,6 +58,7 @@ module Lemans
|
|
|
52
58
|
@restore_paths = []
|
|
53
59
|
@logs_dir = "/logs/verifier"
|
|
54
60
|
@verification = nil
|
|
61
|
+
@environment = Environment.new(network: NetworkPolicy.new("none"))
|
|
55
62
|
end
|
|
56
63
|
|
|
57
64
|
def reward_path = "#{logs_dir.chomp("/")}/reward.txt"
|
|
@@ -64,7 +71,8 @@ module Lemans
|
|
|
64
71
|
"preverify" => preverify,
|
|
65
72
|
"restore" => restore_paths,
|
|
66
73
|
"logs_dir" => logs_dir,
|
|
67
|
-
"verification" => verification&.to_s
|
|
74
|
+
"verification" => verification&.to_s,
|
|
75
|
+
"environment" => { "network" => environment.network.to_h }
|
|
68
76
|
}.compact
|
|
69
77
|
end
|
|
70
78
|
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
# Allowlist proxy
|
|
2
|
+
|
|
3
|
+
This is an HTTP proxy that lets clients reach the allowed hosts only. The Docker backend runs it for the `allowlist` network mode.
|
|
4
|
+
|
|
5
|
+
The proxy supports two request types:
|
|
6
|
+
|
|
7
|
+
- `CONNECT host:port` tunnels (https).
|
|
8
|
+
- Absolute-form requests (`GET http://host/path`) for plain http.
|
|
9
|
+
|
|
10
|
+
The proxy refuses a host that is not on the list with `403 Forbidden`. Host names must match exactly. IP ranges are not supported.
|
|
11
|
+
|
|
12
|
+
## How the Docker backend uses it
|
|
13
|
+
|
|
14
|
+
1. Lemans puts the task container on an internal network. An internal network has no route out.
|
|
15
|
+
2. Lemans connects the proxy container to the internal network and to `bridge`.
|
|
16
|
+
3. Lemans sets `http_proxy` and `https_proxy` for each command in the task container.
|
|
17
|
+
|
|
18
|
+
A tool that ignores the proxy variables cannot reach the network.
|
|
19
|
+
|
|
20
|
+
When the allowed hosts change, lemans replaces the proxy container.
|
|
21
|
+
|
|
22
|
+
## Run it standalone
|
|
23
|
+
|
|
24
|
+
Build the image:
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
docker build --tag lemans-proxy lib/lemans/environments/docker/proxy
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Start the proxy. Give the allowed hosts as arguments:
|
|
31
|
+
|
|
32
|
+
```sh
|
|
33
|
+
docker run --rm --publish 3128:3128 lemans-proxy rubygems.org index.rubygems.org
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Send requests through it:
|
|
37
|
+
|
|
38
|
+
```sh
|
|
39
|
+
curl --proxy http://localhost:3128 https://rubygems.org # allowed
|
|
40
|
+
curl --proxy http://localhost:3128 https://example.com # 403 Forbidden
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
To run it without Docker, use Ruby 3.4 or later:
|
|
44
|
+
|
|
45
|
+
```sh
|
|
46
|
+
ruby lib/lemans/environments/docker/proxy/server.rb rubygems.org
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
The `PORT` variable changes the listen port. The default port is 3128.
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "socket"
|
|
4
|
+
require "uri"
|
|
5
|
+
|
|
6
|
+
# An HTTP proxy that lets clients reach the allowed hosts only:
|
|
7
|
+
# CONNECT tunnels for https, absolute-form requests for plain http.
|
|
8
|
+
class AllowlistProxy
|
|
9
|
+
HOP_HEADERS = /\A(?:connection|keep-alive|proxy-[\w-]+):/i
|
|
10
|
+
|
|
11
|
+
def initialize(hosts, port: 3128)
|
|
12
|
+
@hosts = hosts.map(&:downcase)
|
|
13
|
+
@port = port
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def run
|
|
17
|
+
server = TCPServer.new("0.0.0.0", @port)
|
|
18
|
+
$stdout.puts "listening on #{@port}: #{@hosts.join(", ")}"
|
|
19
|
+
$stdout.flush
|
|
20
|
+
Thread.report_on_exception = false
|
|
21
|
+
loop { Thread.new(server.accept) { handle(it) } }
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
private
|
|
25
|
+
|
|
26
|
+
def handle(client)
|
|
27
|
+
head = client.gets("\r\n\r\n") or return
|
|
28
|
+
request_line, *headers = head.split("\r\n")
|
|
29
|
+
verb, target, version = request_line.split(" ", 3)
|
|
30
|
+
|
|
31
|
+
verb == "CONNECT" ? tunnel(client, target) : forward(client, verb, target, version, headers)
|
|
32
|
+
rescue SystemCallError, IOError, SocketError, URI::Error
|
|
33
|
+
client.write("HTTP/1.1 502 Bad Gateway\r\n\r\n") rescue nil # rubocop:disable Style/RescueModifier
|
|
34
|
+
ensure
|
|
35
|
+
client.close
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def tunnel(client, target)
|
|
39
|
+
host, port = target.split(":", 2)
|
|
40
|
+
return deny(client) unless allowed?(host)
|
|
41
|
+
|
|
42
|
+
upstream = Socket.tcp(host, port.to_i, connect_timeout: 10)
|
|
43
|
+
client.write("HTTP/1.1 200 Connection Established\r\n\r\n")
|
|
44
|
+
pump(client, upstream)
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# One request per connection: the upstream is told to close after the response.
|
|
48
|
+
def forward(client, verb, target, version, headers)
|
|
49
|
+
uri = URI(target)
|
|
50
|
+
return deny(client) unless uri.is_a?(URI::HTTP) && allowed?(uri.host)
|
|
51
|
+
|
|
52
|
+
upstream = Socket.tcp(uri.host, uri.port, connect_timeout: 10)
|
|
53
|
+
headers = headers.grep_v(HOP_HEADERS)
|
|
54
|
+
upstream.write("#{verb} #{uri.request_uri} #{version}\r\n#{headers.join("\r\n")}\r\nConnection: close\r\n\r\n")
|
|
55
|
+
pump(client, upstream)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def allowed?(host) = @hosts.include?(host.to_s.downcase)
|
|
59
|
+
|
|
60
|
+
def deny(client) = client.write("HTTP/1.1 403 Forbidden\r\n\r\n")
|
|
61
|
+
|
|
62
|
+
def pump(client, upstream)
|
|
63
|
+
Thread.new { copy(client, upstream) }
|
|
64
|
+
copy(upstream, client)
|
|
65
|
+
ensure
|
|
66
|
+
upstream.close
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# readpartial drains what gets already buffered, so a request body is not lost.
|
|
70
|
+
def copy(from, to)
|
|
71
|
+
loop { to.write(from.readpartial(65_536)) }
|
|
72
|
+
rescue EOFError, IOError, SystemCallError
|
|
73
|
+
nil
|
|
74
|
+
ensure
|
|
75
|
+
to.close_write rescue nil # rubocop:disable Style/RescueModifier
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
AllowlistProxy.new(ARGV, port: Integer(ENV.fetch("PORT", "3128"))).run if $PROGRAM_NAME == __FILE__
|