aimpg 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aimpg-0.4.2/.github/workflows/test.yml +34 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/PKG-INFO +4 -4
- {aimpg-0.4.0 → aimpg-0.4.2}/README.md +3 -3
- aimpg-0.4.2/TODOS.md +9 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/cli.py +2 -1
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/fake_agent.py +11 -6
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/proxy.py +18 -2
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/run.py +12 -0
- aimpg-0.4.2/aimpg/replay/sandbox.py +324 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/verify.py +103 -7
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/workspace.py +4 -1
- {aimpg-0.4.0 → aimpg-0.4.2}/docs/STRATEGY.md +1 -1
- aimpg-0.4.2/docs/designs/scoreboard-brief.md +46 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/pyproject.toml +1 -1
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_e2e.py +3 -3
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_sandbox.py +33 -14
- aimpg-0.4.2/tests/test_sandbox_required.py +9 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_verify.py +52 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/uv.lock +1 -1
- aimpg-0.4.0/.github/workflows/test.yml +0 -17
- aimpg-0.4.0/TODOS.md +0 -13
- aimpg-0.4.0/aimpg/replay/sandbox.py +0 -175
- {aimpg-0.4.0 → aimpg-0.4.2}/.github/workflows/publish.yml +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/.gitignore +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/LICENSE +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/__init__.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/attribution.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cli.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/coach.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/codex_logs.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cost.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/cursor_usage.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/durable.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/energy.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/equivalence.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/equivalences.json +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/factors.json +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/gitkept.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/ledger.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/live.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/logs.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/model.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/prices.json +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/receipt.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/__init__.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/hints.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/picker.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/select.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/setups.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/replay/stats.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/aimpg/share.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/docs/designs/aimpg-design.md +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/evals/attribution_eval.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/evals/label.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/RESULTS.md +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/allowlist_proxy.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/agent/make_profile.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/RESULTS.md +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/allowlist_proxy.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/spikes/egress/replay.sb +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/conftest.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/gitrepo.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_attribution.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_coach.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_codex_logs.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_cost_and_equivalence.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_cursor_usage.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_durable.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_energy.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_eval_scoring.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_gitkept.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_live.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_logs.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_perf.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_cli.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_hints.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_picker.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_proxy.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_replay_stats.py +0 -0
- {aimpg-0.4.0 → aimpg-0.4.2}/tests/test_share.py +0 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: test
|
|
2
|
+
on:
|
|
3
|
+
push:
|
|
4
|
+
branches: [main, "linux-*"]
|
|
5
|
+
pull_request:
|
|
6
|
+
jobs:
|
|
7
|
+
test:
|
|
8
|
+
strategy:
|
|
9
|
+
matrix:
|
|
10
|
+
os: [ubuntu-latest, macos-latest]
|
|
11
|
+
python: ["3.10", "3.13"]
|
|
12
|
+
runs-on: ${{ matrix.os }}
|
|
13
|
+
steps:
|
|
14
|
+
- uses: actions/checkout@v4
|
|
15
|
+
- uses: astral-sh/setup-uv@v6
|
|
16
|
+
- run: git config --global user.email ci@example.com && git config --global user.name CI
|
|
17
|
+
- run: uv run --python ${{ matrix.python }} --group dev pytest -q
|
|
18
|
+
linux-sandbox:
|
|
19
|
+
# Replays on Linux: bubblewrap + the AppArmor rule Ubuntu 24.04 needs for bwrap only
|
|
20
|
+
runs-on: ubuntu-24.04
|
|
21
|
+
steps:
|
|
22
|
+
- uses: actions/checkout@v4
|
|
23
|
+
- uses: astral-sh/setup-uv@v6
|
|
24
|
+
- run: sudo apt-get update -q && sudo apt-get install -y -q bubblewrap
|
|
25
|
+
- name: allow user namespaces for bwrap only
|
|
26
|
+
run: |
|
|
27
|
+
printf 'abi <abi/4.0>,\ninclude <tunables/global>\n\nprofile bwrap /usr/bin/bwrap flags=(unconfined) {\n userns,\n include if exists <local/bwrap>\n}\n' | sudo tee /etc/apparmor.d/bwrap
|
|
28
|
+
sudo apparmor_parser -r /etc/apparmor.d/bwrap
|
|
29
|
+
- run: git config --global user.email ci@example.com && git config --global user.name CI
|
|
30
|
+
- name: an SSH agent for the sandbox to fail to reach
|
|
31
|
+
run: ssh-agent -a /tmp/aimpg-ci-agent.sock && echo "SSH_AUTH_SOCK=/tmp/aimpg-ci-agent.sock" >> "$GITHUB_ENV"
|
|
32
|
+
- run: uv run --group dev pytest -q -rs tests/test_sandbox_required.py tests/test_replay_sandbox.py tests/test_replay_e2e.py
|
|
33
|
+
env:
|
|
34
|
+
AIMPG_REQUIRE_SANDBOX: "1"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: aimpg
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Real-world energy per solved task for AI coding agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/kumarganduri/aimpg
|
|
6
6
|
Project-URL: Issues, https://github.com/kumarganduri/aimpg/issues
|
|
@@ -113,7 +113,7 @@ VERDICT: NOT PROVEN
|
|
|
113
113
|
```
|
|
114
114
|
That's the real result on the author's Legwork repo: no saving shown, far from the advertised 60–90%. Your repo may differ, and now you can check.
|
|
115
115
|
|
|
116
|
-
Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math for free with `aimpg verify --check record.json
|
|
116
|
+
Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
|
|
117
117
|
|
|
118
118
|
Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
|
|
119
119
|
|
|
@@ -143,11 +143,11 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
|
|
|
143
143
|
|
|
144
144
|
**Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
|
|
145
145
|
|
|
146
|
-
**Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS
|
|
146
|
+
**Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a sandbox (Seatbelt on macOS, bubblewrap on Linux). The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
|
|
147
147
|
|
|
148
148
|
## Limits (honest list)
|
|
149
149
|
|
|
150
|
-
- Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only").
|
|
150
|
+
- Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Replays and `verify` need macOS or Linux and an API key (about $0.10–0.30 per run). On Linux install bubblewrap (`sudo apt install bubblewrap`); Ubuntu 24.04+ also needs a one-time AppArmor rule for bwrap, which aimpg prints if it's missing.
|
|
151
151
|
- Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
|
|
152
152
|
- Tips are upper bounds and overlap; they can't be added together.
|
|
153
153
|
- Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
|
|
@@ -93,7 +93,7 @@ VERDICT: NOT PROVEN
|
|
|
93
93
|
```
|
|
94
94
|
That's the real result on the author's Legwork repo: no saving shown, far from the advertised 60–90%. Your repo may differ, and now you can check.
|
|
95
95
|
|
|
96
|
-
Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math for free with `aimpg verify --check record.json
|
|
96
|
+
Every run writes a **record** (`~/.aimpg/verify/*.record.json`) holding versions, both setups, each run's tokens, $ and outcome, and the verdict, but never code, diffs or commit messages. Repo and commits are hashed unless you add `--public`. Anyone can recheck the math and the record's consistency for free with `aimpg verify --check record.json` (it can't prove the runs happened), and rerun a public record on their own machine with `aimpg verify --rerun record.json --repo <clone>`, which first shows every command, hook and prompt the record would run and asks before running them. Private records keep only fingerprints (SHA-256) of your prompts, CLAUDE.md, settings and commands, never the text. Records also list every redone attempt and excluded commit.
|
|
97
97
|
|
|
98
98
|
Same model: the energy test decides (it must hold at every corner of the energy ranges). Different models or agents: cost per solved task decides, because model sizes are secret. A challenger that solves more than 10 points fewer tasks is never "supported". Codex runs are measured from its own logs and priced from OpenAI's published prices. Codex logs in with an OpenAI **API key** (a ChatGPT subscription login can't be used inside the sandbox); the key is written to the run's throwaway folder and deleted when the run ends. Other commands are judged on solve rate and time only.
|
|
99
99
|
|
|
@@ -123,11 +123,11 @@ One row per AI-assisted commit: date, repo, energy range, CO₂ range and cost.
|
|
|
123
123
|
|
|
124
124
|
**Energy:** a physical formula, not a price proxy. Prefill compute for new input, one KV-cache re-read per output token (so long contexts cost more), plus datacenter overhead. Model sizes aren't public, so every number is a range. Overheads follow Google's full-stack measurement of a median Gemini prompt (chips, host CPU and memory, idle capacity, cooling ≈ 1.7× the chips alone), and a chat-sized prompt on a mid-size model comes out at 0.017–0.34 Wh, bracketing Google's disclosed 0.24 Wh ([arXiv 2508.15734](https://arxiv.org/pdf/2508.15734)). Sources for every factor are in [`aimpg/factors.json`](aimpg/factors.json), [`aimpg/prices.json`](aimpg/prices.json) and [`aimpg/equivalences.json`](aimpg/equivalences.json).
|
|
125
125
|
|
|
126
|
-
**Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a macOS
|
|
126
|
+
**Replays:** each run starts from the code just before your commit, in a fresh, history-free copy, inside a sandbox (Seatbelt on macOS, bubblewrap on Linux). The only network allowed is the model API, through an allowlisting proxy. Your own tests judge the result, and edited tests are always restored before judging. Every token count is cross-checked against Claude Code's own totals. Details are in [docs/designs/aimpg-design.md](docs/designs/aimpg-design.md).
|
|
127
127
|
|
|
128
128
|
## Limits (honest list)
|
|
129
129
|
|
|
130
|
-
- Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only").
|
|
130
|
+
- Claude Code and Codex CLI logs are read automatically. Cursor keeps token usage on its servers, so it needs its usage export (below), and that export doesn't say which folder the work was in: give `--cursor-repo` and requests are matched to your next own commit there by time alone (shown as "by time only"). Replays and `verify` need macOS or Linux and an API key (about $0.10–0.30 per run). On Linux install bubblewrap (`sudo apt install bubblewrap`); Ubuntu 24.04+ also needs a one-time AppArmor rule for bwrap, which aimpg prints if it's missing.
|
|
131
131
|
- Energy is an estimate with a wide range; dollars are close (within about 7% of Claude Code's own session totals on the author's logs).
|
|
132
132
|
- Tips are upper bounds and overlap; they can't be added together.
|
|
133
133
|
- Replays from commit messages alone are hard (12% solved on the author's repo); `--task-mode tests` shows the agent the tests, which makes tasks easier than real work but keeps comparisons fair.
|
aimpg-0.4.2/TODOS.md
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# TODOS
|
|
2
|
+
|
|
3
|
+
## Phase 2
|
|
4
|
+
|
|
5
|
+
The earlier TODOs (test leakage + memorization, statistics gate, network-limit spike) were resolved by the Phase 2 eng review (R3/R4, R7/R15/R16, spikes/egress + spikes/agent).
|
|
6
|
+
|
|
7
|
+
### ~~Linux replay sandbox~~ (done 2026-10-05)
|
|
8
|
+
bubblewrap backend in aimpg/replay/sandbox.py; the proxy is relayed into the sandbox through a Unix socket. CI's `linux-sandbox` job runs the escape and end-to-end tests on Ubuntu 24.04.
|
|
9
|
+
- **Still untested on Linux:** a real paid Claude Code or Codex run (CI has no API keys); only the stand-in agent has run there.
|
|
@@ -21,10 +21,11 @@ from aimpg.replay import run as runner
|
|
|
21
21
|
from aimpg.replay import select, stats
|
|
22
22
|
from aimpg.replay.proxy import AllowlistProxy
|
|
23
23
|
from aimpg.replay import picker
|
|
24
|
+
from aimpg.replay import sandbox
|
|
24
25
|
from aimpg.replay.setups import SETUPS, with_model
|
|
25
26
|
from aimpg.replay.workspace import Layout
|
|
26
27
|
|
|
27
|
-
ROOT =
|
|
28
|
+
ROOT = sandbox.TMP / "aimpg-replay" # outside $HOME; every agent profile denies it
|
|
28
29
|
STATE = Path.home() / ".aimpg" / "replay" # selections + results, unreadable to agents
|
|
29
30
|
DEFAULT_MODEL = "claude-sonnet-5-5"
|
|
30
31
|
DEFAULT_SETUPS = "claude-code,claude-code+terse,claude-code+rtk"
|
|
@@ -53,18 +53,23 @@ def transcript(cfg: Path, cwd: Path, model: str, n: int) -> tuple[dict, dict]:
|
|
|
53
53
|
|
|
54
54
|
|
|
55
55
|
def escape_attempts(cwd: Path, cfg: Path) -> list[str]:
|
|
56
|
-
"""Every attempt that SUCCEEDS is a sandbox bug.
|
|
56
|
+
"""Every attempt that SUCCEEDS is a sandbox bug.
|
|
57
|
+
|
|
58
|
+
Seeing a folder counts only if real content shows: on Linux, home and the
|
|
59
|
+
runs folder are empty private stand-ins, which is the sandbox working.
|
|
60
|
+
"""
|
|
57
61
|
succeeded = []
|
|
58
62
|
real_home = Path(pwd.getpwuid(os.getuid()).pw_dir)
|
|
59
|
-
|
|
63
|
+
own_run = cwd.parent.name
|
|
64
|
+
for target, allowed in ((real_home, set()), (real_home / ".ssh", set()), (cwd.parent.parent, {own_run})):
|
|
60
65
|
try:
|
|
61
|
-
os.listdir(target)
|
|
62
|
-
|
|
66
|
+
if set(os.listdir(target)) - allowed:
|
|
67
|
+
succeeded.append(f"read {target}")
|
|
63
68
|
except OSError:
|
|
64
69
|
pass
|
|
65
70
|
try:
|
|
66
|
-
|
|
67
|
-
succeeded.append("write
|
|
71
|
+
(real_home / "aimpg-escape-probe").write_text("x")
|
|
72
|
+
succeeded.append("write to home")
|
|
68
73
|
except OSError:
|
|
69
74
|
pass
|
|
70
75
|
probe = subprocess.run(["/usr/bin/curl", "-s", "-m", "5", "-o", "/dev/null", "-w", "%{http_code}", "--noproxy", "*", "https://github.com"], capture_output=True, text=True)
|
|
@@ -15,7 +15,11 @@ The allowlist changes per phase:
|
|
|
15
15
|
from __future__ import annotations
|
|
16
16
|
|
|
17
17
|
import asyncio
|
|
18
|
+
import shutil
|
|
19
|
+
import sys
|
|
20
|
+
import tempfile
|
|
18
21
|
import threading
|
|
22
|
+
from pathlib import Path
|
|
19
23
|
from dataclasses import dataclass, field
|
|
20
24
|
|
|
21
25
|
PYPI = frozenset({"pypi.org", "files.pythonhosted.org"})
|
|
@@ -33,6 +37,9 @@ class AllowlistProxy:
|
|
|
33
37
|
port: int = 0
|
|
34
38
|
upstream_port: int = 443 # tests point this at a local echo server
|
|
35
39
|
log: list[tuple[str, str]] = field(default_factory=list) # ("ALLOW" | "BLOCK", "host:port")
|
|
40
|
+
# Linux: also a Unix socket, bound into the sandbox (its private network can't reach host ports)
|
|
41
|
+
socket_path: Path | None = None
|
|
42
|
+
unix: bool = field(default_factory=lambda: sys.platform.startswith("linux"))
|
|
36
43
|
_loop: asyncio.AbstractEventLoop | None = None
|
|
37
44
|
_thread: threading.Thread | None = None
|
|
38
45
|
_ready: threading.Event = field(default_factory=threading.Event)
|
|
@@ -49,6 +56,8 @@ class AllowlistProxy:
|
|
|
49
56
|
self._loop.call_soon_threadsafe(self._loop.stop)
|
|
50
57
|
if self._thread is not None:
|
|
51
58
|
self._thread.join(5)
|
|
59
|
+
if self.socket_path is not None:
|
|
60
|
+
shutil.rmtree(self.socket_path.parent, ignore_errors=True)
|
|
52
61
|
|
|
53
62
|
def set_allow(self, hosts: frozenset[str]) -> None:
|
|
54
63
|
self.allow = frozenset(hosts)
|
|
@@ -65,16 +74,23 @@ class AllowlistProxy:
|
|
|
65
74
|
asyncio.set_event_loop(self._loop)
|
|
66
75
|
server = self._loop.run_until_complete(asyncio.start_server(self._handle, "127.0.0.1", self.port))
|
|
67
76
|
self.port = server.sockets[0].getsockname()[1]
|
|
77
|
+
servers = [server]
|
|
78
|
+
if self.unix:
|
|
79
|
+
# its own folder: the sandbox gets only this folder, nothing else from /tmp
|
|
80
|
+
self.socket_path = Path(tempfile.mkdtemp(prefix="aimpg-proxy-", dir="/tmp")) / "proxy.sock"
|
|
81
|
+
servers.append(self._loop.run_until_complete(asyncio.start_unix_server(self._handle, str(self.socket_path))))
|
|
68
82
|
self._ready.set()
|
|
69
83
|
try:
|
|
70
84
|
self._loop.run_forever()
|
|
71
85
|
finally:
|
|
72
|
-
|
|
86
|
+
for srv in servers:
|
|
87
|
+
srv.close()
|
|
73
88
|
pending = [t for t in asyncio.all_tasks(self._loop) if not t.done()]
|
|
74
89
|
for task in pending:
|
|
75
90
|
task.cancel()
|
|
76
91
|
self._loop.run_until_complete(asyncio.gather(*pending, return_exceptions=True))
|
|
77
|
-
|
|
92
|
+
for srv in servers:
|
|
93
|
+
self._loop.run_until_complete(srv.wait_closed())
|
|
78
94
|
self._loop.close()
|
|
79
95
|
|
|
80
96
|
async def _handle(self, reader: asyncio.StreamReader, writer: asyncio.StreamWriter) -> None:
|
|
@@ -215,6 +215,7 @@ def _run_agent(commit: Commit, setup: Setup, layout: Layout, cfg: Config, work:
|
|
|
215
215
|
# the source repo too: it holds the answer, and may live outside $HOME
|
|
216
216
|
deny_roots=[layout.root, Path(commit.repo)],
|
|
217
217
|
proxy_port=proxy.port,
|
|
218
|
+
proxy_socket=proxy.socket_path,
|
|
218
219
|
)
|
|
219
220
|
res = sandbox.run(
|
|
220
221
|
setup.argv(task_text(commit, cfg.task_mode), cfg.model, cfg.per_run_budget_usd, cfgdir),
|
|
@@ -443,6 +444,17 @@ def load_results(path: Path) -> tuple[list[Record], set[str]]:
|
|
|
443
444
|
return final, excluded
|
|
444
445
|
|
|
445
446
|
|
|
447
|
+
def all_attempts(path: Path) -> list[Record]:
|
|
448
|
+
"""Every run line in a results file, including retries and redone runs (for records)."""
|
|
449
|
+
out = []
|
|
450
|
+
for line in path.read_text().splitlines():
|
|
451
|
+
if line.strip():
|
|
452
|
+
data = json.loads(line)
|
|
453
|
+
if "excluded_commit" not in data:
|
|
454
|
+
out.append(Record(**data))
|
|
455
|
+
return out
|
|
456
|
+
|
|
457
|
+
|
|
446
458
|
def make_solution(commit: Commit, dest: Path) -> Path:
|
|
447
459
|
"""Tar of the commit's non-test files (fake 'solve' agent only)."""
|
|
448
460
|
tmp = dest.with_suffix(".tree")
|
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
"""macOS Seatbelt sandbox for replay runs.
|
|
2
|
+
|
|
3
|
+
Read rules adapted from Legwork's hardened profile
|
|
4
|
+
(github.com/kumarganduri/legwork, legwork/sandbox_runner.py `_generate_profile`,
|
|
5
|
+
MIT, same author): reads are allowed outside $HOME (system libraries) but not
|
|
6
|
+
under it; the workdir is re-allowed. Added for replays:
|
|
7
|
+
|
|
8
|
+
* network only to the local allowlisting proxy port (or none at all);
|
|
9
|
+
* a deny-all replay root, re-allowing only this run's own folders, so parallel
|
|
10
|
+
runs can't read each other's hidden tests (verified: later, more specific
|
|
11
|
+
rules win);
|
|
12
|
+
* read-only access to the agent's and toolchain's own install folders under
|
|
13
|
+
$HOME (Claude Code, uv, node), found by resolving the binaries.
|
|
14
|
+
|
|
15
|
+
Linux: the same policy through bubblewrap (adapted from Legwork's
|
|
16
|
+
`_bwrap_args`): the system read-only; home folders, /tmp, /var/tmp and /run
|
|
17
|
+
replaced by empty private tmpfs (so other runs, the answer repo and local
|
|
18
|
+
sockets such as the SSH agent are gone); only this run's folders bound back
|
|
19
|
+
writable, the agent's install folders read-only. Every namespace is
|
|
20
|
+
unshared, network included. The allowlisting proxy can't be reached as a
|
|
21
|
+
host port from a private network namespace, so it also listens on a Unix
|
|
22
|
+
socket that is bound in, and a tiny relay inside the sandbox listens on
|
|
23
|
+
127.0.0.1:<the same port> and forwards to it. Ubuntu 24.04+ needs an
|
|
24
|
+
AppArmor profile that lets bwrap (only) create user namespaces; the error
|
|
25
|
+
message prints it.
|
|
26
|
+
|
|
27
|
+
Other platforms raise SandboxUnavailable.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
import functools
|
|
33
|
+
import json
|
|
34
|
+
import os
|
|
35
|
+
import shutil
|
|
36
|
+
import signal
|
|
37
|
+
import subprocess
|
|
38
|
+
import sys
|
|
39
|
+
from dataclasses import dataclass, field
|
|
40
|
+
from pathlib import Path
|
|
41
|
+
|
|
42
|
+
_MACH_SERVICES = (
|
|
43
|
+
"com.apple.system.opendirectoryd.libinfo",
|
|
44
|
+
"com.apple.system.opendirectoryd.membership",
|
|
45
|
+
"com.apple.system.notification_center",
|
|
46
|
+
"com.apple.system.logger",
|
|
47
|
+
"com.apple.logd",
|
|
48
|
+
"com.apple.diagnosticd",
|
|
49
|
+
"com.apple.SystemConfiguration.configd",
|
|
50
|
+
"com.apple.SystemConfiguration.DNSConfiguration",
|
|
51
|
+
"com.apple.dnssd.service",
|
|
52
|
+
"com.apple.trustd",
|
|
53
|
+
"com.apple.trustd.agent",
|
|
54
|
+
"com.apple.cfprefsd.daemon",
|
|
55
|
+
"com.apple.cfprefsd.agent",
|
|
56
|
+
)
|
|
57
|
+
# Where replay data lives: outside $HOME on both systems.
|
|
58
|
+
TMP = Path("/private/tmp") if sys.platform == "darwin" else Path("/tmp")
|
|
59
|
+
TOOLS = ("claude", "codex", "uv", "node", "npm", "npx", "git", "rtk")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class SandboxUnavailable(RuntimeError):
|
|
63
|
+
pass
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def available() -> bool:
|
|
67
|
+
return unavailable_reason() is None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
BWRAP_APPARMOR_PROFILE = """abi <abi/4.0>,
|
|
71
|
+
include <tunables/global>
|
|
72
|
+
|
|
73
|
+
profile bwrap /usr/bin/bwrap flags=(unconfined) {
|
|
74
|
+
userns,
|
|
75
|
+
include if exists <local/bwrap>
|
|
76
|
+
}
|
|
77
|
+
"""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def unavailable_reason() -> str | None:
|
|
81
|
+
"""Why replays can't be sandboxed here, with the fix; None if they can."""
|
|
82
|
+
if sys.platform == "darwin":
|
|
83
|
+
return None if _launcher("sandbox-exec") else "sandbox-exec is missing (it ships with macOS)"
|
|
84
|
+
if sys.platform.startswith("linux"):
|
|
85
|
+
if not _launcher("bwrap"):
|
|
86
|
+
return "replays on Linux need bubblewrap: sudo apt install bubblewrap (or dnf/pacman)"
|
|
87
|
+
error = _bwrap_probe()
|
|
88
|
+
if error is None:
|
|
89
|
+
return None
|
|
90
|
+
hint = ""
|
|
91
|
+
try:
|
|
92
|
+
if Path("/proc/sys/kernel/apparmor_restrict_unprivileged_userns").read_text().strip() == "1":
|
|
93
|
+
hint = ("\nUbuntu 24.04+ restricts user namespaces with AppArmor; allow them for bwrap only:\n"
|
|
94
|
+
" sudo tee /etc/apparmor.d/bwrap <<'EOF'\n" + BWRAP_APPARMOR_PROFILE + "EOF\n"
|
|
95
|
+
" sudo apparmor_parser -r /etc/apparmor.d/bwrap")
|
|
96
|
+
except OSError:
|
|
97
|
+
pass
|
|
98
|
+
return f"bubblewrap can't create a sandbox here ({error}){hint}"
|
|
99
|
+
return f"replays need macOS or Linux (not {sys.platform})"
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
@functools.lru_cache(maxsize=None)
|
|
103
|
+
def _launcher(name: str) -> str | None:
|
|
104
|
+
"""Absolute path of the sandbox launcher, looked up in system folders only,
|
|
105
|
+
never in the child's PATH (which a sandboxed step could plant a fake in)."""
|
|
106
|
+
found = shutil.which(name, path="/usr/bin:/bin:/usr/sbin:/sbin:/usr/local/bin")
|
|
107
|
+
return os.path.realpath(found) if found else None
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@functools.lru_cache(maxsize=1)
|
|
111
|
+
def _bwrap_probe() -> str | None:
|
|
112
|
+
try:
|
|
113
|
+
r = subprocess.run([_launcher("bwrap"), "--unshare-all", "--ro-bind", "/", "/", "--dev", "/dev", "--proc", "/proc", "true"],
|
|
114
|
+
stdin=subprocess.DEVNULL, capture_output=True, text=True, timeout=15)
|
|
115
|
+
except (OSError, subprocess.TimeoutExpired) as exc:
|
|
116
|
+
return str(exc)
|
|
117
|
+
return None if r.returncode == 0 else (r.stderr.strip() or f"exit {r.returncode}")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def tool_dirs(tools: tuple[str, ...] = TOOLS) -> list[Path]:
|
|
121
|
+
"""Install folders of the agent/toolchain that live under $HOME (read-only grants).
|
|
122
|
+
|
|
123
|
+
For a binary in some `bin/`, its package root (one level up) is granted so
|
|
124
|
+
node/npm can read their libraries, but never $HOME or ~/.local themselves:
|
|
125
|
+
uv sits directly in ~/.local/bin, and ~/.local holds other apps' data.
|
|
126
|
+
"""
|
|
127
|
+
home = Path.home().resolve()
|
|
128
|
+
too_broad = {home, home / ".local", home / ".local" / "share"}
|
|
129
|
+
found: set[Path] = set()
|
|
130
|
+
for name in tools:
|
|
131
|
+
which = shutil.which(name)
|
|
132
|
+
if not which:
|
|
133
|
+
continue
|
|
134
|
+
for p in (Path(which).absolute(), Path(which).resolve()):
|
|
135
|
+
d = p.parent
|
|
136
|
+
if d.name in ("bin", "versions") and d.parent not in too_broad:
|
|
137
|
+
d = d.parent
|
|
138
|
+
if str(d).startswith(str(home) + os.sep) and d not in too_broad:
|
|
139
|
+
found.add(d)
|
|
140
|
+
uv_python = home / ".local/share/uv/python"
|
|
141
|
+
if uv_python.is_dir():
|
|
142
|
+
found.add(uv_python.resolve())
|
|
143
|
+
return sorted(found)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def tool_path(tools: tuple[str, ...] = TOOLS) -> str:
|
|
147
|
+
"""PATH made of the folders the tools are actually invoked from."""
|
|
148
|
+
dirs: list[str] = []
|
|
149
|
+
for name in tools:
|
|
150
|
+
which = shutil.which(name)
|
|
151
|
+
if which and str(Path(which).parent) not in dirs and str(Path(which).parent) not in ("/usr/bin", "/bin"):
|
|
152
|
+
dirs.append(str(Path(which).parent))
|
|
153
|
+
return ":".join(dirs + ["/usr/bin", "/bin", "/usr/sbin", "/sbin"])
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass
|
|
157
|
+
class Profile:
|
|
158
|
+
writable: list[Path]
|
|
159
|
+
readable: list[Path] = field(default_factory=list)
|
|
160
|
+
deny_roots: list[Path] = field(default_factory=list)
|
|
161
|
+
proxy_port: int | None = None # None: no network at all
|
|
162
|
+
proxy_socket: Path | None = None # Linux: the proxy's Unix socket, relayed to 127.0.0.1:proxy_port inside
|
|
163
|
+
|
|
164
|
+
def render(self) -> str:
|
|
165
|
+
home = str(Path.home().resolve())
|
|
166
|
+
writable = [str(Path(p).resolve()) for p in self.writable]
|
|
167
|
+
readable = [str(Path(p).resolve()) for p in self.readable]
|
|
168
|
+
lines = [
|
|
169
|
+
"(version 1)",
|
|
170
|
+
"(deny default)",
|
|
171
|
+
"(allow process-fork)",
|
|
172
|
+
"(allow process-exec)",
|
|
173
|
+
"(allow sysctl-read)",
|
|
174
|
+
"(allow mach-lookup " + " ".join(f'(global-name "{m}")' for m in _MACH_SERVICES) + ")",
|
|
175
|
+
"(allow iokit-open)",
|
|
176
|
+
# Codex syncs macOS managed preferences at startup and refuses to run
|
|
177
|
+
# without them; cfprefsd shares them read-only through this memory.
|
|
178
|
+
'(allow ipc-posix-shm-read* (ipc-posix-name-prefix "apple.cfprefs"))',
|
|
179
|
+
"(allow signal (target same-sandbox))",
|
|
180
|
+
f'(allow file-read* (require-not (subpath "{home}")))',
|
|
181
|
+
]
|
|
182
|
+
for root in self.deny_roots:
|
|
183
|
+
lines.append(f'(deny file-read* file-write* (subpath "{Path(root).resolve()}"))')
|
|
184
|
+
for d in readable:
|
|
185
|
+
lines.append(f'(allow file-read* (subpath "{d}"))')
|
|
186
|
+
for d in writable:
|
|
187
|
+
lines.append(f'(allow file-read* file-write* (subpath "{d}"))')
|
|
188
|
+
for ancestor in sorted(_ancestors([*writable, *readable], [home, *map(str, self.deny_roots)])):
|
|
189
|
+
lines.append(f'(allow file-read-metadata (literal "{ancestor}"))')
|
|
190
|
+
lines.append('(allow file-write-data (literal "/dev/null") (literal "/dev/zero"))')
|
|
191
|
+
if self.proxy_port is not None:
|
|
192
|
+
lines += ["(allow system-socket)", f'(allow network-outbound (remote ip "localhost:{self.proxy_port}"))']
|
|
193
|
+
return "\n".join(lines) + "\n"
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def bwrap_args(profile: Profile, cwd: Path) -> list[str]:
|
|
197
|
+
"""Linux equivalent of Profile.render (see the module docstring)."""
|
|
198
|
+
home = Path.home().resolve()
|
|
199
|
+
hidden = sorted({p for p in (Path("/home"), Path("/root"), home) if p.is_dir()}, key=lambda p: len(p.parts))
|
|
200
|
+
args = [_launcher("bwrap") or "/usr/bin/bwrap", "--unshare-all", "--die-with-parent", "--new-session",
|
|
201
|
+
"--ro-bind", "/", "/", "--dev", "/dev", "--proc", "/proc",
|
|
202
|
+
"--tmpfs", "/tmp", "--tmpfs", "/var/tmp", "--tmpfs", "/run"]
|
|
203
|
+
if os.path.isdir("/var/run") and not os.path.islink("/var/run"):
|
|
204
|
+
args += ["--tmpfs", "/var/run"]
|
|
205
|
+
for d in hidden:
|
|
206
|
+
args += ["--tmpfs", str(d)]
|
|
207
|
+
for root in profile.deny_roots: # e.g. the answer repo, wherever it lives
|
|
208
|
+
r = Path(root).resolve()
|
|
209
|
+
if r.is_dir() and not any(r == h or h in r.parents for h in (*hidden, Path("/tmp"), Path("/var/tmp"))):
|
|
210
|
+
args += ["--tmpfs", str(r)]
|
|
211
|
+
for d in profile.readable:
|
|
212
|
+
d = Path(d).resolve()
|
|
213
|
+
if d.exists():
|
|
214
|
+
args += ["--ro-bind", str(d), str(d)]
|
|
215
|
+
if profile.proxy_socket is not None:
|
|
216
|
+
python = Path(sys.base_prefix).resolve() # the relay runs on this Python's stdlib
|
|
217
|
+
args += ["--ro-bind", str(python), str(python)]
|
|
218
|
+
sock_dir = Path(profile.proxy_socket).parent.resolve()
|
|
219
|
+
args += ["--bind", str(sock_dir), str(sock_dir)]
|
|
220
|
+
for d in profile.writable:
|
|
221
|
+
d = Path(d).resolve()
|
|
222
|
+
args += ["--bind", str(d), str(d)]
|
|
223
|
+
for d in reversed(hidden): # stand-in homes read-only: writes fail as on macOS
|
|
224
|
+
args += ["--remount-ro", str(d)]
|
|
225
|
+
args += ["--chdir", str(Path(cwd).resolve())]
|
|
226
|
+
return args
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
# Runs inside the Linux sandbox as `python -c`: 127.0.0.1:PORT -> the proxy's Unix socket.
|
|
230
|
+
_RELAY = """
|
|
231
|
+
import socket, subprocess, sys, threading
|
|
232
|
+
sock_path, port, argv = sys.argv[1], int(sys.argv[2]), sys.argv[4:]
|
|
233
|
+
def pipe(a, b):
|
|
234
|
+
try:
|
|
235
|
+
while True:
|
|
236
|
+
data = a.recv(65536)
|
|
237
|
+
if not data:
|
|
238
|
+
break
|
|
239
|
+
b.sendall(data)
|
|
240
|
+
except OSError:
|
|
241
|
+
pass
|
|
242
|
+
finally:
|
|
243
|
+
for s in (a, b):
|
|
244
|
+
try:
|
|
245
|
+
s.shutdown(socket.SHUT_RDWR)
|
|
246
|
+
except OSError:
|
|
247
|
+
pass
|
|
248
|
+
def serve(srv):
|
|
249
|
+
while True:
|
|
250
|
+
client, _ = srv.accept()
|
|
251
|
+
try:
|
|
252
|
+
up = socket.socket(socket.AF_UNIX)
|
|
253
|
+
up.connect(sock_path)
|
|
254
|
+
except OSError:
|
|
255
|
+
client.close()
|
|
256
|
+
continue
|
|
257
|
+
threading.Thread(target=pipe, args=(client, up), daemon=True).start()
|
|
258
|
+
threading.Thread(target=pipe, args=(up, client), daemon=True).start()
|
|
259
|
+
srv = socket.socket()
|
|
260
|
+
srv.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
|
261
|
+
srv.bind(("127.0.0.1", port))
|
|
262
|
+
srv.listen(64)
|
|
263
|
+
threading.Thread(target=serve, args=(srv,), daemon=True).start()
|
|
264
|
+
code = subprocess.call(argv)
|
|
265
|
+
sys.exit(code if code >= 0 else 128 - code)
|
|
266
|
+
"""
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _ancestors(paths: list[str], roots: list[str]) -> set[str]:
|
|
270
|
+
"""Parent dirs between a denied root and an allowed path: metadata only, so realpath() works."""
|
|
271
|
+
out = set()
|
|
272
|
+
for p in paths:
|
|
273
|
+
for root in roots:
|
|
274
|
+
root = str(Path(root).resolve())
|
|
275
|
+
if p.startswith(root + os.sep):
|
|
276
|
+
cur = Path(p).parent
|
|
277
|
+
while str(cur).startswith(root):
|
|
278
|
+
out.add(str(cur))
|
|
279
|
+
if str(cur) == root:
|
|
280
|
+
break
|
|
281
|
+
cur = cur.parent
|
|
282
|
+
return out
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
@dataclass
|
|
286
|
+
class Result:
|
|
287
|
+
returncode: int | None # None when timed out
|
|
288
|
+
stdout: str
|
|
289
|
+
stderr: str
|
|
290
|
+
timed_out: bool = False
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def run(argv: list[str], *, profile: Profile, profile_path: Path, env: dict[str, str], cwd: Path, timeout: float) -> Result:
|
|
294
|
+
"""Run argv inside the sandbox. The whole process group is killed on timeout."""
|
|
295
|
+
reason = unavailable_reason()
|
|
296
|
+
if reason:
|
|
297
|
+
raise SandboxUnavailable(reason)
|
|
298
|
+
if sys.platform == "darwin":
|
|
299
|
+
profile_path.write_text(profile.render())
|
|
300
|
+
full = [_launcher("sandbox-exec"), "-f", str(profile_path), *argv]
|
|
301
|
+
else:
|
|
302
|
+
inner = argv
|
|
303
|
+
if profile.proxy_socket is not None:
|
|
304
|
+
inner = [str(Path(sys.executable).resolve()), "-I", "-c", _RELAY, str(profile.proxy_socket), str(profile.proxy_port), "--", *argv]
|
|
305
|
+
full = [*bwrap_args(profile, cwd), "--", *inner]
|
|
306
|
+
profile_path.write_text(json.dumps(full[:-len(argv)] if argv else full, indent=1)) # audit copy
|
|
307
|
+
proc = subprocess.Popen(
|
|
308
|
+
full,
|
|
309
|
+
cwd=cwd,
|
|
310
|
+
env=env,
|
|
311
|
+
stdin=subprocess.DEVNULL, # Codex waits for "additional input" on an open stdin
|
|
312
|
+
stdout=subprocess.PIPE,
|
|
313
|
+
stderr=subprocess.PIPE,
|
|
314
|
+
text=True,
|
|
315
|
+
errors="replace",
|
|
316
|
+
start_new_session=True,
|
|
317
|
+
)
|
|
318
|
+
try:
|
|
319
|
+
out, err = proc.communicate(timeout=timeout)
|
|
320
|
+
return Result(proc.returncode, out, err)
|
|
321
|
+
except subprocess.TimeoutExpired:
|
|
322
|
+
os.killpg(proc.pid, signal.SIGKILL)
|
|
323
|
+
out, err = proc.communicate()
|
|
324
|
+
return Result(None, out, err, timed_out=True)
|