ssebench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ssebench/__init__.py +5 -0
- ssebench/__main__.py +7 -0
- ssebench/_data/.env.example +33 -0
- ssebench/_data/agents/README.md +26 -0
- ssebench/_data/agents/claude-code/.dockerignore +11 -0
- ssebench/_data/agents/claude-code/Dockerfile +64 -0
- ssebench/_data/agents/claude-code/agent.yaml +1 -0
- ssebench/_data/agents/claude-code/claude-code-sse/main.py +441 -0
- ssebench/_data/agents/claude-code/claude-code-sse/pyproject.toml +13 -0
- ssebench/_data/agents/claude-code/claude-code-sse/run.sh +17 -0
- ssebench/_data/agents/codex/.dockerignore +11 -0
- ssebench/_data/agents/codex/Dockerfile +26 -0
- ssebench/_data/agents/codex/agent.yaml +1 -0
- ssebench/_data/agents/codex/codex-sse/config.py +28 -0
- ssebench/_data/agents/codex/codex-sse/main.py +108 -0
- ssebench/_data/agents/codex/codex-sse/pyproject.toml +14 -0
- ssebench/_data/agents/codex/codex-sse/run.sh +12 -0
- ssebench/_data/agents/dummy/Dockerfile +5 -0
- ssebench/_data/agents/dummy/agent.yaml +1 -0
- ssebench/_data/agents/opencode/.dockerignore +11 -0
- ssebench/_data/agents/opencode/Dockerfile +27 -0
- ssebench/_data/agents/opencode/agent.yaml +1 -0
- ssebench/_data/agents/opencode/opencode-sse/main.py +704 -0
- ssebench/_data/agents/opencode/opencode-sse/pyproject.toml +14 -0
- ssebench/_data/agents/opencode/opencode-sse/run.sh +10 -0
- ssebench/_data/agents/reference/.dockerignore +11 -0
- ssebench/_data/agents/reference/Dockerfile +25 -0
- ssebench/_data/agents/reference/README.md +44 -0
- ssebench/_data/agents/reference/agent.yaml +1 -0
- ssebench/_data/agents/reference/reference-sse/main.py +147 -0
- ssebench/_data/agents/reference/reference-sse/pyproject.toml +11 -0
- ssebench/_data/agents/ruff.toml +3 -0
- ssebench/_data/datasets/pilot/LICENSE +396 -0
- ssebench/_data/datasets/pilot/images.lock.json +5 -0
- ssebench/_data/datasets/pilot/manifest.json +3885 -0
- ssebench/_data/deploy/compose/demo.yaml +58 -0
- ssebench/_data/deploy/compose/docker-compose.yaml +72 -0
- ssebench/_data/images/common/setup-source.sh +50 -0
- ssebench/_data/images/common/share-build-caches.sh +15 -0
- ssebench/_data/images/litellm/Dockerfile +11 -0
- ssebench/_data/images/litellm/config_gen.py +59 -0
- ssebench/_data/images/sandbox/Dockerfile +120 -0
- ssebench/_data/images/sidecar-agent/Dockerfile +67 -0
- ssebench/_data/images/sidecar-case/Dockerfile +46 -0
- ssebench/_data/images/sidecar-case/entrypoint.sh +29 -0
- ssebench/_data/models/anthropic-claude.yaml +63 -0
- ssebench/_data/models/google-gemini.yaml +16 -0
- ssebench/_data/models/openai-gpt.yaml +32 -0
- ssebench/_data/pyproject.toml +70 -0
- ssebench/_data/runtime/evaluator/archive.py +27 -0
- ssebench/_data/runtime/evaluator/main.py +150 -0
- ssebench/_data/runtime/evaluator/pyproject.toml +11 -0
- ssebench/_data/runtime/evaluator/tests/test_main.py +88 -0
- ssebench/_data/runtime/mcp/config.py +77 -0
- ssebench/_data/runtime/mcp/evaluator.py +146 -0
- ssebench/_data/runtime/mcp/patch_report.py +196 -0
- ssebench/_data/runtime/mcp/pyproject.toml +13 -0
- ssebench/_data/runtime/mcp/server.py +41 -0
- ssebench/_data/runtime/mcp/tests/test_config.py +38 -0
- ssebench/_data/runtime/mcp/tests/test_evaluator.py +147 -0
- ssebench/_data/runtime/mcp/tests/test_patch_report.py +254 -0
- ssebench/_data/runtime/plugins/artifact/README.md +18 -0
- ssebench/_data/runtime/plugins/artifact/main.py +102 -0
- ssebench/_data/runtime/plugins/artifact/pyproject.toml +6 -0
- ssebench/_data/runtime/plugins/artifact/run.sh +4 -0
- ssebench/_data/runtime/plugins/oracle/README.md +39 -0
- ssebench/_data/runtime/plugins/oracle/config.py +22 -0
- ssebench/_data/runtime/plugins/oracle/fuzz.py +59 -0
- ssebench/_data/runtime/plugins/oracle/main.py +78 -0
- ssebench/_data/runtime/plugins/oracle/pyproject.toml +12 -0
- ssebench/_data/runtime/plugins/oracle/review.py +120 -0
- ssebench/_data/runtime/plugins/oracle/run.sh +4 -0
- ssebench/_data/runtime/plugins/oracle/skills/fuzzing/SKILL.md +89 -0
- ssebench/_data/runtime/plugins/oracle/skills/harness-generation/SKILL.md +245 -0
- ssebench/_data/runtime/plugins/oracle/skills/seeds-collection/SKILL.md +69 -0
- ssebench/_data/runtime/plugins/oracle/tests/test_review.py +100 -0
- ssebench/_data/runtime/plugins/plugins.yaml +32 -0
- ssebench/_data/runtime/plugins/schema.json +36 -0
- ssebench/_data/sdk/python/README.md +60 -0
- ssebench/_data/sdk/python/pyproject.toml +74 -0
- ssebench/_data/sdk/python/sse/__init__.py +18 -0
- ssebench/_data/sdk/python/sse/ai.py +471 -0
- ssebench/_data/sdk/python/sse/cheating.py +31 -0
- ssebench/_data/sdk/python/sse/daemon.py +144 -0
- ssebench/_data/sdk/python/sse/error.py +18 -0
- ssebench/_data/sdk/python/sse/grading.py +178 -0
- ssebench/_data/sdk/python/sse/helper.py +66 -0
- ssebench/_data/sdk/python/sse/metadata.py +74 -0
- ssebench/_data/sdk/python/sse/project.py +78 -0
- ssebench/_data/sdk/python/sse/prompt.py +148 -0
- ssebench/_data/sdk/python/sse/py.typed +0 -0
- ssebench/_data/sdk/python/sse/reference.py +29 -0
- ssebench/_data/sdk/python/sse/tools/__init__.py +2 -0
- ssebench/_data/sdk/python/sse/tools/bash.py +15 -0
- ssebench/_data/sdk/python/sse/tools/bencher.py +90 -0
- ssebench/_data/uv.lock +2650 -0
- ssebench/agents/__init__.py +3 -0
- ssebench/agents/agent.py +143 -0
- ssebench/arch.py +56 -0
- ssebench/backends/__init__.py +67 -0
- ssebench/backends/base.py +418 -0
- ssebench/backends/docker.py +463 -0
- ssebench/backends/images.py +28 -0
- ssebench/backends/kubernetes/__init__.py +6 -0
- ssebench/backends/kubernetes/api.py +237 -0
- ssebench/backends/kubernetes/archive.py +99 -0
- ssebench/backends/kubernetes/backend.py +622 -0
- ssebench/backends/kubernetes/config.py +150 -0
- ssebench/backends/kubernetes/forward.py +134 -0
- ssebench/backends/kubernetes/manifests.py +320 -0
- ssebench/cli/__init__.py +5 -0
- ssebench/cli/build.py +95 -0
- ssebench/cli/cli.py +568 -0
- ssebench/cli/dataset.py +352 -0
- ssebench/cli/demo.py +91 -0
- ssebench/cli/init.py +36 -0
- ssebench/cli/runs.py +435 -0
- ssebench/cli/tasks.py +69 -0
- ssebench/dataset/__init__.py +1 -0
- ssebench/dataset/dockerfile.py +141 -0
- ssebench/dataset/generate.py +88 -0
- ssebench/dataset/publish.py +244 -0
- ssebench/dataset/schema.py +41 -0
- ssebench/dataset/validate.py +175 -0
- ssebench/dataset/verify.py +513 -0
- ssebench/demo.py +687 -0
- ssebench/doctor.py +329 -0
- ssebench/errors.py +8 -0
- ssebench/extensions.py +266 -0
- ssebench/middleware/__init__.py +15 -0
- ssebench/middleware/tools.py +195 -0
- ssebench/models/__init__.py +3 -0
- ssebench/models/model.py +147 -0
- ssebench/paths.py +131 -0
- ssebench/pipe/__init__.py +5 -0
- ssebench/pipe/build.py +31 -0
- ssebench/pipe/interface.py +10 -0
- ssebench/pipe/registry.py +8 -0
- ssebench/plugins.py +107 -0
- ssebench/runner/__init__.py +3 -0
- ssebench/runner/layout.py +66 -0
- ssebench/runner/lifecycle.py +119 -0
- ssebench/runner/reference.py +72 -0
- ssebench/runner/result.py +107 -0
- ssebench/runner/runner.py +387 -0
- ssebench/settings.py +77 -0
- ssebench/stack.py +292 -0
- ssebench/tasks/__init__.py +5 -0
- ssebench/tasks/catalog.py +281 -0
- ssebench/tasks/local.py +159 -0
- ssebench/tasks/manifest.py +130 -0
- ssebench/tasks/metadata.py +126 -0
- ssebench/tasks/task.py +79 -0
- ssebench/version.py +30 -0
- ssebench/workspace.py +78 -0
- ssebench-1.0.0.dist-info/METADATA +89 -0
- ssebench-1.0.0.dist-info/RECORD +161 -0
- ssebench-1.0.0.dist-info/WHEEL +4 -0
- ssebench-1.0.0.dist-info/entry_points.txt +2 -0
- ssebench-1.0.0.dist-info/licenses/LICENSE +202 -0
- ssebench-1.0.0.dist-info/licenses/NOTICE +10 -0
ssebench/__init__.py
ADDED
ssebench/__main__.py
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Local settings. `ssebench init` and `just setup` copy this file to .env and
|
|
2
|
+
# fill in the two generated secrets. .env is ignored by git; never commit it.
|
|
3
|
+
|
|
4
|
+
# Admin key of the local LiteLLM proxy. The CLI uses it to create a key with a
|
|
5
|
+
# budget for each run.
|
|
6
|
+
LITELLM_MASTER_KEY=
|
|
7
|
+
# Password of the proxy's Postgres database; letters and digits only, as it is
|
|
8
|
+
# part of a URL. Postgres keeps the password it was created with: after you
|
|
9
|
+
# change it, remove the stack's volume (`docker compose -p <project> down -v`).
|
|
10
|
+
POSTGRES_PASSWORD=
|
|
11
|
+
|
|
12
|
+
# Keys of the model providers you use. models/*.yaml refers to them as
|
|
13
|
+
# os.environ/<NAME>; a model without its key fails when it is called. After a
|
|
14
|
+
# change, restart the proxy with `ssebench proxy up` (`just launch`).
|
|
15
|
+
ANTHROPIC_API_KEY=
|
|
16
|
+
OPENAI_API_KEY=
|
|
17
|
+
GOOGLE_API_KEY=
|
|
18
|
+
|
|
19
|
+
# Optional settings, shown with their defaults.
|
|
20
|
+
# Host port of the LiteLLM proxy.
|
|
21
|
+
# LITELLM_PORT=4000
|
|
22
|
+
# Host address the proxy's port is published on. It holds the master key and the
|
|
23
|
+
# provider keys, so it listens on loopback only.
|
|
24
|
+
# LITELLM_BIND=127.0.0.1
|
|
25
|
+
# Compose project of the proxy stack; stacks with different names (and ports)
|
|
26
|
+
# run side by side.
|
|
27
|
+
# COMPOSE_PROJECT_NAME=ssebench
|
|
28
|
+
# Registry prefix of every image SSEBench builds or pulls.
|
|
29
|
+
# SSEBENCH_REGISTRY=ghcr.io/42-b3yond-6ug/ssebench
|
|
30
|
+
# Task catalog of `ssebench run` and the web UI: the path or URL of a
|
|
31
|
+
# manifest.json, a dataset directory, or the URL of a catalog service. Unset,
|
|
32
|
+
# it is the bundled datasets/pilot/manifest.json.
|
|
33
|
+
# SSEBENCH_CATALOG=
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# agents
|
|
2
|
+
|
|
3
|
+
One folder per agent. An agent is the layer that goes on top of the tool layer
|
|
4
|
+
of a run; `ssebench run --agent <name>` builds it from its folder.
|
|
5
|
+
|
|
6
|
+
| Agent | What it is |
|
|
7
|
+
|---|---|
|
|
8
|
+
| [`claude-code/`](claude-code/) | Anthropic's Claude Code. |
|
|
9
|
+
| [`codex/`](codex/) | OpenAI's Codex CLI. |
|
|
10
|
+
| [`opencode/`](opencode/) | The open-source OpenCode agent. |
|
|
11
|
+
| [`dummy/`](dummy/) | Does nothing and makes no model calls; a run with it grades the untouched source, so its patch fails. It tests the pipeline. |
|
|
12
|
+
| [`reference/`](reference/) | Applies the task's known fix and makes no model calls; a sound task passes every check. It checks a task, and powers `just demo`. Its runs never count as a model's score. |
|
|
13
|
+
|
|
14
|
+
Each folder has an `agent.yaml` (the agent's name), a `Dockerfile` and, for
|
|
15
|
+
every agent but `dummy`, a wrapper `<name>-sse/`. The wrapper is a Python project
|
|
16
|
+
that starts the agent, reads the task through the `sse` SDK and writes
|
|
17
|
+
`dialog.jsonl`; every wrapper is a member of the repository's uv workspace. An
|
|
18
|
+
agent never talks to a model provider directly: all LLM traffic goes through the
|
|
19
|
+
LiteLLM proxy, at the endpoint the run passes in `SSE_BASE_URL`.
|
|
20
|
+
|
|
21
|
+
To add an agent, follow [Add an agent](../docs/guides/add-an-agent.md).
|
|
22
|
+
|
|
23
|
+
- [Dialog protocol](../docs/reference/dialog-protocol.md): the format of
|
|
24
|
+
`dialog.jsonl`
|
|
25
|
+
- [Reference runs](../docs/reference/cli.md#reference-runs)
|
|
26
|
+
- [Image layers](../docs/concepts/image-layers.md)
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# Stage 1: Install Claude Code (cached separately from agent code changes)
|
|
2
|
+
FROM ubuntu:22.04 AS claude-installer
|
|
3
|
+
RUN apt-get update && apt-get install -y curl ca-certificates && rm -rf /var/lib/apt/lists/*
|
|
4
|
+
RUN useradd -m -u 1000 model
|
|
5
|
+
USER model
|
|
6
|
+
ENV HOME=/home/model
|
|
7
|
+
ARG TARGETARCH
|
|
8
|
+
ARG CLAUDE_CODE_VERSION=2.1.277
|
|
9
|
+
# Checksums of the linux-x64 and linux-arm64 builds, from the release's
|
|
10
|
+
# manifest.json; update both when bumping the version.
|
|
11
|
+
ARG CLAUDE_CODE_SHA256_AMD64=722210f05ba494d8f6df69423c4d4f2960900f7a007d0532851c7a36e375cab7
|
|
12
|
+
ARG CLAUDE_CODE_SHA256_ARM64=242c4d743beabc822edd8f247101bb800b4036c69e74b9e2a1adb120dfe46f5d
|
|
13
|
+
# `claude install` sets up the launcher in ~/.local/bin for this version.
|
|
14
|
+
RUN case "${TARGETARCH}" in \
|
|
15
|
+
amd64) platform=linux-x64; sha256="${CLAUDE_CODE_SHA256_AMD64}" ;; \
|
|
16
|
+
arm64) platform=linux-arm64; sha256="${CLAUDE_CODE_SHA256_ARM64}" ;; \
|
|
17
|
+
*) echo "no Claude Code release for ${TARGETARCH}" >&2; exit 1 ;; \
|
|
18
|
+
esac && \
|
|
19
|
+
curl -fsSLo /tmp/claude \
|
|
20
|
+
"https://downloads.claude.ai/claude-code-releases/${CLAUDE_CODE_VERSION}/${platform}/claude" && \
|
|
21
|
+
echo "${sha256} /tmp/claude" | sha256sum -c - && \
|
|
22
|
+
chmod +x /tmp/claude && \
|
|
23
|
+
/tmp/claude install "${CLAUDE_CODE_VERSION}" && \
|
|
24
|
+
rm /tmp/claude
|
|
25
|
+
|
|
26
|
+
# Stage 2: Final image
|
|
27
|
+
FROM ssebench-agent
|
|
28
|
+
|
|
29
|
+
# The wrapper is a member of the uv workspace at the repository root, so /app
|
|
30
|
+
# holds the workspace files it needs (from the `workspace` build context) in the
|
|
31
|
+
# repository layout.
|
|
32
|
+
# Create /app owned by model user (WORKDIR creates as root)
|
|
33
|
+
RUN mkdir -p /app/agents/claude-code/claude-code-sse && chown -R model:model /app
|
|
34
|
+
WORKDIR /app/agents/claude-code/claude-code-sse
|
|
35
|
+
|
|
36
|
+
# Copy Claude Code installation from stage 1
|
|
37
|
+
COPY --from=claude-installer --chown=model:model /home/model/.local /home/model/.local
|
|
38
|
+
COPY --from=claude-installer --chown=model:model /home/model/.claude /home/model/.claude
|
|
39
|
+
|
|
40
|
+
# Ensure model user's home and cache directories have correct ownership
|
|
41
|
+
# This prevents "Permission denied" errors when uv runs as model user
|
|
42
|
+
RUN mkdir -p /home/model/.cache && \
|
|
43
|
+
rm -rf /home/model/.cache/uv && \
|
|
44
|
+
chown -R model:model /home/model
|
|
45
|
+
|
|
46
|
+
# Only set CLAUDE path - HOME is set by entrypoint when running as model user
|
|
47
|
+
# Setting HOME globally causes MCP server (running as root) to pollute model's cache
|
|
48
|
+
ENV CLAUDE=/home/model/.local/bin/claude
|
|
49
|
+
|
|
50
|
+
COPY --from=workspace --chown=model:model pyproject.toml uv.lock /app/
|
|
51
|
+
COPY --from=workspace --chown=model:model sdk/python/pyproject.toml sdk/python/README.md /app/sdk/python/
|
|
52
|
+
COPY --from=workspace --chown=model:model sdk/python/sse/ /app/sdk/python/sse/
|
|
53
|
+
|
|
54
|
+
# Copy agent code (changes frequently, but doesn't invalidate cached stages above)
|
|
55
|
+
# .dockerignore excludes .venv, __pycache__, etc.
|
|
56
|
+
COPY --chown=model:model claude-code-sse/ /app/agents/claude-code/claude-code-sse
|
|
57
|
+
|
|
58
|
+
# Install the wrapper's environment, Python included, now: at run time the
|
|
59
|
+
# container reaches the LiteLLM proxy only.
|
|
60
|
+
USER model
|
|
61
|
+
RUN HOME=/home/model uv sync --frozen --no-dev --package claude-code-sse
|
|
62
|
+
USER root
|
|
63
|
+
|
|
64
|
+
CMD ["./run.sh"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
name: claude-code
|
|
@@ -0,0 +1,441 @@
|
|
|
1
|
+
import asyncio
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import subprocess
|
|
5
|
+
import sys
|
|
6
|
+
import time
|
|
7
|
+
from asyncio.streams import StreamReader
|
|
8
|
+
from datetime import UTC, datetime
|
|
9
|
+
from io import TextIOWrapper
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from sse import project
|
|
14
|
+
from sse.prompt import task_prompt
|
|
15
|
+
|
|
16
|
+
# =============================================================================
|
|
17
|
+
# Dialog Writer - Converts Claude Code stream-json to SSEBench dialog format
|
|
18
|
+
# =============================================================================
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class DialogWriter:
|
|
22
|
+
"""
|
|
23
|
+
Writes agent dialog entries to $SSE_ARCHIVE/dialog.jsonl for WebUI consumption.
|
|
24
|
+
|
|
25
|
+
Converts Claude Code's stream-json output to our simplified dialog format:
|
|
26
|
+
- init: Session start metadata
|
|
27
|
+
- prompt: User task given to agent
|
|
28
|
+
- message: Assistant text responses
|
|
29
|
+
- thinking: Reasoning blocks (extended thinking)
|
|
30
|
+
- tool: Tool calls with status (running/success/error)
|
|
31
|
+
- complete: Session end with stats
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(self, log_path: str | None = None):
|
|
35
|
+
# Default to $SSE_ARCHIVE/dialog.jsonl, in the agent's archive directory
|
|
36
|
+
if log_path is None:
|
|
37
|
+
archive = os.environ.get("SSE_ARCHIVE", "/tmp/sse-archive")
|
|
38
|
+
log_path = f"{archive}/dialog.jsonl"
|
|
39
|
+
self.log_path: Path = Path(log_path)
|
|
40
|
+
self.log_path.parent.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
self.seq: int = 0
|
|
42
|
+
self.start_time: float = time.time()
|
|
43
|
+
self.turns: int = 0
|
|
44
|
+
self.total_input_tokens: int = 0
|
|
45
|
+
self.total_output_tokens: int = 0
|
|
46
|
+
self.pending_tools: dict[str, dict[str, str]] = {} # tool_id -> entry data
|
|
47
|
+
self.completed: bool = False
|
|
48
|
+
self.file: TextIOWrapper = open(self.log_path, "w")
|
|
49
|
+
|
|
50
|
+
def _write_entry(self, entry: dict[str, Any]):
|
|
51
|
+
"""Write a single entry to the JSONL file."""
|
|
52
|
+
_ = self.file.write(json.dumps(entry) + "\n")
|
|
53
|
+
self.file.flush()
|
|
54
|
+
|
|
55
|
+
def _now(self) -> str:
|
|
56
|
+
"""Get current timestamp in ISO format."""
|
|
57
|
+
return datetime.now(UTC).isoformat()
|
|
58
|
+
|
|
59
|
+
def _next_seq(self) -> int:
|
|
60
|
+
"""Get next sequence number."""
|
|
61
|
+
seq = self.seq
|
|
62
|
+
self.seq += 1
|
|
63
|
+
return seq
|
|
64
|
+
|
|
65
|
+
def write_init(self, task: str, cwd: str, model: str, agent: str = "claude-code"):
|
|
66
|
+
"""Write session init entry."""
|
|
67
|
+
self._write_entry(
|
|
68
|
+
{
|
|
69
|
+
"seq": self._next_seq(),
|
|
70
|
+
"ts": self._now(),
|
|
71
|
+
"type": "init",
|
|
72
|
+
"data": {
|
|
73
|
+
"task": task,
|
|
74
|
+
"cwd": cwd,
|
|
75
|
+
"model": model,
|
|
76
|
+
"agent": agent,
|
|
77
|
+
},
|
|
78
|
+
}
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
def write_prompt(self, content: str):
|
|
82
|
+
"""Write user prompt entry."""
|
|
83
|
+
self._write_entry(
|
|
84
|
+
{
|
|
85
|
+
"seq": self._next_seq(),
|
|
86
|
+
"ts": self._now(),
|
|
87
|
+
"type": "prompt",
|
|
88
|
+
"content": content,
|
|
89
|
+
}
|
|
90
|
+
)
|
|
91
|
+
|
|
92
|
+
def write_message(
|
|
93
|
+
self, content: str, input_tokens: int = 0, output_tokens: int = 0
|
|
94
|
+
):
|
|
95
|
+
"""Write assistant message entry."""
|
|
96
|
+
entry: dict[str, Any] = {
|
|
97
|
+
"seq": self._next_seq(),
|
|
98
|
+
"ts": self._now(),
|
|
99
|
+
"type": "message",
|
|
100
|
+
"role": "assistant",
|
|
101
|
+
"content": content,
|
|
102
|
+
}
|
|
103
|
+
if input_tokens > 0 or output_tokens > 0:
|
|
104
|
+
entry["tokens"] = {"in": input_tokens, "out": output_tokens}
|
|
105
|
+
self.total_input_tokens += input_tokens
|
|
106
|
+
self.total_output_tokens += output_tokens
|
|
107
|
+
self._write_entry(entry)
|
|
108
|
+
|
|
109
|
+
def write_thinking(self, content: str):
|
|
110
|
+
"""Write thinking/reasoning entry."""
|
|
111
|
+
self._write_entry(
|
|
112
|
+
{
|
|
113
|
+
"seq": self._next_seq(),
|
|
114
|
+
"ts": self._now(),
|
|
115
|
+
"type": "thinking",
|
|
116
|
+
"content": content,
|
|
117
|
+
}
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
def write_tool_start(
|
|
121
|
+
self, tool_id: str, name: str, args: dict[str, Any] | None = None
|
|
122
|
+
):
|
|
123
|
+
"""Write tool call start entry (status=running)."""
|
|
124
|
+
entry: dict[str, Any] = {
|
|
125
|
+
"seq": self._next_seq(),
|
|
126
|
+
"ts": self._now(),
|
|
127
|
+
"type": "tool",
|
|
128
|
+
"tool_id": tool_id,
|
|
129
|
+
"name": name,
|
|
130
|
+
"status": "running",
|
|
131
|
+
}
|
|
132
|
+
if args:
|
|
133
|
+
entry["args"] = args
|
|
134
|
+
self._write_entry(entry)
|
|
135
|
+
self.pending_tools[tool_id] = {"name": name}
|
|
136
|
+
|
|
137
|
+
def write_tool_result(
|
|
138
|
+
self, tool_id: str, result: str | None = None, error: str | None = None
|
|
139
|
+
):
|
|
140
|
+
"""Write tool call result entry (status=success or error)."""
|
|
141
|
+
name = self.pending_tools.pop(tool_id, {}).get("name", "unknown")
|
|
142
|
+
entry = {
|
|
143
|
+
"seq": self._next_seq(),
|
|
144
|
+
"ts": self._now(),
|
|
145
|
+
"type": "tool",
|
|
146
|
+
"tool_id": tool_id,
|
|
147
|
+
"name": name,
|
|
148
|
+
"status": "error" if error else "success",
|
|
149
|
+
}
|
|
150
|
+
if result:
|
|
151
|
+
entry["result"] = result
|
|
152
|
+
if error:
|
|
153
|
+
entry["error"] = error
|
|
154
|
+
self._write_entry(entry)
|
|
155
|
+
|
|
156
|
+
def write_complete(self, status: str = "success", message: str | None = None):
|
|
157
|
+
"""Write session completion entry. Guarded against double-writes."""
|
|
158
|
+
if self.completed:
|
|
159
|
+
return
|
|
160
|
+
self.completed = True
|
|
161
|
+
duration_ms = int((time.time() - self.start_time) * 1000)
|
|
162
|
+
entry = {
|
|
163
|
+
"seq": self._next_seq(),
|
|
164
|
+
"ts": self._now(),
|
|
165
|
+
"type": "complete",
|
|
166
|
+
"status": status,
|
|
167
|
+
"turns": self.turns,
|
|
168
|
+
"duration_ms": duration_ms,
|
|
169
|
+
"total_tokens": {
|
|
170
|
+
"in": self.total_input_tokens,
|
|
171
|
+
"out": self.total_output_tokens,
|
|
172
|
+
},
|
|
173
|
+
}
|
|
174
|
+
if message:
|
|
175
|
+
entry["message"] = message
|
|
176
|
+
self._write_entry(entry)
|
|
177
|
+
|
|
178
|
+
def process_claude_event(self, event: dict[str, Any]):
|
|
179
|
+
"""
|
|
180
|
+
Process a single Claude Code stream-json event and convert to dialog format.
|
|
181
|
+
|
|
182
|
+
Claude Code stream-json events include:
|
|
183
|
+
- {"type": "assistant", "message": {...}}
|
|
184
|
+
- {"type": "user", "message": {...}}
|
|
185
|
+
- {"type": "result", ...}
|
|
186
|
+
- {"type": "system", ...}
|
|
187
|
+
"""
|
|
188
|
+
event_type = event.get("type", "")
|
|
189
|
+
|
|
190
|
+
if event_type == "assistant":
|
|
191
|
+
self.turns += 1
|
|
192
|
+
message = event.get("message", {})
|
|
193
|
+
content_blocks = message.get("content", [])
|
|
194
|
+
|
|
195
|
+
# Extract per-turn token usage
|
|
196
|
+
usage = message.get("usage", {})
|
|
197
|
+
input_tokens = usage.get("input_tokens", 0)
|
|
198
|
+
output_tokens = usage.get("output_tokens", 0)
|
|
199
|
+
tokens_attributed = False
|
|
200
|
+
|
|
201
|
+
for block in content_blocks:
|
|
202
|
+
block_type = block.get("type", "")
|
|
203
|
+
|
|
204
|
+
if block_type == "text":
|
|
205
|
+
text = block.get("text", "")
|
|
206
|
+
if text:
|
|
207
|
+
if not tokens_attributed:
|
|
208
|
+
self.write_message(text, input_tokens, output_tokens)
|
|
209
|
+
tokens_attributed = True
|
|
210
|
+
else:
|
|
211
|
+
self.write_message(text)
|
|
212
|
+
|
|
213
|
+
elif block_type == "thinking":
|
|
214
|
+
thinking = block.get("thinking", "")
|
|
215
|
+
if thinking:
|
|
216
|
+
self.write_thinking(thinking)
|
|
217
|
+
|
|
218
|
+
elif block_type == "tool_use":
|
|
219
|
+
tool_id = block.get("id", str(self._next_seq()))
|
|
220
|
+
tool_name = block.get("name", "unknown")
|
|
221
|
+
tool_input = block.get("input", {})
|
|
222
|
+
self.write_tool_start(tool_id, tool_name, tool_input)
|
|
223
|
+
|
|
224
|
+
# If no message was written, still track token usage globally
|
|
225
|
+
if not tokens_attributed:
|
|
226
|
+
self.total_input_tokens += input_tokens
|
|
227
|
+
self.total_output_tokens += output_tokens
|
|
228
|
+
|
|
229
|
+
elif event_type == "user":
|
|
230
|
+
message = event.get("message", {})
|
|
231
|
+
content_blocks = message.get("content", [])
|
|
232
|
+
|
|
233
|
+
for block in content_blocks:
|
|
234
|
+
block_type = block.get("type", "")
|
|
235
|
+
|
|
236
|
+
if block_type == "tool_result":
|
|
237
|
+
tool_id = block.get("tool_use_id", "")
|
|
238
|
+
is_error = block.get("is_error", False)
|
|
239
|
+
content = block.get("content", "")
|
|
240
|
+
|
|
241
|
+
# Content can be string or list of content blocks
|
|
242
|
+
if isinstance(content, list):
|
|
243
|
+
text_parts = []
|
|
244
|
+
for c in content:
|
|
245
|
+
if isinstance(c, dict) and c.get("type") == "text":
|
|
246
|
+
text_parts.append(c.get("text", ""))
|
|
247
|
+
elif isinstance(c, str):
|
|
248
|
+
text_parts.append(c)
|
|
249
|
+
content = "\n".join(text_parts)
|
|
250
|
+
|
|
251
|
+
if is_error:
|
|
252
|
+
self.write_tool_result(tool_id, error=content)
|
|
253
|
+
else:
|
|
254
|
+
self.write_tool_result(tool_id, result=content)
|
|
255
|
+
|
|
256
|
+
elif event_type == "result":
|
|
257
|
+
# Session ended. `result` holds the final text; `subtype` says how it ended.
|
|
258
|
+
subtype = event.get("subtype", "")
|
|
259
|
+
if subtype == "success" and not event.get("is_error", False):
|
|
260
|
+
self.write_complete("success")
|
|
261
|
+
else:
|
|
262
|
+
self.write_complete(
|
|
263
|
+
"error", message=f"Session ended: {subtype or 'unknown'}"
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
def close(self):
|
|
267
|
+
"""Close the log file."""
|
|
268
|
+
self.file.close()
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
# =============================================================================
|
|
272
|
+
# Process Output Handling
|
|
273
|
+
# =============================================================================
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
async def stream_output_with_dialog(
|
|
277
|
+
stream: StreamReader, prefix: str = "", dialog: DialogWriter | None = None
|
|
278
|
+
):
|
|
279
|
+
"""
|
|
280
|
+
Reads from a stream line-by-line, prints it with a prefix,
|
|
281
|
+
and processes JSONL events for dialog logging.
|
|
282
|
+
"""
|
|
283
|
+
buf = b""
|
|
284
|
+
|
|
285
|
+
while True:
|
|
286
|
+
try:
|
|
287
|
+
chunk = await stream.read(4096)
|
|
288
|
+
buf += chunk
|
|
289
|
+
line_sep = b"\n"
|
|
290
|
+
while line_sep in buf:
|
|
291
|
+
first_line, buf = buf.split(line_sep, maxsplit=1)
|
|
292
|
+
line_str = first_line.decode().strip()
|
|
293
|
+
print(f"{prefix} {line_str}", flush=True)
|
|
294
|
+
|
|
295
|
+
# Try to parse as JSON for dialog processing
|
|
296
|
+
if dialog and prefix == "[STDOUT]" and line_str:
|
|
297
|
+
try:
|
|
298
|
+
event: dict[str, Any] = json.loads(line_str)
|
|
299
|
+
dialog.process_claude_event(event)
|
|
300
|
+
except json.JSONDecodeError:
|
|
301
|
+
pass # Not JSON, skip
|
|
302
|
+
|
|
303
|
+
if len(chunk) == 0:
|
|
304
|
+
# EOF, prints whatever we have in buf
|
|
305
|
+
remaining = buf.decode().strip()
|
|
306
|
+
if remaining:
|
|
307
|
+
print(f"{prefix} {remaining}", flush=True)
|
|
308
|
+
break
|
|
309
|
+
|
|
310
|
+
except Exception as e:
|
|
311
|
+
print(f"Error reading stream {prefix}: {e}", file=sys.stderr)
|
|
312
|
+
continue
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
async def run_claude(dialog: DialogWriter):
|
|
316
|
+
try:
|
|
317
|
+
claude_executable = os.environ["CLAUDE"]
|
|
318
|
+
model_name = os.environ["SSE_MODEL_NAME"]
|
|
319
|
+
base_url = os.environ["SSE_BASE_URL"]
|
|
320
|
+
api_key = os.environ["SSE_API_KEY"]
|
|
321
|
+
except KeyError as _:
|
|
322
|
+
dialog.write_complete("error", message="Missing environment variables")
|
|
323
|
+
return
|
|
324
|
+
|
|
325
|
+
# Write init entry
|
|
326
|
+
dialog.write_init(
|
|
327
|
+
task=project.metadata.id,
|
|
328
|
+
cwd=str(project.source),
|
|
329
|
+
model=model_name,
|
|
330
|
+
agent="claude-code",
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
prompt = task_prompt()
|
|
334
|
+
dialog.write_prompt(prompt)
|
|
335
|
+
|
|
336
|
+
# Build Claude command arguments
|
|
337
|
+
# --dangerously-skip-permissions requires non-root user (handled by entrypoint)
|
|
338
|
+
command_args = [
|
|
339
|
+
claude_executable,
|
|
340
|
+
"--verbose",
|
|
341
|
+
"--max-turns",
|
|
342
|
+
"500",
|
|
343
|
+
"-p",
|
|
344
|
+
"--model",
|
|
345
|
+
model_name,
|
|
346
|
+
"--output-format",
|
|
347
|
+
"stream-json",
|
|
348
|
+
"--dangerously-skip-permissions",
|
|
349
|
+
# Passed here rather than written to the project's .claude/, which
|
|
350
|
+
# would become part of the agent's patch.
|
|
351
|
+
"--settings",
|
|
352
|
+
json.dumps({"permissions": {"defaultMode": "bypassPermissions"}}),
|
|
353
|
+
]
|
|
354
|
+
|
|
355
|
+
# Set up environment for Claude Code
|
|
356
|
+
env = os.environ.copy()
|
|
357
|
+
env["ANTHROPIC_BASE_URL"] = base_url
|
|
358
|
+
env["ANTHROPIC_AUTH_TOKEN"] = api_key
|
|
359
|
+
env["ANTHROPIC_DEFAULT_HAIKU_MODEL"] = model_name # redirects the haiku slot
|
|
360
|
+
env["ANTHROPIC_DEFAULT_SONNET_MODEL"] = model_name # redirects the sonnet slot
|
|
361
|
+
env["ANTHROPIC_DEFAULT_OPUS_MODEL"] = model_name # redirects the opus slot
|
|
362
|
+
|
|
363
|
+
process = await asyncio.create_subprocess_exec(
|
|
364
|
+
*command_args,
|
|
365
|
+
cwd=project.source,
|
|
366
|
+
stdout=asyncio.subprocess.PIPE,
|
|
367
|
+
stderr=asyncio.subprocess.PIPE,
|
|
368
|
+
stdin=asyncio.subprocess.PIPE,
|
|
369
|
+
env=env,
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
if process.stdin:
|
|
373
|
+
process.stdin.write(prompt.encode("utf-8"))
|
|
374
|
+
await process.stdin.drain()
|
|
375
|
+
process.stdin.close()
|
|
376
|
+
|
|
377
|
+
assert process.stdout is not None
|
|
378
|
+
assert process.stderr is not None
|
|
379
|
+
|
|
380
|
+
stdout_task = asyncio.create_task(
|
|
381
|
+
stream_output_with_dialog(process.stdout, prefix="[STDOUT]", dialog=dialog)
|
|
382
|
+
)
|
|
383
|
+
stderr_task = asyncio.create_task(
|
|
384
|
+
stream_output_with_dialog(process.stderr, prefix="[STDERR]", dialog=None)
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
return_code = await process.wait()
|
|
388
|
+
_ = await asyncio.gather(stdout_task, stderr_task)
|
|
389
|
+
|
|
390
|
+
# Write completion if not already written by a result event
|
|
391
|
+
if not dialog.completed:
|
|
392
|
+
if return_code == 0:
|
|
393
|
+
dialog.write_complete("success")
|
|
394
|
+
else:
|
|
395
|
+
dialog.write_complete(
|
|
396
|
+
"error", message=f"Process exited with code {return_code}"
|
|
397
|
+
)
|
|
398
|
+
|
|
399
|
+
print(f"[claude] exited with code {return_code}")
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
async def main():
|
|
403
|
+
"""
|
|
404
|
+
Main function that configures and launches the Claude Code agent.
|
|
405
|
+
Assumes MCP server is already running on http://localhost:3000/mcp
|
|
406
|
+
"""
|
|
407
|
+
claude_executable = os.environ["CLAUDE"]
|
|
408
|
+
|
|
409
|
+
# Register the MCP server with Claude Code
|
|
410
|
+
_ = subprocess.run(
|
|
411
|
+
[
|
|
412
|
+
claude_executable,
|
|
413
|
+
"mcp",
|
|
414
|
+
"add",
|
|
415
|
+
"ssebench",
|
|
416
|
+
"http://localhost:3000/mcp",
|
|
417
|
+
"--transport",
|
|
418
|
+
"http",
|
|
419
|
+
"--scope",
|
|
420
|
+
"user",
|
|
421
|
+
]
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
# Initialize dialog writer
|
|
425
|
+
dialog = DialogWriter()
|
|
426
|
+
|
|
427
|
+
try:
|
|
428
|
+
# Launch Claude Code agent
|
|
429
|
+
process_task = asyncio.create_task(run_claude(dialog))
|
|
430
|
+
await process_task
|
|
431
|
+
except Exception as e:
|
|
432
|
+
dialog.write_complete("error", message=str(e))
|
|
433
|
+
finally:
|
|
434
|
+
dialog.close()
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
if __name__ == "__main__":
|
|
438
|
+
try:
|
|
439
|
+
asyncio.run(main())
|
|
440
|
+
except KeyboardInterrupt:
|
|
441
|
+
print("killed")
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
#!/bin/bash
|
|
2
|
+
set -euo pipefail
|
|
3
|
+
|
|
4
|
+
# change cwd to current folder
|
|
5
|
+
SCRIPT_PATH="$(realpath "${BASH_SOURCE[0]}")"
|
|
6
|
+
SCRIPT_DIR="$(dirname "$SCRIPT_PATH")"
|
|
7
|
+
cd "$SCRIPT_DIR" || exit 1
|
|
8
|
+
|
|
9
|
+
# The run container reaches the LiteLLM proxy only: turn off Claude Code's
|
|
10
|
+
# update checks, telemetry, error reports and plugin marketplace installs.
|
|
11
|
+
export CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1
|
|
12
|
+
export CLAUDE_CODE_DISABLE_OFFICIAL_MARKETPLACE_AUTOINSTALL=1
|
|
13
|
+
|
|
14
|
+
# Launch the agent (runs as model user via entrypoint)
|
|
15
|
+
echo "[run.sh] Launching Claude Code agent..."
|
|
16
|
+
# The image holds the environment, and the run container has no internet.
|
|
17
|
+
exec uv run --offline --no-sync main.py
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
FROM ssebench-agent
|
|
2
|
+
|
|
3
|
+
# The wrapper is a member of the uv workspace at the repository root, so /app
|
|
4
|
+
# holds the workspace files it needs (from the `workspace` build context) in the
|
|
5
|
+
# repository layout.
|
|
6
|
+
RUN mkdir -p /app/agents/codex/codex-sse && chown -R model:model /app
|
|
7
|
+
WORKDIR /app/agents/codex/codex-sse
|
|
8
|
+
|
|
9
|
+
RUN curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \
|
|
10
|
+
apt-get install -y nodejs
|
|
11
|
+
|
|
12
|
+
ARG CODEX_VERSION=0.159.0
|
|
13
|
+
RUN npm install -g "@openai/codex@${CODEX_VERSION}"
|
|
14
|
+
|
|
15
|
+
COPY --from=workspace --chown=model:model pyproject.toml uv.lock /app/
|
|
16
|
+
COPY --from=workspace --chown=model:model sdk/python/pyproject.toml sdk/python/README.md /app/sdk/python/
|
|
17
|
+
COPY --from=workspace --chown=model:model sdk/python/sse/ /app/sdk/python/sse/
|
|
18
|
+
COPY --chown=model:model codex-sse/ /app/agents/codex/codex-sse
|
|
19
|
+
|
|
20
|
+
# Install the wrapper's environment, Python included, now: at run time the
|
|
21
|
+
# container reaches the LiteLLM proxy only.
|
|
22
|
+
USER model
|
|
23
|
+
RUN HOME=/home/model uv sync --frozen --no-dev --package codex-sse
|
|
24
|
+
USER root
|
|
25
|
+
|
|
26
|
+
CMD ["./run.sh"]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
name: codex
|