ssebench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (161) hide show
  1. ssebench/__init__.py +5 -0
  2. ssebench/__main__.py +7 -0
  3. ssebench/_data/.env.example +33 -0
  4. ssebench/_data/agents/README.md +26 -0
  5. ssebench/_data/agents/claude-code/.dockerignore +11 -0
  6. ssebench/_data/agents/claude-code/Dockerfile +64 -0
  7. ssebench/_data/agents/claude-code/agent.yaml +1 -0
  8. ssebench/_data/agents/claude-code/claude-code-sse/main.py +441 -0
  9. ssebench/_data/agents/claude-code/claude-code-sse/pyproject.toml +13 -0
  10. ssebench/_data/agents/claude-code/claude-code-sse/run.sh +17 -0
  11. ssebench/_data/agents/codex/.dockerignore +11 -0
  12. ssebench/_data/agents/codex/Dockerfile +26 -0
  13. ssebench/_data/agents/codex/agent.yaml +1 -0
  14. ssebench/_data/agents/codex/codex-sse/config.py +28 -0
  15. ssebench/_data/agents/codex/codex-sse/main.py +108 -0
  16. ssebench/_data/agents/codex/codex-sse/pyproject.toml +14 -0
  17. ssebench/_data/agents/codex/codex-sse/run.sh +12 -0
  18. ssebench/_data/agents/dummy/Dockerfile +5 -0
  19. ssebench/_data/agents/dummy/agent.yaml +1 -0
  20. ssebench/_data/agents/opencode/.dockerignore +11 -0
  21. ssebench/_data/agents/opencode/Dockerfile +27 -0
  22. ssebench/_data/agents/opencode/agent.yaml +1 -0
  23. ssebench/_data/agents/opencode/opencode-sse/main.py +704 -0
  24. ssebench/_data/agents/opencode/opencode-sse/pyproject.toml +14 -0
  25. ssebench/_data/agents/opencode/opencode-sse/run.sh +10 -0
  26. ssebench/_data/agents/reference/.dockerignore +11 -0
  27. ssebench/_data/agents/reference/Dockerfile +25 -0
  28. ssebench/_data/agents/reference/README.md +44 -0
  29. ssebench/_data/agents/reference/agent.yaml +1 -0
  30. ssebench/_data/agents/reference/reference-sse/main.py +147 -0
  31. ssebench/_data/agents/reference/reference-sse/pyproject.toml +11 -0
  32. ssebench/_data/agents/ruff.toml +3 -0
  33. ssebench/_data/datasets/pilot/LICENSE +396 -0
  34. ssebench/_data/datasets/pilot/images.lock.json +5 -0
  35. ssebench/_data/datasets/pilot/manifest.json +3885 -0
  36. ssebench/_data/deploy/compose/demo.yaml +58 -0
  37. ssebench/_data/deploy/compose/docker-compose.yaml +72 -0
  38. ssebench/_data/images/common/setup-source.sh +50 -0
  39. ssebench/_data/images/common/share-build-caches.sh +15 -0
  40. ssebench/_data/images/litellm/Dockerfile +11 -0
  41. ssebench/_data/images/litellm/config_gen.py +59 -0
  42. ssebench/_data/images/sandbox/Dockerfile +120 -0
  43. ssebench/_data/images/sidecar-agent/Dockerfile +67 -0
  44. ssebench/_data/images/sidecar-case/Dockerfile +46 -0
  45. ssebench/_data/images/sidecar-case/entrypoint.sh +29 -0
  46. ssebench/_data/models/anthropic-claude.yaml +63 -0
  47. ssebench/_data/models/google-gemini.yaml +16 -0
  48. ssebench/_data/models/openai-gpt.yaml +32 -0
  49. ssebench/_data/pyproject.toml +70 -0
  50. ssebench/_data/runtime/evaluator/archive.py +27 -0
  51. ssebench/_data/runtime/evaluator/main.py +150 -0
  52. ssebench/_data/runtime/evaluator/pyproject.toml +11 -0
  53. ssebench/_data/runtime/evaluator/tests/test_main.py +88 -0
  54. ssebench/_data/runtime/mcp/config.py +77 -0
  55. ssebench/_data/runtime/mcp/evaluator.py +146 -0
  56. ssebench/_data/runtime/mcp/patch_report.py +196 -0
  57. ssebench/_data/runtime/mcp/pyproject.toml +13 -0
  58. ssebench/_data/runtime/mcp/server.py +41 -0
  59. ssebench/_data/runtime/mcp/tests/test_config.py +38 -0
  60. ssebench/_data/runtime/mcp/tests/test_evaluator.py +147 -0
  61. ssebench/_data/runtime/mcp/tests/test_patch_report.py +254 -0
  62. ssebench/_data/runtime/plugins/artifact/README.md +18 -0
  63. ssebench/_data/runtime/plugins/artifact/main.py +102 -0
  64. ssebench/_data/runtime/plugins/artifact/pyproject.toml +6 -0
  65. ssebench/_data/runtime/plugins/artifact/run.sh +4 -0
  66. ssebench/_data/runtime/plugins/oracle/README.md +39 -0
  67. ssebench/_data/runtime/plugins/oracle/config.py +22 -0
  68. ssebench/_data/runtime/plugins/oracle/fuzz.py +59 -0
  69. ssebench/_data/runtime/plugins/oracle/main.py +78 -0
  70. ssebench/_data/runtime/plugins/oracle/pyproject.toml +12 -0
  71. ssebench/_data/runtime/plugins/oracle/review.py +120 -0
  72. ssebench/_data/runtime/plugins/oracle/run.sh +4 -0
  73. ssebench/_data/runtime/plugins/oracle/skills/fuzzing/SKILL.md +89 -0
  74. ssebench/_data/runtime/plugins/oracle/skills/harness-generation/SKILL.md +245 -0
  75. ssebench/_data/runtime/plugins/oracle/skills/seeds-collection/SKILL.md +69 -0
  76. ssebench/_data/runtime/plugins/oracle/tests/test_review.py +100 -0
  77. ssebench/_data/runtime/plugins/plugins.yaml +32 -0
  78. ssebench/_data/runtime/plugins/schema.json +36 -0
  79. ssebench/_data/sdk/python/README.md +60 -0
  80. ssebench/_data/sdk/python/pyproject.toml +74 -0
  81. ssebench/_data/sdk/python/sse/__init__.py +18 -0
  82. ssebench/_data/sdk/python/sse/ai.py +471 -0
  83. ssebench/_data/sdk/python/sse/cheating.py +31 -0
  84. ssebench/_data/sdk/python/sse/daemon.py +144 -0
  85. ssebench/_data/sdk/python/sse/error.py +18 -0
  86. ssebench/_data/sdk/python/sse/grading.py +178 -0
  87. ssebench/_data/sdk/python/sse/helper.py +66 -0
  88. ssebench/_data/sdk/python/sse/metadata.py +74 -0
  89. ssebench/_data/sdk/python/sse/project.py +78 -0
  90. ssebench/_data/sdk/python/sse/prompt.py +148 -0
  91. ssebench/_data/sdk/python/sse/py.typed +0 -0
  92. ssebench/_data/sdk/python/sse/reference.py +29 -0
  93. ssebench/_data/sdk/python/sse/tools/__init__.py +2 -0
  94. ssebench/_data/sdk/python/sse/tools/bash.py +15 -0
  95. ssebench/_data/sdk/python/sse/tools/bencher.py +90 -0
  96. ssebench/_data/uv.lock +2650 -0
  97. ssebench/agents/__init__.py +3 -0
  98. ssebench/agents/agent.py +143 -0
  99. ssebench/arch.py +56 -0
  100. ssebench/backends/__init__.py +67 -0
  101. ssebench/backends/base.py +418 -0
  102. ssebench/backends/docker.py +463 -0
  103. ssebench/backends/images.py +28 -0
  104. ssebench/backends/kubernetes/__init__.py +6 -0
  105. ssebench/backends/kubernetes/api.py +237 -0
  106. ssebench/backends/kubernetes/archive.py +99 -0
  107. ssebench/backends/kubernetes/backend.py +622 -0
  108. ssebench/backends/kubernetes/config.py +150 -0
  109. ssebench/backends/kubernetes/forward.py +134 -0
  110. ssebench/backends/kubernetes/manifests.py +320 -0
  111. ssebench/cli/__init__.py +5 -0
  112. ssebench/cli/build.py +95 -0
  113. ssebench/cli/cli.py +568 -0
  114. ssebench/cli/dataset.py +352 -0
  115. ssebench/cli/demo.py +91 -0
  116. ssebench/cli/init.py +36 -0
  117. ssebench/cli/runs.py +435 -0
  118. ssebench/cli/tasks.py +69 -0
  119. ssebench/dataset/__init__.py +1 -0
  120. ssebench/dataset/dockerfile.py +141 -0
  121. ssebench/dataset/generate.py +88 -0
  122. ssebench/dataset/publish.py +244 -0
  123. ssebench/dataset/schema.py +41 -0
  124. ssebench/dataset/validate.py +175 -0
  125. ssebench/dataset/verify.py +513 -0
  126. ssebench/demo.py +687 -0
  127. ssebench/doctor.py +329 -0
  128. ssebench/errors.py +8 -0
  129. ssebench/extensions.py +266 -0
  130. ssebench/middleware/__init__.py +15 -0
  131. ssebench/middleware/tools.py +195 -0
  132. ssebench/models/__init__.py +3 -0
  133. ssebench/models/model.py +147 -0
  134. ssebench/paths.py +131 -0
  135. ssebench/pipe/__init__.py +5 -0
  136. ssebench/pipe/build.py +31 -0
  137. ssebench/pipe/interface.py +10 -0
  138. ssebench/pipe/registry.py +8 -0
  139. ssebench/plugins.py +107 -0
  140. ssebench/runner/__init__.py +3 -0
  141. ssebench/runner/layout.py +66 -0
  142. ssebench/runner/lifecycle.py +119 -0
  143. ssebench/runner/reference.py +72 -0
  144. ssebench/runner/result.py +107 -0
  145. ssebench/runner/runner.py +387 -0
  146. ssebench/settings.py +77 -0
  147. ssebench/stack.py +292 -0
  148. ssebench/tasks/__init__.py +5 -0
  149. ssebench/tasks/catalog.py +281 -0
  150. ssebench/tasks/local.py +159 -0
  151. ssebench/tasks/manifest.py +130 -0
  152. ssebench/tasks/metadata.py +126 -0
  153. ssebench/tasks/task.py +79 -0
  154. ssebench/version.py +30 -0
  155. ssebench/workspace.py +78 -0
  156. ssebench-1.0.0.dist-info/METADATA +89 -0
  157. ssebench-1.0.0.dist-info/RECORD +161 -0
  158. ssebench-1.0.0.dist-info/WHEEL +4 -0
  159. ssebench-1.0.0.dist-info/entry_points.txt +2 -0
  160. ssebench-1.0.0.dist-info/licenses/LICENSE +202 -0
  161. ssebench-1.0.0.dist-info/licenses/NOTICE +10 -0
ssebench/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """SSEBench: run AI agents against real vulnerabilities and grade their patches."""
2
+
3
+ from ssebench.version import __version__
4
+
5
+ __all__ = ["__version__"]
ssebench/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ import logging
2
+
3
+ from ssebench.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ logging.basicConfig(level=logging.INFO)
7
+ main()
@@ -0,0 +1,33 @@
1
+ # Local settings. `ssebench init` and `just setup` copy this file to .env and
2
+ # fill in the two generated secrets. .env is ignored by git; never commit it.
3
+
4
+ # Admin key of the local LiteLLM proxy. The CLI uses it to create a key with a
5
+ # budget for each run.
6
+ LITELLM_MASTER_KEY=
7
+ # Password of the proxy's Postgres database; letters and digits only, as it is
8
+ # part of a URL. Postgres keeps the password it was created with: after you
9
+ # change it, remove the stack's volume (`docker compose -p <project> down -v`).
10
+ POSTGRES_PASSWORD=
11
+
12
+ # Keys of the model providers you use. models/*.yaml refers to them as
13
+ # os.environ/<NAME>; a model without its key fails when it is called. After a
14
+ # change, restart the proxy with `ssebench proxy up` (`just launch`).
15
+ ANTHROPIC_API_KEY=
16
+ OPENAI_API_KEY=
17
+ GOOGLE_API_KEY=
18
+
19
+ # Optional settings, shown with their defaults.
20
+ # Host port of the LiteLLM proxy.
21
+ # LITELLM_PORT=4000
22
+ # Host address the proxy's port is published on. It holds the master key and the
23
+ # provider keys, so it listens on loopback only.
24
+ # LITELLM_BIND=127.0.0.1
25
+ # Compose project of the proxy stack; stacks with different names (and ports)
26
+ # run side by side.
27
+ # COMPOSE_PROJECT_NAME=ssebench
28
+ # Registry prefix of every image SSEBench builds or pulls.
29
+ # SSEBENCH_REGISTRY=ghcr.io/42-b3yond-6ug/ssebench
30
+ # Task catalog of `ssebench run` and the web UI: the path or URL of a
31
+ # manifest.json, a dataset directory, or the URL of a catalog service. Unset,
32
+ # it is the bundled datasets/pilot/manifest.json.
33
+ # SSEBENCH_CATALOG=
@@ -0,0 +1,26 @@
1
+ # agents
2
+
3
+ One folder per agent. An agent is the layer that goes on top of the tool layer
4
+ of a run; `ssebench run --agent <name>` builds it from its folder.
5
+
6
+ | Agent | What it is |
7
+ |---|---|
8
+ | [`claude-code/`](claude-code/) | Anthropic's Claude Code. |
9
+ | [`codex/`](codex/) | OpenAI's Codex CLI. |
10
+ | [`opencode/`](opencode/) | The open-source OpenCode agent. |
11
+ | [`dummy/`](dummy/) | Does nothing and makes no model calls; a run with it grades the untouched source, so its patch fails. It tests the pipeline. |
12
+ | [`reference/`](reference/) | Applies the task's known fix and makes no model calls; a sound task passes every check. It checks a task, and powers `just demo`. Its runs never count as a model's score. |
13
+
14
+ Each folder has an `agent.yaml` (the agent's name), a `Dockerfile` and, for
15
+ every agent but `dummy`, a wrapper `<name>-sse/`. The wrapper is a Python project
16
+ that starts the agent, reads the task through the `sse` SDK and writes
17
+ `dialog.jsonl`; every wrapper is a member of the repository's uv workspace. An
18
+ agent never talks to a model provider directly: all LLM traffic goes through the
19
+ LiteLLM proxy, at the endpoint the run passes in `SSE_BASE_URL`.
20
+
21
+ To add an agent, follow [Add an agent](../docs/guides/add-an-agent.md).
22
+
23
+ - [Dialog protocol](../docs/reference/dialog-protocol.md): the format of
24
+ `dialog.jsonl`
25
+ - [Reference runs](../docs/reference/cli.md#reference-runs)
26
+ - [Image layers](../docs/concepts/image-layers.md)
@@ -0,0 +1,11 @@
1
+ # Python virtual environments
2
+ **/.venv
3
+ **/__pycache__
4
+ **/*.pyc
5
+
6
+ # IDE
7
+ **/.idea
8
+ **/.vscode
9
+
10
+ # Cache
11
+ **/.cache
@@ -0,0 +1,64 @@
1
+ # Stage 1: Install Claude Code (cached separately from agent code changes)
2
+ FROM ubuntu:22.04 AS claude-installer
3
+ RUN apt-get update && apt-get install -y curl ca-certificates && rm -rf /var/lib/apt/lists/*
4
+ RUN useradd -m -u 1000 model
5
+ USER model
6
+ ENV HOME=/home/model
7
+ ARG TARGETARCH
8
+ ARG CLAUDE_CODE_VERSION=2.1.277
9
+ # Checksums of the linux-x64 and linux-arm64 builds, from the release's
10
+ # manifest.json; update both when bumping the version.
11
+ ARG CLAUDE_CODE_SHA256_AMD64=722210f05ba494d8f6df69423c4d4f2960900f7a007d0532851c7a36e375cab7
12
+ ARG CLAUDE_CODE_SHA256_ARM64=242c4d743beabc822edd8f247101bb800b4036c69e74b9e2a1adb120dfe46f5d
13
+ # `claude install` sets up the launcher in ~/.local/bin for this version.
14
+ RUN case "${TARGETARCH}" in \
15
+ amd64) platform=linux-x64; sha256="${CLAUDE_CODE_SHA256_AMD64}" ;; \
16
+ arm64) platform=linux-arm64; sha256="${CLAUDE_CODE_SHA256_ARM64}" ;; \
17
+ *) echo "no Claude Code release for ${TARGETARCH}" >&2; exit 1 ;; \
18
+ esac && \
19
+ curl -fsSLo /tmp/claude \
20
+ "https://downloads.claude.ai/claude-code-releases/${CLAUDE_CODE_VERSION}/${platform}/claude" && \
21
+ echo "${sha256} /tmp/claude" | sha256sum -c - && \
22
+ chmod +x /tmp/claude && \
23
+ /tmp/claude install "${CLAUDE_CODE_VERSION}" && \
24
+ rm /tmp/claude
25
+
26
+ # Stage 2: Final image
27
+ FROM ssebench-agent
28
+
29
+ # The wrapper is a member of the uv workspace at the repository root, so /app
30
+ # holds the workspace files it needs (from the `workspace` build context) in the
31
+ # repository layout.
32
+ # Create /app owned by model user (WORKDIR creates as root)
33
+ RUN mkdir -p /app/agents/claude-code/claude-code-sse && chown -R model:model /app
34
+ WORKDIR /app/agents/claude-code/claude-code-sse
35
+
36
+ # Copy Claude Code installation from stage 1
37
+ COPY --from=claude-installer --chown=model:model /home/model/.local /home/model/.local
38
+ COPY --from=claude-installer --chown=model:model /home/model/.claude /home/model/.claude
39
+
40
+ # Ensure model user's home and cache directories have correct ownership
41
+ # This prevents "Permission denied" errors when uv runs as model user
42
+ RUN mkdir -p /home/model/.cache && \
43
+ rm -rf /home/model/.cache/uv && \
44
+ chown -R model:model /home/model
45
+
46
+ # Only set CLAUDE path - HOME is set by entrypoint when running as model user
47
+ # Setting HOME globally causes MCP server (running as root) to pollute model's cache
48
+ ENV CLAUDE=/home/model/.local/bin/claude
49
+
50
+ COPY --from=workspace --chown=model:model pyproject.toml uv.lock /app/
51
+ COPY --from=workspace --chown=model:model sdk/python/pyproject.toml sdk/python/README.md /app/sdk/python/
52
+ COPY --from=workspace --chown=model:model sdk/python/sse/ /app/sdk/python/sse/
53
+
54
+ # Copy agent code (changes frequently, but doesn't invalidate cached stages above)
55
+ # .dockerignore excludes .venv, __pycache__, etc.
56
+ COPY --chown=model:model claude-code-sse/ /app/agents/claude-code/claude-code-sse
57
+
58
+ # Install the wrapper's environment, Python included, now: at run time the
59
+ # container reaches the LiteLLM proxy only.
60
+ USER model
61
+ RUN HOME=/home/model uv sync --frozen --no-dev --package claude-code-sse
62
+ USER root
63
+
64
+ CMD ["./run.sh"]
@@ -0,0 +1 @@
1
+ name: claude-code
@@ -0,0 +1,441 @@
1
+ import asyncio
2
+ import json
3
+ import os
4
+ import subprocess
5
+ import sys
6
+ import time
7
+ from asyncio.streams import StreamReader
8
+ from datetime import UTC, datetime
9
+ from io import TextIOWrapper
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from sse import project
14
+ from sse.prompt import task_prompt
15
+
16
+ # =============================================================================
17
+ # Dialog Writer - Converts Claude Code stream-json to SSEBench dialog format
18
+ # =============================================================================
19
+
20
+
21
+ class DialogWriter:
22
+ """
23
+ Writes agent dialog entries to $SSE_ARCHIVE/dialog.jsonl for WebUI consumption.
24
+
25
+ Converts Claude Code's stream-json output to our simplified dialog format:
26
+ - init: Session start metadata
27
+ - prompt: User task given to agent
28
+ - message: Assistant text responses
29
+ - thinking: Reasoning blocks (extended thinking)
30
+ - tool: Tool calls with status (running/success/error)
31
+ - complete: Session end with stats
32
+ """
33
+
34
+ def __init__(self, log_path: str | None = None):
35
+ # Default to $SSE_ARCHIVE/dialog.jsonl, in the agent's archive directory
36
+ if log_path is None:
37
+ archive = os.environ.get("SSE_ARCHIVE", "/tmp/sse-archive")
38
+ log_path = f"{archive}/dialog.jsonl"
39
+ self.log_path: Path = Path(log_path)
40
+ self.log_path.parent.mkdir(parents=True, exist_ok=True)
41
+ self.seq: int = 0
42
+ self.start_time: float = time.time()
43
+ self.turns: int = 0
44
+ self.total_input_tokens: int = 0
45
+ self.total_output_tokens: int = 0
46
+ self.pending_tools: dict[str, dict[str, str]] = {} # tool_id -> entry data
47
+ self.completed: bool = False
48
+ self.file: TextIOWrapper = open(self.log_path, "w")
49
+
50
+ def _write_entry(self, entry: dict[str, Any]):
51
+ """Write a single entry to the JSONL file."""
52
+ _ = self.file.write(json.dumps(entry) + "\n")
53
+ self.file.flush()
54
+
55
+ def _now(self) -> str:
56
+ """Get current timestamp in ISO format."""
57
+ return datetime.now(UTC).isoformat()
58
+
59
+ def _next_seq(self) -> int:
60
+ """Get next sequence number."""
61
+ seq = self.seq
62
+ self.seq += 1
63
+ return seq
64
+
65
+ def write_init(self, task: str, cwd: str, model: str, agent: str = "claude-code"):
66
+ """Write session init entry."""
67
+ self._write_entry(
68
+ {
69
+ "seq": self._next_seq(),
70
+ "ts": self._now(),
71
+ "type": "init",
72
+ "data": {
73
+ "task": task,
74
+ "cwd": cwd,
75
+ "model": model,
76
+ "agent": agent,
77
+ },
78
+ }
79
+ )
80
+
81
+ def write_prompt(self, content: str):
82
+ """Write user prompt entry."""
83
+ self._write_entry(
84
+ {
85
+ "seq": self._next_seq(),
86
+ "ts": self._now(),
87
+ "type": "prompt",
88
+ "content": content,
89
+ }
90
+ )
91
+
92
+ def write_message(
93
+ self, content: str, input_tokens: int = 0, output_tokens: int = 0
94
+ ):
95
+ """Write assistant message entry."""
96
+ entry: dict[str, Any] = {
97
+ "seq": self._next_seq(),
98
+ "ts": self._now(),
99
+ "type": "message",
100
+ "role": "assistant",
101
+ "content": content,
102
+ }
103
+ if input_tokens > 0 or output_tokens > 0:
104
+ entry["tokens"] = {"in": input_tokens, "out": output_tokens}
105
+ self.total_input_tokens += input_tokens
106
+ self.total_output_tokens += output_tokens
107
+ self._write_entry(entry)
108
+
109
+ def write_thinking(self, content: str):
110
+ """Write thinking/reasoning entry."""
111
+ self._write_entry(
112
+ {
113
+ "seq": self._next_seq(),
114
+ "ts": self._now(),
115
+ "type": "thinking",
116
+ "content": content,
117
+ }
118
+ )
119
+
120
+ def write_tool_start(
121
+ self, tool_id: str, name: str, args: dict[str, Any] | None = None
122
+ ):
123
+ """Write tool call start entry (status=running)."""
124
+ entry: dict[str, Any] = {
125
+ "seq": self._next_seq(),
126
+ "ts": self._now(),
127
+ "type": "tool",
128
+ "tool_id": tool_id,
129
+ "name": name,
130
+ "status": "running",
131
+ }
132
+ if args:
133
+ entry["args"] = args
134
+ self._write_entry(entry)
135
+ self.pending_tools[tool_id] = {"name": name}
136
+
137
+ def write_tool_result(
138
+ self, tool_id: str, result: str | None = None, error: str | None = None
139
+ ):
140
+ """Write tool call result entry (status=success or error)."""
141
+ name = self.pending_tools.pop(tool_id, {}).get("name", "unknown")
142
+ entry = {
143
+ "seq": self._next_seq(),
144
+ "ts": self._now(),
145
+ "type": "tool",
146
+ "tool_id": tool_id,
147
+ "name": name,
148
+ "status": "error" if error else "success",
149
+ }
150
+ if result:
151
+ entry["result"] = result
152
+ if error:
153
+ entry["error"] = error
154
+ self._write_entry(entry)
155
+
156
+ def write_complete(self, status: str = "success", message: str | None = None):
157
+ """Write session completion entry. Guarded against double-writes."""
158
+ if self.completed:
159
+ return
160
+ self.completed = True
161
+ duration_ms = int((time.time() - self.start_time) * 1000)
162
+ entry = {
163
+ "seq": self._next_seq(),
164
+ "ts": self._now(),
165
+ "type": "complete",
166
+ "status": status,
167
+ "turns": self.turns,
168
+ "duration_ms": duration_ms,
169
+ "total_tokens": {
170
+ "in": self.total_input_tokens,
171
+ "out": self.total_output_tokens,
172
+ },
173
+ }
174
+ if message:
175
+ entry["message"] = message
176
+ self._write_entry(entry)
177
+
178
+ def process_claude_event(self, event: dict[str, Any]):
179
+ """
180
+ Process a single Claude Code stream-json event and convert to dialog format.
181
+
182
+ Claude Code stream-json events include:
183
+ - {"type": "assistant", "message": {...}}
184
+ - {"type": "user", "message": {...}}
185
+ - {"type": "result", ...}
186
+ - {"type": "system", ...}
187
+ """
188
+ event_type = event.get("type", "")
189
+
190
+ if event_type == "assistant":
191
+ self.turns += 1
192
+ message = event.get("message", {})
193
+ content_blocks = message.get("content", [])
194
+
195
+ # Extract per-turn token usage
196
+ usage = message.get("usage", {})
197
+ input_tokens = usage.get("input_tokens", 0)
198
+ output_tokens = usage.get("output_tokens", 0)
199
+ tokens_attributed = False
200
+
201
+ for block in content_blocks:
202
+ block_type = block.get("type", "")
203
+
204
+ if block_type == "text":
205
+ text = block.get("text", "")
206
+ if text:
207
+ if not tokens_attributed:
208
+ self.write_message(text, input_tokens, output_tokens)
209
+ tokens_attributed = True
210
+ else:
211
+ self.write_message(text)
212
+
213
+ elif block_type == "thinking":
214
+ thinking = block.get("thinking", "")
215
+ if thinking:
216
+ self.write_thinking(thinking)
217
+
218
+ elif block_type == "tool_use":
219
+ tool_id = block.get("id", str(self._next_seq()))
220
+ tool_name = block.get("name", "unknown")
221
+ tool_input = block.get("input", {})
222
+ self.write_tool_start(tool_id, tool_name, tool_input)
223
+
224
+ # If no message was written, still track token usage globally
225
+ if not tokens_attributed:
226
+ self.total_input_tokens += input_tokens
227
+ self.total_output_tokens += output_tokens
228
+
229
+ elif event_type == "user":
230
+ message = event.get("message", {})
231
+ content_blocks = message.get("content", [])
232
+
233
+ for block in content_blocks:
234
+ block_type = block.get("type", "")
235
+
236
+ if block_type == "tool_result":
237
+ tool_id = block.get("tool_use_id", "")
238
+ is_error = block.get("is_error", False)
239
+ content = block.get("content", "")
240
+
241
+ # Content can be string or list of content blocks
242
+ if isinstance(content, list):
243
+ text_parts = []
244
+ for c in content:
245
+ if isinstance(c, dict) and c.get("type") == "text":
246
+ text_parts.append(c.get("text", ""))
247
+ elif isinstance(c, str):
248
+ text_parts.append(c)
249
+ content = "\n".join(text_parts)
250
+
251
+ if is_error:
252
+ self.write_tool_result(tool_id, error=content)
253
+ else:
254
+ self.write_tool_result(tool_id, result=content)
255
+
256
+ elif event_type == "result":
257
+ # Session ended. `result` holds the final text; `subtype` says how it ended.
258
+ subtype = event.get("subtype", "")
259
+ if subtype == "success" and not event.get("is_error", False):
260
+ self.write_complete("success")
261
+ else:
262
+ self.write_complete(
263
+ "error", message=f"Session ended: {subtype or 'unknown'}"
264
+ )
265
+
266
+ def close(self):
267
+ """Close the log file."""
268
+ self.file.close()
269
+
270
+
271
+ # =============================================================================
272
+ # Process Output Handling
273
+ # =============================================================================
274
+
275
+
276
+ async def stream_output_with_dialog(
277
+ stream: StreamReader, prefix: str = "", dialog: DialogWriter | None = None
278
+ ):
279
+ """
280
+ Reads from a stream line-by-line, prints it with a prefix,
281
+ and processes JSONL events for dialog logging.
282
+ """
283
+ buf = b""
284
+
285
+ while True:
286
+ try:
287
+ chunk = await stream.read(4096)
288
+ buf += chunk
289
+ line_sep = b"\n"
290
+ while line_sep in buf:
291
+ first_line, buf = buf.split(line_sep, maxsplit=1)
292
+ line_str = first_line.decode().strip()
293
+ print(f"{prefix} {line_str}", flush=True)
294
+
295
+ # Try to parse as JSON for dialog processing
296
+ if dialog and prefix == "[STDOUT]" and line_str:
297
+ try:
298
+ event: dict[str, Any] = json.loads(line_str)
299
+ dialog.process_claude_event(event)
300
+ except json.JSONDecodeError:
301
+ pass # Not JSON, skip
302
+
303
+ if len(chunk) == 0:
304
+ # EOF, prints whatever we have in buf
305
+ remaining = buf.decode().strip()
306
+ if remaining:
307
+ print(f"{prefix} {remaining}", flush=True)
308
+ break
309
+
310
+ except Exception as e:
311
+ print(f"Error reading stream {prefix}: {e}", file=sys.stderr)
312
+ continue
313
+
314
+
315
+ async def run_claude(dialog: DialogWriter):
316
+ try:
317
+ claude_executable = os.environ["CLAUDE"]
318
+ model_name = os.environ["SSE_MODEL_NAME"]
319
+ base_url = os.environ["SSE_BASE_URL"]
320
+ api_key = os.environ["SSE_API_KEY"]
321
+ except KeyError as _:
322
+ dialog.write_complete("error", message="Missing environment variables")
323
+ return
324
+
325
+ # Write init entry
326
+ dialog.write_init(
327
+ task=project.metadata.id,
328
+ cwd=str(project.source),
329
+ model=model_name,
330
+ agent="claude-code",
331
+ )
332
+
333
+ prompt = task_prompt()
334
+ dialog.write_prompt(prompt)
335
+
336
+ # Build Claude command arguments
337
+ # --dangerously-skip-permissions requires non-root user (handled by entrypoint)
338
+ command_args = [
339
+ claude_executable,
340
+ "--verbose",
341
+ "--max-turns",
342
+ "500",
343
+ "-p",
344
+ "--model",
345
+ model_name,
346
+ "--output-format",
347
+ "stream-json",
348
+ "--dangerously-skip-permissions",
349
+ # Passed here rather than written to the project's .claude/, which
350
+ # would become part of the agent's patch.
351
+ "--settings",
352
+ json.dumps({"permissions": {"defaultMode": "bypassPermissions"}}),
353
+ ]
354
+
355
+ # Set up environment for Claude Code
356
+ env = os.environ.copy()
357
+ env["ANTHROPIC_BASE_URL"] = base_url
358
+ env["ANTHROPIC_AUTH_TOKEN"] = api_key
359
+ env["ANTHROPIC_DEFAULT_HAIKU_MODEL"] = model_name # redirects the haiku slot
360
+ env["ANTHROPIC_DEFAULT_SONNET_MODEL"] = model_name # redirects the sonnet slot
361
+ env["ANTHROPIC_DEFAULT_OPUS_MODEL"] = model_name # redirects the opus slot
362
+
363
+ process = await asyncio.create_subprocess_exec(
364
+ *command_args,
365
+ cwd=project.source,
366
+ stdout=asyncio.subprocess.PIPE,
367
+ stderr=asyncio.subprocess.PIPE,
368
+ stdin=asyncio.subprocess.PIPE,
369
+ env=env,
370
+ )
371
+
372
+ if process.stdin:
373
+ process.stdin.write(prompt.encode("utf-8"))
374
+ await process.stdin.drain()
375
+ process.stdin.close()
376
+
377
+ assert process.stdout is not None
378
+ assert process.stderr is not None
379
+
380
+ stdout_task = asyncio.create_task(
381
+ stream_output_with_dialog(process.stdout, prefix="[STDOUT]", dialog=dialog)
382
+ )
383
+ stderr_task = asyncio.create_task(
384
+ stream_output_with_dialog(process.stderr, prefix="[STDERR]", dialog=None)
385
+ )
386
+
387
+ return_code = await process.wait()
388
+ _ = await asyncio.gather(stdout_task, stderr_task)
389
+
390
+ # Write completion if not already written by a result event
391
+ if not dialog.completed:
392
+ if return_code == 0:
393
+ dialog.write_complete("success")
394
+ else:
395
+ dialog.write_complete(
396
+ "error", message=f"Process exited with code {return_code}"
397
+ )
398
+
399
+ print(f"[claude] exited with code {return_code}")
400
+
401
+
402
+ async def main():
403
+ """
404
+ Main function that configures and launches the Claude Code agent.
405
+ Assumes MCP server is already running on http://localhost:3000/mcp
406
+ """
407
+ claude_executable = os.environ["CLAUDE"]
408
+
409
+ # Register the MCP server with Claude Code
410
+ _ = subprocess.run(
411
+ [
412
+ claude_executable,
413
+ "mcp",
414
+ "add",
415
+ "ssebench",
416
+ "http://localhost:3000/mcp",
417
+ "--transport",
418
+ "http",
419
+ "--scope",
420
+ "user",
421
+ ]
422
+ )
423
+
424
+ # Initialize dialog writer
425
+ dialog = DialogWriter()
426
+
427
+ try:
428
+ # Launch Claude Code agent
429
+ process_task = asyncio.create_task(run_claude(dialog))
430
+ await process_task
431
+ except Exception as e:
432
+ dialog.write_complete("error", message=str(e))
433
+ finally:
434
+ dialog.close()
435
+
436
+
437
+ if __name__ == "__main__":
438
+ try:
439
+ asyncio.run(main())
440
+ except KeyboardInterrupt:
441
+ print("killed")
@@ -0,0 +1,13 @@
1
+ [project]
2
+ name = "claude-code-sse"
3
+ version = "1.0.0"
4
+ description = "Claude"
5
+ readme = "README.md"
6
+ requires-python = ">=3.12"
7
+ dependencies = [
8
+ "fastmcp",
9
+ "ssebench-sdk",
10
+ ]
11
+
12
+ [tool.uv.sources]
13
+ ssebench-sdk = { workspace = true }
@@ -0,0 +1,17 @@
1
+ #!/bin/bash
2
+ set -euo pipefail
3
+
4
+ # change cwd to current folder
5
+ SCRIPT_PATH="$(realpath "${BASH_SOURCE[0]}")"
6
+ SCRIPT_DIR="$(dirname "$SCRIPT_PATH")"
7
+ cd "$SCRIPT_DIR" || exit 1
8
+
9
+ # The run container reaches the LiteLLM proxy only: turn off Claude Code's
10
+ # update checks, telemetry, error reports and plugin marketplace installs.
11
+ export CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC=1
12
+ export CLAUDE_CODE_DISABLE_OFFICIAL_MARKETPLACE_AUTOINSTALL=1
13
+
14
+ # Launch the agent (runs as model user via entrypoint)
15
+ echo "[run.sh] Launching Claude Code agent..."
16
+ # The image holds the environment, and the run container has no internet.
17
+ exec uv run --offline --no-sync main.py
@@ -0,0 +1,11 @@
1
+ # Python virtual environments
2
+ **/.venv
3
+ **/__pycache__
4
+ **/*.pyc
5
+
6
+ # IDE
7
+ **/.idea
8
+ **/.vscode
9
+
10
+ # Cache
11
+ **/.cache
@@ -0,0 +1,26 @@
1
+ FROM ssebench-agent
2
+
3
+ # The wrapper is a member of the uv workspace at the repository root, so /app
4
+ # holds the workspace files it needs (from the `workspace` build context) in the
5
+ # repository layout.
6
+ RUN mkdir -p /app/agents/codex/codex-sse && chown -R model:model /app
7
+ WORKDIR /app/agents/codex/codex-sse
8
+
9
+ RUN curl -fsSL https://deb.nodesource.com/setup_22.x | bash - && \
10
+ apt-get install -y nodejs
11
+
12
+ ARG CODEX_VERSION=0.159.0
13
+ RUN npm install -g "@openai/codex@${CODEX_VERSION}"
14
+
15
+ COPY --from=workspace --chown=model:model pyproject.toml uv.lock /app/
16
+ COPY --from=workspace --chown=model:model sdk/python/pyproject.toml sdk/python/README.md /app/sdk/python/
17
+ COPY --from=workspace --chown=model:model sdk/python/sse/ /app/sdk/python/sse/
18
+ COPY --chown=model:model codex-sse/ /app/agents/codex/codex-sse
19
+
20
+ # Install the wrapper's environment, Python included, now: at run time the
21
+ # container reaches the LiteLLM proxy only.
22
+ USER model
23
+ RUN HOME=/home/model uv sync --frozen --no-dev --package codex-sse
24
+ USER root
25
+
26
+ CMD ["./run.sh"]
@@ -0,0 +1 @@
1
+ name: codex