ai-push-hooks 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -1
- package/README.md +80 -525
- package/SECURITY.md +102 -14
- package/ai-push-hooks.toml +9 -2
- package/bin/ai-push-hooks.js +6 -6
- package/package.json +3 -2
- package/pyproject.toml +1 -1
- package/src/ai_push_hooks/artifacts.py +67 -13
- package/src/ai_push_hooks/config.py +575 -22
- package/src/ai_push_hooks/engine.py +116 -7
- package/src/ai_push_hooks/executors/apply.py +75 -36
- package/src/ai_push_hooks/executors/ask.py +224 -0
- package/src/ai_push_hooks/executors/exec.py +17 -801
- package/src/ai_push_hooks/executors/runner_workflow.py +478 -0
- package/src/ai_push_hooks/executors/runners/__init__.py +78 -0
- package/src/ai_push_hooks/executors/runners/claude.py +286 -0
- package/src/ai_push_hooks/executors/runners/codex.py +254 -0
- package/src/ai_push_hooks/executors/runners/command.py +178 -0
- package/src/ai_push_hooks/executors/runners/contracts.py +597 -0
- package/src/ai_push_hooks/executors/runners/opencode.py +528 -0
- package/src/ai_push_hooks/executors/runners/opencode_support.py +276 -0
- package/src/ai_push_hooks/executors/runners/process.py +464 -0
- package/src/ai_push_hooks/executors/runners/registry.py +117 -0
- package/src/ai_push_hooks/executors/step_commands.py +478 -0
- package/src/ai_push_hooks/git_utils.py +834 -0
- package/src/ai_push_hooks/hook.py +1 -1
- package/src/ai_push_hooks/modules/beads.py +1 -1
- package/src/ai_push_hooks/modules/docs.py +129 -89
- package/src/ai_push_hooks/modules/pr.py +1 -1
- package/src/ai_push_hooks/plugin_loader.py +422 -0
- package/src/ai_push_hooks/plugins.py +134 -0
- package/src/ai_push_hooks/prompts_builtin.py +9 -2
- package/src/ai_push_hooks/types.py +407 -75
- package/vendor/README.md +15 -0
- package/vendor/requirements.txt +1 -0
- package/vendor/tomli-2.4.0-py3-none-any.whl +0 -0
- package/src/ai_push_hooks/executors/llm.py +0 -624
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Claude Code CLI runner.
|
|
2
|
+
|
|
3
|
+
The adapter deliberately uses Claude's print-mode JSON result rather than its
|
|
4
|
+
internal transcript/event files. Capability discovery is performed once when
|
|
5
|
+
the adapter is constructed so a changed CLI cannot silently receive a weaker
|
|
6
|
+
permission or persistence policy.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import pathlib
|
|
14
|
+
import shutil
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from .contracts import (
|
|
18
|
+
RunnerAdapterUnavailableError,
|
|
19
|
+
RunnerCapabilities,
|
|
20
|
+
RunnerContractError,
|
|
21
|
+
RunnerError,
|
|
22
|
+
RunnerProtocolError,
|
|
23
|
+
RunnerRequest,
|
|
24
|
+
RunnerResult,
|
|
25
|
+
SessionMetadata,
|
|
26
|
+
bounded_redacted_diagnostics,
|
|
27
|
+
request_sensitive_diagnostics,
|
|
28
|
+
require_final_text,
|
|
29
|
+
require_zero_exit,
|
|
30
|
+
)
|
|
31
|
+
from .process import ProcessResult, run_process
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
CAPABILITY_CHECK_TIMEOUT_SECONDS = 10
|
|
35
|
+
CLAUDE_EXECUTABLE = "claude"
|
|
36
|
+
ANALYSIS_TOOLS = "Read,Glob,Grep"
|
|
37
|
+
APPLY_TOOLS = "Read,Edit,Write"
|
|
38
|
+
|
|
39
|
+
# These are intentionally the long, documented spellings used by the
|
|
40
|
+
# invocation. In particular, --allowed-tools alone is not enough evidence
|
|
41
|
+
# for using --allowedTools: an installed CLI must advertise the exact spelling
|
|
42
|
+
# we execute.
|
|
43
|
+
REQUIRED_HELP_MARKERS = (
|
|
44
|
+
"--print",
|
|
45
|
+
"--output-format",
|
|
46
|
+
"json",
|
|
47
|
+
"--no-session-persistence",
|
|
48
|
+
"--model",
|
|
49
|
+
"--permission-mode",
|
|
50
|
+
"dontAsk",
|
|
51
|
+
"acceptEdits",
|
|
52
|
+
"--tools",
|
|
53
|
+
"--allowedTools",
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def resolve_claude_executable() -> str:
|
|
58
|
+
"""Resolve the user-managed Claude executable without invoking it."""
|
|
59
|
+
|
|
60
|
+
executable = shutil.which(CLAUDE_EXECUTABLE)
|
|
61
|
+
if executable:
|
|
62
|
+
return executable
|
|
63
|
+
raise RunnerAdapterUnavailableError(
|
|
64
|
+
"Claude Code CLI is required but is not installed"
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _capability_error(reason: str, *, details: str = "") -> RunnerAdapterUnavailableError:
|
|
69
|
+
return RunnerAdapterUnavailableError(
|
|
70
|
+
"Claude Code CLI does not satisfy the required non-interactive contract",
|
|
71
|
+
details=f"{reason}{(': ' + details) if details else ''}",
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def check_claude_capabilities(
|
|
76
|
+
executable: str,
|
|
77
|
+
*,
|
|
78
|
+
cwd: pathlib.Path | None = None,
|
|
79
|
+
) -> None:
|
|
80
|
+
"""Check required flags and permission modes using ``claude --help`` only.
|
|
81
|
+
|
|
82
|
+
This function never supplies a prompt, model, or authentication operation.
|
|
83
|
+
A truncated help response is rejected because it cannot prove that every
|
|
84
|
+
required safety flag is supported.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
check_cwd = pathlib.Path.cwd() if cwd is None else pathlib.Path(cwd)
|
|
88
|
+
try:
|
|
89
|
+
help_result = run_process(
|
|
90
|
+
[executable, "--help"],
|
|
91
|
+
cwd=check_cwd,
|
|
92
|
+
input_text=None,
|
|
93
|
+
timeout_seconds=CAPABILITY_CHECK_TIMEOUT_SECONDS,
|
|
94
|
+
)
|
|
95
|
+
except RunnerError as exc:
|
|
96
|
+
raise _capability_error(type(exc).__name__) from exc
|
|
97
|
+
|
|
98
|
+
if help_result.returncode != 0:
|
|
99
|
+
details = bounded_redacted_diagnostics(help_result.stdout, help_result.stderr)
|
|
100
|
+
raise _capability_error("help command failed", details=details)
|
|
101
|
+
if help_result.stdout_truncated or help_result.stderr_truncated:
|
|
102
|
+
raise _capability_error("help output was truncated")
|
|
103
|
+
|
|
104
|
+
help_text = f"{help_result.stdout}\n{help_result.stderr}"
|
|
105
|
+
missing = tuple(marker for marker in REQUIRED_HELP_MARKERS if marker not in help_text)
|
|
106
|
+
if missing:
|
|
107
|
+
raise _capability_error("missing required flags or modes", details=", ".join(missing))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _protocol_failure(
|
|
111
|
+
request: RunnerRequest,
|
|
112
|
+
process_result: ProcessResult,
|
|
113
|
+
reason: str,
|
|
114
|
+
*,
|
|
115
|
+
extra_secrets: tuple[str, ...] = (),
|
|
116
|
+
) -> RunnerProtocolError:
|
|
117
|
+
details = request_sensitive_diagnostics(
|
|
118
|
+
request,
|
|
119
|
+
process_result.stdout,
|
|
120
|
+
process_result.stderr,
|
|
121
|
+
env=os.environ,
|
|
122
|
+
extra=extra_secrets,
|
|
123
|
+
)
|
|
124
|
+
return RunnerProtocolError(
|
|
125
|
+
f"Claude returned an invalid result for {request.profile_id!r} "
|
|
126
|
+
f"({request.runner_type}) at {request.stage!r}: {reason}",
|
|
127
|
+
details=details,
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def parse_claude_result(
|
|
132
|
+
request: RunnerRequest,
|
|
133
|
+
process_result: ProcessResult,
|
|
134
|
+
) -> RunnerResult:
|
|
135
|
+
"""Parse one documented top-level Claude print-mode JSON result.
|
|
136
|
+
|
|
137
|
+
The parser is additive with respect to metadata: fields such as usage,
|
|
138
|
+
cost, or future provider metadata are ignored. It is deliberately strict
|
|
139
|
+
about framing and the final result so a bounded/truncated stream cannot be
|
|
140
|
+
mistaken for a successful response.
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
if process_result.stdout_truncated or process_result.stderr_truncated:
|
|
144
|
+
streams = []
|
|
145
|
+
if process_result.stdout_truncated:
|
|
146
|
+
streams.append("stdout")
|
|
147
|
+
if process_result.stderr_truncated:
|
|
148
|
+
streams.append("stderr")
|
|
149
|
+
raise _protocol_failure(
|
|
150
|
+
request,
|
|
151
|
+
process_result,
|
|
152
|
+
"output stream was truncated: " + ", ".join(streams),
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
raw = process_result.stdout.strip()
|
|
156
|
+
if not raw:
|
|
157
|
+
raise _protocol_failure(request, process_result, "stdout was empty")
|
|
158
|
+
try:
|
|
159
|
+
payload: Any = json.loads(raw)
|
|
160
|
+
except json.JSONDecodeError as exc:
|
|
161
|
+
raise _protocol_failure(request, process_result, "stdout was not valid JSON") from exc
|
|
162
|
+
if not isinstance(payload, dict):
|
|
163
|
+
raise _protocol_failure(request, process_result, "top-level JSON value was not an object")
|
|
164
|
+
|
|
165
|
+
message_type = payload.get("type")
|
|
166
|
+
if message_type != "result":
|
|
167
|
+
raise _protocol_failure(request, process_result, "top-level JSON type was not result")
|
|
168
|
+
|
|
169
|
+
is_error = payload.get("is_error")
|
|
170
|
+
if not isinstance(is_error, bool):
|
|
171
|
+
raise _protocol_failure(request, process_result, "is_error was not boolean")
|
|
172
|
+
subtype = payload.get("subtype")
|
|
173
|
+
if not isinstance(subtype, str):
|
|
174
|
+
raise _protocol_failure(request, process_result, "subtype was not a string")
|
|
175
|
+
|
|
176
|
+
# Claude documents success as subtype=success and uses error_* subtypes for
|
|
177
|
+
# terminal failures. Treat every explicit non-success subtype as a
|
|
178
|
+
# failure, including future subtypes, rather than accepting an unknown
|
|
179
|
+
# terminal state as a response.
|
|
180
|
+
if is_error or subtype != "success":
|
|
181
|
+
# Never copy provider/model-controlled subtype text into a diagnostic.
|
|
182
|
+
raise _protocol_failure(
|
|
183
|
+
request,
|
|
184
|
+
process_result,
|
|
185
|
+
"terminal result was not successful",
|
|
186
|
+
extra_secrets=(subtype,),
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
final_text = payload.get("result")
|
|
190
|
+
if not isinstance(final_text, str):
|
|
191
|
+
raise _protocol_failure(request, process_result, "result was not a string")
|
|
192
|
+
|
|
193
|
+
session_id = payload.get("session_id")
|
|
194
|
+
if session_id is not None and not isinstance(session_id, str):
|
|
195
|
+
raise _protocol_failure(request, process_result, "session_id was not a string")
|
|
196
|
+
if isinstance(session_id, str) and not session_id.strip():
|
|
197
|
+
raise _protocol_failure(request, process_result, "session_id was empty")
|
|
198
|
+
|
|
199
|
+
return RunnerResult(
|
|
200
|
+
final_text=require_final_text(final_text, mode=request.mode),
|
|
201
|
+
returncode=process_result.returncode,
|
|
202
|
+
stdout=process_result.stdout,
|
|
203
|
+
stderr=process_result.stderr,
|
|
204
|
+
session=SessionMetadata(
|
|
205
|
+
session_id=session_id,
|
|
206
|
+
state="ephemeral",
|
|
207
|
+
resumable=False,
|
|
208
|
+
),
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def build_claude_argv(executable: str, request: RunnerRequest) -> list[str]:
|
|
213
|
+
"""Build the exact shell-free argv contract for one Claude invocation."""
|
|
214
|
+
|
|
215
|
+
if not isinstance(request, RunnerRequest):
|
|
216
|
+
raise RunnerContractError("Claude runner requires a RunnerRequest")
|
|
217
|
+
if request.runner_type != "claude":
|
|
218
|
+
raise RunnerContractError(
|
|
219
|
+
f"Claude runner cannot handle runner type {request.runner_type!r}"
|
|
220
|
+
)
|
|
221
|
+
if request.mode not in {"ask", "apply"}:
|
|
222
|
+
raise RunnerContractError("Claude runner request mode must be 'ask' or 'apply'")
|
|
223
|
+
|
|
224
|
+
tools = ANALYSIS_TOOLS if request.mode == "ask" else APPLY_TOOLS
|
|
225
|
+
argv = [
|
|
226
|
+
executable,
|
|
227
|
+
"-p",
|
|
228
|
+
"--output-format",
|
|
229
|
+
"json",
|
|
230
|
+
"--no-session-persistence",
|
|
231
|
+
]
|
|
232
|
+
if request.model is not None and request.model.strip():
|
|
233
|
+
argv.extend(["--model", request.model])
|
|
234
|
+
argv.extend(
|
|
235
|
+
[
|
|
236
|
+
"--permission-mode",
|
|
237
|
+
"dontAsk" if request.mode == "ask" else "acceptEdits",
|
|
238
|
+
"--tools",
|
|
239
|
+
tools,
|
|
240
|
+
"--allowedTools",
|
|
241
|
+
tools,
|
|
242
|
+
]
|
|
243
|
+
)
|
|
244
|
+
return argv
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
class ClaudeRunner:
|
|
248
|
+
"""Run Claude Code with an ephemeral, stage-specific tool policy."""
|
|
249
|
+
|
|
250
|
+
capabilities = RunnerCapabilities()
|
|
251
|
+
|
|
252
|
+
def __init__(self, executable: str) -> None:
|
|
253
|
+
self.executable = executable
|
|
254
|
+
|
|
255
|
+
def run(self, request: RunnerRequest) -> RunnerResult:
|
|
256
|
+
argv = build_claude_argv(self.executable, request)
|
|
257
|
+
packet = request.prompt_packet().render()
|
|
258
|
+
process_result = run_process(
|
|
259
|
+
argv,
|
|
260
|
+
cwd=request.cwd,
|
|
261
|
+
input_text=packet,
|
|
262
|
+
timeout_seconds=request.timeout_seconds,
|
|
263
|
+
# None explicitly means inherit the user's normal Claude login and
|
|
264
|
+
# provider environment. The environment is never logged.
|
|
265
|
+
env=None,
|
|
266
|
+
)
|
|
267
|
+
normalized = RunnerResult(
|
|
268
|
+
final_text="",
|
|
269
|
+
returncode=process_result.returncode,
|
|
270
|
+
stdout=process_result.stdout,
|
|
271
|
+
stderr=process_result.stderr,
|
|
272
|
+
)
|
|
273
|
+
require_zero_exit(
|
|
274
|
+
request,
|
|
275
|
+
normalized,
|
|
276
|
+
env=os.environ,
|
|
277
|
+
)
|
|
278
|
+
return parse_claude_result(request, process_result)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def create_runner() -> ClaudeRunner:
|
|
282
|
+
"""Construct a Claude runner only after proving its CLI capabilities."""
|
|
283
|
+
|
|
284
|
+
executable = resolve_claude_executable()
|
|
285
|
+
check_claude_capabilities(executable)
|
|
286
|
+
return ClaudeRunner(executable)
|
|
@@ -0,0 +1,254 @@
|
|
|
1
|
+
"""Codex CLI runner adapter.
|
|
2
|
+
|
|
3
|
+
The Codex JSON mode is a JSONL event stream rather than a single response.
|
|
4
|
+
Only completed agent-message items are considered model output, and a process
|
|
5
|
+
is successful only after a successful terminal turn and a zero exit status.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
from .contracts import (
|
|
15
|
+
RunnerCapabilities,
|
|
16
|
+
RunnerContractError,
|
|
17
|
+
RunnerMissingOutputError,
|
|
18
|
+
RunnerProtocolError,
|
|
19
|
+
RunnerRequest,
|
|
20
|
+
RunnerResult,
|
|
21
|
+
SessionMetadata,
|
|
22
|
+
request_sensitive_diagnostics,
|
|
23
|
+
require_final_text,
|
|
24
|
+
require_zero_exit,
|
|
25
|
+
)
|
|
26
|
+
from .process import ProcessResult, run_process
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
CODEX_EXECUTABLE = "codex"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass(frozen=True)
|
|
33
|
+
class _CodexOutput:
|
|
34
|
+
final_text: str
|
|
35
|
+
thread_id: str | None
|
|
36
|
+
terminal_success: bool
|
|
37
|
+
terminal_failure: bool
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _protocol_error(
|
|
41
|
+
request: RunnerRequest,
|
|
42
|
+
process: ProcessResult,
|
|
43
|
+
reason: str,
|
|
44
|
+
) -> RunnerProtocolError:
|
|
45
|
+
"""Build a bounded diagnostic without echoing prompt or environment data."""
|
|
46
|
+
|
|
47
|
+
details = request_sensitive_diagnostics(
|
|
48
|
+
request,
|
|
49
|
+
process.stdout,
|
|
50
|
+
process.stderr,
|
|
51
|
+
env=os.environ,
|
|
52
|
+
)
|
|
53
|
+
return RunnerProtocolError(
|
|
54
|
+
f"Codex runner {request.profile_id!r} ({request.runner_type}) "
|
|
55
|
+
f"failed at {request.stage!r}: {reason}",
|
|
56
|
+
details=details,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _parse_codex_jsonl(request: RunnerRequest, process: ProcessResult) -> _CodexOutput:
|
|
61
|
+
"""Parse Codex's additive JSONL event stream.
|
|
62
|
+
|
|
63
|
+
Unknown event types are intentionally ignored. Known event types with an
|
|
64
|
+
invalid shape are protocol failures, since accepting them could turn a
|
|
65
|
+
partial stream into a false success.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
final_text = ""
|
|
69
|
+
thread_id: str | None = None
|
|
70
|
+
terminal_success = False
|
|
71
|
+
terminal_failure = False
|
|
72
|
+
turn_open = False
|
|
73
|
+
|
|
74
|
+
for line_number, line in enumerate(process.stdout.splitlines(), start=1):
|
|
75
|
+
payload = line.strip()
|
|
76
|
+
if not payload:
|
|
77
|
+
continue
|
|
78
|
+
try:
|
|
79
|
+
event = json.loads(payload)
|
|
80
|
+
except json.JSONDecodeError as exc:
|
|
81
|
+
raise _protocol_error(
|
|
82
|
+
request,
|
|
83
|
+
process,
|
|
84
|
+
f"Codex emitted malformed JSONL at line {line_number}",
|
|
85
|
+
) from exc
|
|
86
|
+
if not isinstance(event, dict):
|
|
87
|
+
raise _protocol_error(
|
|
88
|
+
request,
|
|
89
|
+
process,
|
|
90
|
+
f"Codex JSONL event at line {line_number} was not an object",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
event_type = event.get("type")
|
|
94
|
+
if not isinstance(event_type, str):
|
|
95
|
+
# Future event formats may add fields, but every event still needs
|
|
96
|
+
# the discriminator used by the documented JSONL protocol.
|
|
97
|
+
raise _protocol_error(
|
|
98
|
+
request,
|
|
99
|
+
process,
|
|
100
|
+
f"Codex JSONL event at line {line_number} had no type",
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
if event_type == "thread.started":
|
|
104
|
+
value = event.get("thread_id")
|
|
105
|
+
if not isinstance(value, str) or not value.strip():
|
|
106
|
+
raise _protocol_error(
|
|
107
|
+
request,
|
|
108
|
+
process,
|
|
109
|
+
"Codex thread.started event had no thread id",
|
|
110
|
+
)
|
|
111
|
+
thread_id = value
|
|
112
|
+
elif event_type == "turn.started":
|
|
113
|
+
if turn_open:
|
|
114
|
+
raise _protocol_error(request, process, "Codex JSONL turn order was invalid")
|
|
115
|
+
turn_open = True
|
|
116
|
+
if not terminal_failure:
|
|
117
|
+
terminal_success = False
|
|
118
|
+
elif event_type == "item.completed":
|
|
119
|
+
item = event.get("item")
|
|
120
|
+
if not isinstance(item, dict):
|
|
121
|
+
raise _protocol_error(request, process, "Codex item.completed event had no item")
|
|
122
|
+
if item.get("type") == "agent_message":
|
|
123
|
+
text = item.get("text")
|
|
124
|
+
if not isinstance(text, str):
|
|
125
|
+
raise _protocol_error(
|
|
126
|
+
request,
|
|
127
|
+
process,
|
|
128
|
+
"Codex agent_message item had no text",
|
|
129
|
+
)
|
|
130
|
+
# Multiple completed messages are additive; the last one is
|
|
131
|
+
# the final response for this invocation.
|
|
132
|
+
final_text = text
|
|
133
|
+
elif event_type == "turn.completed":
|
|
134
|
+
status = event.get("status")
|
|
135
|
+
turn_open = False
|
|
136
|
+
if isinstance(status, str) and status.lower() not in {
|
|
137
|
+
"completed",
|
|
138
|
+
"success",
|
|
139
|
+
"succeeded",
|
|
140
|
+
}:
|
|
141
|
+
terminal_failure = True
|
|
142
|
+
terminal_success = False
|
|
143
|
+
elif event.get("error"):
|
|
144
|
+
terminal_failure = True
|
|
145
|
+
terminal_success = False
|
|
146
|
+
elif not terminal_failure:
|
|
147
|
+
terminal_success = True
|
|
148
|
+
elif event_type in {"turn.failed", "error"}:
|
|
149
|
+
terminal_failure = True
|
|
150
|
+
terminal_success = False
|
|
151
|
+
turn_open = False
|
|
152
|
+
|
|
153
|
+
return _CodexOutput(final_text, thread_id, terminal_success, terminal_failure)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
@dataclass(frozen=True)
|
|
157
|
+
class CodexRunner:
|
|
158
|
+
"""Run one fresh, ephemeral Codex CLI invocation."""
|
|
159
|
+
|
|
160
|
+
executable: str = CODEX_EXECUTABLE
|
|
161
|
+
capabilities: RunnerCapabilities = RunnerCapabilities()
|
|
162
|
+
|
|
163
|
+
def run(self, request: RunnerRequest) -> RunnerResult:
|
|
164
|
+
if request.runner_type != "codex":
|
|
165
|
+
raise RunnerContractError(
|
|
166
|
+
f"Codex runner cannot handle runner type {request.runner_type!r}"
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
sandbox = "read-only" if request.mode == "ask" else "workspace-write"
|
|
170
|
+
argv = [
|
|
171
|
+
self.executable,
|
|
172
|
+
"exec",
|
|
173
|
+
"--json",
|
|
174
|
+
"--color",
|
|
175
|
+
"never",
|
|
176
|
+
"--sandbox",
|
|
177
|
+
sandbox,
|
|
178
|
+
"--ephemeral",
|
|
179
|
+
"--cd",
|
|
180
|
+
str(request.cwd),
|
|
181
|
+
]
|
|
182
|
+
if request.model:
|
|
183
|
+
argv.extend(("--model", request.model))
|
|
184
|
+
|
|
185
|
+
# Artifact-only analysis runs in a scratch directory and apply always
|
|
186
|
+
# runs in a non-VCS staging projection. Both need this flag; a
|
|
187
|
+
# project-aware analysis run normally has a Git repository available.
|
|
188
|
+
if request.mode == "apply" or (
|
|
189
|
+
request.mode == "ask" and request.project_access == "artifacts"
|
|
190
|
+
):
|
|
191
|
+
argv.append("--skip-git-repo-check")
|
|
192
|
+
argv.append("-")
|
|
193
|
+
|
|
194
|
+
process = run_process(
|
|
195
|
+
argv,
|
|
196
|
+
cwd=request.cwd,
|
|
197
|
+
input_text=request.prompt_packet().render(),
|
|
198
|
+
timeout_seconds=request.timeout_seconds,
|
|
199
|
+
env=None,
|
|
200
|
+
)
|
|
201
|
+
result = RunnerResult(
|
|
202
|
+
final_text="",
|
|
203
|
+
returncode=process.returncode,
|
|
204
|
+
stdout=process.stdout,
|
|
205
|
+
stderr=process.stderr,
|
|
206
|
+
)
|
|
207
|
+
require_zero_exit(
|
|
208
|
+
request,
|
|
209
|
+
result,
|
|
210
|
+
env=os.environ,
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
if process.stdout_truncated or process.stderr_truncated:
|
|
214
|
+
raise _protocol_error(request, process, "Codex output stream was truncated")
|
|
215
|
+
if not process.stdout.strip():
|
|
216
|
+
raise _protocol_error(request, process, "Codex emitted an empty JSONL stream")
|
|
217
|
+
|
|
218
|
+
parsed = _parse_codex_jsonl(request, process)
|
|
219
|
+
if not parsed.terminal_success:
|
|
220
|
+
reason = (
|
|
221
|
+
"Codex terminal turn failed"
|
|
222
|
+
if parsed.terminal_failure
|
|
223
|
+
else "Codex JSONL stream had no successful terminal turn"
|
|
224
|
+
)
|
|
225
|
+
raise _protocol_error(request, process, reason)
|
|
226
|
+
|
|
227
|
+
try:
|
|
228
|
+
final_text = require_final_text(parsed.final_text, mode=request.mode)
|
|
229
|
+
except RunnerMissingOutputError as exc:
|
|
230
|
+
details = request_sensitive_diagnostics(
|
|
231
|
+
request,
|
|
232
|
+
process.stdout,
|
|
233
|
+
process.stderr,
|
|
234
|
+
env=os.environ,
|
|
235
|
+
)
|
|
236
|
+
raise RunnerMissingOutputError(str(exc), details=details) from exc
|
|
237
|
+
|
|
238
|
+
return RunnerResult(
|
|
239
|
+
final_text=final_text,
|
|
240
|
+
returncode=process.returncode,
|
|
241
|
+
stdout=process.stdout,
|
|
242
|
+
stderr=process.stderr,
|
|
243
|
+
session=SessionMetadata(
|
|
244
|
+
session_id=parsed.thread_id,
|
|
245
|
+
state="ephemeral",
|
|
246
|
+
resumable=False,
|
|
247
|
+
),
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def create_runner() -> CodexRunner:
|
|
252
|
+
"""Construct the built-in Codex adapter for the static runner registry."""
|
|
253
|
+
|
|
254
|
+
return CodexRunner()
|