ai-push-hooks 0.2.1 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +73 -1
  2. package/README.md +80 -525
  3. package/SECURITY.md +102 -14
  4. package/ai-push-hooks.toml +9 -2
  5. package/bin/ai-push-hooks.js +6 -6
  6. package/package.json +3 -2
  7. package/pyproject.toml +1 -1
  8. package/src/ai_push_hooks/artifacts.py +67 -13
  9. package/src/ai_push_hooks/config.py +575 -22
  10. package/src/ai_push_hooks/engine.py +116 -7
  11. package/src/ai_push_hooks/executors/apply.py +75 -36
  12. package/src/ai_push_hooks/executors/ask.py +224 -0
  13. package/src/ai_push_hooks/executors/exec.py +17 -801
  14. package/src/ai_push_hooks/executors/runner_workflow.py +478 -0
  15. package/src/ai_push_hooks/executors/runners/__init__.py +78 -0
  16. package/src/ai_push_hooks/executors/runners/claude.py +286 -0
  17. package/src/ai_push_hooks/executors/runners/codex.py +254 -0
  18. package/src/ai_push_hooks/executors/runners/command.py +178 -0
  19. package/src/ai_push_hooks/executors/runners/contracts.py +597 -0
  20. package/src/ai_push_hooks/executors/runners/opencode.py +528 -0
  21. package/src/ai_push_hooks/executors/runners/opencode_support.py +276 -0
  22. package/src/ai_push_hooks/executors/runners/process.py +464 -0
  23. package/src/ai_push_hooks/executors/runners/registry.py +117 -0
  24. package/src/ai_push_hooks/executors/step_commands.py +478 -0
  25. package/src/ai_push_hooks/git_utils.py +834 -0
  26. package/src/ai_push_hooks/hook.py +1 -1
  27. package/src/ai_push_hooks/modules/beads.py +1 -1
  28. package/src/ai_push_hooks/modules/docs.py +129 -89
  29. package/src/ai_push_hooks/modules/pr.py +1 -1
  30. package/src/ai_push_hooks/plugin_loader.py +422 -0
  31. package/src/ai_push_hooks/plugins.py +134 -0
  32. package/src/ai_push_hooks/prompts_builtin.py +9 -2
  33. package/src/ai_push_hooks/types.py +407 -75
  34. package/vendor/README.md +15 -0
  35. package/vendor/requirements.txt +1 -0
  36. package/vendor/tomli-2.4.0-py3-none-any.whl +0 -0
  37. package/src/ai_push_hooks/executors/llm.py +0 -624
@@ -0,0 +1,286 @@
1
+ """Claude Code CLI runner.
2
+
3
+ The adapter deliberately uses Claude's print-mode JSON result rather than its
4
+ internal transcript/event files. Capability discovery is performed once when
5
+ the adapter is constructed so a changed CLI cannot silently receive a weaker
6
+ permission or persistence policy.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import os
13
+ import pathlib
14
+ import shutil
15
+ from typing import Any
16
+
17
+ from .contracts import (
18
+ RunnerAdapterUnavailableError,
19
+ RunnerCapabilities,
20
+ RunnerContractError,
21
+ RunnerError,
22
+ RunnerProtocolError,
23
+ RunnerRequest,
24
+ RunnerResult,
25
+ SessionMetadata,
26
+ bounded_redacted_diagnostics,
27
+ request_sensitive_diagnostics,
28
+ require_final_text,
29
+ require_zero_exit,
30
+ )
31
+ from .process import ProcessResult, run_process
32
+
33
+
34
+ CAPABILITY_CHECK_TIMEOUT_SECONDS = 10
35
+ CLAUDE_EXECUTABLE = "claude"
36
+ ANALYSIS_TOOLS = "Read,Glob,Grep"
37
+ APPLY_TOOLS = "Read,Edit,Write"
38
+
39
+ # These are intentionally the long, documented spellings used by the
40
+ # invocation. In particular, --allowed-tools alone is not enough evidence
41
+ # for using --allowedTools: an installed CLI must advertise the exact spelling
42
+ # we execute.
43
+ REQUIRED_HELP_MARKERS = (
44
+ "--print",
45
+ "--output-format",
46
+ "json",
47
+ "--no-session-persistence",
48
+ "--model",
49
+ "--permission-mode",
50
+ "dontAsk",
51
+ "acceptEdits",
52
+ "--tools",
53
+ "--allowedTools",
54
+ )
55
+
56
+
57
+ def resolve_claude_executable() -> str:
58
+ """Resolve the user-managed Claude executable without invoking it."""
59
+
60
+ executable = shutil.which(CLAUDE_EXECUTABLE)
61
+ if executable:
62
+ return executable
63
+ raise RunnerAdapterUnavailableError(
64
+ "Claude Code CLI is required but is not installed"
65
+ )
66
+
67
+
68
+ def _capability_error(reason: str, *, details: str = "") -> RunnerAdapterUnavailableError:
69
+ return RunnerAdapterUnavailableError(
70
+ "Claude Code CLI does not satisfy the required non-interactive contract",
71
+ details=f"{reason}{(': ' + details) if details else ''}",
72
+ )
73
+
74
+
75
+ def check_claude_capabilities(
76
+ executable: str,
77
+ *,
78
+ cwd: pathlib.Path | None = None,
79
+ ) -> None:
80
+ """Check required flags and permission modes using ``claude --help`` only.
81
+
82
+ This function never supplies a prompt, model, or authentication operation.
83
+ A truncated help response is rejected because it cannot prove that every
84
+ required safety flag is supported.
85
+ """
86
+
87
+ check_cwd = pathlib.Path.cwd() if cwd is None else pathlib.Path(cwd)
88
+ try:
89
+ help_result = run_process(
90
+ [executable, "--help"],
91
+ cwd=check_cwd,
92
+ input_text=None,
93
+ timeout_seconds=CAPABILITY_CHECK_TIMEOUT_SECONDS,
94
+ )
95
+ except RunnerError as exc:
96
+ raise _capability_error(type(exc).__name__) from exc
97
+
98
+ if help_result.returncode != 0:
99
+ details = bounded_redacted_diagnostics(help_result.stdout, help_result.stderr)
100
+ raise _capability_error("help command failed", details=details)
101
+ if help_result.stdout_truncated or help_result.stderr_truncated:
102
+ raise _capability_error("help output was truncated")
103
+
104
+ help_text = f"{help_result.stdout}\n{help_result.stderr}"
105
+ missing = tuple(marker for marker in REQUIRED_HELP_MARKERS if marker not in help_text)
106
+ if missing:
107
+ raise _capability_error("missing required flags or modes", details=", ".join(missing))
108
+
109
+
110
+ def _protocol_failure(
111
+ request: RunnerRequest,
112
+ process_result: ProcessResult,
113
+ reason: str,
114
+ *,
115
+ extra_secrets: tuple[str, ...] = (),
116
+ ) -> RunnerProtocolError:
117
+ details = request_sensitive_diagnostics(
118
+ request,
119
+ process_result.stdout,
120
+ process_result.stderr,
121
+ env=os.environ,
122
+ extra=extra_secrets,
123
+ )
124
+ return RunnerProtocolError(
125
+ f"Claude returned an invalid result for {request.profile_id!r} "
126
+ f"({request.runner_type}) at {request.stage!r}: {reason}",
127
+ details=details,
128
+ )
129
+
130
+
131
+ def parse_claude_result(
132
+ request: RunnerRequest,
133
+ process_result: ProcessResult,
134
+ ) -> RunnerResult:
135
+ """Parse one documented top-level Claude print-mode JSON result.
136
+
137
+ The parser is additive with respect to metadata: fields such as usage,
138
+ cost, or future provider metadata are ignored. It is deliberately strict
139
+ about framing and the final result so a bounded/truncated stream cannot be
140
+ mistaken for a successful response.
141
+ """
142
+
143
+ if process_result.stdout_truncated or process_result.stderr_truncated:
144
+ streams = []
145
+ if process_result.stdout_truncated:
146
+ streams.append("stdout")
147
+ if process_result.stderr_truncated:
148
+ streams.append("stderr")
149
+ raise _protocol_failure(
150
+ request,
151
+ process_result,
152
+ "output stream was truncated: " + ", ".join(streams),
153
+ )
154
+
155
+ raw = process_result.stdout.strip()
156
+ if not raw:
157
+ raise _protocol_failure(request, process_result, "stdout was empty")
158
+ try:
159
+ payload: Any = json.loads(raw)
160
+ except json.JSONDecodeError as exc:
161
+ raise _protocol_failure(request, process_result, "stdout was not valid JSON") from exc
162
+ if not isinstance(payload, dict):
163
+ raise _protocol_failure(request, process_result, "top-level JSON value was not an object")
164
+
165
+ message_type = payload.get("type")
166
+ if message_type != "result":
167
+ raise _protocol_failure(request, process_result, "top-level JSON type was not result")
168
+
169
+ is_error = payload.get("is_error")
170
+ if not isinstance(is_error, bool):
171
+ raise _protocol_failure(request, process_result, "is_error was not boolean")
172
+ subtype = payload.get("subtype")
173
+ if not isinstance(subtype, str):
174
+ raise _protocol_failure(request, process_result, "subtype was not a string")
175
+
176
+ # Claude documents success as subtype=success and uses error_* subtypes for
177
+ # terminal failures. Treat every explicit non-success subtype as a
178
+ # failure, including future subtypes, rather than accepting an unknown
179
+ # terminal state as a response.
180
+ if is_error or subtype != "success":
181
+ # Never copy provider/model-controlled subtype text into a diagnostic.
182
+ raise _protocol_failure(
183
+ request,
184
+ process_result,
185
+ "terminal result was not successful",
186
+ extra_secrets=(subtype,),
187
+ )
188
+
189
+ final_text = payload.get("result")
190
+ if not isinstance(final_text, str):
191
+ raise _protocol_failure(request, process_result, "result was not a string")
192
+
193
+ session_id = payload.get("session_id")
194
+ if session_id is not None and not isinstance(session_id, str):
195
+ raise _protocol_failure(request, process_result, "session_id was not a string")
196
+ if isinstance(session_id, str) and not session_id.strip():
197
+ raise _protocol_failure(request, process_result, "session_id was empty")
198
+
199
+ return RunnerResult(
200
+ final_text=require_final_text(final_text, mode=request.mode),
201
+ returncode=process_result.returncode,
202
+ stdout=process_result.stdout,
203
+ stderr=process_result.stderr,
204
+ session=SessionMetadata(
205
+ session_id=session_id,
206
+ state="ephemeral",
207
+ resumable=False,
208
+ ),
209
+ )
210
+
211
+
212
+ def build_claude_argv(executable: str, request: RunnerRequest) -> list[str]:
213
+ """Build the exact shell-free argv contract for one Claude invocation."""
214
+
215
+ if not isinstance(request, RunnerRequest):
216
+ raise RunnerContractError("Claude runner requires a RunnerRequest")
217
+ if request.runner_type != "claude":
218
+ raise RunnerContractError(
219
+ f"Claude runner cannot handle runner type {request.runner_type!r}"
220
+ )
221
+ if request.mode not in {"ask", "apply"}:
222
+ raise RunnerContractError("Claude runner request mode must be 'ask' or 'apply'")
223
+
224
+ tools = ANALYSIS_TOOLS if request.mode == "ask" else APPLY_TOOLS
225
+ argv = [
226
+ executable,
227
+ "-p",
228
+ "--output-format",
229
+ "json",
230
+ "--no-session-persistence",
231
+ ]
232
+ if request.model is not None and request.model.strip():
233
+ argv.extend(["--model", request.model])
234
+ argv.extend(
235
+ [
236
+ "--permission-mode",
237
+ "dontAsk" if request.mode == "ask" else "acceptEdits",
238
+ "--tools",
239
+ tools,
240
+ "--allowedTools",
241
+ tools,
242
+ ]
243
+ )
244
+ return argv
245
+
246
+
247
+ class ClaudeRunner:
248
+ """Run Claude Code with an ephemeral, stage-specific tool policy."""
249
+
250
+ capabilities = RunnerCapabilities()
251
+
252
+ def __init__(self, executable: str) -> None:
253
+ self.executable = executable
254
+
255
+ def run(self, request: RunnerRequest) -> RunnerResult:
256
+ argv = build_claude_argv(self.executable, request)
257
+ packet = request.prompt_packet().render()
258
+ process_result = run_process(
259
+ argv,
260
+ cwd=request.cwd,
261
+ input_text=packet,
262
+ timeout_seconds=request.timeout_seconds,
263
+ # None explicitly means inherit the user's normal Claude login and
264
+ # provider environment. The environment is never logged.
265
+ env=None,
266
+ )
267
+ normalized = RunnerResult(
268
+ final_text="",
269
+ returncode=process_result.returncode,
270
+ stdout=process_result.stdout,
271
+ stderr=process_result.stderr,
272
+ )
273
+ require_zero_exit(
274
+ request,
275
+ normalized,
276
+ env=os.environ,
277
+ )
278
+ return parse_claude_result(request, process_result)
279
+
280
+
281
+ def create_runner() -> ClaudeRunner:
282
+ """Construct a Claude runner only after proving its CLI capabilities."""
283
+
284
+ executable = resolve_claude_executable()
285
+ check_claude_capabilities(executable)
286
+ return ClaudeRunner(executable)
@@ -0,0 +1,254 @@
1
+ """Codex CLI runner adapter.
2
+
3
+ The Codex JSON mode is a JSONL event stream rather than a single response.
4
+ Only completed agent-message items are considered model output, and a process
5
+ is successful only after a successful terminal turn and a zero exit status.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import os
12
+ from dataclasses import dataclass
13
+
14
+ from .contracts import (
15
+ RunnerCapabilities,
16
+ RunnerContractError,
17
+ RunnerMissingOutputError,
18
+ RunnerProtocolError,
19
+ RunnerRequest,
20
+ RunnerResult,
21
+ SessionMetadata,
22
+ request_sensitive_diagnostics,
23
+ require_final_text,
24
+ require_zero_exit,
25
+ )
26
+ from .process import ProcessResult, run_process
27
+
28
+
29
+ CODEX_EXECUTABLE = "codex"
30
+
31
+
32
+ @dataclass(frozen=True)
33
+ class _CodexOutput:
34
+ final_text: str
35
+ thread_id: str | None
36
+ terminal_success: bool
37
+ terminal_failure: bool
38
+
39
+
40
+ def _protocol_error(
41
+ request: RunnerRequest,
42
+ process: ProcessResult,
43
+ reason: str,
44
+ ) -> RunnerProtocolError:
45
+ """Build a bounded diagnostic without echoing prompt or environment data."""
46
+
47
+ details = request_sensitive_diagnostics(
48
+ request,
49
+ process.stdout,
50
+ process.stderr,
51
+ env=os.environ,
52
+ )
53
+ return RunnerProtocolError(
54
+ f"Codex runner {request.profile_id!r} ({request.runner_type}) "
55
+ f"failed at {request.stage!r}: {reason}",
56
+ details=details,
57
+ )
58
+
59
+
60
+ def _parse_codex_jsonl(request: RunnerRequest, process: ProcessResult) -> _CodexOutput:
61
+ """Parse Codex's additive JSONL event stream.
62
+
63
+ Unknown event types are intentionally ignored. Known event types with an
64
+ invalid shape are protocol failures, since accepting them could turn a
65
+ partial stream into a false success.
66
+ """
67
+
68
+ final_text = ""
69
+ thread_id: str | None = None
70
+ terminal_success = False
71
+ terminal_failure = False
72
+ turn_open = False
73
+
74
+ for line_number, line in enumerate(process.stdout.splitlines(), start=1):
75
+ payload = line.strip()
76
+ if not payload:
77
+ continue
78
+ try:
79
+ event = json.loads(payload)
80
+ except json.JSONDecodeError as exc:
81
+ raise _protocol_error(
82
+ request,
83
+ process,
84
+ f"Codex emitted malformed JSONL at line {line_number}",
85
+ ) from exc
86
+ if not isinstance(event, dict):
87
+ raise _protocol_error(
88
+ request,
89
+ process,
90
+ f"Codex JSONL event at line {line_number} was not an object",
91
+ )
92
+
93
+ event_type = event.get("type")
94
+ if not isinstance(event_type, str):
95
+ # Future event formats may add fields, but every event still needs
96
+ # the discriminator used by the documented JSONL protocol.
97
+ raise _protocol_error(
98
+ request,
99
+ process,
100
+ f"Codex JSONL event at line {line_number} had no type",
101
+ )
102
+
103
+ if event_type == "thread.started":
104
+ value = event.get("thread_id")
105
+ if not isinstance(value, str) or not value.strip():
106
+ raise _protocol_error(
107
+ request,
108
+ process,
109
+ "Codex thread.started event had no thread id",
110
+ )
111
+ thread_id = value
112
+ elif event_type == "turn.started":
113
+ if turn_open:
114
+ raise _protocol_error(request, process, "Codex JSONL turn order was invalid")
115
+ turn_open = True
116
+ if not terminal_failure:
117
+ terminal_success = False
118
+ elif event_type == "item.completed":
119
+ item = event.get("item")
120
+ if not isinstance(item, dict):
121
+ raise _protocol_error(request, process, "Codex item.completed event had no item")
122
+ if item.get("type") == "agent_message":
123
+ text = item.get("text")
124
+ if not isinstance(text, str):
125
+ raise _protocol_error(
126
+ request,
127
+ process,
128
+ "Codex agent_message item had no text",
129
+ )
130
+ # Multiple completed messages are additive; the last one is
131
+ # the final response for this invocation.
132
+ final_text = text
133
+ elif event_type == "turn.completed":
134
+ status = event.get("status")
135
+ turn_open = False
136
+ if isinstance(status, str) and status.lower() not in {
137
+ "completed",
138
+ "success",
139
+ "succeeded",
140
+ }:
141
+ terminal_failure = True
142
+ terminal_success = False
143
+ elif event.get("error"):
144
+ terminal_failure = True
145
+ terminal_success = False
146
+ elif not terminal_failure:
147
+ terminal_success = True
148
+ elif event_type in {"turn.failed", "error"}:
149
+ terminal_failure = True
150
+ terminal_success = False
151
+ turn_open = False
152
+
153
+ return _CodexOutput(final_text, thread_id, terminal_success, terminal_failure)
154
+
155
+
156
+ @dataclass(frozen=True)
157
+ class CodexRunner:
158
+ """Run one fresh, ephemeral Codex CLI invocation."""
159
+
160
+ executable: str = CODEX_EXECUTABLE
161
+ capabilities: RunnerCapabilities = RunnerCapabilities()
162
+
163
+ def run(self, request: RunnerRequest) -> RunnerResult:
164
+ if request.runner_type != "codex":
165
+ raise RunnerContractError(
166
+ f"Codex runner cannot handle runner type {request.runner_type!r}"
167
+ )
168
+
169
+ sandbox = "read-only" if request.mode == "ask" else "workspace-write"
170
+ argv = [
171
+ self.executable,
172
+ "exec",
173
+ "--json",
174
+ "--color",
175
+ "never",
176
+ "--sandbox",
177
+ sandbox,
178
+ "--ephemeral",
179
+ "--cd",
180
+ str(request.cwd),
181
+ ]
182
+ if request.model:
183
+ argv.extend(("--model", request.model))
184
+
185
+ # Artifact-only analysis runs in a scratch directory and apply always
186
+ # runs in a non-VCS staging projection. Both need this flag; a
187
+ # project-aware analysis run normally has a Git repository available.
188
+ if request.mode == "apply" or (
189
+ request.mode == "ask" and request.project_access == "artifacts"
190
+ ):
191
+ argv.append("--skip-git-repo-check")
192
+ argv.append("-")
193
+
194
+ process = run_process(
195
+ argv,
196
+ cwd=request.cwd,
197
+ input_text=request.prompt_packet().render(),
198
+ timeout_seconds=request.timeout_seconds,
199
+ env=None,
200
+ )
201
+ result = RunnerResult(
202
+ final_text="",
203
+ returncode=process.returncode,
204
+ stdout=process.stdout,
205
+ stderr=process.stderr,
206
+ )
207
+ require_zero_exit(
208
+ request,
209
+ result,
210
+ env=os.environ,
211
+ )
212
+
213
+ if process.stdout_truncated or process.stderr_truncated:
214
+ raise _protocol_error(request, process, "Codex output stream was truncated")
215
+ if not process.stdout.strip():
216
+ raise _protocol_error(request, process, "Codex emitted an empty JSONL stream")
217
+
218
+ parsed = _parse_codex_jsonl(request, process)
219
+ if not parsed.terminal_success:
220
+ reason = (
221
+ "Codex terminal turn failed"
222
+ if parsed.terminal_failure
223
+ else "Codex JSONL stream had no successful terminal turn"
224
+ )
225
+ raise _protocol_error(request, process, reason)
226
+
227
+ try:
228
+ final_text = require_final_text(parsed.final_text, mode=request.mode)
229
+ except RunnerMissingOutputError as exc:
230
+ details = request_sensitive_diagnostics(
231
+ request,
232
+ process.stdout,
233
+ process.stderr,
234
+ env=os.environ,
235
+ )
236
+ raise RunnerMissingOutputError(str(exc), details=details) from exc
237
+
238
+ return RunnerResult(
239
+ final_text=final_text,
240
+ returncode=process.returncode,
241
+ stdout=process.stdout,
242
+ stderr=process.stderr,
243
+ session=SessionMetadata(
244
+ session_id=parsed.thread_id,
245
+ state="ephemeral",
246
+ resumable=False,
247
+ ),
248
+ )
249
+
250
+
251
+ def create_runner() -> CodexRunner:
252
+ """Construct the built-in Codex adapter for the static runner registry."""
253
+
254
+ return CodexRunner()