ai-push-hooks 0.2.1 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/CHANGELOG.md +73 -1
  2. package/README.md +80 -525
  3. package/SECURITY.md +102 -14
  4. package/ai-push-hooks.toml +9 -2
  5. package/bin/ai-push-hooks.js +6 -6
  6. package/package.json +3 -2
  7. package/pyproject.toml +1 -1
  8. package/src/ai_push_hooks/artifacts.py +67 -13
  9. package/src/ai_push_hooks/config.py +575 -22
  10. package/src/ai_push_hooks/engine.py +116 -7
  11. package/src/ai_push_hooks/executors/apply.py +75 -36
  12. package/src/ai_push_hooks/executors/ask.py +224 -0
  13. package/src/ai_push_hooks/executors/exec.py +17 -801
  14. package/src/ai_push_hooks/executors/runner_workflow.py +478 -0
  15. package/src/ai_push_hooks/executors/runners/__init__.py +78 -0
  16. package/src/ai_push_hooks/executors/runners/claude.py +286 -0
  17. package/src/ai_push_hooks/executors/runners/codex.py +254 -0
  18. package/src/ai_push_hooks/executors/runners/command.py +178 -0
  19. package/src/ai_push_hooks/executors/runners/contracts.py +597 -0
  20. package/src/ai_push_hooks/executors/runners/opencode.py +528 -0
  21. package/src/ai_push_hooks/executors/runners/opencode_support.py +276 -0
  22. package/src/ai_push_hooks/executors/runners/process.py +464 -0
  23. package/src/ai_push_hooks/executors/runners/registry.py +117 -0
  24. package/src/ai_push_hooks/executors/step_commands.py +478 -0
  25. package/src/ai_push_hooks/git_utils.py +834 -0
  26. package/src/ai_push_hooks/hook.py +1 -1
  27. package/src/ai_push_hooks/modules/beads.py +1 -1
  28. package/src/ai_push_hooks/modules/docs.py +129 -89
  29. package/src/ai_push_hooks/modules/pr.py +1 -1
  30. package/src/ai_push_hooks/plugin_loader.py +422 -0
  31. package/src/ai_push_hooks/plugins.py +134 -0
  32. package/src/ai_push_hooks/prompts_builtin.py +9 -2
  33. package/src/ai_push_hooks/types.py +407 -75
  34. package/vendor/README.md +15 -0
  35. package/vendor/requirements.txt +1 -0
  36. package/vendor/tomli-2.4.0-py3-none-any.whl +0 -0
  37. package/src/ai_push_hooks/executors/llm.py +0 -624
@@ -10,7 +10,7 @@ from .artifacts import ArtifactStore, generate_run_id
10
10
  from .config import load_config
11
11
  from .engine import WorkflowEngine
12
12
  from .paths import ensure_private_directory, resolve_contained_path, write_text_no_follow
13
- from .executors.exec import (
13
+ from .git_utils import (
14
14
  collect_changed_files,
15
15
  collect_diff,
16
16
  collect_revision_ranges,
@@ -2,7 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  from typing import Any
4
4
 
5
- from ..executors.exec import collect_commit_messages_for_ranges, is_feature_branch
5
+ from ..git_utils import collect_commit_messages_for_ranges, is_feature_branch
6
6
  from ..types import CollectorResult, RuntimeContext
7
7
 
8
8
 
@@ -1,6 +1,5 @@
1
1
  from __future__ import annotations
2
2
 
3
- import json
4
3
  import os
5
4
  import pathlib
6
5
  import re
@@ -10,7 +9,7 @@ from pathlib import PurePosixPath
10
9
  from typing import Any
11
10
 
12
11
  from ..types import CollectorResult, RuntimeContext
13
- from ..executors.exec import collect_commit_messages_for_ranges, git, path_matches, run_command
12
+ from ..git_utils import collect_commit_messages_for_ranges, git, path_matches
14
13
 
15
14
  DOC_INCLUDE_PATTERNS = ("README.md", "docs/**/*.md")
16
15
  DOC_IGNORE_PATTERNS = ("docs/archive/**",)
@@ -18,6 +17,10 @@ DOC_CONTEXT_LINES = 2
18
17
  DOC_MAX_BYTES = 64 * 1024
19
18
  DOC_CONTEXT_BUDGET = 32000
20
19
  DOC_FALLBACK_FILE_LIMIT = 8
20
+ # Matching metadata is deliberately bounded independently of the document
21
+ # inventory. The normal context budget is reached much earlier, while this
22
+ # cap prevents a pathological repeated-match file from retaining every hit.
23
+ DOC_MAX_QUERY_MATCHES = 4096
21
24
 
22
25
 
23
26
  def _path_matches(path: str, patterns: tuple[str, ...]) -> bool:
@@ -126,25 +129,70 @@ def _read_bounded_text(path: pathlib.Path, max_bytes: int | None = None) -> str:
126
129
  os.close(descriptor)
127
130
 
128
131
 
129
- def _append_context_chunk(chunks: list[str], chunk: str, budget: int) -> bool:
130
- current_size = sum(len(item) for item in chunks) + max(0, len(chunks) - 1)
131
- remaining = budget - current_size
132
- if remaining <= 0:
133
- return False
134
- truncated = len(chunk) > remaining
135
- if truncated:
136
- if chunks:
132
+ class _ContextAccumulator:
133
+ """Collect bounded snippets without rescanning all prior snippets."""
134
+
135
+ def __init__(self) -> None:
136
+ self.chunks: list[str] = []
137
+ self.size = 0
138
+
139
+ def append(self, chunk: str, budget: int) -> bool:
140
+ remaining = budget - self.size
141
+ separator_size = 1 if self.chunks else 0
142
+ remaining -= separator_size
143
+ if remaining <= 0:
137
144
  return False
138
- chunk = chunk[:remaining]
139
- chunks.append(chunk)
140
- return not truncated
145
+ truncated = len(chunk) > remaining
146
+ if truncated:
147
+ if self.chunks:
148
+ return False
149
+ chunk = chunk[:remaining]
150
+ self.chunks.append(chunk)
151
+ self.size += separator_size + len(chunk)
152
+ return not truncated
141
153
 
154
+ def render(self) -> str:
155
+ return "\n".join(self.chunks)
142
156
 
143
- def _fallback_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path]) -> str:
157
+
158
+ class _QueryMatchBuffer:
159
+ """Bounded query-ranked snippet metadata for one document scan."""
160
+
161
+ def __init__(self, max_matches: int = DOC_MAX_QUERY_MATCHES) -> None:
162
+ self.max_matches = max_matches
163
+ self.matches: list[tuple[tuple[int, int], str]] = []
164
+ self._seen: set[tuple[int, int]] = set()
165
+ self.characters = 0
166
+ self.saturated = False
167
+
168
+ def add(
169
+ self,
170
+ key: tuple[int, int],
171
+ relative: str,
172
+ line_number: int,
173
+ line: str,
174
+ ) -> None:
175
+ if key in self._seen or self.saturated or len(self.matches) >= self.max_matches:
176
+ return
177
+ chunk = f"{relative}:{line_number}: {line}"
178
+ self._seen.add(key)
179
+ self.matches.append((key, chunk))
180
+ self.characters += len(chunk)
181
+ self.saturated = (
182
+ len(self.matches) >= self.max_matches
183
+ or self.characters >= DOC_CONTEXT_BUDGET
184
+ )
185
+
186
+
187
+ def _fallback_docs_context(
188
+ repo_root: pathlib.Path,
189
+ doc_files: list[pathlib.Path],
190
+ contents: dict[pathlib.Path, str] | None = None,
191
+ ) -> str:
144
192
  snippets: list[str] = []
145
193
  for path in doc_files[:DOC_FALLBACK_FILE_LIMIT]:
146
194
  relative = path.relative_to(repo_root).as_posix()
147
- content = _read_bounded_text(path)
195
+ content = _read_bounded_text(path) if contents is None else contents[path]
148
196
  block = f"--- {relative} ---\n{content}"
149
197
  current_size = len("\n\n".join(snippets))
150
198
  remaining = DOC_CONTEXT_BUDGET - current_size
@@ -156,6 +204,54 @@ def _fallback_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path
156
204
  return "\n\n".join(snippets)
157
205
 
158
206
 
207
+ def _collect_query_matches(
208
+ repo_root: pathlib.Path,
209
+ doc_files: list[pathlib.Path],
210
+ queries: list[str],
211
+ *,
212
+ rg_available: bool,
213
+ ) -> tuple[list[_QueryMatchBuffer], dict[pathlib.Path, str]]:
214
+ """Read each document once and retain only bounded ranked snippet metadata.
215
+
216
+ The old rg path skipped files larger than ``DOC_MAX_BYTES`` while its
217
+ no-rg fallback read a bounded prefix. Keep that environment-dependent
218
+ compatibility behavior explicit while avoiding N query subprocesses and
219
+ without caching the whole documentation tree.
220
+ """
221
+
222
+ buffers = [_QueryMatchBuffer() for _query in queries]
223
+ fallback_contents: dict[pathlib.Path, str] = {}
224
+ for path_index, path in enumerate(doc_files):
225
+ if rg_available:
226
+ try:
227
+ if path.stat().st_size > DOC_MAX_BYTES:
228
+ continue
229
+ except OSError:
230
+ continue
231
+ content = _read_bounded_text(path)
232
+ if not rg_available and path_index < DOC_FALLBACK_FILE_LIMIT:
233
+ fallback_contents[path] = content
234
+ lines = content.splitlines()
235
+ relative = path.relative_to(repo_root).as_posix()
236
+ for line_index, line in enumerate(lines):
237
+ for query_index, query in enumerate(queries):
238
+ if query not in line:
239
+ continue
240
+ first = max(0, line_index - DOC_CONTEXT_LINES)
241
+ last = min(len(lines), line_index + DOC_CONTEXT_LINES + 1)
242
+ buffer = buffers[query_index]
243
+ for index in range(first, last):
244
+ buffer.add(
245
+ (path_index, index),
246
+ relative,
247
+ index + 1,
248
+ lines[index],
249
+ )
250
+ if all(buffer.saturated for buffer in buffers):
251
+ return buffers, fallback_contents
252
+ return buffers, fallback_contents
253
+
254
+
159
255
  def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path], queries: list[str]) -> str:
160
256
  repo_root = repo_root.resolve(strict=True)
161
257
  if not doc_files:
@@ -163,83 +259,27 @@ def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path],
163
259
  if not queries:
164
260
  return _fallback_docs_context(repo_root, doc_files)
165
261
 
166
- if shutil.which("rg") is None:
167
- chunks: list[str] = []
168
- seen: set[tuple[str, int]] = set()
169
- for query in queries:
170
- for path in doc_files:
171
- relative = path.relative_to(repo_root).as_posix()
172
- lines = _read_bounded_text(path).splitlines()
173
- matching_lines = [index for index, line in enumerate(lines) if query in line]
174
- for matching_index in matching_lines:
175
- first = max(0, matching_index - DOC_CONTEXT_LINES)
176
- last = min(len(lines), matching_index + DOC_CONTEXT_LINES + 1)
177
- for index in range(first, last):
178
- key = (relative, index + 1)
179
- if key in seen:
180
- continue
181
- seen.add(key)
182
- chunk = f"{relative}:{index + 1}: {lines[index]}"
183
- if not _append_context_chunk(chunks, chunk, DOC_CONTEXT_BUDGET):
184
- return "\n".join(chunks)
185
- return "\n".join(chunks) if chunks else _fallback_docs_context(repo_root, doc_files)
186
-
187
- files = [path.relative_to(repo_root).as_posix() for path in doc_files]
188
- allowed_files = set(files)
189
- chunks: list[str] = []
190
- seen: set[tuple[str, int]] = set()
191
- for query in queries:
192
- completed = run_command(
193
- [
194
- "rg",
195
- "--json",
196
- "--fixed-strings",
197
- "--with-filename",
198
- "--color=never",
199
- "--context",
200
- str(DOC_CONTEXT_LINES),
201
- "--max-filesize",
202
- str(DOC_MAX_BYTES),
203
- "--",
204
- query,
205
- *files,
206
- ],
207
- cwd=repo_root,
208
- check=False,
209
- )
210
- if completed.returncode not in {0, 1}:
211
- continue
212
- for line in completed.stdout.splitlines():
213
- try:
214
- message = json.loads(line)
215
- except json.JSONDecodeError:
216
- continue
217
- if message.get("type") not in {"match", "context"}:
218
- continue
219
- data = message.get("data")
220
- if not isinstance(data, dict):
221
- continue
222
- path_data = data.get("path")
223
- lines_data = data.get("lines")
224
- file_name = path_data.get("text") if isinstance(path_data, dict) else None
225
- line_number = data.get("line_number")
226
- content = lines_data.get("text") if isinstance(lines_data, dict) else None
227
- if (
228
- not isinstance(file_name, str)
229
- or file_name not in allowed_files
230
- or not isinstance(line_number, int)
231
- or not isinstance(content, str)
232
- ):
233
- continue
234
- key = (file_name, line_number)
262
+ rg_available = shutil.which("rg") is not None
263
+ matches_by_query, fallback_contents = _collect_query_matches(
264
+ repo_root,
265
+ doc_files,
266
+ queries,
267
+ rg_available=rg_available,
268
+ )
269
+ accumulator = _ContextAccumulator()
270
+ seen: set[tuple[int, int]] = set()
271
+ for query_matches in matches_by_query:
272
+ for key, chunk in query_matches.matches:
235
273
  if key in seen:
236
274
  continue
237
275
  seen.add(key)
238
- clean_content = content.rstrip("\r\n")
239
- chunk = f"{file_name}:{line_number}: {clean_content}"
240
- if not _append_context_chunk(chunks, chunk, DOC_CONTEXT_BUDGET):
241
- return "\n".join(chunks)
242
- return "\n".join(chunks)
276
+ if not accumulator.append(chunk, DOC_CONTEXT_BUDGET):
277
+ return accumulator.render()
278
+ if accumulator.chunks:
279
+ return accumulator.render()
280
+ if rg_available:
281
+ return ""
282
+ return _fallback_docs_context(repo_root, doc_files, fallback_contents)
243
283
 
244
284
 
245
285
  def collect_docs_context(context: RuntimeContext, _state: Any) -> CollectorResult:
@@ -2,7 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  from typing import Any
4
4
 
5
- from ..executors.exec import (
5
+ from ..git_utils import (
6
6
  collect_commit_messages_for_ranges,
7
7
  env_bool,
8
8
  initial_pr_defer_reason,