ai-push-hooks 0.2.1 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +73 -1
- package/README.md +80 -525
- package/SECURITY.md +102 -14
- package/ai-push-hooks.toml +9 -2
- package/bin/ai-push-hooks.js +6 -6
- package/package.json +3 -2
- package/pyproject.toml +1 -1
- package/src/ai_push_hooks/artifacts.py +67 -13
- package/src/ai_push_hooks/config.py +575 -22
- package/src/ai_push_hooks/engine.py +116 -7
- package/src/ai_push_hooks/executors/apply.py +75 -36
- package/src/ai_push_hooks/executors/ask.py +224 -0
- package/src/ai_push_hooks/executors/exec.py +17 -801
- package/src/ai_push_hooks/executors/runner_workflow.py +478 -0
- package/src/ai_push_hooks/executors/runners/__init__.py +78 -0
- package/src/ai_push_hooks/executors/runners/claude.py +286 -0
- package/src/ai_push_hooks/executors/runners/codex.py +254 -0
- package/src/ai_push_hooks/executors/runners/command.py +178 -0
- package/src/ai_push_hooks/executors/runners/contracts.py +597 -0
- package/src/ai_push_hooks/executors/runners/opencode.py +528 -0
- package/src/ai_push_hooks/executors/runners/opencode_support.py +276 -0
- package/src/ai_push_hooks/executors/runners/process.py +464 -0
- package/src/ai_push_hooks/executors/runners/registry.py +117 -0
- package/src/ai_push_hooks/executors/step_commands.py +478 -0
- package/src/ai_push_hooks/git_utils.py +834 -0
- package/src/ai_push_hooks/hook.py +1 -1
- package/src/ai_push_hooks/modules/beads.py +1 -1
- package/src/ai_push_hooks/modules/docs.py +129 -89
- package/src/ai_push_hooks/modules/pr.py +1 -1
- package/src/ai_push_hooks/plugin_loader.py +422 -0
- package/src/ai_push_hooks/plugins.py +134 -0
- package/src/ai_push_hooks/prompts_builtin.py +9 -2
- package/src/ai_push_hooks/types.py +407 -75
- package/vendor/README.md +15 -0
- package/vendor/requirements.txt +1 -0
- package/vendor/tomli-2.4.0-py3-none-any.whl +0 -0
- package/src/ai_push_hooks/executors/llm.py +0 -624
|
@@ -10,7 +10,7 @@ from .artifacts import ArtifactStore, generate_run_id
|
|
|
10
10
|
from .config import load_config
|
|
11
11
|
from .engine import WorkflowEngine
|
|
12
12
|
from .paths import ensure_private_directory, resolve_contained_path, write_text_no_follow
|
|
13
|
-
from .
|
|
13
|
+
from .git_utils import (
|
|
14
14
|
collect_changed_files,
|
|
15
15
|
collect_diff,
|
|
16
16
|
collect_revision_ranges,
|
|
@@ -2,7 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
from typing import Any
|
|
4
4
|
|
|
5
|
-
from ..
|
|
5
|
+
from ..git_utils import collect_commit_messages_for_ranges, is_feature_branch
|
|
6
6
|
from ..types import CollectorResult, RuntimeContext
|
|
7
7
|
|
|
8
8
|
|
|
@@ -1,6 +1,5 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
-
import json
|
|
4
3
|
import os
|
|
5
4
|
import pathlib
|
|
6
5
|
import re
|
|
@@ -10,7 +9,7 @@ from pathlib import PurePosixPath
|
|
|
10
9
|
from typing import Any
|
|
11
10
|
|
|
12
11
|
from ..types import CollectorResult, RuntimeContext
|
|
13
|
-
from ..
|
|
12
|
+
from ..git_utils import collect_commit_messages_for_ranges, git, path_matches
|
|
14
13
|
|
|
15
14
|
DOC_INCLUDE_PATTERNS = ("README.md", "docs/**/*.md")
|
|
16
15
|
DOC_IGNORE_PATTERNS = ("docs/archive/**",)
|
|
@@ -18,6 +17,10 @@ DOC_CONTEXT_LINES = 2
|
|
|
18
17
|
DOC_MAX_BYTES = 64 * 1024
|
|
19
18
|
DOC_CONTEXT_BUDGET = 32000
|
|
20
19
|
DOC_FALLBACK_FILE_LIMIT = 8
|
|
20
|
+
# Matching metadata is deliberately bounded independently of the document
|
|
21
|
+
# inventory. The normal context budget is reached much earlier, while this
|
|
22
|
+
# cap prevents a pathological repeated-match file from retaining every hit.
|
|
23
|
+
DOC_MAX_QUERY_MATCHES = 4096
|
|
21
24
|
|
|
22
25
|
|
|
23
26
|
def _path_matches(path: str, patterns: tuple[str, ...]) -> bool:
|
|
@@ -126,25 +129,70 @@ def _read_bounded_text(path: pathlib.Path, max_bytes: int | None = None) -> str:
|
|
|
126
129
|
os.close(descriptor)
|
|
127
130
|
|
|
128
131
|
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
132
|
+
class _ContextAccumulator:
|
|
133
|
+
"""Collect bounded snippets without rescanning all prior snippets."""
|
|
134
|
+
|
|
135
|
+
def __init__(self) -> None:
|
|
136
|
+
self.chunks: list[str] = []
|
|
137
|
+
self.size = 0
|
|
138
|
+
|
|
139
|
+
def append(self, chunk: str, budget: int) -> bool:
|
|
140
|
+
remaining = budget - self.size
|
|
141
|
+
separator_size = 1 if self.chunks else 0
|
|
142
|
+
remaining -= separator_size
|
|
143
|
+
if remaining <= 0:
|
|
137
144
|
return False
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
145
|
+
truncated = len(chunk) > remaining
|
|
146
|
+
if truncated:
|
|
147
|
+
if self.chunks:
|
|
148
|
+
return False
|
|
149
|
+
chunk = chunk[:remaining]
|
|
150
|
+
self.chunks.append(chunk)
|
|
151
|
+
self.size += separator_size + len(chunk)
|
|
152
|
+
return not truncated
|
|
141
153
|
|
|
154
|
+
def render(self) -> str:
|
|
155
|
+
return "\n".join(self.chunks)
|
|
142
156
|
|
|
143
|
-
|
|
157
|
+
|
|
158
|
+
class _QueryMatchBuffer:
|
|
159
|
+
"""Bounded query-ranked snippet metadata for one document scan."""
|
|
160
|
+
|
|
161
|
+
def __init__(self, max_matches: int = DOC_MAX_QUERY_MATCHES) -> None:
|
|
162
|
+
self.max_matches = max_matches
|
|
163
|
+
self.matches: list[tuple[tuple[int, int], str]] = []
|
|
164
|
+
self._seen: set[tuple[int, int]] = set()
|
|
165
|
+
self.characters = 0
|
|
166
|
+
self.saturated = False
|
|
167
|
+
|
|
168
|
+
def add(
|
|
169
|
+
self,
|
|
170
|
+
key: tuple[int, int],
|
|
171
|
+
relative: str,
|
|
172
|
+
line_number: int,
|
|
173
|
+
line: str,
|
|
174
|
+
) -> None:
|
|
175
|
+
if key in self._seen or self.saturated or len(self.matches) >= self.max_matches:
|
|
176
|
+
return
|
|
177
|
+
chunk = f"{relative}:{line_number}: {line}"
|
|
178
|
+
self._seen.add(key)
|
|
179
|
+
self.matches.append((key, chunk))
|
|
180
|
+
self.characters += len(chunk)
|
|
181
|
+
self.saturated = (
|
|
182
|
+
len(self.matches) >= self.max_matches
|
|
183
|
+
or self.characters >= DOC_CONTEXT_BUDGET
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _fallback_docs_context(
|
|
188
|
+
repo_root: pathlib.Path,
|
|
189
|
+
doc_files: list[pathlib.Path],
|
|
190
|
+
contents: dict[pathlib.Path, str] | None = None,
|
|
191
|
+
) -> str:
|
|
144
192
|
snippets: list[str] = []
|
|
145
193
|
for path in doc_files[:DOC_FALLBACK_FILE_LIMIT]:
|
|
146
194
|
relative = path.relative_to(repo_root).as_posix()
|
|
147
|
-
content = _read_bounded_text(path)
|
|
195
|
+
content = _read_bounded_text(path) if contents is None else contents[path]
|
|
148
196
|
block = f"--- {relative} ---\n{content}"
|
|
149
197
|
current_size = len("\n\n".join(snippets))
|
|
150
198
|
remaining = DOC_CONTEXT_BUDGET - current_size
|
|
@@ -156,6 +204,54 @@ def _fallback_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path
|
|
|
156
204
|
return "\n\n".join(snippets)
|
|
157
205
|
|
|
158
206
|
|
|
207
|
+
def _collect_query_matches(
|
|
208
|
+
repo_root: pathlib.Path,
|
|
209
|
+
doc_files: list[pathlib.Path],
|
|
210
|
+
queries: list[str],
|
|
211
|
+
*,
|
|
212
|
+
rg_available: bool,
|
|
213
|
+
) -> tuple[list[_QueryMatchBuffer], dict[pathlib.Path, str]]:
|
|
214
|
+
"""Read each document once and retain only bounded ranked snippet metadata.
|
|
215
|
+
|
|
216
|
+
The old rg path skipped files larger than ``DOC_MAX_BYTES`` while its
|
|
217
|
+
no-rg fallback read a bounded prefix. Keep that environment-dependent
|
|
218
|
+
compatibility behavior explicit while avoiding N query subprocesses and
|
|
219
|
+
without caching the whole documentation tree.
|
|
220
|
+
"""
|
|
221
|
+
|
|
222
|
+
buffers = [_QueryMatchBuffer() for _query in queries]
|
|
223
|
+
fallback_contents: dict[pathlib.Path, str] = {}
|
|
224
|
+
for path_index, path in enumerate(doc_files):
|
|
225
|
+
if rg_available:
|
|
226
|
+
try:
|
|
227
|
+
if path.stat().st_size > DOC_MAX_BYTES:
|
|
228
|
+
continue
|
|
229
|
+
except OSError:
|
|
230
|
+
continue
|
|
231
|
+
content = _read_bounded_text(path)
|
|
232
|
+
if not rg_available and path_index < DOC_FALLBACK_FILE_LIMIT:
|
|
233
|
+
fallback_contents[path] = content
|
|
234
|
+
lines = content.splitlines()
|
|
235
|
+
relative = path.relative_to(repo_root).as_posix()
|
|
236
|
+
for line_index, line in enumerate(lines):
|
|
237
|
+
for query_index, query in enumerate(queries):
|
|
238
|
+
if query not in line:
|
|
239
|
+
continue
|
|
240
|
+
first = max(0, line_index - DOC_CONTEXT_LINES)
|
|
241
|
+
last = min(len(lines), line_index + DOC_CONTEXT_LINES + 1)
|
|
242
|
+
buffer = buffers[query_index]
|
|
243
|
+
for index in range(first, last):
|
|
244
|
+
buffer.add(
|
|
245
|
+
(path_index, index),
|
|
246
|
+
relative,
|
|
247
|
+
index + 1,
|
|
248
|
+
lines[index],
|
|
249
|
+
)
|
|
250
|
+
if all(buffer.saturated for buffer in buffers):
|
|
251
|
+
return buffers, fallback_contents
|
|
252
|
+
return buffers, fallback_contents
|
|
253
|
+
|
|
254
|
+
|
|
159
255
|
def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path], queries: list[str]) -> str:
|
|
160
256
|
repo_root = repo_root.resolve(strict=True)
|
|
161
257
|
if not doc_files:
|
|
@@ -163,83 +259,27 @@ def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path],
|
|
|
163
259
|
if not queries:
|
|
164
260
|
return _fallback_docs_context(repo_root, doc_files)
|
|
165
261
|
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
for index in range(first, last):
|
|
178
|
-
key = (relative, index + 1)
|
|
179
|
-
if key in seen:
|
|
180
|
-
continue
|
|
181
|
-
seen.add(key)
|
|
182
|
-
chunk = f"{relative}:{index + 1}: {lines[index]}"
|
|
183
|
-
if not _append_context_chunk(chunks, chunk, DOC_CONTEXT_BUDGET):
|
|
184
|
-
return "\n".join(chunks)
|
|
185
|
-
return "\n".join(chunks) if chunks else _fallback_docs_context(repo_root, doc_files)
|
|
186
|
-
|
|
187
|
-
files = [path.relative_to(repo_root).as_posix() for path in doc_files]
|
|
188
|
-
allowed_files = set(files)
|
|
189
|
-
chunks: list[str] = []
|
|
190
|
-
seen: set[tuple[str, int]] = set()
|
|
191
|
-
for query in queries:
|
|
192
|
-
completed = run_command(
|
|
193
|
-
[
|
|
194
|
-
"rg",
|
|
195
|
-
"--json",
|
|
196
|
-
"--fixed-strings",
|
|
197
|
-
"--with-filename",
|
|
198
|
-
"--color=never",
|
|
199
|
-
"--context",
|
|
200
|
-
str(DOC_CONTEXT_LINES),
|
|
201
|
-
"--max-filesize",
|
|
202
|
-
str(DOC_MAX_BYTES),
|
|
203
|
-
"--",
|
|
204
|
-
query,
|
|
205
|
-
*files,
|
|
206
|
-
],
|
|
207
|
-
cwd=repo_root,
|
|
208
|
-
check=False,
|
|
209
|
-
)
|
|
210
|
-
if completed.returncode not in {0, 1}:
|
|
211
|
-
continue
|
|
212
|
-
for line in completed.stdout.splitlines():
|
|
213
|
-
try:
|
|
214
|
-
message = json.loads(line)
|
|
215
|
-
except json.JSONDecodeError:
|
|
216
|
-
continue
|
|
217
|
-
if message.get("type") not in {"match", "context"}:
|
|
218
|
-
continue
|
|
219
|
-
data = message.get("data")
|
|
220
|
-
if not isinstance(data, dict):
|
|
221
|
-
continue
|
|
222
|
-
path_data = data.get("path")
|
|
223
|
-
lines_data = data.get("lines")
|
|
224
|
-
file_name = path_data.get("text") if isinstance(path_data, dict) else None
|
|
225
|
-
line_number = data.get("line_number")
|
|
226
|
-
content = lines_data.get("text") if isinstance(lines_data, dict) else None
|
|
227
|
-
if (
|
|
228
|
-
not isinstance(file_name, str)
|
|
229
|
-
or file_name not in allowed_files
|
|
230
|
-
or not isinstance(line_number, int)
|
|
231
|
-
or not isinstance(content, str)
|
|
232
|
-
):
|
|
233
|
-
continue
|
|
234
|
-
key = (file_name, line_number)
|
|
262
|
+
rg_available = shutil.which("rg") is not None
|
|
263
|
+
matches_by_query, fallback_contents = _collect_query_matches(
|
|
264
|
+
repo_root,
|
|
265
|
+
doc_files,
|
|
266
|
+
queries,
|
|
267
|
+
rg_available=rg_available,
|
|
268
|
+
)
|
|
269
|
+
accumulator = _ContextAccumulator()
|
|
270
|
+
seen: set[tuple[int, int]] = set()
|
|
271
|
+
for query_matches in matches_by_query:
|
|
272
|
+
for key, chunk in query_matches.matches:
|
|
235
273
|
if key in seen:
|
|
236
274
|
continue
|
|
237
275
|
seen.add(key)
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
276
|
+
if not accumulator.append(chunk, DOC_CONTEXT_BUDGET):
|
|
277
|
+
return accumulator.render()
|
|
278
|
+
if accumulator.chunks:
|
|
279
|
+
return accumulator.render()
|
|
280
|
+
if rg_available:
|
|
281
|
+
return ""
|
|
282
|
+
return _fallback_docs_context(repo_root, doc_files, fallback_contents)
|
|
243
283
|
|
|
244
284
|
|
|
245
285
|
def collect_docs_context(context: RuntimeContext, _state: Any) -> CollectorResult:
|