ai-push-hooks 0.3.0 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,7 +10,7 @@ from .artifacts import ArtifactStore, generate_run_id
10
10
  from .config import load_config
11
11
  from .engine import WorkflowEngine
12
12
  from .paths import ensure_private_directory, resolve_contained_path, write_text_no_follow
13
- from .executors.exec import (
13
+ from .git_utils import (
14
14
  collect_changed_files,
15
15
  collect_diff,
16
16
  collect_revision_ranges,
@@ -2,7 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  from typing import Any
4
4
 
5
- from ..executors.exec import collect_commit_messages_for_ranges, is_feature_branch
5
+ from ..git_utils import collect_commit_messages_for_ranges, is_feature_branch
6
6
  from ..types import CollectorResult, RuntimeContext
7
7
 
8
8
 
@@ -1,6 +1,5 @@
1
1
  from __future__ import annotations
2
2
 
3
- import json
4
3
  import os
5
4
  import pathlib
6
5
  import re
@@ -10,7 +9,7 @@ from pathlib import PurePosixPath
10
9
  from typing import Any
11
10
 
12
11
  from ..types import CollectorResult, RuntimeContext
13
- from ..executors.exec import collect_commit_messages_for_ranges, git, path_matches, run_command
12
+ from ..git_utils import collect_commit_messages_for_ranges, git, path_matches
14
13
 
15
14
  DOC_INCLUDE_PATTERNS = ("README.md", "docs/**/*.md")
16
15
  DOC_IGNORE_PATTERNS = ("docs/archive/**",)
@@ -18,6 +17,10 @@ DOC_CONTEXT_LINES = 2
18
17
  DOC_MAX_BYTES = 64 * 1024
19
18
  DOC_CONTEXT_BUDGET = 32000
20
19
  DOC_FALLBACK_FILE_LIMIT = 8
20
+ # Matching metadata is deliberately bounded independently of the document
21
+ # inventory. The normal context budget is reached much earlier, while this
22
+ # cap prevents a pathological repeated-match file from retaining every hit.
23
+ DOC_MAX_QUERY_MATCHES = 4096
21
24
 
22
25
 
23
26
  def _path_matches(path: str, patterns: tuple[str, ...]) -> bool:
@@ -126,25 +129,70 @@ def _read_bounded_text(path: pathlib.Path, max_bytes: int | None = None) -> str:
126
129
  os.close(descriptor)
127
130
 
128
131
 
129
- def _append_context_chunk(chunks: list[str], chunk: str, budget: int) -> bool:
130
- current_size = sum(len(item) for item in chunks) + max(0, len(chunks) - 1)
131
- remaining = budget - current_size
132
- if remaining <= 0:
133
- return False
134
- truncated = len(chunk) > remaining
135
- if truncated:
136
- if chunks:
132
+ class _ContextAccumulator:
133
+ """Collect bounded snippets without rescanning all prior snippets."""
134
+
135
+ def __init__(self) -> None:
136
+ self.chunks: list[str] = []
137
+ self.size = 0
138
+
139
+ def append(self, chunk: str, budget: int) -> bool:
140
+ remaining = budget - self.size
141
+ separator_size = 1 if self.chunks else 0
142
+ remaining -= separator_size
143
+ if remaining <= 0:
137
144
  return False
138
- chunk = chunk[:remaining]
139
- chunks.append(chunk)
140
- return not truncated
145
+ truncated = len(chunk) > remaining
146
+ if truncated:
147
+ if self.chunks:
148
+ return False
149
+ chunk = chunk[:remaining]
150
+ self.chunks.append(chunk)
151
+ self.size += separator_size + len(chunk)
152
+ return not truncated
141
153
 
154
+ def render(self) -> str:
155
+ return "\n".join(self.chunks)
142
156
 
143
- def _fallback_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path]) -> str:
157
+
158
+ class _QueryMatchBuffer:
159
+ """Bounded query-ranked snippet metadata for one document scan."""
160
+
161
+ def __init__(self, max_matches: int = DOC_MAX_QUERY_MATCHES) -> None:
162
+ self.max_matches = max_matches
163
+ self.matches: list[tuple[tuple[int, int], str]] = []
164
+ self._seen: set[tuple[int, int]] = set()
165
+ self.characters = 0
166
+ self.saturated = False
167
+
168
+ def add(
169
+ self,
170
+ key: tuple[int, int],
171
+ relative: str,
172
+ line_number: int,
173
+ line: str,
174
+ ) -> None:
175
+ if key in self._seen or self.saturated or len(self.matches) >= self.max_matches:
176
+ return
177
+ chunk = f"{relative}:{line_number}: {line}"
178
+ self._seen.add(key)
179
+ self.matches.append((key, chunk))
180
+ self.characters += len(chunk)
181
+ self.saturated = (
182
+ len(self.matches) >= self.max_matches
183
+ or self.characters >= DOC_CONTEXT_BUDGET
184
+ )
185
+
186
+
187
+ def _fallback_docs_context(
188
+ repo_root: pathlib.Path,
189
+ doc_files: list[pathlib.Path],
190
+ contents: dict[pathlib.Path, str] | None = None,
191
+ ) -> str:
144
192
  snippets: list[str] = []
145
193
  for path in doc_files[:DOC_FALLBACK_FILE_LIMIT]:
146
194
  relative = path.relative_to(repo_root).as_posix()
147
- content = _read_bounded_text(path)
195
+ content = _read_bounded_text(path) if contents is None else contents[path]
148
196
  block = f"--- {relative} ---\n{content}"
149
197
  current_size = len("\n\n".join(snippets))
150
198
  remaining = DOC_CONTEXT_BUDGET - current_size
@@ -156,6 +204,54 @@ def _fallback_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path
156
204
  return "\n\n".join(snippets)
157
205
 
158
206
 
207
+ def _collect_query_matches(
208
+ repo_root: pathlib.Path,
209
+ doc_files: list[pathlib.Path],
210
+ queries: list[str],
211
+ *,
212
+ rg_available: bool,
213
+ ) -> tuple[list[_QueryMatchBuffer], dict[pathlib.Path, str]]:
214
+ """Read each document once and retain only bounded ranked snippet metadata.
215
+
216
+ The old rg path skipped files larger than ``DOC_MAX_BYTES`` while its
217
+ no-rg fallback read a bounded prefix. Keep that environment-dependent
218
+ compatibility behavior explicit while avoiding N query subprocesses and
219
+ without caching the whole documentation tree.
220
+ """
221
+
222
+ buffers = [_QueryMatchBuffer() for _query in queries]
223
+ fallback_contents: dict[pathlib.Path, str] = {}
224
+ for path_index, path in enumerate(doc_files):
225
+ if rg_available:
226
+ try:
227
+ if path.stat().st_size > DOC_MAX_BYTES:
228
+ continue
229
+ except OSError:
230
+ continue
231
+ content = _read_bounded_text(path)
232
+ if not rg_available and path_index < DOC_FALLBACK_FILE_LIMIT:
233
+ fallback_contents[path] = content
234
+ lines = content.splitlines()
235
+ relative = path.relative_to(repo_root).as_posix()
236
+ for line_index, line in enumerate(lines):
237
+ for query_index, query in enumerate(queries):
238
+ if query not in line:
239
+ continue
240
+ first = max(0, line_index - DOC_CONTEXT_LINES)
241
+ last = min(len(lines), line_index + DOC_CONTEXT_LINES + 1)
242
+ buffer = buffers[query_index]
243
+ for index in range(first, last):
244
+ buffer.add(
245
+ (path_index, index),
246
+ relative,
247
+ index + 1,
248
+ lines[index],
249
+ )
250
+ if all(buffer.saturated for buffer in buffers):
251
+ return buffers, fallback_contents
252
+ return buffers, fallback_contents
253
+
254
+
159
255
  def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path], queries: list[str]) -> str:
160
256
  repo_root = repo_root.resolve(strict=True)
161
257
  if not doc_files:
@@ -163,83 +259,27 @@ def _search_docs_context(repo_root: pathlib.Path, doc_files: list[pathlib.Path],
163
259
  if not queries:
164
260
  return _fallback_docs_context(repo_root, doc_files)
165
261
 
166
- if shutil.which("rg") is None:
167
- chunks: list[str] = []
168
- seen: set[tuple[str, int]] = set()
169
- for query in queries:
170
- for path in doc_files:
171
- relative = path.relative_to(repo_root).as_posix()
172
- lines = _read_bounded_text(path).splitlines()
173
- matching_lines = [index for index, line in enumerate(lines) if query in line]
174
- for matching_index in matching_lines:
175
- first = max(0, matching_index - DOC_CONTEXT_LINES)
176
- last = min(len(lines), matching_index + DOC_CONTEXT_LINES + 1)
177
- for index in range(first, last):
178
- key = (relative, index + 1)
179
- if key in seen:
180
- continue
181
- seen.add(key)
182
- chunk = f"{relative}:{index + 1}: {lines[index]}"
183
- if not _append_context_chunk(chunks, chunk, DOC_CONTEXT_BUDGET):
184
- return "\n".join(chunks)
185
- return "\n".join(chunks) if chunks else _fallback_docs_context(repo_root, doc_files)
186
-
187
- files = [path.relative_to(repo_root).as_posix() for path in doc_files]
188
- allowed_files = set(files)
189
- chunks: list[str] = []
190
- seen: set[tuple[str, int]] = set()
191
- for query in queries:
192
- completed = run_command(
193
- [
194
- "rg",
195
- "--json",
196
- "--fixed-strings",
197
- "--with-filename",
198
- "--color=never",
199
- "--context",
200
- str(DOC_CONTEXT_LINES),
201
- "--max-filesize",
202
- str(DOC_MAX_BYTES),
203
- "--",
204
- query,
205
- *files,
206
- ],
207
- cwd=repo_root,
208
- check=False,
209
- )
210
- if completed.returncode not in {0, 1}:
211
- continue
212
- for line in completed.stdout.splitlines():
213
- try:
214
- message = json.loads(line)
215
- except json.JSONDecodeError:
216
- continue
217
- if message.get("type") not in {"match", "context"}:
218
- continue
219
- data = message.get("data")
220
- if not isinstance(data, dict):
221
- continue
222
- path_data = data.get("path")
223
- lines_data = data.get("lines")
224
- file_name = path_data.get("text") if isinstance(path_data, dict) else None
225
- line_number = data.get("line_number")
226
- content = lines_data.get("text") if isinstance(lines_data, dict) else None
227
- if (
228
- not isinstance(file_name, str)
229
- or file_name not in allowed_files
230
- or not isinstance(line_number, int)
231
- or not isinstance(content, str)
232
- ):
233
- continue
234
- key = (file_name, line_number)
262
+ rg_available = shutil.which("rg") is not None
263
+ matches_by_query, fallback_contents = _collect_query_matches(
264
+ repo_root,
265
+ doc_files,
266
+ queries,
267
+ rg_available=rg_available,
268
+ )
269
+ accumulator = _ContextAccumulator()
270
+ seen: set[tuple[int, int]] = set()
271
+ for query_matches in matches_by_query:
272
+ for key, chunk in query_matches.matches:
235
273
  if key in seen:
236
274
  continue
237
275
  seen.add(key)
238
- clean_content = content.rstrip("\r\n")
239
- chunk = f"{file_name}:{line_number}: {clean_content}"
240
- if not _append_context_chunk(chunks, chunk, DOC_CONTEXT_BUDGET):
241
- return "\n".join(chunks)
242
- return "\n".join(chunks)
276
+ if not accumulator.append(chunk, DOC_CONTEXT_BUDGET):
277
+ return accumulator.render()
278
+ if accumulator.chunks:
279
+ return accumulator.render()
280
+ if rg_available:
281
+ return ""
282
+ return _fallback_docs_context(repo_root, doc_files, fallback_contents)
243
283
 
244
284
 
245
285
  def collect_docs_context(context: RuntimeContext, _state: Any) -> CollectorResult:
@@ -2,7 +2,7 @@ from __future__ import annotations
2
2
 
3
3
  from typing import Any
4
4
 
5
- from ..executors.exec import (
5
+ from ..git_utils import (
6
6
  collect_commit_messages_for_ranges,
7
7
  env_bool,
8
8
  initial_pr_defer_reason,
@@ -10,7 +10,6 @@ completed.
10
10
 
11
11
  from __future__ import annotations # noqa: I001
12
12
 
13
- import copy
14
13
  import errno
15
14
  import hashlib
16
15
  import inspect
@@ -174,96 +173,121 @@ class PluginLoader:
174
173
  def __init__(self) -> None:
175
174
  self._cache: dict[tuple[pathlib.Path, pathlib.Path], ModuleType] = {}
176
175
  self._lock = threading.RLock()
176
+ self._source_locks: dict[tuple[pathlib.Path, pathlib.Path], threading.Lock] = {}
177
177
  self._module_number = 0
178
178
 
179
+ def _source_lock(
180
+ self, key: tuple[pathlib.Path, pathlib.Path]
181
+ ) -> threading.Lock:
182
+ with self._lock:
183
+ lock = self._source_locks.get(key)
184
+ if lock is None:
185
+ lock = threading.Lock()
186
+ self._source_locks[key] = lock
187
+ return lock
188
+
189
+ def _load_module(
190
+ self,
191
+ root: pathlib.Path,
192
+ relative_path: str,
193
+ callable_name: str,
194
+ ) -> ModuleType:
195
+ try:
196
+ callback_path, source = _open_source(root, relative_path)
197
+ except HookError as exc:
198
+ raise HookError(
199
+ f"Python plugin {relative_path}:{callable_name} could not be loaded safely: "
200
+ f"{str(exc).removeprefix('Python plugin ')}"
201
+ ) from None
202
+ # ``callback_path`` is canonical after containment checks and is equal
203
+ # to the lexical cache key for a safe repository file.
204
+ try:
205
+ source_text = source.decode(_SOURCE_ENCODING)
206
+ code = compile(source_text, str(callback_path), "exec")
207
+ except UnicodeDecodeError:
208
+ raise HookError(
209
+ f"Python plugin {relative_path}:{callable_name} source is not valid UTF-8"
210
+ ) from None
211
+ except (SyntaxError, ValueError, TypeError):
212
+ raise HookError(
213
+ f"Python plugin {relative_path}:{callable_name} could not be compiled"
214
+ ) from None
215
+
216
+ with self._lock:
217
+ self._module_number += 1
218
+ module_number = self._module_number
219
+ digest = hashlib.sha256(f"{root}\0{callback_path}".encode()).hexdigest()[:16]
220
+ module_name = f"ai_push_hooks_plugin_{digest}_{module_number}"
221
+ module = ModuleType(module_name)
222
+ module.__file__ = str(callback_path)
223
+ module.__package__ = ""
224
+ try:
225
+ exec(code, module.__dict__) # noqa: S102
226
+ except KeyboardInterrupt:
227
+ raise
228
+ except SystemExit:
229
+ raise HookError(
230
+ f"Python plugin {relative_path}:{callable_name} exited during import"
231
+ ) from None
232
+ except ModuleNotFoundError as exc:
233
+ if exc.name:
234
+ detail = f"is missing dependency {exc.name!r}"
235
+ else:
236
+ detail = "could not import a dependency"
237
+ raise HookError(
238
+ f"Python plugin {relative_path}:{callable_name} {detail}"
239
+ ) from None
240
+ except ImportError:
241
+ raise HookError(
242
+ f"Python plugin {relative_path}:{callable_name} could not import a dependency"
243
+ ) from None
244
+ except Exception: # noqa: BLE001
245
+ raise HookError(
246
+ f"Python plugin {relative_path}:{callable_name} failed during import"
247
+ ) from None
248
+ return module
249
+
179
250
  def load(self, repo_root: pathlib.Path, reference: str) -> Callable[..., Any]:
180
251
  """Return the named top-level callable from a safe source snapshot.
181
252
 
182
- The lock includes descriptor open, source read, compilation, and module
183
- execution. Thus concurrent collectors cannot execute the same source
184
- twice. The cache key includes both canonical repository root and
185
- canonical file path, so identical relative names in two repositories
186
- remain independent.
253
+ A per-source lock includes descriptor open, source read, compilation,
254
+ and module execution. Thus concurrent collectors cannot execute the
255
+ same source twice, while different sources can import concurrently.
256
+ The cache key includes both canonical repository root and canonical
257
+ file path, so identical relative names in two repositories remain
258
+ independent.
187
259
  """
188
260
 
189
261
  relative_path, callable_name = _reference_parts(reference)
190
262
  root = _canonical_root(repo_root)
263
+ lexical_callback_path = root.joinpath(*relative_path.split("/"))
264
+ key = (root, lexical_callback_path)
191
265
  with self._lock:
192
- lexical_callback_path = root.joinpath(*relative_path.split("/"))
193
- key = (root, lexical_callback_path)
194
266
  module = self._cache.get(key)
195
- if module is None:
196
- try:
197
- callback_path, source = _open_source(root, relative_path)
198
- except HookError as exc:
199
- raise HookError(
200
- f"Python plugin {relative_path}:{callable_name} could not be loaded safely: "
201
- f"{str(exc).removeprefix('Python plugin ')}"
202
- ) from None
203
- # ``callback_path`` is canonical after containment checks and
204
- # is equal to the lexical path for a safe repository file.
205
- key = (root, callback_path)
206
- try:
207
- source_text = source.decode(_SOURCE_ENCODING)
208
- code = compile(source_text, str(callback_path), "exec")
209
- except UnicodeDecodeError:
210
- raise HookError(
211
- f"Python plugin {relative_path}:{callable_name} source is not valid UTF-8"
212
- ) from None
213
- except (SyntaxError, ValueError, TypeError):
214
- raise HookError(
215
- f"Python plugin {relative_path}:{callable_name} could not be compiled"
216
- ) from None
217
-
218
- self._module_number += 1
219
- digest = hashlib.sha256(
220
- f"{root}\0{callback_path}".encode()
221
- ).hexdigest()[:16]
222
- module_name = f"ai_push_hooks_plugin_{digest}_{self._module_number}"
223
- module = ModuleType(module_name)
224
- module.__file__ = str(callback_path)
225
- module.__package__ = ""
226
- try:
227
- exec(code, module.__dict__) # noqa: S102
228
- except KeyboardInterrupt:
229
- raise
230
- except SystemExit:
231
- raise HookError(
232
- f"Python plugin {relative_path}:{callable_name} exited during import"
233
- ) from None
234
- except ModuleNotFoundError as exc:
235
- if exc.name:
236
- detail = f"is missing dependency {exc.name!r}"
237
- else:
238
- detail = "could not import a dependency"
239
- raise HookError(
240
- f"Python plugin {relative_path}:{callable_name} {detail}"
241
- ) from None
242
- except ImportError:
243
- raise HookError(
244
- f"Python plugin {relative_path}:{callable_name} could not import a dependency"
245
- ) from None
246
- except Exception: # noqa: BLE001
247
- raise HookError(
248
- f"Python plugin {relative_path}:{callable_name} failed during import"
249
- ) from None
250
- self._cache[key] = module
251
-
252
- try:
253
- callback = getattr(module, callable_name)
254
- except AttributeError:
255
- raise _failure(
256
- "callback", relative_path, callable_name, "was not defined"
257
- ) from None
258
- if not callable(callback):
259
- raise _failure(
260
- "callback", relative_path, callable_name, "is not callable"
261
- ) from None
262
- if inspect.iscoroutinefunction(callback):
263
- raise _failure(
264
- "callback", relative_path, callable_name, "must be synchronous"
265
- ) from None
266
- return callback
267
+ if module is None:
268
+ with self._source_lock(key):
269
+ with self._lock:
270
+ module = self._cache.get(key)
271
+ if module is None:
272
+ module = self._load_module(root, relative_path, callable_name)
273
+ with self._lock:
274
+ self._cache[key] = module
275
+
276
+ try:
277
+ callback = getattr(module, callable_name)
278
+ except AttributeError:
279
+ raise _failure(
280
+ "callback", relative_path, callable_name, "was not defined"
281
+ ) from None
282
+ if not callable(callback):
283
+ raise _failure(
284
+ "callback", relative_path, callable_name, "is not callable"
285
+ ) from None
286
+ if inspect.iscoroutinefunction(callback):
287
+ raise _failure(
288
+ "callback", relative_path, callable_name, "must be synchronous"
289
+ ) from None
290
+ return callback
267
291
 
268
292
  def invoke(
269
293
  self,
@@ -358,8 +382,8 @@ def build_plugin_context(
358
382
  module_id=state.module.id,
359
383
  step_id=step.id,
360
384
  inputs=_ordered_inputs(step, input_paths),
361
- options=copy.deepcopy(step.options),
362
- prior_module_metadata=copy.deepcopy(state.metadata),
385
+ options=step.options,
386
+ prior_module_metadata=state.metadata,
363
387
  push=push,
364
388
  logger=runtime.logger,
365
389
  )
@@ -14,6 +14,7 @@ from typing import Any, ClassVar
14
14
  READ_ONLY_STEP_TYPES = frozenset({"collect", "ask"})
15
15
  PROMPTABLE_STEP_TYPES = frozenset({"ask", "apply"})
16
16
  SUPPORTED_STEP_TYPES = frozenset({"collect", "ask", "apply", "exec", "assert"})
17
+ DEFAULT_STEP_COMMAND_TIMEOUT_SECONDS = 60
17
18
  FEATURE_BRANCH_PREFIXES = ("feat/", "feature/")
18
19
  ZERO_OID_LENGTHS = frozenset({40, 64})
19
20
 
@@ -0,0 +1,15 @@
1
+ # Bundled Python Dependency
2
+
3
+ The npm package includes the unmodified pure-Python wheel for [Tomli](https://pypi.org/project/tomli/2.4.0/). Python 3.10 uses it to parse TOML; Python 3.11+ uses the standard-library `tomllib` module.
4
+
5
+ The wrapper adds the wheel to `PYTHONPATH`, so no pip install, install script, or runtime download is needed. The wheel includes Tomli's MIT license under `tomli-2.4.0.dist-info/licenses/LICENSE`.
6
+
7
+ To reproduce the bundled download from the repository root:
8
+
9
+ ```bash
10
+ python -m pip download --index-url https://pypi.org/simple --require-hashes \
11
+ --only-binary=:all: --no-deps --platform any --implementation py --abi none \
12
+ --python-version 3.10 --dest vendor --requirement vendor/requirements.txt
13
+ ```
14
+
15
+ When updating, verify the pure-Python wheel's SHA-256 against PyPI, update the pin and wrapper path, and run the npm installed-hook tests on Python 3.10.
@@ -0,0 +1 @@
1
+ tomli==2.4.0 --hash=sha256:1f776e7d669ebceb01dee46484485f43a4048746235e683bcdffacdf1fb4785a