aipr-py 0.2.4__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
aipr/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """aipr: read an open-source repository's AI contribution policy."""
2
+
3
+ __version__ = "0.2.3"
aipr/cache.py ADDED
@@ -0,0 +1,84 @@
1
+ """Thread-safe TTL cache for policy detection results."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import copy
6
+ import os
7
+ import threading
8
+ import time
9
+ from dataclasses import dataclass
10
+ from typing import TYPE_CHECKING, Optional
11
+
12
+ if TYPE_CHECKING:
13
+ from aipr.detector import Policy
14
+
15
+
16
+ # Default TTL: 24 hours (in seconds)
17
+ DEFAULT_CACHE_TTL = int(os.environ.get("AIPR_CACHE_TTL", "86400"))
18
+ MAX_CACHE_SIZE = int(os.environ.get("AIPR_CACHE_SIZE", "1024"))
19
+
20
+
21
+ @dataclass
22
+ class CacheEntry:
23
+ ts: float
24
+ policy: object # Policy dataclass
25
+
26
+
27
+ class TTLCache:
28
+ """Thread-safe TTL cache with LRU eviction and deep copy on read.
29
+
30
+ Replaces lru_cache for Policy objects to ensure:
31
+ 1. Thread safety via threading.Lock
32
+ 2. TTL-based expiration (configurable via AIPR_CACHE_TTL env var)
33
+ 3. Deep copy on read to prevent mutation of cached state
34
+ """
35
+
36
+ def __init__(self, ttl: int = DEFAULT_CACHE_TTL, maxsize: int = MAX_CACHE_SIZE):
37
+ self._ttl = ttl
38
+ self._maxsize = maxsize
39
+ self._cache: dict[str, CacheEntry] = {}
40
+ self._lock = threading.Lock()
41
+
42
+ def get(self, key: str) -> Optional[object]:
43
+ """Get a cached Policy, returning a deep copy.
44
+
45
+ Returns None if key not found or entry expired.
46
+ """
47
+ with self._lock:
48
+ entry = self._cache.get(key)
49
+ if entry is None:
50
+ return None
51
+ if time.time() - entry.ts > self._ttl:
52
+ del self._cache[key]
53
+ return None
54
+ return copy.deepcopy(entry.policy)
55
+
56
+ def put(self, key: str, policy: object) -> None:
57
+ """Store a Policy with current timestamp.
58
+
59
+ Evicts oldest entries if cache exceeds maxsize.
60
+ """
61
+ with self._lock:
62
+ while len(self._cache) >= self._maxsize:
63
+ oldest_key = min(self._cache, key=lambda k: self._cache[k].ts)
64
+ del self._cache[oldest_key]
65
+ self._cache[key] = CacheEntry(ts=time.time(), policy=copy.deepcopy(policy))
66
+
67
+ def clear(self) -> None:
68
+ """Clear all cached entries."""
69
+ with self._lock:
70
+ self._cache.clear()
71
+
72
+ def __len__(self) -> int:
73
+ with self._lock:
74
+ return len(self._cache)
75
+
76
+ def __contains__(self, key: str) -> bool:
77
+ with self._lock:
78
+ entry = self._cache.get(key)
79
+ if entry is None:
80
+ return False
81
+ if time.time() - entry.ts > self._ttl:
82
+ del self._cache[key]
83
+ return False
84
+ return True
aipr/cli.py ADDED
@@ -0,0 +1,448 @@
1
+ """aipr CLI: read an open-source repository's AI contribution policy.
2
+
3
+ Usage:
4
+ aipr <owner/repo> # fetch governance files from GitHub and classify
5
+ aipr --text <file> # classify a local file
6
+ aipr --json <owner/repo> # machine-readable output
7
+ aipr init # scaffold AI_POLICY.md and AI_TOOL_POLICY.md
8
+
9
+ Exit codes: 0 = autonomous-safe, 1 = not safe / restricted, 2 = unknown, 64 = usage error.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import argparse
15
+ import json
16
+ import logging
17
+ import os
18
+ import sys
19
+ import time
20
+ from pathlib import Path
21
+
22
+ from . import __version__
23
+ from .detector import Verdict, detect_policy
24
+ from .http import fetch_with_retry, RateLimitError
25
+
26
+ logger = logging.getLogger(__name__)
27
+
28
+ # Files that commonly carry AI policy, in priority order. The org-level
29
+ # .github repo is also probed because many foundations centralize there.
30
+ CANDIDATE_FILES = [
31
+ "AI_POLICY.md",
32
+ "AI_POLICY.rst",
33
+ "AI_TOOL_POLICY.md",
34
+ "CONTRIBUTING.md",
35
+ "CONTRIBUTING.rst",
36
+ "CONTRIBUTING-BEGINNERS.md",
37
+ ".github/AI_POLICY.md",
38
+ ".github/CONTRIBUTING.md",
39
+ ".github/copilot-instructions.md",
40
+ "docs/CONTRIBUTING.md",
41
+ "docs/CONTRIBUTING-BEGINNERS.md",
42
+ "AGENTS.md",
43
+ "CLAUDE.md",
44
+ ".cursorrules",
45
+ ".windsurfrules",
46
+ ".aider.conf.yml",
47
+ "README.md",
48
+ ]
49
+
50
+ ORG_FALLBACK_FILES = [".github/AI_POLICY.md", ".github/CONTRIBUTING.md"]
51
+
52
+ EXIT_OK = 0
53
+ EXIT_UNSAFE = 1
54
+ EXIT_UNKNOWN = 2
55
+ EXIT_USAGE = 64
56
+
57
+ # --- on-disk cache ----------------------------------------------------------
58
+ # Each governance file fetch is cached under AIPR_CACHE_DIR (default:
59
+ # ~/.cache/aipr) with a TTL (AIPR_CACHE_TTL seconds, default 86400 = 24h).
60
+ # The cron scans ~10 repos/day and each repo costs up to 11 API calls; the
61
+ # cache keeps repeat scans free of charge. --no-cache bypasses it entirely.
62
+
63
+
64
+ def _cache_dir() -> Path:
65
+ return Path(os.environ.get("AIPR_CACHE_DIR", str(Path.home() / ".cache" / "aipr")))
66
+
67
+
68
+ def _cache_ttl() -> int:
69
+ try:
70
+ return int(os.environ.get("AIPR_CACHE_TTL", "86400"))
71
+ except ValueError:
72
+ return 86400
73
+
74
+
75
+ def clear_cache() -> None:
76
+ """Delete every cached policy entry."""
77
+ d = _cache_dir()
78
+ if d.exists():
79
+ for f in d.glob("*.json"):
80
+ f.unlink(missing_ok=True)
81
+
82
+
83
+ def _cache_get(key: str):
84
+ path = _cache_dir() / f"{key}.json"
85
+ try:
86
+ entry = json.loads(path.read_text())
87
+ if time.time() - entry["ts"] <= _cache_ttl():
88
+ return entry["files"]
89
+ path.unlink(missing_ok=True)
90
+ except Exception as e:
91
+ logger.warning("Cache read failed for %s: %s", key, e)
92
+ return None
93
+
94
+
95
+ def _cache_put(key: str, files: list) -> None:
96
+ d = _cache_dir()
97
+ d.mkdir(parents=True, exist_ok=True)
98
+ path = d / f"{key}.json"
99
+ try:
100
+ path.write_text(json.dumps({"ts": time.time(), "files": files}))
101
+ except OSError:
102
+ pass
103
+
104
+
105
+ def _fetch_gh(url: str) -> str | None:
106
+ """Fetch raw content via the GitHub API (honors GH_TOKEN; no hard dep on gh).
107
+
108
+ Delegates to :func:`aipr.http.fetch_with_retry` for retry + rate-limit
109
+ handling. Transient errors (502/503/504/408/429) are retried up to 3 times
110
+ with exponential backoff. HTTP 403 (rate limit) raises RateLimitError.
111
+ """
112
+ return fetch_with_retry(url)
113
+
114
+
115
+ def fetch_policy_text(repo: str, use_cache: bool = True) -> list[tuple[str, str]]:
116
+ """Return [(filename, text), ...] for every candidate file found in owner/repo.
117
+
118
+ Raises :class:`RateLimitError` if rate-limited by the GitHub API. In that case,
119
+ no caching occurs — a later retry can succeed.
120
+ """
121
+ key = repo.replace("/", "__")
122
+ if use_cache:
123
+ cached = _cache_get(key)
124
+ if cached is not None:
125
+ return [(name, text) for name, text in cached]
126
+
127
+ results: list[tuple[str, str]] = []
128
+ for name in CANDIDATE_FILES:
129
+ try:
130
+ text = _fetch_gh(f"https://api.github.com/repos/{repo}/contents/{name}")
131
+ except RateLimitError:
132
+ raise # propagate — caller handles, no caching
133
+ if text and text.strip():
134
+ results.append((name, text))
135
+ if not results and "/" in repo:
136
+ org = repo.split("/")[0]
137
+ for name in ORG_FALLBACK_FILES:
138
+ try:
139
+ text = _fetch_gh(f"https://api.github.com/repos/{org}/.github/contents/{name}")
140
+ except RateLimitError:
141
+ raise
142
+ if text and text.strip():
143
+ results.append((f"{org}/.github/{name}", text))
144
+ break
145
+ if use_cache and results:
146
+ _cache_put(key, [(name, text) for name, text in results])
147
+ return results
148
+
149
+
150
+ def _validate_repo(repo: str) -> bool:
151
+ """Validate that repo matches the expected OWNER/REPO format."""
152
+ import re
153
+ return bool(re.match(r'^[a-zA-Z0-9_.-]+/[a-zA-Z0-9_.-]+$', repo))
154
+
155
+
156
+ def classify_repo(repo: str, use_cache: bool = True) -> dict:
157
+ """Fetch + classify all governance files of one repository.
158
+
159
+ On :class:`RateLimitError`, the cache is NOT populated with an empty result
160
+ so that a later retry can succeed. The returned dict contains
161
+ ``rate_limited: true`` to allow callers to distinguish this case from
162
+ genuine "no policy found" unknowns.
163
+ """
164
+ if not _validate_repo(repo):
165
+ raise ValueError(f"Invalid repository format: {repo!r}. Expected OWNER/REPO with alphanumeric, hyphen, underscore, dot characters.")
166
+ try:
167
+ files = fetch_policy_text(repo, use_cache=use_cache)
168
+ except RateLimitError as e:
169
+ logger.warning("Rate-limited on %s: %s", repo, e)
170
+ return {
171
+ "repo": repo,
172
+ "verdict": Verdict.UNKNOWN.value,
173
+ "files": [],
174
+ "rate_limited": True,
175
+ "autonomous_safe": False,
176
+ "confidence": 0.0,
177
+ "score": 0.0,
178
+ }
179
+ if not files:
180
+ return {"repo": repo, "verdict": Verdict.UNKNOWN.value, "files": [],
181
+ "autonomous_safe": False, "confidence": 0.0, "score": 0.0}
182
+
183
+ combined = "\n\n".join(text for _, text in files)
184
+ policy = detect_policy(combined)
185
+ return {
186
+ "repo": repo,
187
+ "verdict": policy.verdict.value,
188
+ "files": [name for name, _ in files],
189
+ "evidence": policy.evidence,
190
+ "autonomous_safe": policy.autonomous_safe,
191
+ "confidence": policy.confidence,
192
+ "score": policy.score,
193
+ }
194
+
195
+
196
+ def _build_parser() -> argparse.ArgumentParser:
197
+ """Build the argument parser for aipr."""
198
+ parser = argparse.ArgumentParser(
199
+ prog="aipr",
200
+ description="Read a repository's AI contribution policy before contributing.",
201
+ )
202
+ parser.add_argument("repo", nargs="*", help="owner/repo to inspect (accepts several for batch)")
203
+ parser.add_argument("--text", metavar="FILE", help="classify a local file instead")
204
+ parser.add_argument("--json", action="store_true", dest="as_json", help="JSON output")
205
+ parser.add_argument("--sarif", action="store_true", dest="as_sarif", help="SARIF 2.1.0 output (for GitHub Code Scanning)")
206
+ parser.add_argument(
207
+ "--explain",
208
+ action="store_true",
209
+ dest="explain",
210
+ help="explain verdict with contributing policy sources and snippets",
211
+ )
212
+ parser.add_argument(
213
+ "--no-cache",
214
+ action="store_true",
215
+ dest="no_cache",
216
+ help="fetch fresh policy files even if a cached copy exists",
217
+ )
218
+ parser.add_argument(
219
+ "--version",
220
+ action="version",
221
+ version=f"%(prog)s {__version__}",
222
+ )
223
+ return parser
224
+
225
+
226
+ def _parse_init_args(argv: list[str]) -> argparse.Namespace:
227
+ """Parse args for the 'init' subcommand."""
228
+ parser = argparse.ArgumentParser(prog="aipr init", description="Scaffold AI policy files")
229
+ parser.add_argument(
230
+ "--dir",
231
+ type=Path,
232
+ default=Path("."),
233
+ help="Directory to create policy files (default: current dir)",
234
+ )
235
+ parser.add_argument(
236
+ "--type",
237
+ choices=["permissive", "disclose", "human_only"],
238
+ default="disclose",
239
+ help="Policy preset (default: disclose)",
240
+ )
241
+ parser.add_argument(
242
+ "--org",
243
+ help="Organization name for centralized policies",
244
+ )
245
+ return parser.parse_args(argv)
246
+
247
+
248
+ def _cmd_init(args: argparse.Namespace) -> int:
249
+ """Scaffold AI policy files in the target directory."""
250
+ target_dir: Path = args.dir
251
+ target_dir.mkdir(parents=True, exist_ok=True)
252
+
253
+ policy_file = target_dir / "AI_POLICY.md"
254
+ tool_policy_file = target_dir / "AI_TOOL_POLICY.md"
255
+
256
+ org_line = f" for {args.org}" if args.org else ""
257
+
258
+ if args.type == "permissive":
259
+ policy_text = f"""# AI Usage Policy{org_line}
260
+
261
+ This repository welcomes contributions made with the assistance of AI tools.
262
+
263
+ ## Guidelines
264
+
265
+ - AI-assisted contributions are encouraged.
266
+ - Contributors are responsible for verifying the correctness of all submitted code.
267
+ - No disclosure is required for AI-assisted work.
268
+
269
+ ## Scope
270
+
271
+ This policy applies to all contributions: code, documentation, tests, and design.
272
+ """
273
+ elif args.type == "human_only":
274
+ policy_text = f"""# AI Usage Policy{org_line}
275
+
276
+ This repository does NOT accept contributions generated by AI tools.
277
+
278
+ ## Policy
279
+
280
+ - All contributions must be authored by humans.
281
+ - AI-generated code, documentation, or design submissions will be rejected.
282
+ - Automated pull requests from bots or AI agents are not permitted.
283
+
284
+ ## Rationale
285
+
286
+ [Explain why human authorship is required: e.g., IP concerns, code quality, regulatory requirements.]
287
+ """
288
+ else: # disclose
289
+ policy_text = f"""# AI Usage Policy{org_line}
290
+
291
+ This repository accepts AI-assisted contributions with disclosure.
292
+
293
+ ## Guidelines
294
+
295
+ - Contributions made with AI assistance are welcome.
296
+ - You MUST disclose the use of AI tools in your pull request description.
297
+ - Add the following trailer to your commit messages when AI was involved:
298
+
299
+ `Assisted-by: AI`
300
+
301
+ - You are responsible for reviewing and verifying all AI-generated content.
302
+
303
+ ## Scope
304
+
305
+ This policy applies to all contributions: code, documentation, tests, and design.
306
+ """
307
+
308
+ tool_policy_text = f"""# AI Tool Policy{org_line}
309
+
310
+ Classification of AI tools and their permitted usage.
311
+
312
+ ## Permitted Tools
313
+
314
+ | Tool | Usage | Notes |
315
+ |------|-------|-------|
316
+ | GitHub Copilot | Code completion | Must review all suggestions |
317
+ | ChatGPT / Claude | Code generation | Must disclose in PR |
318
+ | aipr | Policy checking | Encouraged before contributing |
319
+
320
+ ## Restricted Tools
321
+
322
+ - Fully autonomous coding agents (without human review) are not permitted.
323
+ - Tools that submit PRs automatically without human oversight.
324
+
325
+ ## Disclosure Format
326
+
327
+ In your PR description, include:
328
+
329
+ - Tool name and version
330
+ - What it was used for
331
+ - How you verified the output
332
+ """
333
+
334
+ created = []
335
+ if not policy_file.exists():
336
+ policy_file.write_text(policy_text)
337
+ created.append(str(policy_file))
338
+ else:
339
+ print(f"WARNING: {policy_file} already exists — skipping", file=sys.stderr)
340
+
341
+ if not tool_policy_file.exists():
342
+ tool_policy_file.write_text(tool_policy_text)
343
+ created.append(str(tool_policy_file))
344
+ else:
345
+ print(f"WARNING: {tool_policy_file} already exists — skipping", file=sys.stderr)
346
+
347
+ if created:
348
+ print(f"✓ Created {', '.join(created)}")
349
+ return 0
350
+ else:
351
+ print("No files created (already exist).", file=sys.stderr)
352
+ return 1
353
+
354
+
355
+ def main(argv: list[str] | None = None) -> int:
356
+ # Fast-path: if first arg is "init", dispatch to init subcommand
357
+ # before building the main parser (avoids argparse subparser conflicts
358
+ # with "owner/repo" positional args).
359
+ if argv is not None and argv and argv[0] == "init":
360
+ args = _parse_init_args(argv[1:])
361
+ return _cmd_init(args)
362
+
363
+ parser = _build_parser()
364
+ args = parser.parse_args(argv)
365
+
366
+ if not args.repo and not args.text:
367
+ parser.print_usage(sys.stderr)
368
+ return EXIT_USAGE
369
+
370
+ if args.text:
371
+ if args.repo:
372
+ parser.error("--text cannot be combined with repo")
373
+ text = Path(args.text).read_text(encoding="utf-8", errors="replace")
374
+ result = detect_policy(text)
375
+ payload = {
376
+ "source": args.text,
377
+ "files": [args.text],
378
+ "verdict": result.verdict.value,
379
+ "evidence": result.evidence,
380
+ "autonomous_safe": result.autonomous_safe,
381
+ "confidence": result.confidence,
382
+ "score": result.score,
383
+ }
384
+ exit_code = {
385
+ Verdict.UNKNOWN: EXIT_UNKNOWN,
386
+ }.get(result.verdict, EXIT_OK if result.autonomous_safe else EXIT_UNSAFE)
387
+ if args.as_sarif:
388
+ from .sarif import to_sarif
389
+ print(json.dumps(to_sarif(payload), indent=2))
390
+ else:
391
+ print(json.dumps(payload, indent=2) if args.as_json else _render(payload))
392
+ return exit_code
393
+
394
+ # Batch mode: classify every repo, aggregate the exit code, and emit either
395
+ # a JSON array or a per-repo human-readable block.
396
+ try:
397
+ if args.no_cache:
398
+ from .detector import clear_policy_cache
399
+ clear_policy_cache()
400
+ results = [classify_repo(repo, use_cache=not args.no_cache) for repo in args.repo]
401
+ except ValueError as e:
402
+ print(f"ERROR: {e}", file=sys.stderr)
403
+ return EXIT_USAGE
404
+
405
+ def _exit_code(r: dict) -> int:
406
+ if r["verdict"] == Verdict.UNKNOWN.value:
407
+ return EXIT_UNKNOWN
408
+ return EXIT_OK if r["autonomous_safe"] else EXIT_UNSAFE
409
+
410
+ worst = max(_exit_code(r) for r in results)
411
+
412
+ if args.as_sarif:
413
+ from .sarif import to_sarif
414
+ print(json.dumps(to_sarif(results), indent=2))
415
+ elif args.as_json:
416
+ print(json.dumps(results if len(results) > 1 else results[0], indent=2))
417
+ else:
418
+ for i, r in enumerate(results):
419
+ if i:
420
+ print()
421
+ print(_render(r))
422
+ return worst
423
+
424
+
425
+ def _render(payload: dict) -> str:
426
+ icon = {
427
+ Verdict.HUMAN_ONLY.value: "[BLOCKED] human-only policy",
428
+ Verdict.RESTRICTIVE.value: "[CAUTION] restrictive policy",
429
+ Verdict.DISCLOSE_OK.value: "[OK] allowed with disclosure",
430
+ Verdict.PERMISSIVE.value: "[OK] permissive",
431
+ Verdict.UNKNOWN.value: "[UNKNOWN] no explicit AI policy found",
432
+ }[payload["verdict"]]
433
+ lines = [f"aipr: {payload.get('repo') or payload.get('source')}", icon]
434
+ if "files" in payload and payload["files"]:
435
+ lines.append(f"sources: {', '.join(payload['files'])}")
436
+ if payload.get("confidence") is not None:
437
+ lines.append(f"confidence: {payload['confidence']} score: {payload.get('score')}")
438
+ if payload.get("autonomous_safe"):
439
+ lines.append("autonomous contribution: SAFE (still follow disclosure rules)")
440
+ elif payload["verdict"] != Verdict.UNKNOWN.value:
441
+ lines.append("autonomous contribution: NOT SAFE - require human co-authorship")
442
+ for ev in payload.get("evidence", [])[:3]:
443
+ lines.append(f" {ev}")
444
+ return "\n".join(lines)
445
+
446
+
447
+ if __name__ == "__main__":
448
+ raise SystemExit(main())
aipr/detector.py ADDED
@@ -0,0 +1,169 @@
1
+ """Detect a repository's AI contribution policy from its governance text.
2
+
3
+ Scoring model: weighted phrase matching over the files most likely to carry
4
+ governance (CONTRIBUTING.md, AI_POLICY.md, README.md, AGENTS.md, CLAUDE.md).
5
+ Restrictive signals outweigh permissive ones; silence yields UNKNOWN.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import re as _re
11
+ import regex
12
+ from dataclasses import dataclass, field
13
+ from enum import Enum
14
+
15
+ from aipr.cache import TTLCache
16
+
17
+
18
+ class Verdict(Enum):
19
+ HUMAN_ONLY = "human_only" # AI must not be the main author / human-written only
20
+ RESTRICTIVE = "restrictive" # heavy limits: mandatory process, bans on parts
21
+ DISCLOSE_OK = "disclose_ok" # allowed with disclosure / trailer
22
+ PERMISSIVE = "permissive" # explicitly welcomes AI contributions
23
+ UNKNOWN = "unknown" # no policy text found
24
+
25
+
26
+ # Maximum input length before truncation (prevents ReDoS on adversarial input)
27
+ MAX_INPUT_LENGTH = 1_000_000 # 1 MB
28
+
29
+ # Timeout in milliseconds for regex operations (prevents catastrophic backtracking)
30
+ REGEX_TIMEOUT_MS = 500 # 0.5 seconds
31
+
32
+ # Global TTL cache instance (replaces lru_cache)
33
+ _policy_cache = TTLCache()
34
+
35
+ # (compiled pattern, weight). Positive = restrictive signal, negative = permissive.
36
+ # Note: patterns use regex module (not re) for timeout support.
37
+ RULES: list[tuple[regex.Pattern[str], float]] = [
38
+ # --- human-only / ban level (strongest) ---
39
+ (regex.compile(r"must\s+be\s+fully\s+human[- ]written", regex.I), 5.0),
40
+ (regex.compile(r"ai\s+should\s+never\s+be\s+the\s+main\s+author", regex.I), 5.0),
41
+ (regex.compile(r"human[\s-]+authored\s+only", regex.I), 5.0),
42
+ (regex.compile(r"(?:\bwe\s+)?(?:do\s+not|don't)\s+accept\s+(?:any\s+)?ai", regex.I), 4.5),
43
+ (regex.compile(r"(?:will\s+not\s+be\s+accepted|not\s+accepted\s+here)[^.]*\bai\b", regex.I), 4.5),
44
+ (regex.compile(r"no\s+ai[- ]generated\s+(?:code|content|contributions)", regex.I), 4.5),
45
+ (regex.compile(r"ai\s+contributions?\s+are\s+(?:strictly\s+)?(?:forbidden|prohibited|banned)", regex.I), 4.5),
46
+ (regex.compile(r"(?:full(?:y|)\s+)?ai[- ]generated\s+contributions?[^.]{0,80}are\s+not\s+(?:allowed|permitted)", regex.I), 4.5),
47
+ (regex.compile(r"fully\s+generated\s+code\s+is\s+not\s+allowed", regex.I), 4.5),
48
+ (regex.compile(r"agents?\s+are\s+strictly\s+forbidden", regex.I), 5.0),
49
+ (regex.compile(r"bad\s+ai\s+\w+\s+will\s+be\s+(?:denounced|blocked)", regex.I), 3.0),
50
+ (regex.compile(r"human[\s-]+in[\s-]+the[\s-]+loop\s+is\s+(?:required|mandatory)", regex.I), 2.0),
51
+ (regex.compile(r"(?:\b)?ai[- ]automation\s+without\s+human\s+review\s+is\s+not\s+(?:currently\s+)?permitted", regex.I), 3.0),
52
+ (regex.compile(r"write\s+pr\s+descriptions?\s+yourself", regex.I), 1.5),
53
+ # --- restrictive ---
54
+ (regex.compile(r"all\s+ai\s+usage[^.]{0,60}must\s+be\s+disclosed", regex.I), 2.5),
55
+ (regex.compile(r"mandatory\s+disclosure", regex.I), 2.0),
56
+ (regex.compile(r"must\s+state\s+the\s+tool\s+you\s+used", regex.I), 2.0),
57
+ (regex.compile(r"may\s+not\s+use\s+ai\s+for\s+['\"]?good\s+first\s+issues?", regex.I), 2.0),
58
+ (regex.compile(r"extractive\s+contribution", regex.I), 1.5),
59
+ (regex.compile(r"assisted[- ]by:\s*ai\s+(?:trailer\s+)?is\s+(?:required|mandatory)", regex.I), 1.5),
60
+ # --- understanding / human-in-the-loop mandate (e.g. alibaba/open-code-review AGENTS.md) ---
61
+ (regex.compile(r"(?:you\s+must|contributors?\s+must)\s+(?:disclose|report|declare)[^.]{0,80}\b(?:ai|artificial\s+intelligence|llm|copilot|claude|gpt|coding\s+agent)\b", regex.I), 2.5),
62
+ (regex.compile(r"(?:review|understand)\s+(?:every\s+line|all\s+(?:code|content|text))\s+(?:written|generated)\s+by\s+ai", regex.I), 2.0),
63
+ (regex.compile(r"(?:must\s+not|shall\s+not)\s+attribute\s+commits?\s+(?:to\s+(?:ai|llm)|through\s+(?:assisted[- ]by|co[- ]developed))", regex.I), 1.5),
64
+ # --- local AI tool policy rules (.cursorrules, copilot-instructions, etc.) ---
65
+ (regex.compile(r"never\s+(?:generate|write|create)(?:\s+or\s+(?:generate|write|create))?\s+(?:any\s+)?(?:code|content)\s+(?:for|in|without)\b", regex.I), 2.5),
66
+ (regex.compile(r"all\s+(?:(?:code|ai)\s+)?(?:changes|edits|modifications)\s+must\s+be\s+(?:human[- ]?)?reviewed\b", regex.I), 2.0),
67
+ (regex.compile(r"(?:do\s+not|don't|never|may\s+not)\s+use\s+(?:ai|copilot|cursor|windsurf|aider|llms?)\s+(?:for|to)\b", regex.I), 2.5),
68
+ # --- disclose-ok ---
69
+ (regex.compile(r"assisted[- ]by:\s*ai", regex.I), -1.5),
70
+ (regex.compile(r"disclos\w+[^.]{0,40}\b(?:is|are)\s+(?:required|expected)\b", regex.I), -1.0),
71
+ (regex.compile(r"ai[- ]assisted\s+contributions?\s+are\s+(?:welcome|allowed|accepted)", regex.I), -3.0),
72
+ (regex.compile(r"ai\s+(?:usage|assistance)\s+is\s+(?:welcome|allowed|fine|ok)\b", regex.I), -3.0),
73
+ # --- permissive ---
74
+ (regex.compile(r"(?:\bwe\s+)?(?:warmly\s+)?welcome\s+ai[- ](?:assisted|generated)", regex.I), -3.5),
75
+ (regex.compile(r"feel\s+free\s+to\s+use\s+(?:claude|copilot|chatgpt|llms?|ai\s+tools)", regex.I), -3.0),
76
+ (regex.compile(r"agents?\s+are\s+welcome", regex.I), -3.0),
77
+ # --- permissive (agent-guide patterns seen in the wild: openhuman etc.) ---
78
+ (regex.compile(r"let\s+an\s+ai\s+coding\s+agent\s+guide\s+you", regex.I), -3.0),
79
+ (regex.compile(r"if\s+you\s+use\s+(?:claude\s+code|cursor|ampcode|codex).*?coding\s+agent", regex.I | regex.S), -3.0),
80
+ (regex.compile(r"paste\s+this\s+prompt.*?(?:agents\.md|claude\.md)", regex.I | regex.S), -2.5),
81
+ ]
82
+
83
+ RESTRICTIVE_THRESHOLD = 2.0
84
+ DISCLOSE_THRESHOLD = -0.5
85
+
86
+
87
+ @dataclass
88
+ class Policy:
89
+ verdict: Verdict
90
+ confidence: float
91
+ evidence: list[str] = field(default_factory=list)
92
+ score: float = 0.0
93
+
94
+ @property
95
+ def autonomous_safe(self) -> bool:
96
+ """True when an autonomous agent may contribute without human co-authorship."""
97
+ return self.verdict in (Verdict.DISCLOSE_OK, Verdict.PERMISSIVE)
98
+
99
+
100
+ def _score_text(text: str) -> Policy:
101
+ """Score a single blob of governance text (pure computation, no caching)."""
102
+ score = 0.0
103
+ evidence: list[str] = []
104
+ matched_strong = False
105
+
106
+ for pattern, weight in RULES:
107
+ try:
108
+ match = pattern.search(text, timeout=REGEX_TIMEOUT_MS)
109
+ except TimeoutError:
110
+ continue
111
+ if not match:
112
+ continue
113
+ score += weight
114
+ start = max(0, match.start() - 30)
115
+ end = min(len(text), match.end() + 50)
116
+ snippet = _re.sub(r"\s+", " ", text[start:end]).strip()
117
+ evidence.append(f"[{weight:+.1f}] ...{snippet}...")
118
+ if abs(weight) >= 3.0:
119
+ matched_strong = True
120
+
121
+ if not evidence or score == 0.0:
122
+ return Policy(Verdict.UNKNOWN, 0.0, evidence, score)
123
+
124
+ if score >= 4.0:
125
+ verdict = Verdict.HUMAN_ONLY
126
+ elif score >= RESTRICTIVE_THRESHOLD:
127
+ verdict = Verdict.RESTRICTIVE
128
+ elif score <= DISCLOSE_THRESHOLD:
129
+ verdict = (
130
+ Verdict.PERMISSIVE if score <= -3.0 else Verdict.DISCLOSE_OK
131
+ )
132
+ else:
133
+ verdict = Verdict.RESTRICTIVE if score > 0 else Verdict.DISCLOSE_OK
134
+
135
+ confidence = min(1.0, abs(score) / 5.0)
136
+ if matched_strong:
137
+ confidence = max(confidence, 0.7)
138
+ return Policy(verdict, round(confidence, 2), evidence, round(score, 2))
139
+
140
+
141
+ def clear_policy_cache() -> None:
142
+ """Clear the in-memory cache for policy scoring."""
143
+ _policy_cache.clear()
144
+
145
+
146
+ def detect_policy(text: str) -> Policy:
147
+ """Score one blob of governance text and classify the stance.
148
+
149
+ Uses a thread-safe TTL cache with deep copy on read to prevent
150
+ mutation of shared state across concurrent callers.
151
+
152
+ Input text is truncated to MAX_INPUT_LENGTH to prevent ReDoS on
153
+ adversarial input.
154
+ """
155
+ if not text or not text.strip():
156
+ return Policy(Verdict.UNKNOWN, 0.0)
157
+ # Truncate to prevent catastrophic backtracking on adversarial input
158
+ if len(text) > MAX_INPUT_LENGTH:
159
+ text = text[:MAX_INPUT_LENGTH]
160
+
161
+ # Check cache first (returns deep copy on hit)
162
+ cached = _policy_cache.get(text)
163
+ if cached is not None:
164
+ return cached
165
+
166
+ # Compute and cache
167
+ result = _score_text(text)
168
+ _policy_cache.put(text, result)
169
+ return result
aipr/http.py ADDED
@@ -0,0 +1,84 @@
1
+ """HTTP helpers for aipr: retry + rate-limit awareness for GitHub API calls."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ import os
7
+ import time
8
+ import urllib.error
9
+ import urllib.request
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ class RateLimitError(Exception):
15
+ """Raised when the GitHub API returns HTTP 403 (rate limit)."""
16
+
17
+ def __init__(self, message: str = "GitHub API rate limit exceeded"):
18
+ super().__init__(message)
19
+
20
+
21
+ def fetch_with_retry(
22
+ url: str,
23
+ max_retries: int = 3,
24
+ base_delay: float = 1.0,
25
+ timeout: int = 15,
26
+ ) -> str | None:
27
+ """Fetch raw content via the GitHub API with retry + rate-limit handling.
28
+
29
+ Transient errors (502/503/504/408/429) are retried up to *max_retries* times
30
+ with exponential backoff. HTTP 403 (rate limit) raises :class:`RateLimitError`.
31
+
32
+ Returns the decoded text on success, or ``None`` if the request fails
33
+ for any non-rate-limit reason (e.g. 404 Not Found).
34
+ """
35
+ token = os.environ.get("GH_TOKEN") or os.environ.get("GITHUB_TOKEN")
36
+
37
+ for attempt in range(max_retries + 1):
38
+ from . import __version__
39
+ req = urllib.request.Request(
40
+ url, headers={
41
+ "Accept": "application/vnd.github.raw+json",
42
+ "User-Agent": f"aipr/{__version__}",
43
+ }
44
+ )
45
+ if token:
46
+ req.add_header("Authorization", f"Bearer {token}")
47
+
48
+ try:
49
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
50
+ return resp.read().decode("utf-8", errors="replace")
51
+ except urllib.error.HTTPError as e:
52
+ if e.code == 403:
53
+ raise RateLimitError(
54
+ f"GitHub API rate limit exceeded (HTTP 403) for {url}"
55
+ )
56
+ if e.code in (502, 503, 504, 408, 429) and attempt < max_retries:
57
+ delay = base_delay * (2**attempt)
58
+ logger.warning(
59
+ "Transient HTTP %s on attempt %s/%s, retrying in %.1fs",
60
+ e.code,
61
+ attempt + 1,
62
+ max_retries + 1,
63
+ delay,
64
+ )
65
+ time.sleep(delay)
66
+ continue
67
+ logger.warning("HTTP error %s fetching %s", e.code, url)
68
+ return None
69
+ except Exception as e:
70
+ if attempt < max_retries:
71
+ delay = base_delay * (2**attempt)
72
+ logger.warning(
73
+ "Fetch error on attempt %s/%s, retrying in %.1fs: %s",
74
+ attempt + 1,
75
+ max_retries + 1,
76
+ delay,
77
+ e,
78
+ )
79
+ time.sleep(delay)
80
+ continue
81
+ logger.warning("GitHub fetch failed for %s: %s", url, e)
82
+ return None
83
+
84
+ return None
aipr/py.typed ADDED
File without changes
aipr/sarif.py ADDED
@@ -0,0 +1,186 @@
1
+ """SARIF output generation for aipr.
2
+
3
+ Converts aipr classification results to SARIF 2.1.0 for ingestion by GitHub Code Scanning,
4
+ GitLab Vulnerability Reports, and any other consumer that speaks SARIF.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from typing import Any
9
+
10
+ SARIF_SCHEMA = "https://json.schemastore.org/sarif-2.1.0.json"
11
+
12
+ # Verdict metadata: (rule_id, rule_name, rule_description)
13
+ # Maps aipr verdict strings to SARIF rule metadata.
14
+ VERDICT_RULES = {
15
+ "human_only": (
16
+ "ai-policy-human-only",
17
+ "AI Contributions Prohibited",
18
+ "Repository policy forbids AI-generated contributions without human co-authorship.",
19
+ ),
20
+ "restrictive": (
21
+ "ai-policy-restrictive",
22
+ "AI Contributions Restricted",
23
+ "Repository policy imposes heavy limits on AI contributions (mandatory process, partial bans).",
24
+ ),
25
+ "disclose_ok": (
26
+ "ai-policy-disclose-required",
27
+ "AI Disclosure Required",
28
+ "Repository allows AI-assisted contributions but requires disclosure in PR/commit.",
29
+ ),
30
+ "permissive": (
31
+ "ai-policy-permissive",
32
+ "AI Contributions Allowed",
33
+ "Repository explicitly welcomes AI-assisted contributions.",
34
+ ),
35
+ "unknown": (
36
+ "ai-policy-unknown",
37
+ "No AI Policy Found",
38
+ "Repository has no explicit AI contribution policy — contribution risk is unknown.",
39
+ ),
40
+ }
41
+
42
+ # Map verdict to SARIF level.
43
+ # human_only / restrictive = error (blocks contribution)
44
+ # unknown = warning (needs manual review)
45
+ # permissive / disclose_ok = note (safe to proceed)
46
+ VERDICT_LEVEL = {
47
+ "human_only": "error",
48
+ "restrictive": "error",
49
+ "disclose_ok": "note",
50
+ "permissive": "note",
51
+ "unknown": "warning",
52
+ }
53
+
54
+
55
+ def _make_rule(rule_id: str, name: str, description: str) -> dict:
56
+ return {
57
+ "id": rule_id,
58
+ "name": name,
59
+ "shortDescription": {"text": description},
60
+ "fullDescription": {"text": description},
61
+ "helpUri": "https://github.com/yunaremaia/aipr",
62
+ }
63
+
64
+
65
+ def _make_result(
66
+ rule_id: str,
67
+ message: str,
68
+ *,
69
+ level: str = "warning",
70
+ repo: str | None = None,
71
+ source: str | None = None,
72
+ ) -> dict:
73
+ result: dict[str, Any] = {
74
+ "ruleId": rule_id,
75
+ "message": {"text": message},
76
+ "level": level,
77
+ }
78
+ # repo for remote, source for local text mode
79
+ qualified_name = repo or source or "unknown"
80
+ result["locations"] = [
81
+ {
82
+ "logicalLocations": [
83
+ {
84
+ "fullyQualifiedName": qualified_name,
85
+ "kind": "repository" if repo else "file",
86
+ }
87
+ ]
88
+ }
89
+ ]
90
+ return result
91
+
92
+
93
+ def to_sarif(results: list[dict] | dict, version: str | None = None) -> dict:
94
+ """Convert aipr classification results to SARIF 2.1.0 document.
95
+
96
+ Args:
97
+ results: list of classification dicts from classify_repo() or single-mode payload.
98
+ version: aipr version string (for the tool metadata).
99
+
100
+ Returns:
101
+ SARIF 2.1.0 document as a dict.
102
+ """
103
+ if version is None:
104
+ try:
105
+ from . import __version__ as version
106
+ except ImportError:
107
+ version = "0.2.2"
108
+
109
+ # Normalize to list
110
+ if isinstance(results, dict):
111
+ results = [results]
112
+
113
+ rules: list[dict] = []
114
+ sarif_results: list[dict] = []
115
+ rule_set: set[str] = set()
116
+
117
+ for r in results:
118
+ verdict = r.get("verdict", "unknown")
119
+ repo = r.get("repo")
120
+ source = r.get("source")
121
+ meta = VERDICT_RULES.get(verdict)
122
+ if not meta:
123
+ continue
124
+ rule_id, rule_name, rule_desc = meta
125
+
126
+ if rule_id not in rule_set:
127
+ rules.append(_make_rule(rule_id, rule_name, rule_desc))
128
+ rule_set.add(rule_id)
129
+
130
+ level = VERDICT_LEVEL.get(verdict, "warning")
131
+
132
+ # Build message
133
+ evidence = r.get("evidence", [])
134
+ evidence_text = ""
135
+ if evidence:
136
+ evidence_text = " | Evidence: " + "; ".join(evidence[:3])
137
+ files = r.get("files", [])
138
+ source_text = f" Sources: {', '.join(files)}" if files else ""
139
+
140
+ target = repo or source or "repository"
141
+ if verdict == "human_only":
142
+ message = (
143
+ f"Autonomous AI contributions NOT SAFE for {target}. "
144
+ f"Policy requires human authorship.{evidence_text}{source_text}"
145
+ )
146
+ elif verdict == "restrictive":
147
+ message = (
148
+ f"Autonomous AI contributions NOT SAFE for {target}. "
149
+ f"Policy has restrictive AI requirements.{evidence_text}{source_text}"
150
+ )
151
+ elif verdict == "disclose_ok":
152
+ message = (
153
+ f"AI contributions allowed with disclosure for {target}. "
154
+ f"Remember to add 'Assisted-by: AI' trailer.{evidence_text}{source_text}"
155
+ )
156
+ elif verdict == "permissive":
157
+ message = (
158
+ f"AI contributions welcome for {target}.{evidence_text}{source_text}"
159
+ )
160
+ else: # unknown
161
+ message = (
162
+ f"No explicit AI policy found for {target}. "
163
+ f"Manual review recommended before contributing.{source_text}"
164
+ )
165
+
166
+ sarif_results.append(
167
+ _make_result(rule_id, message, level=level, repo=repo, source=source)
168
+ )
169
+
170
+ return {
171
+ "$schema": SARIF_SCHEMA,
172
+ "version": "2.1.0",
173
+ "runs": [
174
+ {
175
+ "tool": {
176
+ "driver": {
177
+ "name": "aipr",
178
+ "version": version,
179
+ "informationUri": "https://github.com/yunaremaia/aipr",
180
+ "rules": rules,
181
+ }
182
+ },
183
+ "results": sarif_results,
184
+ }
185
+ ],
186
+ }
@@ -0,0 +1,239 @@
1
+ Metadata-Version: 2.4
2
+ Name: aipr-py
3
+ Version: 0.2.4
4
+ Summary: Read a repository's AI contribution policy before you (or your agent) contribute.
5
+ Author-email: Yunare Maia <yunare@gmail.com>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/yunaremaia/aipr
8
+ Keywords: ai-policy,open-source,contributing,cli,agents
9
+ Classifier: Development Status :: 4 - Beta
10
+ Classifier: Environment :: Console
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Software Development :: Version Control :: Git
15
+ Requires-Python: >=3.10
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: regex>=2024.1.0
19
+ Dynamic: license-file
20
+
21
+ # aipr
22
+
23
+ [![CI](https://github.com/yunaremaia/aipr/actions/workflows/ci.yml/badge.svg)](https://github.com/yunaremaia/aipr/actions/workflows/ci.yml)
24
+ [![Python](https://img.shields.io/badge/python-3.10%2B-blue.svg)](https://www.python.org/downloads/)
25
+ [![License](https://img.shields.io/badge/license-MIT-green.svg)](https://github.com/yunaremaia/aipr/blob/main/LICENSE)
26
+
27
+ **AI Policy Read** - read an open-source repository's AI contribution policy
28
+ before you (or your agent) contribute.
29
+
30
+ `aipr` fetches the governance files that usually carry AI rules
31
+ (`CONTRIBUTING.md`, `AI_POLICY.md`, `AGENTS.md`, `CLAUDE.md`, ...), classifies
32
+ the repository's stance with weighted phrase matching, and answers one
33
+ question: **can an AI-assisted or autonomous contribution land here?**
34
+
35
+ ```
36
+ $ aipr asciimoo/hister
37
+ aipr: asciimoo/hister
38
+ [BLOCKED] human-only policy
39
+ sources: CONTRIBUTING.md, README.md
40
+ confidence: 1.0 score: 19.0
41
+ autonomous contribution: NOT SAFE - require human co-authorship
42
+ [+5.0] ...Issues and PR descriptions must be fully human-written...
43
+ [+5.0] ...AI should never be the main author of the PR...
44
+
45
+ $ aipr apache/maka
46
+ aipr: apache/maka
47
+ [UNKNOWN] no explicit AI policy found
48
+ exit=2
49
+ ```
50
+
51
+ ## Why
52
+
53
+ More repositories are publishing explicit AI policies - from "we welcome
54
+ AI-assisted work" to "agents are strictly forbidden". Violating one burns the
55
+ contributor (and, for autonomous agents, the operator): rejected PRs at best,
56
+ blocks at worst. `aipr` makes the check mechanical and cheap, for humans
57
+ deciding where to spend review effort and for agents deciding where to spend
58
+ their quota.
59
+
60
+ ## Install
61
+
62
+ ```bash
63
+ # 1. From PyPI
64
+ pip install aipr-py
65
+
66
+ # 2. Standalone from GitHub
67
+ pip install git+https://github.com/yunaremaia/aipr.git
68
+ # requires Python 3.10+; GH_TOKEN recommended (anonymous API calls rate-limit fast)
69
+ export GH_TOKEN=ghp_xxx # classic token with public repo read access
70
+
71
+ # 3. As a GitHub CLI extension (recommended)
72
+ gh extension install yunaremaia/aipr
73
+ ```
74
+
75
+ The `gh extension install` method is the easiest — after install, `gh aipr OWNER/REPO` works immediately.
76
+
77
+ No dependencies beyond the standard library. `pytest` only to develop.
78
+
79
+ ## Usage
80
+
81
+ ### Standalone
82
+
83
+ ```bash
84
+ aipr OWNER/REPO # classify a GitHub repository
85
+ aipr --text FILE # classify a local governance file
86
+ aipr --json OWNER/REPO # machine-readable output
87
+ aipr --sarif OWNER/REPO # SARIF 2.1.0 output for GitHub Code Scanning
88
+ ```
89
+
90
+ ### As a GitHub CLI extension
91
+
92
+ After `gh extension install yunaremaia/aipr`, use `gh aipr` identically:
93
+
94
+ ```bash
95
+ gh aipr OWNER/REPO
96
+ gh aipr --json OWNER/REPO
97
+ gh aipr --sarif OWNER/REPO
98
+ gh aipr --text FILE
99
+ ```
100
+
101
+ Exit codes are preserved (0/1/2) for CI conditionals.
102
+
103
+ ### `init` — scaffold AI policy files
104
+
105
+ Generate `AI_POLICY.md` and `AI_TOOL_POLICY.md` in your repo:
106
+
107
+ ```bash
108
+ aipr init [--dir .] [--type disclose|permissive|human_only] [--org ORG]
109
+ ```
110
+
111
+ Presets:
112
+ - `permissive` — explicitly welcomes AI-assisted contributions (aipr-safe)
113
+ - `disclose_ok` (default) — allowed with `Assisted-by: AI` disclosure trailer
114
+ - `human_only` — AI must not be the main author (NOT autonomous-safe)
115
+
116
+ ### Verdicts
117
+
118
+ | Verdict | Meaning | Autonomous-safe? |
119
+ |---|---|---|
120
+ | `human_only` | AI must not be the main author / human-written only / bans agents | no |
121
+ | `restrictive` | heavy process: mandatory disclosure + human-in-the-loop requirements | no |
122
+ | `disclose_ok` | allowed with a disclosure trailer (`Assisted-by: AI`) | yes* |
123
+ | `permissive` | explicitly welcomes AI-assisted contributions | yes |
124
+ | `unknown` | no explicit policy found | ask first |
125
+
126
+ \* still follow the disclosure rules - "safe" means *no human co-authorship
127
+ required by policy*, not *no obligations*.
128
+
129
+ ### Exit codes (for CI and agents)
130
+
131
+ | Code | Meaning |
132
+ |---|---|
133
+ | 0 | all inspected repos are autonomous-safe |
134
+ | 1 | at least one repo is restricted or human-only |
135
+ | 2 | at least one repo is unknown / no policy found (ranks worse than 1) |
136
+ | 64 | usage error |
137
+
138
+ Batch mode: `aipr owner/repo1 owner/repo2 ...` prints one block per repo
139
+ (JSON array with `--json`) and the exit code reflects the worst result —
140
+ so an unverified repo can never pass a gate silently.
141
+
142
+ Policy fetches are cached on disk for 24h (`~/.cache/aipr`, configurable via
143
+ `AIPR_CACHE_DIR` / `AIPR_CACHE_TTL`), so repeated scans cost zero API calls.
144
+ Use `--no-cache` to force a fresh fetch.
145
+
146
+ ## How classification works
147
+
148
+ Weighted regex matching over concatenated governance text. Restrictive phrases
149
+ score positive ("must be fully human-written" +5), permissive ones negative
150
+ ("we warmly welcome AI-assisted" -3.5). The strongest signals force the
151
+ verdict; weak mixed signals lean restrictive on purpose - when in doubt, do
152
+ not send a bot.
153
+
154
+ Known limits: English-only patterns; phrase matching cannot understand nuance;
155
+ a repo can carry policy in unusual files we don't probe. Treat UNKNOWN as
156
+ "read it yourself".
157
+
158
+ ## CI/CD Integration
159
+
160
+ ### Block PRs Violating AI Policy
161
+
162
+ Drop `.github/workflows/aipr.yml` into your repository to automatically block pull requests targeting repos with human-only or restrictive AI policies:
163
+
164
+ ```yaml
165
+ # .github/workflows/aipr.yml – block PRs against repos with human-only AI policies
166
+ name: AI Policy Check
167
+ on:
168
+ pull_request:
169
+ branches: [main, master]
170
+
171
+ jobs:
172
+ aipr:
173
+ runs-on: ubuntu-latest
174
+ steps:
175
+ - uses: actions/checkout@v5
176
+ - uses: actions/setup-python@v5
177
+ with:
178
+ python-version: "3.10"
179
+ - run: pip install git+https://github.com/yunaremaia/aipr.git
180
+ - name: Check AI policy
181
+ run: |
182
+ aipr "$GITHUB_REPOSITORY" --json || true
183
+ # Exit 1 = human_only/restrictive (block)
184
+ # Exit 2 = unknown (warn, not block)
185
+ aipr "$GITHUB_REPOSITORY" --json | jq -e '.verdict == "human_only" or .verdict == "restrictive"' && exit 1 || exit 0
186
+ ```
187
+
188
+ ### GitHub Code Scanning (SARIF)
189
+
190
+ Use `--sarif` to output SARIF 2.1.0 (Static Analysis Results Interchange Format)
191
+ and upload to GitHub Code Scanning. This surfaces AI policy compliance as
192
+ alerts in the GitHub Security tab.
193
+
194
+ ```bash
195
+ # Generate SARIF output
196
+ aipr --sarif OWNER/REPO > aipr-results.sarif
197
+ # Upload to GitHub Code Scanning via GitHub Actions:
198
+ # github/codeql-action/upload-sarif with sarif_file: aipr-results.sarif
199
+ ```
200
+
201
+ Verdict mapping to SARIF levels:
202
+ - `human_only` / `restrictive` → `error` (blocks contribution)
203
+ - `unknown` → `warning` (needs manual review)
204
+ - `permissive` / `disclose_ok` → `note` (safe to proceed)
205
+
206
+ Example GitHub Actions workflow snippet:
207
+
208
+ ```yaml
209
+ name: AI Policy Check (SARIF)
210
+ on:
211
+ pull_request:
212
+ branches: [main]
213
+
214
+ jobs:
215
+ aipr-check:
216
+ runs-on: ubuntu-latest
217
+ permissions:
218
+ security-events: write
219
+ steps:
220
+ - uses: actions/checkout@v4
221
+ - uses: actions/setup-python@v5
222
+ with:
223
+ python-version: '3.11'
224
+ - run: pip install git+https://github.com/yunaremaia/aipr.git
225
+ - run: aipr --sarif ${{ github.event.pull_request.head.repo.full_name }} > aipr-results.sarif
226
+ - uses: github/codeql-action/upload-sarif@v3
227
+ with:
228
+ sarif_file: aipr-results.sarif
229
+ ```
230
+
231
+ ## Status
232
+
233
+ Early beta - battle-tested against a handful of real policies (hister,
234
+ modular, polars, MDAnalysis, maka). Rule additions welcome: open an issue with
235
+ the policy text and the verdict you expected.
236
+
237
+ ## License
238
+
239
+ MIT
@@ -0,0 +1,13 @@
1
+ aipr/__init__.py,sha256=Wc0wOm-z7mHTzyWVPxf_OmgVyfxlywkefPQhfHOugSw,92
2
+ aipr/cache.py,sha256=YB9etlQyNgPB2P0J5hjDMfBTOqWf3f3WF2lKsvmS2BQ,2557
3
+ aipr/cli.py,sha256=OOYhgz14gACTcjpUhBbe3aAi6BqcN-nmjtB0ogz8nf0,15050
4
+ aipr/detector.py,sha256=I2_PEftqLx5pJ-EPP0JG6dzd8_IF3Ule9emHBPBY7jc,8193
5
+ aipr/http.py,sha256=igNk2gOVVVWbhse8-J14LmtwQLuNuA_5uFLR9Sq3t7Y,2820
6
+ aipr/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ aipr/sarif.py,sha256=6VwjDFIxONppmgJjVHjGUp5cKvuYQLA7Stb4EA1ViAY,5849
8
+ aipr_py-0.2.4.dist-info/licenses/LICENSE,sha256=jaJ4N5zOiHxuIAaPgohynRXOjhcZsCwBBU6lpDNG-IE,1068
9
+ aipr_py-0.2.4.dist-info/METADATA,sha256=3FYKMmzKeBeB7VoNyaOje3ftpFbqk5OuXOfWRORHrYE,8077
10
+ aipr_py-0.2.4.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
11
+ aipr_py-0.2.4.dist-info/entry_points.txt,sha256=39kVAlqlSgRaNNmA3jMfOj3TTaViXnQiBcI2Q_VsXdM,39
12
+ aipr_py-0.2.4.dist-info/top_level.txt,sha256=3iYRlJDj8wYDnTXX2xK3SwMUaWDxLv2UXY_9CDHd6hY,5
13
+ aipr_py-0.2.4.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ aipr = aipr.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Yunare Maia
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ aipr