llm-sunset 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
llm_sunset/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """llm-sunset: find deprecated and soon-to-be-retired AI model IDs in your code."""
2
+
3
+ __version__ = "0.1.0"
llm_sunset/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ sys.exit(main())
llm_sunset/cli.py ADDED
@@ -0,0 +1,171 @@
1
+ """Command line interface for llm-sunset."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import os
7
+ import sys
8
+ from datetime import date
9
+ from typing import List, Optional
10
+
11
+ from . import __version__, report
12
+ from .data import load
13
+ from .scanner import Matcher, scan
14
+
15
+ EPILOG = """examples:
16
+ llm-sunset scan the current directory
17
+ llm-sunset src/ config/ scan specific paths
18
+ llm-sunset --provider openai only check OpenAI deprecations
19
+ llm-sunset --format sarif > out.sarif
20
+ llm-sunset info gpt-4o-2024-05-13 look up a single model
21
+ llm-sunset upcoming --days 90 list every shutdown in the next 90 days
22
+
23
+ Ignore a line with a comment containing: llm-sunset: ignore
24
+ Ignore a whole file with: llm-sunset: ignore-file
25
+ """
26
+
27
+
28
+ def _today(value: Optional[str]) -> date:
29
+ return date.fromisoformat(value) if value else date.today()
30
+
31
+
32
+ def _common(p: argparse.ArgumentParser) -> None:
33
+ p.add_argument("--provider", action="append", metavar="NAME",
34
+ help="only consider this provider (repeatable; e.g. openai, anthropic, google, azure, groq). "
35
+ "Default: all direct providers; Azure/Vertex/Bedrock need an explicit "
36
+ "--provider (or --provider all)")
37
+ p.add_argument("--offline", action="store_true", help="don't fetch live data; use cache or bundled snapshot")
38
+ p.add_argument("--data-file", metavar="PATH", help="use a local deprecations JSON file")
39
+ p.add_argument("--today", metavar="YYYY-MM-DD", help=argparse.SUPPRESS)
40
+
41
+
42
+ def build_parser() -> argparse.ArgumentParser:
43
+ p = argparse.ArgumentParser(
44
+ prog="llm-sunset",
45
+ description="Find AI model IDs in your code that are deprecated or about to be shut down.",
46
+ epilog=EPILOG,
47
+ formatter_class=argparse.RawDescriptionHelpFormatter,
48
+ )
49
+ p.add_argument("--version", action="version", version=f"llm-sunset {__version__}")
50
+ sub = p.add_subparsers(dest="command")
51
+
52
+ s = sub.add_parser("scan", help="scan files (default command)")
53
+ s.add_argument("paths", nargs="*", default=["."])
54
+ s.add_argument("--format", choices=["text", "json", "markdown", "sarif", "github"], default="text")
55
+ s.add_argument("--fail-within", type=int, default=90, metavar="DAYS",
56
+ help="treat models retiring within DAYS as errors (default: 90)")
57
+ s.add_argument("--warn-within", type=int, default=365, metavar="DAYS",
58
+ help="hide models retiring more than DAYS from now (default: 365, -1 = show all)")
59
+ s.add_argument("--no-fail", action="store_true", help="always exit 0")
60
+ s.add_argument("--exclude", action="append", default=[], metavar="GLOB", help="exclude paths (repeatable)")
61
+ s.add_argument("--include-docs", action="store_true", help="also scan .md/.rst/.txt files")
62
+ _common(s)
63
+
64
+ i = sub.add_parser("info", help="show deprecation info for model IDs")
65
+ i.add_argument("models", nargs="+")
66
+ _common(i)
67
+
68
+ u = sub.add_parser("upcoming", help="list upcoming shutdowns")
69
+ u.add_argument("--days", type=int, default=180, help="look ahead this many days (default: 180)")
70
+ _common(u)
71
+ return p
72
+
73
+
74
+ def cmd_scan(a: argparse.Namespace) -> int:
75
+ today = _today(a.today)
76
+ deps, source = load(offline=a.offline, data_file=a.data_file)
77
+ findings = scan(a.paths, Matcher(deps, a.provider), exclude=a.exclude, include_docs=a.include_docs)
78
+ if a.warn_within >= 0:
79
+ findings = [
80
+ f for f in findings
81
+ if f.deprecation.days_left(today) is None or f.deprecation.days_left(today) <= a.warn_within
82
+ ]
83
+
84
+ if a.format == "json":
85
+ print(report.render_json(findings, today, a.fail_within, source))
86
+ elif a.format == "markdown":
87
+ print(report.render_markdown(findings, today, a.fail_within, source))
88
+ elif a.format == "sarif":
89
+ print(report.render_sarif(findings, today, a.fail_within))
90
+ elif a.format == "github":
91
+ out = report.render_github(findings, today, a.fail_within)
92
+ if out:
93
+ print(out)
94
+ print(report.render_text(findings, today, a.fail_within, source))
95
+ summary = os.environ.get("GITHUB_STEP_SUMMARY")
96
+ if summary:
97
+ with open(summary, "a", encoding="utf-8") as fh:
98
+ fh.write(report.render_markdown(findings, today, a.fail_within, source) + "\n")
99
+ else:
100
+ print(report.render_text(findings, today, a.fail_within, source))
101
+
102
+ if a.no_fail:
103
+ return 0
104
+ return 1 if any(report.severity(f, today, a.fail_within) == "error" for f in findings) else 0
105
+
106
+
107
+ def cmd_info(a: argparse.Namespace) -> int:
108
+ today = _today(a.today)
109
+ deps, source = load(offline=a.offline, data_file=a.data_file)
110
+ matcher = Matcher(deps, a.provider)
111
+ status = 0
112
+ for model in a.models:
113
+ hits = matcher.by_id.get(model, [])
114
+ if not hits:
115
+ print(f"{model}: no deprecation notice found")
116
+ continue
117
+ status = 1
118
+ for d in hits:
119
+ left = d.days_left(today)
120
+ if left is None:
121
+ when = "deprecated, no shutdown date announced"
122
+ elif left <= 0:
123
+ when = f"RETIRED on {d.shutdown_date}"
124
+ else:
125
+ when = f"retires {d.shutdown_date} ({left} days left)"
126
+ print(f"{model} [{d.provider}]: {when}")
127
+ if d.replacements:
128
+ print(f" replace with: {', '.join(d.replacements)}")
129
+ if d.url:
130
+ print(f" source: {d.url}")
131
+ print(f"(data: {source})", file=sys.stderr)
132
+ return status
133
+
134
+
135
+ def cmd_upcoming(a: argparse.Namespace) -> int:
136
+ today = _today(a.today)
137
+ deps, source = load(offline=a.offline, data_file=a.data_file)
138
+ matcher = Matcher(deps, a.provider)
139
+ rows = []
140
+ for items in matcher.by_id.values():
141
+ for d in items:
142
+ left = d.days_left(today)
143
+ if left is not None and 0 < left <= a.days:
144
+ rows.append((d.shutdown_date, d))
145
+ rows.sort(key=lambda r: (r[0], r[1].provider, r[1].model_id))
146
+ if not rows:
147
+ print(f"No shutdowns in the next {a.days} days.")
148
+ for when, d in rows:
149
+ repl = f" -> {', '.join(d.replacements)}" if d.replacements else ""
150
+ print(f"{when} {d.provider:<14} {d.model_id}{repl}")
151
+ print(f"(data: {source})", file=sys.stderr)
152
+ return 0
153
+
154
+
155
+ def main(argv: Optional[List[str]] = None) -> int:
156
+ argv = list(sys.argv[1:] if argv is None else argv)
157
+ known = {"scan", "info", "upcoming", "-h", "--help", "--version"}
158
+ if not argv or argv[0] not in known:
159
+ argv = ["scan"] + argv
160
+ a = build_parser().parse_args(argv)
161
+ handler = {"scan": cmd_scan, "info": cmd_info, "upcoming": cmd_upcoming}[a.command]
162
+ try:
163
+ return handler(a)
164
+ except BrokenPipeError:
165
+ return 0
166
+ except KeyboardInterrupt:
167
+ return 130
168
+
169
+
170
+ if __name__ == "__main__":
171
+ sys.exit(main())
llm_sunset/data.py ADDED
@@ -0,0 +1,139 @@
1
+ """Load the AI model deprecation dataset.
2
+
3
+ Data comes from https://deprecations.info (MIT licensed, refreshed daily by
4
+ https://github.com/deprecations/deprecations-rss). We try, in order:
5
+
6
+ 1. a fresh local cache (< 24h old)
7
+ 2. the live feed
8
+ 3. a stale local cache
9
+ 4. the snapshot bundled with this package (works fully offline)
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import json
15
+ import os
16
+ import time
17
+ import urllib.request
18
+ from dataclasses import dataclass, field
19
+ from datetime import date
20
+ from pathlib import Path
21
+ from typing import Dict, Iterable, List, Optional, Tuple
22
+
23
+ FEED_URLS = (
24
+ "https://deprecations.info/v1/deprecations.json",
25
+ "https://raw.githubusercontent.com/deprecations/deprecations-rss/main/data.json",
26
+ )
27
+ CACHE_TTL_SECONDS = 24 * 60 * 60
28
+ SNAPSHOT_PATH = Path(__file__).with_name("snapshot.json")
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class Deprecation:
33
+ provider: str
34
+ model_id: str
35
+ shutdown_date: Optional[date]
36
+ deprecation_date: Optional[date]
37
+ replacements: Tuple[str, ...] = field(default_factory=tuple)
38
+ url: str = ""
39
+
40
+ def days_left(self, today: date) -> Optional[int]:
41
+ if self.shutdown_date is None:
42
+ return None
43
+ return (self.shutdown_date - today).days
44
+
45
+
46
+ def _parse_date(value: object) -> Optional[date]:
47
+ if not value or not isinstance(value, str):
48
+ return None
49
+ try:
50
+ return date.fromisoformat(value[:10])
51
+ except ValueError:
52
+ return None
53
+
54
+
55
+ def cache_path() -> Path:
56
+ base = os.environ.get("XDG_CACHE_HOME") or os.path.join(Path.home(), ".cache")
57
+ return Path(base) / "llm-sunset" / "deprecations.json"
58
+
59
+
60
+ def _download(timeout: float) -> Optional[list]:
61
+ for url in FEED_URLS:
62
+ try:
63
+ req = urllib.request.Request(url, headers={"User-Agent": "llm-sunset"})
64
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
65
+ data = json.loads(resp.read().decode("utf-8"))
66
+ if isinstance(data, list) and data:
67
+ return data
68
+ except Exception:
69
+ continue
70
+ return None
71
+
72
+
73
+ def _read_json(path: Path) -> Optional[list]:
74
+ try:
75
+ data = json.loads(path.read_text(encoding="utf-8"))
76
+ return data if isinstance(data, list) else None
77
+ except Exception:
78
+ return None
79
+
80
+
81
+ def load_raw(offline: bool = False, data_file: Optional[str] = None, timeout: float = 10.0) -> Tuple[list, str]:
82
+ """Return (records, source_description)."""
83
+ if data_file:
84
+ data = _read_json(Path(data_file))
85
+ if data is None:
86
+ raise SystemExit(f"llm-sunset: could not read data file {data_file}")
87
+ return data, data_file
88
+
89
+ cache = cache_path()
90
+ if not offline:
91
+ fresh = cache.exists() and time.time() - cache.stat().st_mtime < CACHE_TTL_SECONDS
92
+ if fresh:
93
+ data = _read_json(cache)
94
+ if data:
95
+ return data, "cache"
96
+ data = _download(timeout)
97
+ if data:
98
+ try:
99
+ cache.parent.mkdir(parents=True, exist_ok=True)
100
+ cache.write_text(json.dumps(data), encoding="utf-8")
101
+ except OSError:
102
+ pass
103
+ return data, "live"
104
+ data = _read_json(cache)
105
+ if data:
106
+ return data, "cache (stale)"
107
+
108
+ data = _read_json(SNAPSHOT_PATH)
109
+ if not data:
110
+ raise SystemExit("llm-sunset: bundled snapshot missing and feed unreachable")
111
+ return data, "bundled snapshot"
112
+
113
+
114
+ def normalize(records: Iterable[dict]) -> List[Deprecation]:
115
+ """Turn raw records into Deprecation objects, one per (provider, model_id)."""
116
+ best: Dict[Tuple[str, str], Tuple[str, Deprecation]] = {}
117
+ for r in records:
118
+ model_id = (r.get("model_id") or "").strip()
119
+ provider = (r.get("provider") or "Unknown").strip()
120
+ if not model_id:
121
+ continue
122
+ dep = Deprecation(
123
+ provider=provider,
124
+ model_id=model_id,
125
+ shutdown_date=_parse_date(r.get("shutdown_date")),
126
+ deprecation_date=_parse_date(r.get("deprecation_date")),
127
+ replacements=tuple(x for x in (r.get("replacement_models") or []) if isinstance(x, str)),
128
+ url=r.get("url") or "",
129
+ )
130
+ key = (provider, model_id)
131
+ seen = r.get("last_observed") or r.get("announcement_date") or ""
132
+ if key not in best or seen >= best[key][0]:
133
+ best[key] = (seen, dep)
134
+ return [d for _, d in best.values()]
135
+
136
+
137
+ def load(offline: bool = False, data_file: Optional[str] = None) -> Tuple[List[Deprecation], str]:
138
+ raw, source = load_raw(offline=offline, data_file=data_file)
139
+ return normalize(raw), source
llm_sunset/report.py ADDED
@@ -0,0 +1,191 @@
1
+ """Render findings as text, JSON, Markdown, SARIF or GitHub annotations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import sys
8
+ from datetime import date
9
+ from typing import Dict, List
10
+
11
+ from . import __version__
12
+ from .scanner import Finding
13
+
14
+ STATUS_ORDER = {"retired": 0, "retiring": 1, "deprecated": 2}
15
+
16
+
17
+ def severity(f: Finding, today: date, fail_within: int) -> str:
18
+ left = f.deprecation.days_left(today)
19
+ if left is not None and left <= fail_within:
20
+ return "error"
21
+ return "warning"
22
+
23
+
24
+ def describe(f: Finding, today: date) -> str:
25
+ d = f.deprecation
26
+ left = d.days_left(today)
27
+ if left is None:
28
+ when = "deprecated (no shutdown date announced)"
29
+ elif left <= 0:
30
+ when = f"RETIRED on {d.shutdown_date} ({-left} days ago)"
31
+ else:
32
+ when = f"retires {d.shutdown_date} ({left} days left)"
33
+ msg = f"{d.provider} model '{d.model_id}' {when}"
34
+ if d.replacements:
35
+ msg += f"; replace with: {', '.join(d.replacements)}"
36
+ return msg
37
+
38
+
39
+ def _use_color(stream) -> bool:
40
+ return hasattr(stream, "isatty") and stream.isatty() and not os.environ.get("NO_COLOR")
41
+
42
+
43
+ def render_text(findings: List[Finding], today: date, fail_within: int, source: str, files_note: str = "") -> str:
44
+ color = _use_color(sys.stdout)
45
+
46
+ def c(code: str, s: str) -> str:
47
+ return f"\033[{code}m{s}\033[0m" if color else s
48
+
49
+ if not findings:
50
+ return c("32", "✓ No deprecated AI model IDs found.") + f" (data: {source})"
51
+
52
+ lines = []
53
+ for f in findings:
54
+ sev = severity(f, today, fail_within)
55
+ tag = c("31;1", "error ") if sev == "error" else c("33", "warning")
56
+ lines.append(f"{c('1', f'{f.path}:{f.line}:{f.column}')} {tag} {describe(f, today)}")
57
+ errors = sum(1 for f in findings if severity(f, today, fail_within) == "error")
58
+ warnings = len(findings) - errors
59
+ lines.append("")
60
+ lines.append(
61
+ f"{len(findings)} finding(s): {c('31;1', str(errors) + ' error(s)')}, "
62
+ f"{c('33', str(warnings) + ' warning(s)')} (errors = retired or retiring within {fail_within} days; data: {source})"
63
+ )
64
+ return "\n".join(lines)
65
+
66
+
67
+ def _as_dict(f: Finding, today: date, fail_within: int) -> Dict:
68
+ d = f.deprecation
69
+ return {
70
+ "path": f.path,
71
+ "line": f.line,
72
+ "column": f.column,
73
+ "match": f.text,
74
+ "provider": d.provider,
75
+ "model_id": d.model_id,
76
+ "status": f.status(today),
77
+ "severity": severity(f, today, fail_within),
78
+ "shutdown_date": d.shutdown_date.isoformat() if d.shutdown_date else None,
79
+ "deprecation_date": d.deprecation_date.isoformat() if d.deprecation_date else None,
80
+ "days_left": d.days_left(today),
81
+ "replacements": list(d.replacements),
82
+ "url": d.url,
83
+ }
84
+
85
+
86
+ def render_json(findings: List[Finding], today: date, fail_within: int, source: str) -> str:
87
+ return json.dumps(
88
+ {
89
+ "tool": "llm-sunset",
90
+ "version": __version__,
91
+ "date": today.isoformat(),
92
+ "data_source": source,
93
+ "findings": [_as_dict(f, today, fail_within) for f in findings],
94
+ },
95
+ indent=2,
96
+ )
97
+
98
+
99
+ def render_markdown(findings: List[Finding], today: date, fail_within: int, source: str) -> str:
100
+ if not findings:
101
+ return "### 🌅 llm-sunset\n\n✅ No deprecated AI model IDs found.\n"
102
+ out = [
103
+ "### 🌅 llm-sunset: deprecated AI models found",
104
+ "",
105
+ "| | Location | Model | Provider | Shutdown | Replace with |",
106
+ "|---|---|---|---|---|---|",
107
+ ]
108
+ for f in findings:
109
+ d = f.deprecation
110
+ left = d.days_left(today)
111
+ if left is None:
112
+ when = "deprecated"
113
+ elif left <= 0:
114
+ when = f"**retired** {d.shutdown_date}"
115
+ else:
116
+ when = f"{d.shutdown_date} ({left}d)"
117
+ icon = "🔴" if severity(f, today, fail_within) == "error" else "🟡"
118
+ model = f"[`{d.model_id}`]({d.url})" if d.url else f"`{d.model_id}`"
119
+ repl = ", ".join(f"`{r}`" for r in d.replacements) or "—"
120
+ out.append(f"| {icon} | `{f.path}:{f.line}` | {model} | {d.provider} | {when} | {repl} |")
121
+ out += ["", f"<sub>Data: [deprecations.info](https://deprecations.info) ({source}) · llm-sunset {__version__}</sub>", ""]
122
+ return "\n".join(out)
123
+
124
+
125
+ def render_github(findings: List[Finding], today: date, fail_within: int) -> str:
126
+ def esc(s: str) -> str:
127
+ return s.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
128
+
129
+ lines = []
130
+ for f in findings:
131
+ level = severity(f, today, fail_within)
132
+ title = f"Deprecated model: {f.deprecation.model_id}"
133
+ lines.append(
134
+ f"::{level} file={esc(f.path)},line={f.line},col={f.column},title={esc(title)}::{esc(describe(f, today))}"
135
+ )
136
+ return "\n".join(lines)
137
+
138
+
139
+ def _uri(path: str) -> str:
140
+ p = path.replace(os.sep, "/")
141
+ while p.startswith("./"):
142
+ p = p[2:]
143
+ return p
144
+
145
+
146
+ def render_sarif(findings: List[Finding], today: date, fail_within: int) -> str:
147
+ rules = {}
148
+ results = []
149
+ for f in findings:
150
+ d = f.deprecation
151
+ rule_id = f"{d.provider}/{d.model_id}"
152
+ if rule_id not in rules:
153
+ rules[rule_id] = {
154
+ "id": rule_id,
155
+ "name": "DeprecatedAIModel",
156
+ "shortDescription": {"text": f"Deprecated AI model {d.model_id} ({d.provider})"},
157
+ "helpUri": d.url or "https://deprecations.info",
158
+ }
159
+ results.append(
160
+ {
161
+ "ruleId": rule_id,
162
+ "level": severity(f, today, fail_within),
163
+ "message": {"text": describe(f, today)},
164
+ "locations": [
165
+ {
166
+ "physicalLocation": {
167
+ "artifactLocation": {"uri": _uri(f.path)},
168
+ "region": {"startLine": f.line, "startColumn": f.column},
169
+ }
170
+ }
171
+ ],
172
+ }
173
+ )
174
+ sarif = {
175
+ "$schema": "https://json.schemastore.org/sarif-2.1.0.json",
176
+ "version": "2.1.0",
177
+ "runs": [
178
+ {
179
+ "tool": {
180
+ "driver": {
181
+ "name": "llm-sunset",
182
+ "version": __version__,
183
+ "informationUri": "https://github.com/Ashveil1/llm-sunset",
184
+ "rules": list(rules.values()),
185
+ }
186
+ },
187
+ "results": results,
188
+ }
189
+ ],
190
+ }
191
+ return json.dumps(sarif, indent=2)
llm_sunset/scanner.py ADDED
@@ -0,0 +1,173 @@
1
+ """Find references to deprecated AI model IDs in source files."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import bisect
6
+ import fnmatch
7
+ import os
8
+ import re
9
+ from dataclasses import dataclass
10
+ from datetime import date
11
+ from pathlib import Path
12
+ from typing import Dict, Iterable, Iterator, List, Optional, Sequence
13
+
14
+ from .data import Deprecation
15
+
16
+ DEFAULT_EXCLUDE_DIRS = {
17
+ ".git", ".hg", ".svn", "node_modules", "bower_components", "vendor",
18
+ ".venv", "venv", "env", ".env.d", "__pycache__", ".mypy_cache", ".pytest_cache",
19
+ ".ruff_cache", ".tox", ".nox", "dist", "build", "out", "target", ".next",
20
+ ".nuxt", ".svelte-kit", ".turbo", ".cache", "coverage", ".idea", ".vscode",
21
+ "site-packages",
22
+ }
23
+ DEFAULT_EXCLUDE_GLOBS = [
24
+ "*.lock", "package-lock.json", "pnpm-lock.yaml", "yarn.lock", "poetry.lock",
25
+ "uv.lock", "Cargo.lock", "go.sum", "*.min.js", "*.map", "*.svg",
26
+ "*.png", "*.jpg", "*.jpeg", "*.gif", "*.webp", "*.ico", "*.pdf", "*.zip",
27
+ "*.gz", "*.tar", "*.woff", "*.woff2", "*.ttf", "*.mp3", "*.mp4", "*.wasm",
28
+ "*.so", "*.dylib", "*.dll", "*.exe", "*.bin", "*.pyc", "*.ipynb_checkpoints",
29
+ ]
30
+ # Prose files mention old models all the time ("we migrated from gpt-3.5-turbo").
31
+ DOC_GLOBS = ["*.md", "*.mdx", "*.rst", "*.txt", "*.adoc", "CHANGELOG*", "HISTORY*"]
32
+
33
+ MAX_FILE_BYTES = 2 * 1024 * 1024
34
+ IGNORE_LINE = "llm-sunset: ignore"
35
+ IGNORE_FILE = "llm-sunset: ignore-file"
36
+
37
+ # Characters that may legally sit right before / after a model ID. "/" ":" and
38
+ # "@" are allowed so that "openai/gpt-4", "ft:gpt-4o:org::id" and
39
+ # "claude-3-5-sonnet@20240620" still match.
40
+ _LEFT = r"(?<![A-Za-z0-9._+-])"
41
+ _RIGHT = r"(?![A-Za-z0-9_+-])(?!\.[A-Za-z0-9])"
42
+
43
+ # Same character classes as _LEFT/_RIGHT; a token never ends with ".".
44
+ _TOKEN_RE = re.compile(r"[A-Za-z0-9_+-](?:[A-Za-z0-9._+-]*[A-Za-z0-9_+-])?")
45
+
46
+ # Platforms that resell models under their own (often very different) lifecycle
47
+ # dates. Off by default so OpenAI/Anthropic users don't get Azure dates.
48
+ PLATFORM_PROVIDERS = ("azure", "google vertex", "bedrock")
49
+
50
+
51
+ @dataclass(frozen=True)
52
+ class Finding:
53
+ path: str
54
+ line: int
55
+ column: int
56
+ text: str # exact text matched in the file
57
+ deprecation: Deprecation
58
+
59
+ def status(self, today: date) -> str:
60
+ left = self.deprecation.days_left(today)
61
+ if left is None:
62
+ return "deprecated"
63
+ return "retired" if left <= 0 else "retiring"
64
+
65
+
66
+ class Matcher:
67
+ def __init__(self, deprecations: Sequence[Deprecation], providers: Optional[Sequence[str]] = None):
68
+ wanted = [p.lower() for p in providers] if providers else []
69
+ everything = "all" in wanted
70
+ self.by_id: Dict[str, List[Deprecation]] = {}
71
+ for d in deprecations:
72
+ prov = d.provider.lower()
73
+ if everything:
74
+ pass
75
+ elif wanted:
76
+ if not any(w in prov for w in wanted):
77
+ continue
78
+ elif any(p in prov for p in PLATFORM_PROVIDERS):
79
+ continue
80
+ self.by_id.setdefault(d.model_id, []).append(d)
81
+
82
+ self._ids = frozenset(self.by_id)
83
+ # IDs like "babbage", "davinci" or "command" are ordinary words; only
84
+ # match them when they are the entire contents of a string literal.
85
+ self._words = frozenset(i for i in self.by_id if not any(c.isdigit() for c in i))
86
+
87
+ def find(self, text: str) -> Iterator[tuple]:
88
+ """Yield (offset, matched_id, [Deprecation]) for every hit in text."""
89
+ # Fast path: tokenize + set intersection both run in C.
90
+ hits = self._ids.intersection(_TOKEN_RE.findall(text))
91
+ if not hits:
92
+ return
93
+ ids = sorted(hits - self._words, key=len, reverse=True)
94
+ words = sorted(hits & self._words, key=len, reverse=True)
95
+ if ids:
96
+ rx = re.compile(_LEFT + "(" + "|".join(map(re.escape, ids)) + ")" + _RIGHT)
97
+ for m in rx.finditer(text):
98
+ yield m.start(1), m.group(1), self.by_id[m.group(1)]
99
+ if words:
100
+ rx = re.compile(r"""(["'`])(""" + "|".join(map(re.escape, words)) + r")\1")
101
+ for m in rx.finditer(text):
102
+ yield m.start(2), m.group(2), self.by_id[m.group(2)]
103
+
104
+ def match_line(self, line: str) -> Iterator[tuple]:
105
+ return self.find(line)
106
+
107
+
108
+ def _compile_globs(globs: Iterable[str]):
109
+ globs = list(globs)
110
+ if not globs:
111
+ return lambda rel, name: False
112
+ rx = re.compile("|".join(fnmatch.translate(g) for g in globs))
113
+ return lambda rel, name: bool(rx.match(name) or rx.match(rel.replace(os.sep, "/")))
114
+
115
+
116
+ def iter_files(paths: Sequence[str], exclude: Sequence[str], include_docs: bool) -> Iterator[Path]:
117
+ globs = list(DEFAULT_EXCLUDE_GLOBS) + list(exclude)
118
+ if not include_docs:
119
+ globs += DOC_GLOBS
120
+ file_excluded = _compile_globs(globs)
121
+ dir_excluded = _compile_globs(exclude)
122
+ for p in paths:
123
+ root = Path(p)
124
+ if root.is_file():
125
+ yield root # explicitly named files are always scanned
126
+ continue
127
+ for dirpath, dirnames, filenames in os.walk(root):
128
+ rel_dir = os.path.relpath(dirpath, root)
129
+ dirnames[:] = [
130
+ d for d in dirnames
131
+ if d not in DEFAULT_EXCLUDE_DIRS
132
+ and not dir_excluded(os.path.normpath(os.path.join(rel_dir, d)), d)
133
+ ]
134
+ for f in filenames:
135
+ if not file_excluded(os.path.normpath(os.path.join(rel_dir, f)), f):
136
+ yield Path(dirpath) / f
137
+
138
+
139
+ def scan_file(path: Path, matcher: Matcher) -> List[Finding]:
140
+ try:
141
+ with open(path, "rb") as fh:
142
+ raw = fh.read(MAX_FILE_BYTES + 1)
143
+ except OSError:
144
+ return []
145
+ if len(raw) > MAX_FILE_BYTES:
146
+ return []
147
+ if b"\0" in raw[:8192]:
148
+ return []
149
+ text = raw.decode("utf-8", errors="replace")
150
+ if IGNORE_FILE in text:
151
+ return []
152
+ findings = []
153
+ line_starts = None
154
+ for offset, matched, deps in matcher.find(text):
155
+ if line_starts is None:
156
+ line_starts = [0] + [m.end() for m in re.finditer(r"\n", text)]
157
+ idx = bisect.bisect_right(line_starts, offset) - 1
158
+ start = line_starts[idx]
159
+ end = text.find("\n", start)
160
+ line = text[start:] if end == -1 else text[start:end]
161
+ if IGNORE_LINE in line:
162
+ continue
163
+ for d in deps:
164
+ findings.append(Finding(str(path), idx + 1, offset - start + 1, matched, d))
165
+ return findings
166
+
167
+
168
+ def scan(paths: Sequence[str], matcher: Matcher, exclude: Sequence[str] = (), include_docs: bool = False) -> List[Finding]:
169
+ out: List[Finding] = []
170
+ for f in iter_files(paths, exclude, include_docs):
171
+ out.extend(scan_file(f, matcher))
172
+ out.sort(key=lambda x: (x.path, x.line, x.column, x.deprecation.provider))
173
+ return out