llm-sunset 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- llm_sunset/__init__.py +3 -0
- llm_sunset/__main__.py +5 -0
- llm_sunset/cli.py +171 -0
- llm_sunset/data.py +139 -0
- llm_sunset/report.py +191 -0
- llm_sunset/scanner.py +173 -0
- llm_sunset/snapshot.json +4497 -0
- llm_sunset-0.1.0.dist-info/METADATA +124 -0
- llm_sunset-0.1.0.dist-info/RECORD +13 -0
- llm_sunset-0.1.0.dist-info/WHEEL +5 -0
- llm_sunset-0.1.0.dist-info/entry_points.txt +2 -0
- llm_sunset-0.1.0.dist-info/licenses/LICENSE +21 -0
- llm_sunset-0.1.0.dist-info/top_level.txt +1 -0
llm_sunset/__init__.py
ADDED
llm_sunset/__main__.py
ADDED
llm_sunset/cli.py
ADDED
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
"""Command line interface for llm-sunset."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from datetime import date
|
|
9
|
+
from typing import List, Optional
|
|
10
|
+
|
|
11
|
+
from . import __version__, report
|
|
12
|
+
from .data import load
|
|
13
|
+
from .scanner import Matcher, scan
|
|
14
|
+
|
|
15
|
+
EPILOG = """examples:
|
|
16
|
+
llm-sunset scan the current directory
|
|
17
|
+
llm-sunset src/ config/ scan specific paths
|
|
18
|
+
llm-sunset --provider openai only check OpenAI deprecations
|
|
19
|
+
llm-sunset --format sarif > out.sarif
|
|
20
|
+
llm-sunset info gpt-4o-2024-05-13 look up a single model
|
|
21
|
+
llm-sunset upcoming --days 90 list every shutdown in the next 90 days
|
|
22
|
+
|
|
23
|
+
Ignore a line with a comment containing: llm-sunset: ignore
|
|
24
|
+
Ignore a whole file with: llm-sunset: ignore-file
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _today(value: Optional[str]) -> date:
|
|
29
|
+
return date.fromisoformat(value) if value else date.today()
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _common(p: argparse.ArgumentParser) -> None:
|
|
33
|
+
p.add_argument("--provider", action="append", metavar="NAME",
|
|
34
|
+
help="only consider this provider (repeatable; e.g. openai, anthropic, google, azure, groq). "
|
|
35
|
+
"Default: all direct providers; Azure/Vertex/Bedrock need an explicit "
|
|
36
|
+
"--provider (or --provider all)")
|
|
37
|
+
p.add_argument("--offline", action="store_true", help="don't fetch live data; use cache or bundled snapshot")
|
|
38
|
+
p.add_argument("--data-file", metavar="PATH", help="use a local deprecations JSON file")
|
|
39
|
+
p.add_argument("--today", metavar="YYYY-MM-DD", help=argparse.SUPPRESS)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
43
|
+
p = argparse.ArgumentParser(
|
|
44
|
+
prog="llm-sunset",
|
|
45
|
+
description="Find AI model IDs in your code that are deprecated or about to be shut down.",
|
|
46
|
+
epilog=EPILOG,
|
|
47
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
48
|
+
)
|
|
49
|
+
p.add_argument("--version", action="version", version=f"llm-sunset {__version__}")
|
|
50
|
+
sub = p.add_subparsers(dest="command")
|
|
51
|
+
|
|
52
|
+
s = sub.add_parser("scan", help="scan files (default command)")
|
|
53
|
+
s.add_argument("paths", nargs="*", default=["."])
|
|
54
|
+
s.add_argument("--format", choices=["text", "json", "markdown", "sarif", "github"], default="text")
|
|
55
|
+
s.add_argument("--fail-within", type=int, default=90, metavar="DAYS",
|
|
56
|
+
help="treat models retiring within DAYS as errors (default: 90)")
|
|
57
|
+
s.add_argument("--warn-within", type=int, default=365, metavar="DAYS",
|
|
58
|
+
help="hide models retiring more than DAYS from now (default: 365, -1 = show all)")
|
|
59
|
+
s.add_argument("--no-fail", action="store_true", help="always exit 0")
|
|
60
|
+
s.add_argument("--exclude", action="append", default=[], metavar="GLOB", help="exclude paths (repeatable)")
|
|
61
|
+
s.add_argument("--include-docs", action="store_true", help="also scan .md/.rst/.txt files")
|
|
62
|
+
_common(s)
|
|
63
|
+
|
|
64
|
+
i = sub.add_parser("info", help="show deprecation info for model IDs")
|
|
65
|
+
i.add_argument("models", nargs="+")
|
|
66
|
+
_common(i)
|
|
67
|
+
|
|
68
|
+
u = sub.add_parser("upcoming", help="list upcoming shutdowns")
|
|
69
|
+
u.add_argument("--days", type=int, default=180, help="look ahead this many days (default: 180)")
|
|
70
|
+
_common(u)
|
|
71
|
+
return p
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def cmd_scan(a: argparse.Namespace) -> int:
|
|
75
|
+
today = _today(a.today)
|
|
76
|
+
deps, source = load(offline=a.offline, data_file=a.data_file)
|
|
77
|
+
findings = scan(a.paths, Matcher(deps, a.provider), exclude=a.exclude, include_docs=a.include_docs)
|
|
78
|
+
if a.warn_within >= 0:
|
|
79
|
+
findings = [
|
|
80
|
+
f for f in findings
|
|
81
|
+
if f.deprecation.days_left(today) is None or f.deprecation.days_left(today) <= a.warn_within
|
|
82
|
+
]
|
|
83
|
+
|
|
84
|
+
if a.format == "json":
|
|
85
|
+
print(report.render_json(findings, today, a.fail_within, source))
|
|
86
|
+
elif a.format == "markdown":
|
|
87
|
+
print(report.render_markdown(findings, today, a.fail_within, source))
|
|
88
|
+
elif a.format == "sarif":
|
|
89
|
+
print(report.render_sarif(findings, today, a.fail_within))
|
|
90
|
+
elif a.format == "github":
|
|
91
|
+
out = report.render_github(findings, today, a.fail_within)
|
|
92
|
+
if out:
|
|
93
|
+
print(out)
|
|
94
|
+
print(report.render_text(findings, today, a.fail_within, source))
|
|
95
|
+
summary = os.environ.get("GITHUB_STEP_SUMMARY")
|
|
96
|
+
if summary:
|
|
97
|
+
with open(summary, "a", encoding="utf-8") as fh:
|
|
98
|
+
fh.write(report.render_markdown(findings, today, a.fail_within, source) + "\n")
|
|
99
|
+
else:
|
|
100
|
+
print(report.render_text(findings, today, a.fail_within, source))
|
|
101
|
+
|
|
102
|
+
if a.no_fail:
|
|
103
|
+
return 0
|
|
104
|
+
return 1 if any(report.severity(f, today, a.fail_within) == "error" for f in findings) else 0
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def cmd_info(a: argparse.Namespace) -> int:
|
|
108
|
+
today = _today(a.today)
|
|
109
|
+
deps, source = load(offline=a.offline, data_file=a.data_file)
|
|
110
|
+
matcher = Matcher(deps, a.provider)
|
|
111
|
+
status = 0
|
|
112
|
+
for model in a.models:
|
|
113
|
+
hits = matcher.by_id.get(model, [])
|
|
114
|
+
if not hits:
|
|
115
|
+
print(f"{model}: no deprecation notice found")
|
|
116
|
+
continue
|
|
117
|
+
status = 1
|
|
118
|
+
for d in hits:
|
|
119
|
+
left = d.days_left(today)
|
|
120
|
+
if left is None:
|
|
121
|
+
when = "deprecated, no shutdown date announced"
|
|
122
|
+
elif left <= 0:
|
|
123
|
+
when = f"RETIRED on {d.shutdown_date}"
|
|
124
|
+
else:
|
|
125
|
+
when = f"retires {d.shutdown_date} ({left} days left)"
|
|
126
|
+
print(f"{model} [{d.provider}]: {when}")
|
|
127
|
+
if d.replacements:
|
|
128
|
+
print(f" replace with: {', '.join(d.replacements)}")
|
|
129
|
+
if d.url:
|
|
130
|
+
print(f" source: {d.url}")
|
|
131
|
+
print(f"(data: {source})", file=sys.stderr)
|
|
132
|
+
return status
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def cmd_upcoming(a: argparse.Namespace) -> int:
|
|
136
|
+
today = _today(a.today)
|
|
137
|
+
deps, source = load(offline=a.offline, data_file=a.data_file)
|
|
138
|
+
matcher = Matcher(deps, a.provider)
|
|
139
|
+
rows = []
|
|
140
|
+
for items in matcher.by_id.values():
|
|
141
|
+
for d in items:
|
|
142
|
+
left = d.days_left(today)
|
|
143
|
+
if left is not None and 0 < left <= a.days:
|
|
144
|
+
rows.append((d.shutdown_date, d))
|
|
145
|
+
rows.sort(key=lambda r: (r[0], r[1].provider, r[1].model_id))
|
|
146
|
+
if not rows:
|
|
147
|
+
print(f"No shutdowns in the next {a.days} days.")
|
|
148
|
+
for when, d in rows:
|
|
149
|
+
repl = f" -> {', '.join(d.replacements)}" if d.replacements else ""
|
|
150
|
+
print(f"{when} {d.provider:<14} {d.model_id}{repl}")
|
|
151
|
+
print(f"(data: {source})", file=sys.stderr)
|
|
152
|
+
return 0
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
156
|
+
argv = list(sys.argv[1:] if argv is None else argv)
|
|
157
|
+
known = {"scan", "info", "upcoming", "-h", "--help", "--version"}
|
|
158
|
+
if not argv or argv[0] not in known:
|
|
159
|
+
argv = ["scan"] + argv
|
|
160
|
+
a = build_parser().parse_args(argv)
|
|
161
|
+
handler = {"scan": cmd_scan, "info": cmd_info, "upcoming": cmd_upcoming}[a.command]
|
|
162
|
+
try:
|
|
163
|
+
return handler(a)
|
|
164
|
+
except BrokenPipeError:
|
|
165
|
+
return 0
|
|
166
|
+
except KeyboardInterrupt:
|
|
167
|
+
return 130
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
sys.exit(main())
|
llm_sunset/data.py
ADDED
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
"""Load the AI model deprecation dataset.
|
|
2
|
+
|
|
3
|
+
Data comes from https://deprecations.info (MIT licensed, refreshed daily by
|
|
4
|
+
https://github.com/deprecations/deprecations-rss). We try, in order:
|
|
5
|
+
|
|
6
|
+
1. a fresh local cache (< 24h old)
|
|
7
|
+
2. the live feed
|
|
8
|
+
3. a stale local cache
|
|
9
|
+
4. the snapshot bundled with this package (works fully offline)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
import urllib.request
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from datetime import date
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Dict, Iterable, List, Optional, Tuple
|
|
22
|
+
|
|
23
|
+
FEED_URLS = (
|
|
24
|
+
"https://deprecations.info/v1/deprecations.json",
|
|
25
|
+
"https://raw.githubusercontent.com/deprecations/deprecations-rss/main/data.json",
|
|
26
|
+
)
|
|
27
|
+
CACHE_TTL_SECONDS = 24 * 60 * 60
|
|
28
|
+
SNAPSHOT_PATH = Path(__file__).with_name("snapshot.json")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Deprecation:
|
|
33
|
+
provider: str
|
|
34
|
+
model_id: str
|
|
35
|
+
shutdown_date: Optional[date]
|
|
36
|
+
deprecation_date: Optional[date]
|
|
37
|
+
replacements: Tuple[str, ...] = field(default_factory=tuple)
|
|
38
|
+
url: str = ""
|
|
39
|
+
|
|
40
|
+
def days_left(self, today: date) -> Optional[int]:
|
|
41
|
+
if self.shutdown_date is None:
|
|
42
|
+
return None
|
|
43
|
+
return (self.shutdown_date - today).days
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _parse_date(value: object) -> Optional[date]:
|
|
47
|
+
if not value or not isinstance(value, str):
|
|
48
|
+
return None
|
|
49
|
+
try:
|
|
50
|
+
return date.fromisoformat(value[:10])
|
|
51
|
+
except ValueError:
|
|
52
|
+
return None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def cache_path() -> Path:
|
|
56
|
+
base = os.environ.get("XDG_CACHE_HOME") or os.path.join(Path.home(), ".cache")
|
|
57
|
+
return Path(base) / "llm-sunset" / "deprecations.json"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _download(timeout: float) -> Optional[list]:
|
|
61
|
+
for url in FEED_URLS:
|
|
62
|
+
try:
|
|
63
|
+
req = urllib.request.Request(url, headers={"User-Agent": "llm-sunset"})
|
|
64
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
65
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
66
|
+
if isinstance(data, list) and data:
|
|
67
|
+
return data
|
|
68
|
+
except Exception:
|
|
69
|
+
continue
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _read_json(path: Path) -> Optional[list]:
|
|
74
|
+
try:
|
|
75
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
76
|
+
return data if isinstance(data, list) else None
|
|
77
|
+
except Exception:
|
|
78
|
+
return None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def load_raw(offline: bool = False, data_file: Optional[str] = None, timeout: float = 10.0) -> Tuple[list, str]:
|
|
82
|
+
"""Return (records, source_description)."""
|
|
83
|
+
if data_file:
|
|
84
|
+
data = _read_json(Path(data_file))
|
|
85
|
+
if data is None:
|
|
86
|
+
raise SystemExit(f"llm-sunset: could not read data file {data_file}")
|
|
87
|
+
return data, data_file
|
|
88
|
+
|
|
89
|
+
cache = cache_path()
|
|
90
|
+
if not offline:
|
|
91
|
+
fresh = cache.exists() and time.time() - cache.stat().st_mtime < CACHE_TTL_SECONDS
|
|
92
|
+
if fresh:
|
|
93
|
+
data = _read_json(cache)
|
|
94
|
+
if data:
|
|
95
|
+
return data, "cache"
|
|
96
|
+
data = _download(timeout)
|
|
97
|
+
if data:
|
|
98
|
+
try:
|
|
99
|
+
cache.parent.mkdir(parents=True, exist_ok=True)
|
|
100
|
+
cache.write_text(json.dumps(data), encoding="utf-8")
|
|
101
|
+
except OSError:
|
|
102
|
+
pass
|
|
103
|
+
return data, "live"
|
|
104
|
+
data = _read_json(cache)
|
|
105
|
+
if data:
|
|
106
|
+
return data, "cache (stale)"
|
|
107
|
+
|
|
108
|
+
data = _read_json(SNAPSHOT_PATH)
|
|
109
|
+
if not data:
|
|
110
|
+
raise SystemExit("llm-sunset: bundled snapshot missing and feed unreachable")
|
|
111
|
+
return data, "bundled snapshot"
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def normalize(records: Iterable[dict]) -> List[Deprecation]:
|
|
115
|
+
"""Turn raw records into Deprecation objects, one per (provider, model_id)."""
|
|
116
|
+
best: Dict[Tuple[str, str], Tuple[str, Deprecation]] = {}
|
|
117
|
+
for r in records:
|
|
118
|
+
model_id = (r.get("model_id") or "").strip()
|
|
119
|
+
provider = (r.get("provider") or "Unknown").strip()
|
|
120
|
+
if not model_id:
|
|
121
|
+
continue
|
|
122
|
+
dep = Deprecation(
|
|
123
|
+
provider=provider,
|
|
124
|
+
model_id=model_id,
|
|
125
|
+
shutdown_date=_parse_date(r.get("shutdown_date")),
|
|
126
|
+
deprecation_date=_parse_date(r.get("deprecation_date")),
|
|
127
|
+
replacements=tuple(x for x in (r.get("replacement_models") or []) if isinstance(x, str)),
|
|
128
|
+
url=r.get("url") or "",
|
|
129
|
+
)
|
|
130
|
+
key = (provider, model_id)
|
|
131
|
+
seen = r.get("last_observed") or r.get("announcement_date") or ""
|
|
132
|
+
if key not in best or seen >= best[key][0]:
|
|
133
|
+
best[key] = (seen, dep)
|
|
134
|
+
return [d for _, d in best.values()]
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def load(offline: bool = False, data_file: Optional[str] = None) -> Tuple[List[Deprecation], str]:
|
|
138
|
+
raw, source = load_raw(offline=offline, data_file=data_file)
|
|
139
|
+
return normalize(raw), source
|
llm_sunset/report.py
ADDED
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
"""Render findings as text, JSON, Markdown, SARIF or GitHub annotations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import sys
|
|
8
|
+
from datetime import date
|
|
9
|
+
from typing import Dict, List
|
|
10
|
+
|
|
11
|
+
from . import __version__
|
|
12
|
+
from .scanner import Finding
|
|
13
|
+
|
|
14
|
+
STATUS_ORDER = {"retired": 0, "retiring": 1, "deprecated": 2}
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def severity(f: Finding, today: date, fail_within: int) -> str:
|
|
18
|
+
left = f.deprecation.days_left(today)
|
|
19
|
+
if left is not None and left <= fail_within:
|
|
20
|
+
return "error"
|
|
21
|
+
return "warning"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def describe(f: Finding, today: date) -> str:
|
|
25
|
+
d = f.deprecation
|
|
26
|
+
left = d.days_left(today)
|
|
27
|
+
if left is None:
|
|
28
|
+
when = "deprecated (no shutdown date announced)"
|
|
29
|
+
elif left <= 0:
|
|
30
|
+
when = f"RETIRED on {d.shutdown_date} ({-left} days ago)"
|
|
31
|
+
else:
|
|
32
|
+
when = f"retires {d.shutdown_date} ({left} days left)"
|
|
33
|
+
msg = f"{d.provider} model '{d.model_id}' {when}"
|
|
34
|
+
if d.replacements:
|
|
35
|
+
msg += f"; replace with: {', '.join(d.replacements)}"
|
|
36
|
+
return msg
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _use_color(stream) -> bool:
|
|
40
|
+
return hasattr(stream, "isatty") and stream.isatty() and not os.environ.get("NO_COLOR")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def render_text(findings: List[Finding], today: date, fail_within: int, source: str, files_note: str = "") -> str:
|
|
44
|
+
color = _use_color(sys.stdout)
|
|
45
|
+
|
|
46
|
+
def c(code: str, s: str) -> str:
|
|
47
|
+
return f"\033[{code}m{s}\033[0m" if color else s
|
|
48
|
+
|
|
49
|
+
if not findings:
|
|
50
|
+
return c("32", "✓ No deprecated AI model IDs found.") + f" (data: {source})"
|
|
51
|
+
|
|
52
|
+
lines = []
|
|
53
|
+
for f in findings:
|
|
54
|
+
sev = severity(f, today, fail_within)
|
|
55
|
+
tag = c("31;1", "error ") if sev == "error" else c("33", "warning")
|
|
56
|
+
lines.append(f"{c('1', f'{f.path}:{f.line}:{f.column}')} {tag} {describe(f, today)}")
|
|
57
|
+
errors = sum(1 for f in findings if severity(f, today, fail_within) == "error")
|
|
58
|
+
warnings = len(findings) - errors
|
|
59
|
+
lines.append("")
|
|
60
|
+
lines.append(
|
|
61
|
+
f"{len(findings)} finding(s): {c('31;1', str(errors) + ' error(s)')}, "
|
|
62
|
+
f"{c('33', str(warnings) + ' warning(s)')} (errors = retired or retiring within {fail_within} days; data: {source})"
|
|
63
|
+
)
|
|
64
|
+
return "\n".join(lines)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _as_dict(f: Finding, today: date, fail_within: int) -> Dict:
|
|
68
|
+
d = f.deprecation
|
|
69
|
+
return {
|
|
70
|
+
"path": f.path,
|
|
71
|
+
"line": f.line,
|
|
72
|
+
"column": f.column,
|
|
73
|
+
"match": f.text,
|
|
74
|
+
"provider": d.provider,
|
|
75
|
+
"model_id": d.model_id,
|
|
76
|
+
"status": f.status(today),
|
|
77
|
+
"severity": severity(f, today, fail_within),
|
|
78
|
+
"shutdown_date": d.shutdown_date.isoformat() if d.shutdown_date else None,
|
|
79
|
+
"deprecation_date": d.deprecation_date.isoformat() if d.deprecation_date else None,
|
|
80
|
+
"days_left": d.days_left(today),
|
|
81
|
+
"replacements": list(d.replacements),
|
|
82
|
+
"url": d.url,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def render_json(findings: List[Finding], today: date, fail_within: int, source: str) -> str:
|
|
87
|
+
return json.dumps(
|
|
88
|
+
{
|
|
89
|
+
"tool": "llm-sunset",
|
|
90
|
+
"version": __version__,
|
|
91
|
+
"date": today.isoformat(),
|
|
92
|
+
"data_source": source,
|
|
93
|
+
"findings": [_as_dict(f, today, fail_within) for f in findings],
|
|
94
|
+
},
|
|
95
|
+
indent=2,
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def render_markdown(findings: List[Finding], today: date, fail_within: int, source: str) -> str:
|
|
100
|
+
if not findings:
|
|
101
|
+
return "### 🌅 llm-sunset\n\n✅ No deprecated AI model IDs found.\n"
|
|
102
|
+
out = [
|
|
103
|
+
"### 🌅 llm-sunset: deprecated AI models found",
|
|
104
|
+
"",
|
|
105
|
+
"| | Location | Model | Provider | Shutdown | Replace with |",
|
|
106
|
+
"|---|---|---|---|---|---|",
|
|
107
|
+
]
|
|
108
|
+
for f in findings:
|
|
109
|
+
d = f.deprecation
|
|
110
|
+
left = d.days_left(today)
|
|
111
|
+
if left is None:
|
|
112
|
+
when = "deprecated"
|
|
113
|
+
elif left <= 0:
|
|
114
|
+
when = f"**retired** {d.shutdown_date}"
|
|
115
|
+
else:
|
|
116
|
+
when = f"{d.shutdown_date} ({left}d)"
|
|
117
|
+
icon = "🔴" if severity(f, today, fail_within) == "error" else "🟡"
|
|
118
|
+
model = f"[`{d.model_id}`]({d.url})" if d.url else f"`{d.model_id}`"
|
|
119
|
+
repl = ", ".join(f"`{r}`" for r in d.replacements) or "—"
|
|
120
|
+
out.append(f"| {icon} | `{f.path}:{f.line}` | {model} | {d.provider} | {when} | {repl} |")
|
|
121
|
+
out += ["", f"<sub>Data: [deprecations.info](https://deprecations.info) ({source}) · llm-sunset {__version__}</sub>", ""]
|
|
122
|
+
return "\n".join(out)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def render_github(findings: List[Finding], today: date, fail_within: int) -> str:
|
|
126
|
+
def esc(s: str) -> str:
|
|
127
|
+
return s.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
|
|
128
|
+
|
|
129
|
+
lines = []
|
|
130
|
+
for f in findings:
|
|
131
|
+
level = severity(f, today, fail_within)
|
|
132
|
+
title = f"Deprecated model: {f.deprecation.model_id}"
|
|
133
|
+
lines.append(
|
|
134
|
+
f"::{level} file={esc(f.path)},line={f.line},col={f.column},title={esc(title)}::{esc(describe(f, today))}"
|
|
135
|
+
)
|
|
136
|
+
return "\n".join(lines)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _uri(path: str) -> str:
|
|
140
|
+
p = path.replace(os.sep, "/")
|
|
141
|
+
while p.startswith("./"):
|
|
142
|
+
p = p[2:]
|
|
143
|
+
return p
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def render_sarif(findings: List[Finding], today: date, fail_within: int) -> str:
|
|
147
|
+
rules = {}
|
|
148
|
+
results = []
|
|
149
|
+
for f in findings:
|
|
150
|
+
d = f.deprecation
|
|
151
|
+
rule_id = f"{d.provider}/{d.model_id}"
|
|
152
|
+
if rule_id not in rules:
|
|
153
|
+
rules[rule_id] = {
|
|
154
|
+
"id": rule_id,
|
|
155
|
+
"name": "DeprecatedAIModel",
|
|
156
|
+
"shortDescription": {"text": f"Deprecated AI model {d.model_id} ({d.provider})"},
|
|
157
|
+
"helpUri": d.url or "https://deprecations.info",
|
|
158
|
+
}
|
|
159
|
+
results.append(
|
|
160
|
+
{
|
|
161
|
+
"ruleId": rule_id,
|
|
162
|
+
"level": severity(f, today, fail_within),
|
|
163
|
+
"message": {"text": describe(f, today)},
|
|
164
|
+
"locations": [
|
|
165
|
+
{
|
|
166
|
+
"physicalLocation": {
|
|
167
|
+
"artifactLocation": {"uri": _uri(f.path)},
|
|
168
|
+
"region": {"startLine": f.line, "startColumn": f.column},
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
],
|
|
172
|
+
}
|
|
173
|
+
)
|
|
174
|
+
sarif = {
|
|
175
|
+
"$schema": "https://json.schemastore.org/sarif-2.1.0.json",
|
|
176
|
+
"version": "2.1.0",
|
|
177
|
+
"runs": [
|
|
178
|
+
{
|
|
179
|
+
"tool": {
|
|
180
|
+
"driver": {
|
|
181
|
+
"name": "llm-sunset",
|
|
182
|
+
"version": __version__,
|
|
183
|
+
"informationUri": "https://github.com/Ashveil1/llm-sunset",
|
|
184
|
+
"rules": list(rules.values()),
|
|
185
|
+
}
|
|
186
|
+
},
|
|
187
|
+
"results": results,
|
|
188
|
+
}
|
|
189
|
+
],
|
|
190
|
+
}
|
|
191
|
+
return json.dumps(sarif, indent=2)
|
llm_sunset/scanner.py
ADDED
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
"""Find references to deprecated AI model IDs in source files."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import bisect
|
|
6
|
+
import fnmatch
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from datetime import date
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Dict, Iterable, Iterator, List, Optional, Sequence
|
|
13
|
+
|
|
14
|
+
from .data import Deprecation
|
|
15
|
+
|
|
16
|
+
DEFAULT_EXCLUDE_DIRS = {
|
|
17
|
+
".git", ".hg", ".svn", "node_modules", "bower_components", "vendor",
|
|
18
|
+
".venv", "venv", "env", ".env.d", "__pycache__", ".mypy_cache", ".pytest_cache",
|
|
19
|
+
".ruff_cache", ".tox", ".nox", "dist", "build", "out", "target", ".next",
|
|
20
|
+
".nuxt", ".svelte-kit", ".turbo", ".cache", "coverage", ".idea", ".vscode",
|
|
21
|
+
"site-packages",
|
|
22
|
+
}
|
|
23
|
+
DEFAULT_EXCLUDE_GLOBS = [
|
|
24
|
+
"*.lock", "package-lock.json", "pnpm-lock.yaml", "yarn.lock", "poetry.lock",
|
|
25
|
+
"uv.lock", "Cargo.lock", "go.sum", "*.min.js", "*.map", "*.svg",
|
|
26
|
+
"*.png", "*.jpg", "*.jpeg", "*.gif", "*.webp", "*.ico", "*.pdf", "*.zip",
|
|
27
|
+
"*.gz", "*.tar", "*.woff", "*.woff2", "*.ttf", "*.mp3", "*.mp4", "*.wasm",
|
|
28
|
+
"*.so", "*.dylib", "*.dll", "*.exe", "*.bin", "*.pyc", "*.ipynb_checkpoints",
|
|
29
|
+
]
|
|
30
|
+
# Prose files mention old models all the time ("we migrated from gpt-3.5-turbo").
|
|
31
|
+
DOC_GLOBS = ["*.md", "*.mdx", "*.rst", "*.txt", "*.adoc", "CHANGELOG*", "HISTORY*"]
|
|
32
|
+
|
|
33
|
+
MAX_FILE_BYTES = 2 * 1024 * 1024
|
|
34
|
+
IGNORE_LINE = "llm-sunset: ignore"
|
|
35
|
+
IGNORE_FILE = "llm-sunset: ignore-file"
|
|
36
|
+
|
|
37
|
+
# Characters that may legally sit right before / after a model ID. "/" ":" and
|
|
38
|
+
# "@" are allowed so that "openai/gpt-4", "ft:gpt-4o:org::id" and
|
|
39
|
+
# "claude-3-5-sonnet@20240620" still match.
|
|
40
|
+
_LEFT = r"(?<![A-Za-z0-9._+-])"
|
|
41
|
+
_RIGHT = r"(?![A-Za-z0-9_+-])(?!\.[A-Za-z0-9])"
|
|
42
|
+
|
|
43
|
+
# Same character classes as _LEFT/_RIGHT; a token never ends with ".".
|
|
44
|
+
_TOKEN_RE = re.compile(r"[A-Za-z0-9_+-](?:[A-Za-z0-9._+-]*[A-Za-z0-9_+-])?")
|
|
45
|
+
|
|
46
|
+
# Platforms that resell models under their own (often very different) lifecycle
|
|
47
|
+
# dates. Off by default so OpenAI/Anthropic users don't get Azure dates.
|
|
48
|
+
PLATFORM_PROVIDERS = ("azure", "google vertex", "bedrock")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True)
|
|
52
|
+
class Finding:
|
|
53
|
+
path: str
|
|
54
|
+
line: int
|
|
55
|
+
column: int
|
|
56
|
+
text: str # exact text matched in the file
|
|
57
|
+
deprecation: Deprecation
|
|
58
|
+
|
|
59
|
+
def status(self, today: date) -> str:
|
|
60
|
+
left = self.deprecation.days_left(today)
|
|
61
|
+
if left is None:
|
|
62
|
+
return "deprecated"
|
|
63
|
+
return "retired" if left <= 0 else "retiring"
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class Matcher:
|
|
67
|
+
def __init__(self, deprecations: Sequence[Deprecation], providers: Optional[Sequence[str]] = None):
|
|
68
|
+
wanted = [p.lower() for p in providers] if providers else []
|
|
69
|
+
everything = "all" in wanted
|
|
70
|
+
self.by_id: Dict[str, List[Deprecation]] = {}
|
|
71
|
+
for d in deprecations:
|
|
72
|
+
prov = d.provider.lower()
|
|
73
|
+
if everything:
|
|
74
|
+
pass
|
|
75
|
+
elif wanted:
|
|
76
|
+
if not any(w in prov for w in wanted):
|
|
77
|
+
continue
|
|
78
|
+
elif any(p in prov for p in PLATFORM_PROVIDERS):
|
|
79
|
+
continue
|
|
80
|
+
self.by_id.setdefault(d.model_id, []).append(d)
|
|
81
|
+
|
|
82
|
+
self._ids = frozenset(self.by_id)
|
|
83
|
+
# IDs like "babbage", "davinci" or "command" are ordinary words; only
|
|
84
|
+
# match them when they are the entire contents of a string literal.
|
|
85
|
+
self._words = frozenset(i for i in self.by_id if not any(c.isdigit() for c in i))
|
|
86
|
+
|
|
87
|
+
def find(self, text: str) -> Iterator[tuple]:
|
|
88
|
+
"""Yield (offset, matched_id, [Deprecation]) for every hit in text."""
|
|
89
|
+
# Fast path: tokenize + set intersection both run in C.
|
|
90
|
+
hits = self._ids.intersection(_TOKEN_RE.findall(text))
|
|
91
|
+
if not hits:
|
|
92
|
+
return
|
|
93
|
+
ids = sorted(hits - self._words, key=len, reverse=True)
|
|
94
|
+
words = sorted(hits & self._words, key=len, reverse=True)
|
|
95
|
+
if ids:
|
|
96
|
+
rx = re.compile(_LEFT + "(" + "|".join(map(re.escape, ids)) + ")" + _RIGHT)
|
|
97
|
+
for m in rx.finditer(text):
|
|
98
|
+
yield m.start(1), m.group(1), self.by_id[m.group(1)]
|
|
99
|
+
if words:
|
|
100
|
+
rx = re.compile(r"""(["'`])(""" + "|".join(map(re.escape, words)) + r")\1")
|
|
101
|
+
for m in rx.finditer(text):
|
|
102
|
+
yield m.start(2), m.group(2), self.by_id[m.group(2)]
|
|
103
|
+
|
|
104
|
+
def match_line(self, line: str) -> Iterator[tuple]:
|
|
105
|
+
return self.find(line)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _compile_globs(globs: Iterable[str]):
|
|
109
|
+
globs = list(globs)
|
|
110
|
+
if not globs:
|
|
111
|
+
return lambda rel, name: False
|
|
112
|
+
rx = re.compile("|".join(fnmatch.translate(g) for g in globs))
|
|
113
|
+
return lambda rel, name: bool(rx.match(name) or rx.match(rel.replace(os.sep, "/")))
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def iter_files(paths: Sequence[str], exclude: Sequence[str], include_docs: bool) -> Iterator[Path]:
|
|
117
|
+
globs = list(DEFAULT_EXCLUDE_GLOBS) + list(exclude)
|
|
118
|
+
if not include_docs:
|
|
119
|
+
globs += DOC_GLOBS
|
|
120
|
+
file_excluded = _compile_globs(globs)
|
|
121
|
+
dir_excluded = _compile_globs(exclude)
|
|
122
|
+
for p in paths:
|
|
123
|
+
root = Path(p)
|
|
124
|
+
if root.is_file():
|
|
125
|
+
yield root # explicitly named files are always scanned
|
|
126
|
+
continue
|
|
127
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
128
|
+
rel_dir = os.path.relpath(dirpath, root)
|
|
129
|
+
dirnames[:] = [
|
|
130
|
+
d for d in dirnames
|
|
131
|
+
if d not in DEFAULT_EXCLUDE_DIRS
|
|
132
|
+
and not dir_excluded(os.path.normpath(os.path.join(rel_dir, d)), d)
|
|
133
|
+
]
|
|
134
|
+
for f in filenames:
|
|
135
|
+
if not file_excluded(os.path.normpath(os.path.join(rel_dir, f)), f):
|
|
136
|
+
yield Path(dirpath) / f
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def scan_file(path: Path, matcher: Matcher) -> List[Finding]:
|
|
140
|
+
try:
|
|
141
|
+
with open(path, "rb") as fh:
|
|
142
|
+
raw = fh.read(MAX_FILE_BYTES + 1)
|
|
143
|
+
except OSError:
|
|
144
|
+
return []
|
|
145
|
+
if len(raw) > MAX_FILE_BYTES:
|
|
146
|
+
return []
|
|
147
|
+
if b"\0" in raw[:8192]:
|
|
148
|
+
return []
|
|
149
|
+
text = raw.decode("utf-8", errors="replace")
|
|
150
|
+
if IGNORE_FILE in text:
|
|
151
|
+
return []
|
|
152
|
+
findings = []
|
|
153
|
+
line_starts = None
|
|
154
|
+
for offset, matched, deps in matcher.find(text):
|
|
155
|
+
if line_starts is None:
|
|
156
|
+
line_starts = [0] + [m.end() for m in re.finditer(r"\n", text)]
|
|
157
|
+
idx = bisect.bisect_right(line_starts, offset) - 1
|
|
158
|
+
start = line_starts[idx]
|
|
159
|
+
end = text.find("\n", start)
|
|
160
|
+
line = text[start:] if end == -1 else text[start:end]
|
|
161
|
+
if IGNORE_LINE in line:
|
|
162
|
+
continue
|
|
163
|
+
for d in deps:
|
|
164
|
+
findings.append(Finding(str(path), idx + 1, offset - start + 1, matched, d))
|
|
165
|
+
return findings
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def scan(paths: Sequence[str], matcher: Matcher, exclude: Sequence[str] = (), include_docs: bool = False) -> List[Finding]:
|
|
169
|
+
out: List[Finding] = []
|
|
170
|
+
for f in iter_files(paths, exclude, include_docs):
|
|
171
|
+
out.extend(scan_file(f, matcher))
|
|
172
|
+
out.sort(key=lambda x: (x.path, x.line, x.column, x.deprecation.provider))
|
|
173
|
+
return out
|