github-to-sqlite 2.8.3__py3-none-any.whl → 2.9.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
github_to_sqlite/cli.py CHANGED
@@ -1,5 +1,6 @@
1
1
  import click
2
2
  import datetime
3
+ import itertools
3
4
  import pathlib
4
5
  import textwrap
5
6
  import os
@@ -104,19 +105,54 @@ def issues(db_path, repo, issue_ids, auth, load):
104
105
  type=click.Path(file_okay=True, dir_okay=False, allow_dash=True, exists=True),
105
106
  help="Load pull-requests JSON from this file instead of the API",
106
107
  )
107
- def pull_requests(db_path, repo, pull_request_ids, auth, load):
108
+ @click.option(
109
+ "--org",
110
+ "orgs",
111
+ help="Fetch all pull requests from this GitHub organization",
112
+ multiple=True,
113
+ )
114
+ @click.option(
115
+ "--state",
116
+ help="Only fetch pull requests in this state",
117
+ )
118
+ @click.option(
119
+ "--search",
120
+ help="Find pull requests with a search query",
121
+ )
122
+ def pull_requests(db_path, repo, pull_request_ids, auth, load, orgs, state, search):
108
123
  "Save pull_requests for a specified repository, e.g. simonw/datasette"
109
124
  db = sqlite_utils.Database(db_path)
110
125
  token = load_token(auth)
111
- repo_full = utils.fetch_repo(repo, token)
112
- utils.save_repo(db, repo_full)
113
126
  if load:
127
+ repo_full = utils.fetch_repo(repo, token)
128
+ utils.save_repo(db, repo_full)
114
129
  pull_requests = json.load(open(load))
130
+ utils.save_pull_requests(db, pull_requests, repo_full)
131
+ elif search:
132
+ repos_seen = set()
133
+ search += " is:pr"
134
+ pull_requests = utils.fetch_searched_pulls_or_issues(search, token)
135
+ for pull_request in pull_requests:
136
+ pr_repo_url = pull_request["repository_url"]
137
+ if pr_repo_url not in repos_seen:
138
+ pr_repo = utils.fetch_repo(url=pr_repo_url)
139
+ utils.save_repo(db, pr_repo)
140
+ repos_seen.add(pr_repo_url)
141
+ utils.save_pull_requests(db, [pull_request], pr_repo)
115
142
  else:
116
- pull_requests = utils.fetch_pull_requests(repo, token, pull_request_ids)
117
-
118
- pull_requests = list(pull_requests)
119
- utils.save_pull_requests(db, pull_requests, repo_full)
143
+ if orgs:
144
+ repos = itertools.chain.from_iterable(
145
+ utils.fetch_all_repos(token=token, org=org) for org in orgs
146
+ )
147
+ else:
148
+ repos = [utils.fetch_repo(repo, token)]
149
+ for repo_full in repos:
150
+ utils.save_repo(db, repo_full)
151
+ repo = repo_full["full_name"]
152
+ pull_requests = utils.fetch_pull_requests(
153
+ repo, state, token, pull_request_ids
154
+ )
155
+ utils.save_pull_requests(db, pull_requests, repo_full)
120
156
  utils.ensure_db_shape(db)
121
157
 
122
158
 
@@ -464,7 +500,9 @@ def scrape_dependents(db_path, repos, auth, verbose):
464
500
  {
465
501
  "repo": repo_full["id"],
466
502
  "dependent": dependent_id,
467
- "first_seen_utc": datetime.datetime.utcnow().isoformat(),
503
+ "first_seen_utc": datetime.datetime.now(datetime.timezone.utc)
504
+ .replace(tzinfo=None)
505
+ .isoformat(),
468
506
  },
469
507
  pk=("repo", "dependent"),
470
508
  foreign_keys=(
github_to_sqlite/utils.py CHANGED
@@ -2,6 +2,7 @@ import base64
2
2
  import requests
3
3
  import re
4
4
  import time
5
+ import urllib.parse
5
6
  import yaml
6
7
 
7
8
  FTS_CONFIG = {
@@ -74,16 +75,17 @@ FOREIGN_KEYS = [
74
75
 
75
76
 
76
77
  class GitHubError(Exception):
77
- def __init__(self, message, status_code):
78
+ def __init__(self, message, status_code, headers=None):
78
79
  self.message = message
79
80
  self.status_code = status_code
81
+ self.headers = headers
80
82
 
81
83
  @classmethod
82
84
  def from_response(cls, response):
83
85
  message = response.json()["message"]
84
86
  if "git repository is empty" in message.lower():
85
87
  cls = GitHubRepositoryEmpty
86
- return cls(message, response.status_code)
88
+ return cls(message, response.status_code, response.headers)
87
89
 
88
90
 
89
91
  class GitHubRepositoryEmpty(GitHubError):
@@ -169,8 +171,11 @@ def save_pull_requests(db, pull_requests, repo):
169
171
  # Add repo key
170
172
  pull_request["repo"] = repo["id"]
171
173
  # Pull request _links can be flattened to just their URL
172
- pull_request["url"] = pull_request["_links"]["html"]["href"]
173
- pull_request.pop("_links")
174
+ if "_links" in pull_request:
175
+ pull_request["url"] = pull_request["_links"]["html"]["href"]
176
+ pull_request.pop("_links")
177
+ else:
178
+ pull_request["url"] = pull_request["pull_request"]["html_url"]
174
179
  # Extract user
175
180
  pull_request["user"] = save_user(db, pull_request["user"])
176
181
  labels = pull_request.pop("labels")
@@ -178,8 +183,9 @@ def save_pull_requests(db, pull_requests, repo):
178
183
  if pull_request.get("merged_by"):
179
184
  pull_request["merged_by"] = save_user(db, pull_request["merged_by"])
180
185
  # Head sha
181
- pull_request["head"] = pull_request["head"]["sha"]
182
- pull_request["base"] = pull_request["base"]["sha"]
186
+ if "head" in pull_request:
187
+ pull_request["head"] = pull_request["head"]["sha"]
188
+ pull_request["base"] = pull_request["base"]["sha"]
183
189
  # Extract milestone
184
190
  if pull_request["milestone"]:
185
191
  pull_request["milestone"] = save_milestone(
@@ -223,6 +229,11 @@ def save_pull_requests(db, pull_requests, repo):
223
229
 
224
230
 
225
231
  def save_user(db, user):
232
+ # Under some conditions, GitHub caches removed repositories with
233
+ # stars and ends up leaving dangling `None` user references.
234
+ if user is None:
235
+ return None
236
+
226
237
  # Remove all url fields except avatar_url and html_url
227
238
  to_save = {
228
239
  key: value
@@ -286,12 +297,13 @@ def save_issue_comment(db, comment):
286
297
  return last_pk
287
298
 
288
299
 
289
- def fetch_repo(full_name, token=None):
300
+ def fetch_repo(full_name=None, token=None, url=None):
290
301
  headers = make_headers(token)
291
302
  # Get topics:
292
303
  headers["Accept"] = "application/vnd.github.mercy-preview+json"
293
- owner, slug = full_name.split("/")
294
- url = "https://api.github.com/repos/{}/{}".format(owner, slug)
304
+ if url is None:
305
+ owner, slug = full_name.split("/")
306
+ url = "https://api.github.com/repos/{}/{}".format(owner, slug)
295
307
  response = requests.get(url, headers=headers)
296
308
  response.raise_for_status()
297
309
  return response.json()
@@ -352,7 +364,7 @@ def fetch_issues(repo, token=None, issue_ids=None):
352
364
  yield from issues
353
365
 
354
366
 
355
- def fetch_pull_requests(repo, token=None, pull_request_ids=None):
367
+ def fetch_pull_requests(repo, state=None, token=None, pull_request_ids=None):
356
368
  headers = make_headers(token)
357
369
  headers["accept"] = "application/vnd.github.v3+json"
358
370
  if pull_request_ids:
@@ -364,11 +376,20 @@ def fetch_pull_requests(repo, token=None, pull_request_ids=None):
364
376
  response.raise_for_status()
365
377
  yield response.json()
366
378
  else:
367
- url = "https://api.github.com/repos/{}/pulls?state=all&filter=all".format(repo)
379
+ state = state or "all"
380
+ url = f"https://api.github.com/repos/{repo}/pulls?state={state}"
368
381
  for pull_requests in paginate(url, headers):
369
382
  yield from pull_requests
370
383
 
371
384
 
385
+ def fetch_searched_pulls_or_issues(query, token=None):
386
+ headers = make_headers(token)
387
+ url = "https://api.github.com/search/issues?"
388
+ url += urllib.parse.urlencode({"q": query})
389
+ for pulls_or_issues in paginate(url, headers):
390
+ yield from pulls_or_issues["items"]
391
+
392
+
372
393
  def fetch_issue_comments(repo, token=None, issue=None):
373
394
  assert "/" in repo
374
395
  headers = make_headers(token)
@@ -439,13 +460,17 @@ def fetch_stargazers(repo, token=None):
439
460
  yield from stargazers
440
461
 
441
462
 
442
- def fetch_all_repos(username=None, token=None):
443
- assert username or token, "Must provide username= or token= or both"
463
+ def fetch_all_repos(username=None, token=None, org=None):
464
+ assert (
465
+ username or token or org
466
+ ), "Must provide username= or token= or org= or a combination"
444
467
  headers = make_headers(token)
445
468
  # Get topics for each repo:
446
469
  headers["Accept"] = "application/vnd.github.mercy-preview+json"
447
470
  if username:
448
471
  url = "https://api.github.com/users/{}/repos".format(username)
472
+ elif org:
473
+ url = "https://api.github.com/orgs/{}/repos".format(org)
449
474
  else:
450
475
  url = "https://api.github.com/user/repos"
451
476
  for repos in paginate(url, headers):
@@ -463,6 +488,7 @@ def fetch_user(username=None, token=None):
463
488
 
464
489
 
465
490
  def paginate(url, headers=None):
491
+ url += ("&" if "?" in url else "?") + "per_page=100"
466
492
  while url:
467
493
  response = requests.get(url, headers=headers)
468
494
  # For HTTP 204 no-content this yields an empty list
@@ -671,7 +697,10 @@ def ensure_foreign_keys(db):
671
697
  for expected_foreign_key in FOREIGN_KEYS:
672
698
  table, column, table2, column2 = expected_foreign_key
673
699
  if (
674
- expected_foreign_key not in db[table].foreign_keys
700
+ expected_foreign_key not in {
701
+ (fk.table, fk.column, fk.other_table, fk.other_column)
702
+ for fk in db[table].foreign_keys
703
+ }
675
704
  and
676
705
  # Ensure all tables and columns exist
677
706
  db[table].exists()
@@ -726,7 +755,7 @@ def scrape_dependents(repo, verbose=False):
726
755
  yield from repos
727
756
  # next page?
728
757
  try:
729
- next_link = soup.select(".paginate-container")[0].find("a", text="Next")
758
+ next_link = soup.select(".paginate-container")[0].find("a", string="Next")
730
759
  except IndexError:
731
760
  break
732
761
  if next_link is not None:
@@ -1,20 +1,19 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: github-to-sqlite
3
- Version: 2.8.3
3
+ Version: 2.9.1
4
4
  Summary: Save data from GitHub to a SQLite database
5
- Home-page: https://github.com/dogsheep/github-to-sqlite
6
5
  Author: Simon Willison
7
- License: Apache License, Version 2.0
8
- Platform: UNKNOWN
9
- Description-Content-Type: text/markdown
6
+ License-Expression: Apache-2.0
10
7
  License-File: LICENSE
11
- Requires-Dist: sqlite-utils (>=2.7.2)
8
+ Requires-Dist: sqlite-utils>4
12
9
  Requires-Dist: requests
13
- Requires-Dist: PyYAML
14
- Provides-Extra: test
15
- Requires-Dist: pytest ; extra == 'test'
16
- Requires-Dist: requests-mock ; extra == 'test'
17
- Requires-Dist: bs4 ; extra == 'test'
10
+ Requires-Dist: pyyaml
11
+ Requires-Python: >=3.10
12
+ Project-URL: Homepage, https://github.com/dogsheep/github-to-sqlite
13
+ Project-URL: Changelog, https://github.com/dogsheep/github-to-sqlite/releases
14
+ Project-URL: Issues, https://github.com/dogsheep/github-to-sqlite/issues
15
+ Project-URL: CI, https://github.com/dogsheep/github-to-sqlite/actions
16
+ Description-Content-Type: text/markdown
18
17
 
19
18
  # github-to-sqlite
20
19
 
@@ -100,13 +99,25 @@ You can use the `--pull-request` option one or more times to load specific pull
100
99
 
101
100
  Note that the `merged_by` column on the `pull_requests` table will only be populated for pull requests that are loaded using the `--pull-request` option - the GitHub API does not return this field for pull requests that are loaded in bulk.
102
101
 
102
+ You can load only pull requests in a certain state with the `--state` option:
103
+
104
+ $ github-to-sqlite pull-requests --state=open github.db simonw/datasette
105
+
106
+ Pull requests across an entire organization (or more than one) can be loaded with `--org`:
107
+
108
+ $ github-to-sqlite pull-requests --state=open --org=psf --org=python github.db
109
+
110
+ You can use a search query to find pull requests. Note that no more than 1000 will be loaded (this is a GitHub API limitation), and some data will be missing (base and head SHAs). When using searches, other filters are ignored; put all criteria into the search itself:
111
+
112
+ $ github-to-sqlite pull-requests --search='org:python defaultdict state:closed created:<2023-09-01' github.db
113
+
103
114
  Example: [pull_requests table](https://github-to-sqlite.dogsheep.net/github/pull_requests)
104
115
 
105
116
  ## Fetching issue comments for a repository
106
117
 
107
118
  The `issue-comments` command retrieves all of the comments on all of the issues in a repository.
108
119
 
109
- It is recommended you run `issues` first, so that each imported comment can have a foreign key poining to its issue.
120
+ It is recommended you run `issues` first, so that each imported comment can have a foreign key pointing to its issue.
110
121
 
111
122
  $ github-to-sqlite issues github.db simonw/datasette
112
123
  $ github-to-sqlite issue-comments github.db simonw/datasette
@@ -119,7 +130,7 @@ Example: [issue_comments table](https://github-to-sqlite.dogsheep.net/github/iss
119
130
 
120
131
  ## Fetching commits for a repository
121
132
 
122
- The `commits` command retrieves details of all of the commits for one or more repositories. It currently fetches the sha, commit message and author and committer details - it does no retrieve the full commit body.
133
+ The `commits` command retrieves details of all of the commits for one or more repositories. It currently fetches the SHA, commit message and author and committer details; it does not retrieve the full commit body.
123
134
 
124
135
  $ github-to-sqlite commits github.db simonw/datasette simonw/sqlite-utils
125
136
 
@@ -174,7 +185,7 @@ You can pass more than one username to fetch for multiple users or organizations
174
185
 
175
186
  $ github-to-sqlite repos github.db simonw dogsheep
176
187
 
177
- Add the `--readme` option to save the README for the repo in a column called `readme`. Add `--readme-html` to save the HTML rendered version of the README into a collumn called `readme_html`.
188
+ Add the `--readme` option to save the README for the repo in a column called `readme`. Add `--readme-html` to save the HTML rendered version of the README into a column called `readme_html`.
178
189
 
179
190
  Example: [repos table](https://github-to-sqlite.dogsheep.net/github/repos)
180
191
 
@@ -234,7 +245,7 @@ You can fetch a list of every emoji supported by GitHub using the `emojis` comma
234
245
 
235
246
  $ github-to-sqlite emojis github.db
236
247
 
237
- This will create a table callad `emojis` with a primary key `name` and a `url` column.
248
+ This will create a table called `emojis` with a primary key `name` and a `url` column.
238
249
 
239
250
  If you add the `--fetch` option the command will also fetch the binary content of the images and place them in an `image` column:
240
251
 
@@ -253,7 +264,7 @@ The `github-to-sqlite get` command provides a convenient shortcut for making aut
253
264
 
254
265
  This will make an authenticated call to the URL you provide and pretty-print the resulting JSON to the console.
255
266
 
256
- You can ommit the `https://api.github.com/` prefix, for example:
267
+ You can omit the `https://api.github.com/` prefix, for example:
257
268
 
258
269
  $ github-to-sqlite get /gists
259
270
 
@@ -264,5 +275,3 @@ Many GitHub APIs are [paginated using the HTTP Link header](https://docs.github.
264
275
  You can outline newline-delimited JSON for each item using `--nl`. This can be useful for streaming items into another tool.
265
276
 
266
277
  $ github-to-sqlite get /users/simonw/repos --nl
267
-
268
-
@@ -0,0 +1,8 @@
1
+ github_to_sqlite/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
+ github_to_sqlite/cli.py,sha256=Qk6AFMgW2hYuJOASZ5wP2_mQEwl04RmKS25C143AWVI,19505
3
+ github_to_sqlite/utils.py,sha256=nqixL6NNvK1kiOkkjGwvSwcQNw-3mormqp68lLXNmSo,29321
4
+ github_to_sqlite-2.9.1.dist-info/licenses/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
5
+ github_to_sqlite-2.9.1.dist-info/WHEEL,sha256=-i9oRNYVXXZJUIYl5zclLIg6onEb0NLibTX34uln84w,81
6
+ github_to_sqlite-2.9.1.dist-info/entry_points.txt,sha256=zr29qE6GkeNBMRB4dBewDVokC920g7K_5r6q5mtcctM,63
7
+ github_to_sqlite-2.9.1.dist-info/METADATA,sha256=tlLiKD_UoTlqmLXg5Y7wcGvEVvNbgQYCkOBY72l2LHA,13443
8
+ github_to_sqlite-2.9.1.dist-info/RECORD,,
@@ -1,5 +1,4 @@
1
1
  Wheel-Version: 1.0
2
- Generator: bdist_wheel (0.37.0)
2
+ Generator: uv 0.12.13
3
3
  Root-Is-Purelib: true
4
4
  Tag: py3-none-any
5
-
@@ -0,0 +1,3 @@
1
+ [console_scripts]
2
+ github-to-sqlite = github_to_sqlite.cli:cli
3
+
@@ -1,9 +0,0 @@
1
- github_to_sqlite/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
2
- github_to_sqlite/cli.py,sha256=uOk9fOZQfNGKZ0mmdhOMKGbNbc5TutJqLEdAhWW3EWM,18156
3
- github_to_sqlite/utils.py,sha256=XyBCNlnePY3k6MALRacaAUAvKjaBPAm7JfbjJRYO3WY,28229
4
- github_to_sqlite-2.8.3.dist-info/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
5
- github_to_sqlite-2.8.3.dist-info/METADATA,sha256=vkvbH9E0BA9D8MK1ZF7pIrWCZkvV5iUtofqGBV-YD4k,12646
6
- github_to_sqlite-2.8.3.dist-info/WHEEL,sha256=ewwEueio1C2XeHTvT17n8dZUJgOvyCWCt0WVNLClP9o,92
7
- github_to_sqlite-2.8.3.dist-info/entry_points.txt,sha256=XEmCvWXu8ObFt_ZKeuXIqJo_feZ0tDDrugFvhG8RYyU,81
8
- github_to_sqlite-2.8.3.dist-info/top_level.txt,sha256=bur6DmbPzE5A4FKDQpi9AkSa-AIjAGKdRor_kyLlllA,17
9
- github_to_sqlite-2.8.3.dist-info/RECORD,,
@@ -1,4 +0,0 @@
1
-
2
- [console_scripts]
3
- github-to-sqlite=github_to_sqlite.cli:cli
4
-
@@ -1 +0,0 @@
1
- github_to_sqlite