airbyte-source-github 2.6.0__py3-none-any.whl → 2.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/METADATA +2 -3
- airbyte_source_github-2.7.0.dist-info/RECORD +11 -0
- source_github/components.py +558 -2
- source_github/manifest.yaml +9403 -695
- source_github/source.py +7 -37
- airbyte_source_github-2.6.0.dist-info/RECORD +0 -36
- source_github/backoff_strategies.py +0 -52
- source_github/errors_handlers.py +0 -223
- source_github/github_schema.py +0 -41035
- source_github/graphql.py +0 -375
- source_github/schemas/commit_comments.json +0 -68
- source_github/schemas/issue_reactions.json +0 -35
- source_github/schemas/issue_timeline_events.json +0 -1197
- source_github/schemas/projects.json +0 -64
- source_github/schemas/projects_v2.json +0 -103
- source_github/schemas/pull_request_comment_reactions.json +0 -35
- source_github/schemas/pull_request_stats.json +0 -105
- source_github/schemas/pull_requests.json +0 -432
- source_github/schemas/releases.json +0 -153
- source_github/schemas/reviews.json +0 -87
- source_github/schemas/shared/events/comment.json +0 -188
- source_github/schemas/shared/events/commented.json +0 -118
- source_github/schemas/shared/events/committed.json +0 -56
- source_github/schemas/shared/events/cross_referenced.json +0 -822
- source_github/schemas/shared/events/reviewed.json +0 -139
- source_github/schemas/shared/reaction.json +0 -27
- source_github/schemas/shared/reactions.json +0 -35
- source_github/schemas/shared/user.json +0 -59
- source_github/schemas/shared/user_graphql.json +0 -26
- source_github/streams.py +0 -909
- source_github/utils.py +0 -24
- {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/WHEEL +0 -0
- {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/entry_points.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: airbyte-source-github
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.7.0
|
|
4
4
|
Summary: Source implementation for GitHub.
|
|
5
5
|
Home-page: https://airbyte.com
|
|
6
6
|
License: ELv2
|
|
@@ -13,8 +13,7 @@ Classifier: Programming Language :: Python :: 3.10
|
|
|
13
13
|
Classifier: Programming Language :: Python :: 3.11
|
|
14
14
|
Classifier: Programming Language :: Python :: 3.12
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
-
Requires-Dist: airbyte-cdk (>=7.
|
|
17
|
-
Requires-Dist: sgqlc (==16.3)
|
|
16
|
+
Requires-Dist: airbyte-cdk (>=7.30.0,<8.0.0)
|
|
18
17
|
Project-URL: Documentation, https://docs.airbyte.com/integrations/sources/github
|
|
19
18
|
Project-URL: Repository, https://github.com/airbytehq/airbyte
|
|
20
19
|
Description-Content-Type: text/markdown
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
source_github/__init__.py,sha256=8hLLjFC1wI1mMU442Y5qCcND9qIwyhaANQwT-GAIK0s,1135
|
|
2
|
+
source_github/components.py,sha256=iqhaoVq5Sao1fD86GyvCyMyM8-QlNCQYelZFC7BZzmw,35433
|
|
3
|
+
source_github/config_migrations.py,sha256=guUJAdNP-liUciVaJB4ackEEMCN4jd7bNd85QFmDNlU,3932
|
|
4
|
+
source_github/constants.py,sha256=Hj3Q4y7OoU-Iff4m9gEC2CjwmWJYXhNbHVNjg8EBLmQ,238
|
|
5
|
+
source_github/manifest.yaml,sha256=tohJ8Sx_yOPKeVeRCNMwGd-TD3IOib7zDpLf2QT_M0k,668630
|
|
6
|
+
source_github/run.py,sha256=gfWy8TcMHa0XRwaVUyq4KmslooST38fvfCLuP5D-5Fo,2391
|
|
7
|
+
source_github/source.py,sha256=5VvGvnuj2FhvtHkHDzjiGZIBfPgSg-LMqyw5liGGdP4,18139
|
|
8
|
+
airbyte_source_github-2.7.0.dist-info/METADATA,sha256=EjShSrKSQFQ0d9phj_VZJUgRviOZO3WR3Lszu2C-iss,1218
|
|
9
|
+
airbyte_source_github-2.7.0.dist-info/WHEEL,sha256=Nq82e9rUAnEjt98J6MlVmMCZb-t9cYE2Ir1kpBmnWfs,88
|
|
10
|
+
airbyte_source_github-2.7.0.dist-info/entry_points.txt,sha256=gYhqVrTAZvMwuYByg0b_-o115yUFLLcfNxMrLZmiW9k,55
|
|
11
|
+
airbyte_source_github-2.7.0.dist-info/RECORD,,
|
source_github/components.py
CHANGED
|
@@ -2,19 +2,575 @@
|
|
|
2
2
|
# Copyright (c) 2023 Airbyte, Inc., all rights reserved.
|
|
3
3
|
#
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
"""Custom low-code components for source-github.
|
|
6
|
+
|
|
7
|
+
Everything here exists because the GitHub GraphQL responses cannot be reshaped into the
|
|
8
|
+
connector's long-standing REST-compatible record shape with declarative transformations
|
|
9
|
+
alone. Pagination, page-size reduction, error handling and incremental behavior are all
|
|
10
|
+
handled by the manifest.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import base64
|
|
14
|
+
import binascii
|
|
15
|
+
import logging
|
|
16
|
+
import struct
|
|
17
|
+
from dataclasses import InitVar, dataclass
|
|
6
18
|
from datetime import timedelta
|
|
7
19
|
from itertools import groupby
|
|
8
20
|
from typing import Any, Iterable, List, Mapping, MutableMapping, Optional
|
|
9
21
|
|
|
10
22
|
import requests
|
|
11
23
|
|
|
24
|
+
from airbyte_cdk.models import FailureType
|
|
12
25
|
from airbyte_cdk.sources.declarative.extractors.record_extractor import RecordExtractor
|
|
26
|
+
from airbyte_cdk.sources.declarative.interpolation.interpolated_string import InterpolatedString
|
|
13
27
|
from airbyte_cdk.sources.declarative.migrations.state_migration import StateMigration
|
|
14
28
|
from airbyte_cdk.sources.declarative.partition_routers.substream_partition_router import SubstreamPartitionRouter
|
|
15
29
|
from airbyte_cdk.sources.declarative.requesters.paginators.strategies.cursor_pagination_strategy import CursorPaginationStrategy
|
|
16
|
-
from airbyte_cdk.sources.
|
|
30
|
+
from airbyte_cdk.sources.declarative.requesters.paginators.strategies.pagination_strategy import (
|
|
31
|
+
PaginationStrategy,
|
|
32
|
+
)
|
|
33
|
+
from airbyte_cdk.sources.declarative.transformations import RecordTransformation
|
|
34
|
+
from airbyte_cdk.sources.types import Config, Record, StreamSlice, StreamState
|
|
17
35
|
from airbyte_cdk.utils.datetime_helpers import ab_datetime_parse
|
|
36
|
+
from airbyte_cdk.utils.traced_exception import AirbyteTracedException
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
LOGGER = logging.getLogger("airbyte")
|
|
40
|
+
|
|
41
|
+
# GitHub's GraphQL reaction content enum, mapped to the field names the REST API used and
|
|
42
|
+
# therefore to the names already present in the `releases` schema and in user warehouses.
|
|
43
|
+
GRAPHQL_REACTION_TO_REST = {
|
|
44
|
+
"THUMBS_UP": "plus_one",
|
|
45
|
+
"THUMBS_DOWN": "minus_one",
|
|
46
|
+
"LAUGH": "laugh",
|
|
47
|
+
"HOORAY": "hooray",
|
|
48
|
+
"CONFUSED": "confused",
|
|
49
|
+
"HEART": "heart",
|
|
50
|
+
"ROCKET": "rocket",
|
|
51
|
+
"EYES": "eyes",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _extract_database_id_from_node_id(node_id: Optional[str]) -> Optional[int]:
|
|
56
|
+
"""Extract the numeric database ID from a GitHub GraphQL Node ID.
|
|
57
|
+
|
|
58
|
+
GitHub Node IDs with type prefixes (e.g. 'RA_...') are URL-safe base64 encodings of a
|
|
59
|
+
msgpack array: [type_flag, repo_database_id, entity_database_id]. The last 4 bytes encode
|
|
60
|
+
the entity's numeric database ID as a big-endian uint32.
|
|
61
|
+
|
|
62
|
+
Release assets are the only place this is needed: the GraphQL `ReleaseAsset` type exposes
|
|
63
|
+
no `databaseId`, but the REST-shaped schema has always carried a numeric `id`.
|
|
64
|
+
"""
|
|
65
|
+
if not node_id or "_" not in node_id:
|
|
66
|
+
return None
|
|
67
|
+
try:
|
|
68
|
+
encoded = node_id.split("_", 1)[1]
|
|
69
|
+
decoded = base64.urlsafe_b64decode(encoded + "==")
|
|
70
|
+
if len(decoded) >= 4:
|
|
71
|
+
return struct.unpack(">I", decoded[-4:])[0]
|
|
72
|
+
except (ValueError, struct.error, binascii.Error):
|
|
73
|
+
return None
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _resolve_page_size(page_size: Any, config: Config) -> Optional[int]:
|
|
78
|
+
"""Interpolate and coerce a page size supplied to a custom component.
|
|
79
|
+
|
|
80
|
+
A custom component's fields are handed over uninterpolated, so a manifest value like
|
|
81
|
+
`"{{ config['page_size_for_large_streams'] }}"` arrives as that literal string. The CDK
|
|
82
|
+
reduces the page size arithmetically, so it has to be an int by the time it is returned
|
|
83
|
+
from `get_page_size`.
|
|
84
|
+
|
|
85
|
+
`page_size_for_large_streams` left the spec in 1.0.1 but is still honored, so a value that
|
|
86
|
+
is not a whole number can still arrive from the API, Terraform or an embedded config
|
|
87
|
+
without a form to validate it. Coercing it with a bare `int()` would surface as a
|
|
88
|
+
`ValueError` from the middle of a sync, so it is reported as the configuration error it is.
|
|
89
|
+
"""
|
|
90
|
+
if page_size is None:
|
|
91
|
+
return None
|
|
92
|
+
if isinstance(page_size, str):
|
|
93
|
+
page_size = InterpolatedString.create(page_size, parameters={}).eval(config)
|
|
94
|
+
try:
|
|
95
|
+
if isinstance(page_size, bool) or (isinstance(page_size, float) and not page_size.is_integer()):
|
|
96
|
+
raise ValueError(page_size)
|
|
97
|
+
resolved = int(page_size)
|
|
98
|
+
except (TypeError, ValueError):
|
|
99
|
+
raise AirbyteTracedException(
|
|
100
|
+
internal_message=f"page_size_for_large_streams resolved to {page_size!r}, which is not a whole number",
|
|
101
|
+
message=f'"Page size for large streams" (page_size_for_large_streams) must be a whole number. ' f"Got {page_size!r}.",
|
|
102
|
+
failure_type=FailureType.config_error,
|
|
103
|
+
)
|
|
104
|
+
if resolved < 1:
|
|
105
|
+
raise AirbyteTracedException(
|
|
106
|
+
internal_message=f"page_size_for_large_streams resolved to {resolved}, which is not strictly positive",
|
|
107
|
+
message=f'"Page size for large streams" (page_size_for_large_streams) must be at least 1. ' f"Got {resolved}.",
|
|
108
|
+
failure_type=FailureType.config_error,
|
|
109
|
+
)
|
|
110
|
+
return resolved
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass
|
|
114
|
+
class ReleasesRecordTransformation(RecordTransformation):
|
|
115
|
+
"""Reshape a GraphQL `Release` node into the REST-compatible `releases` record.
|
|
116
|
+
|
|
117
|
+
Ported verbatim from the legacy `streams.Releases.parse_response`. Five separate
|
|
118
|
+
concerns, none of them expressible as AddFields/RemoveFields:
|
|
119
|
+
|
|
120
|
+
- `assets`: unwrap the connection, flatten `uploader` to `uploader_id`, and recover each
|
|
121
|
+
asset's numeric `id` from its node ID.
|
|
122
|
+
- `reactions`: collapse `reactionGroups` into the REST reaction-count object, including
|
|
123
|
+
the zero entries for reactions nobody used and the `total_count` sum.
|
|
124
|
+
- `mentions_count`: unwrap a `totalCount`-only connection.
|
|
125
|
+
- `target_commitish`: unwrap `tagCommit.oid`.
|
|
126
|
+
- `url`/`assets_url`/`upload_url`/`tarball_url`/`zipball_url`: GraphQL does not return
|
|
127
|
+
these, so they are synthesized from the repository, release ID and tag, exactly as the
|
|
128
|
+
REST payload had them.
|
|
129
|
+
"""
|
|
130
|
+
|
|
131
|
+
def transform(
|
|
132
|
+
self,
|
|
133
|
+
record: MutableMapping[str, Any],
|
|
134
|
+
config: Optional[Config] = None,
|
|
135
|
+
stream_state: Optional[StreamState] = None,
|
|
136
|
+
stream_slice: Optional[StreamSlice] = None,
|
|
137
|
+
) -> None:
|
|
138
|
+
repository = (stream_slice or {}).get("repository")
|
|
139
|
+
record["repository"] = repository
|
|
140
|
+
|
|
141
|
+
if record.get("author"):
|
|
142
|
+
record["author"]["type"] = record["author"].pop("__typename", "User")
|
|
143
|
+
|
|
144
|
+
record["assets"] = self._assets(record)
|
|
145
|
+
record["reactions"] = self._reactions(record)
|
|
146
|
+
|
|
147
|
+
mentions_connection = record.pop("mentions_connection", None)
|
|
148
|
+
if mentions_connection is not None:
|
|
149
|
+
record["mentions_count"] = mentions_connection.get("totalCount", 0)
|
|
150
|
+
|
|
151
|
+
tag_commit = record.pop("tagCommit", None)
|
|
152
|
+
record["target_commitish"] = tag_commit.get("target_commitish") if tag_commit else None
|
|
153
|
+
|
|
154
|
+
api_url = (config or {}).get("api_url") or "https://api.github.com"
|
|
155
|
+
record.update(
|
|
156
|
+
self._rest_urls(
|
|
157
|
+
api_url=api_url.rstrip("/"),
|
|
158
|
+
repository=repository,
|
|
159
|
+
release_id=record.get("id"),
|
|
160
|
+
tag_name=record.get("tag_name"),
|
|
161
|
+
)
|
|
162
|
+
)
|
|
163
|
+
|
|
164
|
+
def _assets(self, record: Mapping[str, Any]) -> list:
|
|
165
|
+
assets_data = record.get("assets") or {}
|
|
166
|
+
if (assets_data.get("pageInfo") or {}).get("hasNextPage"):
|
|
167
|
+
# The query asks for `releaseAssets(first: 100)` and the manifest paginates the
|
|
168
|
+
# releases connection only, so a release with more than 100 assets is truncated.
|
|
169
|
+
# Warn rather than fail, which is what the Python stream did.
|
|
170
|
+
LOGGER.warning(
|
|
171
|
+
"Release %s in %s has >100 assets; only the first 100 were synced. "
|
|
172
|
+
"Sub-pagination for release assets is not yet implemented.",
|
|
173
|
+
record.get("id"),
|
|
174
|
+
record.get("repository"),
|
|
175
|
+
)
|
|
176
|
+
assets = assets_data.get("nodes", [])
|
|
177
|
+
for asset in assets:
|
|
178
|
+
uploader = asset.pop("uploader", None)
|
|
179
|
+
asset["uploader_id"] = uploader.get("id") if uploader else None
|
|
180
|
+
asset["id"] = _extract_database_id_from_node_id(asset.get("node_id"))
|
|
181
|
+
return assets
|
|
182
|
+
|
|
183
|
+
def _reactions(self, record: MutableMapping[str, Any]) -> Optional[Mapping[str, Any]]:
|
|
184
|
+
reaction_groups = record.pop("reaction_groups", None)
|
|
185
|
+
if reaction_groups is None:
|
|
186
|
+
return None
|
|
187
|
+
reactions: MutableMapping[str, Any] = {key: 0 for key in GRAPHQL_REACTION_TO_REST.values()}
|
|
188
|
+
total = 0
|
|
189
|
+
for group in reaction_groups:
|
|
190
|
+
rest_key = GRAPHQL_REACTION_TO_REST.get(group.get("content"))
|
|
191
|
+
if rest_key:
|
|
192
|
+
count = (group.get("reactors") or {}).get("totalCount", 0)
|
|
193
|
+
reactions[rest_key] = count
|
|
194
|
+
total += count
|
|
195
|
+
reactions["total_count"] = total
|
|
196
|
+
return reactions
|
|
197
|
+
|
|
198
|
+
@staticmethod
|
|
199
|
+
def _rest_urls(api_url: str, repository: Optional[str], release_id: Optional[int], tag_name: Optional[str]) -> Mapping[str, Any]:
|
|
200
|
+
upload_url = api_url.replace("api.github.com", "uploads.github.com")
|
|
201
|
+
return {
|
|
202
|
+
"url": f"{api_url}/repos/{repository}/releases/{release_id}",
|
|
203
|
+
"assets_url": f"{api_url}/repos/{repository}/releases/{release_id}/assets",
|
|
204
|
+
"upload_url": f"{upload_url}/repos/{repository}/releases/{release_id}/assets{{?name,label}}",
|
|
205
|
+
"tarball_url": f"{api_url}/repos/{repository}/tarball/{tag_name}" if tag_name else None,
|
|
206
|
+
"zipball_url": f"{api_url}/repos/{repository}/zipball/{tag_name}" if tag_name else None,
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
@dataclass
|
|
211
|
+
class NestedGraphQLPaginationStrategy(PaginationStrategy):
|
|
212
|
+
"""Two-level cursor traversal for `reviews` and `issue_reactions`.
|
|
213
|
+
|
|
214
|
+
Both streams walk a repository-level connection (`pullRequests` / `issues`) whose nodes
|
|
215
|
+
each carry a child connection (`reviews` / `reactions`). A child connection that has more
|
|
216
|
+
pages cannot be paginated in place, so the legacy streams switched the query to a
|
|
217
|
+
drill-down rooted at that single parent (`repository.pullRequest(number:)`) and came back
|
|
218
|
+
to the parent listing afterwards.
|
|
219
|
+
|
|
220
|
+
All traversal state lives in the page token rather than on this object. That is not a
|
|
221
|
+
stylistic choice: one `PaginationStrategy` instance is shared by every partition of a
|
|
222
|
+
stream, and the partitions are read concurrently, so the legacy `self.reviews_cursors` /
|
|
223
|
+
`self.pull_requests_cursor` dicts were keyed by repository precisely to work around state
|
|
224
|
+
that should never have been shared. A self-contained token removes the sharing instead.
|
|
225
|
+
|
|
226
|
+
The token is `{document, after, number, pending, list_after}`:
|
|
227
|
+
|
|
228
|
+
- `document` is the query to send next, so the retriever's `request_body` only has to
|
|
229
|
+
choose between "the token's document" and the root listing document.
|
|
230
|
+
- `pending` is the queue of `(parent_number, child_cursor)` pairs still to drill into,
|
|
231
|
+
popped LIFO to match the legacy `dict.popitem()`.
|
|
232
|
+
- `list_after` is where to resume the parent listing once `pending` drains.
|
|
233
|
+
|
|
234
|
+
`first` stays a GraphQL variable in both documents so `REDUCE_PAGE_SIZE` still works; only
|
|
235
|
+
the parent's `number` is inlined into the drill-down document, since it is not a page size.
|
|
236
|
+
"""
|
|
237
|
+
|
|
238
|
+
config: Config
|
|
239
|
+
parameters: InitVar[Mapping[str, Any]]
|
|
240
|
+
list_connection: str = ""
|
|
241
|
+
drilldown_field: str = ""
|
|
242
|
+
child_connection: str = ""
|
|
243
|
+
page_size: Optional[int] = None
|
|
244
|
+
|
|
245
|
+
# The drill-down document carries the parent's number inline. A literal marker rather than
|
|
246
|
+
# `str.format`, because the GraphQL body is full of braces.
|
|
247
|
+
NUMBER_PLACEHOLDER = "__NUMBER__"
|
|
248
|
+
|
|
249
|
+
def __post_init__(self, parameters: Mapping[str, Any]) -> None:
|
|
250
|
+
for field in ("list_connection", "drilldown_field", "child_connection"):
|
|
251
|
+
if not getattr(self, field):
|
|
252
|
+
raise ValueError(f"NestedGraphQLPaginationStrategy requires `{field}`")
|
|
253
|
+
# Read from `$parameters` rather than from a manifest field: a custom component's
|
|
254
|
+
# string fields are not interpolated, so a `{{ parameters[...] }}` reference would
|
|
255
|
+
# arrive verbatim. The stream declares each document once and both the requester's
|
|
256
|
+
# `request_body` and this strategy read that one declaration.
|
|
257
|
+
for field in ("list_document", "drilldown_document"):
|
|
258
|
+
value = parameters.get(field)
|
|
259
|
+
if not value:
|
|
260
|
+
raise ValueError(f"NestedGraphQLPaginationStrategy requires `{field}` in $parameters")
|
|
261
|
+
setattr(self, field, value)
|
|
262
|
+
|
|
263
|
+
@property
|
|
264
|
+
def initial_token(self) -> Optional[Any]:
|
|
265
|
+
return None
|
|
266
|
+
|
|
267
|
+
def get_page_size(self) -> Optional[int]:
|
|
268
|
+
return _resolve_page_size(self.page_size, self.config)
|
|
269
|
+
|
|
270
|
+
def next_page_token(
|
|
271
|
+
self,
|
|
272
|
+
response: requests.Response,
|
|
273
|
+
last_page_size: int,
|
|
274
|
+
last_record: Optional[Any],
|
|
275
|
+
last_page_token_value: Optional[Any] = None,
|
|
276
|
+
page_size_override: Optional[int] = None,
|
|
277
|
+
) -> Optional[Mapping[str, Any]]:
|
|
278
|
+
previous = last_page_token_value if isinstance(last_page_token_value, Mapping) else {}
|
|
279
|
+
pending = [list(item) for item in previous.get("pending", [])]
|
|
280
|
+
list_after = previous.get("list_after")
|
|
281
|
+
|
|
282
|
+
repository = (response.json().get("data") or {}).get("repository")
|
|
283
|
+
if repository:
|
|
284
|
+
if self.list_connection in repository:
|
|
285
|
+
connection = repository[self.list_connection] or {}
|
|
286
|
+
page_info = connection.get("pageInfo") or {}
|
|
287
|
+
if page_info.get("hasNextPage"):
|
|
288
|
+
list_after = page_info.get("endCursor")
|
|
289
|
+
for node in connection.get("nodes") or []:
|
|
290
|
+
self._queue_child(node, pending)
|
|
291
|
+
elif self.drilldown_field in repository:
|
|
292
|
+
self._queue_child(repository[self.drilldown_field] or {}, pending)
|
|
293
|
+
|
|
294
|
+
if pending:
|
|
295
|
+
number, after = pending.pop()
|
|
296
|
+
return {
|
|
297
|
+
"document": self.drilldown_document.replace(self.NUMBER_PLACEHOLDER, str(number)),
|
|
298
|
+
"after": after,
|
|
299
|
+
"number": number,
|
|
300
|
+
"pending": pending,
|
|
301
|
+
"list_after": list_after,
|
|
302
|
+
}
|
|
303
|
+
if list_after:
|
|
304
|
+
return {
|
|
305
|
+
"document": self.list_document,
|
|
306
|
+
"after": list_after,
|
|
307
|
+
"number": None,
|
|
308
|
+
"pending": [],
|
|
309
|
+
"list_after": None,
|
|
310
|
+
}
|
|
311
|
+
return None
|
|
312
|
+
|
|
313
|
+
def _queue_child(self, node: Mapping[str, Any], pending: list) -> None:
|
|
314
|
+
child = node.get(self.child_connection) or {}
|
|
315
|
+
if (child.get("pageInfo") or {}).get("hasNextPage"):
|
|
316
|
+
pending.append([node.get("number"), child["pageInfo"]["endCursor"]])
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
@dataclass
|
|
320
|
+
class NestedGraphQLRecordExtractor(RecordExtractor):
|
|
321
|
+
"""Extract child records from either shape a two-level GraphQL traversal can return.
|
|
322
|
+
|
|
323
|
+
The listing query nests the child connection under every parent node:
|
|
324
|
+
|
|
325
|
+
data.repository.<list_connection>.nodes[*].<child_connection>.nodes[*]
|
|
326
|
+
|
|
327
|
+
the drill-down query returns a single parent:
|
|
328
|
+
|
|
329
|
+
data.repository.<drilldown_field>.<child_connection>.nodes[*]
|
|
330
|
+
|
|
331
|
+
A `DpathExtractor` can express either path but not both, and the records also need fields
|
|
332
|
+
that only exist on the parent node (`reviews.pull_request_url` comes from the pull
|
|
333
|
+
request's `url`), which a path-based extractor cannot reach at all. `parent_fields` maps a
|
|
334
|
+
field on the parent node to the field name to copy it into.
|
|
335
|
+
"""
|
|
336
|
+
|
|
337
|
+
config: Config
|
|
338
|
+
parameters: InitVar[Mapping[str, Any]]
|
|
339
|
+
list_connection: str = ""
|
|
340
|
+
drilldown_field: str = ""
|
|
341
|
+
child_connection: str = ""
|
|
342
|
+
parent_fields: Optional[Mapping[str, str]] = None
|
|
343
|
+
|
|
344
|
+
def __post_init__(self, parameters: Mapping[str, Any]) -> None:
|
|
345
|
+
for field in ("list_connection", "drilldown_field", "child_connection"):
|
|
346
|
+
if not getattr(self, field):
|
|
347
|
+
raise ValueError(f"NestedGraphQLRecordExtractor requires `{field}`")
|
|
348
|
+
|
|
349
|
+
def extract_records(self, response: requests.Response) -> Iterable[MutableMapping[Any, Any]]:
|
|
350
|
+
repository = (response.json().get("data") or {}).get("repository")
|
|
351
|
+
if not repository:
|
|
352
|
+
# GitHub answers 200 with a null repository when the token cannot see it.
|
|
353
|
+
return
|
|
354
|
+
repository_name = f"{(repository.get('owner') or {}).get('login')}/{repository.get('name')}"
|
|
355
|
+
if self.list_connection in repository:
|
|
356
|
+
parents = ((repository[self.list_connection] or {}).get("nodes")) or []
|
|
357
|
+
else:
|
|
358
|
+
parent = repository.get(self.drilldown_field)
|
|
359
|
+
parents = [parent] if parent else []
|
|
360
|
+
for parent in parents:
|
|
361
|
+
children = ((parent.get(self.child_connection) or {}).get("nodes")) or []
|
|
362
|
+
for record in children:
|
|
363
|
+
record["repository"] = repository_name
|
|
364
|
+
for source, destination in (self.parent_fields or {}).items():
|
|
365
|
+
record[destination] = parent.get(source)
|
|
366
|
+
yield record
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
# Depth-first order for the four-level reaction traversal: the deepest pending connection is
|
|
370
|
+
# always drilled into first, so a comment's reactions are finished before the next pull
|
|
371
|
+
# request is opened. Ported from `graphql.CursorStorage`, which built the same ordering out of
|
|
372
|
+
# a heap keyed on this list reversed.
|
|
373
|
+
_REACTION_TRAVERSAL_PRIORITY = {
|
|
374
|
+
"Reaction": 0,
|
|
375
|
+
"PullRequestReviewComment": 1,
|
|
376
|
+
"PullRequestReview": 2,
|
|
377
|
+
"PullRequest": 3,
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
# Which child connection each queued object type needs paginated, and which document roots at
|
|
381
|
+
# it. `PullRequest` is the repository-level listing; the rest root at `node(id:)`.
|
|
382
|
+
_REACTION_CONNECTION_OF = {
|
|
383
|
+
"PullRequest": "pullRequests",
|
|
384
|
+
"PullRequestReview": "reviews",
|
|
385
|
+
"PullRequestReviewComment": "comments",
|
|
386
|
+
"Reaction": "reactions",
|
|
387
|
+
}
|
|
388
|
+
|
|
389
|
+
|
|
390
|
+
@dataclass
|
|
391
|
+
class DeepNestedGraphQLPaginationStrategy(PaginationStrategy):
|
|
392
|
+
"""Four-level depth-first traversal for `pull_request_comment_reactions`.
|
|
393
|
+
|
|
394
|
+
`repository.pullRequests -> reviews -> comments -> reactions`. Every level can have more
|
|
395
|
+
pages than the query asked for, and none of them can be paginated in place, so each
|
|
396
|
+
overflowing connection is queued and later re-rooted with its own document:
|
|
397
|
+
|
|
398
|
+
- `PullRequest` -> the repository listing, paginated by `pullRequests`
|
|
399
|
+
- `PullRequestReview`-> `node(id: <pull request>)`, paginated by `reviews`
|
|
400
|
+
- `PullRequestReviewComment` -> `node(id: <review>)`, paginated by `comments`
|
|
401
|
+
- `Reaction` -> `node(id: <comment>)`, paginated by `reactions`
|
|
402
|
+
|
|
403
|
+
The queue is ordered deepest-first, which is what makes the traversal depth-first: the
|
|
404
|
+
reactions of a comment are drained before the next review is opened.
|
|
405
|
+
|
|
406
|
+
Like `NestedGraphQLPaginationStrategy`, the queue lives in the page token rather than on
|
|
407
|
+
this object, because one strategy instance is shared by every partition of the stream and
|
|
408
|
+
the partitions are read concurrently. The legacy `self.cursor_storage` was a single heap
|
|
409
|
+
shared across repositories.
|
|
410
|
+
|
|
411
|
+
One legacy behavior is deliberately not carried over: the legacy `request_body_json` used to send
|
|
412
|
+
`first = min(page_size, total_count)` to avoid paying for pages larger than what remained.
|
|
413
|
+
`first` has to stay a GraphQL variable for `REDUCE_PAGE_SIZE` to be able to shrink it, and
|
|
414
|
+
a variable cannot be per-token, so the connector may now over-ask on the last page of a
|
|
415
|
+
connection. GitHub returns fewer records; the cost is a slightly higher query score.
|
|
416
|
+
"""
|
|
417
|
+
|
|
418
|
+
config: Config
|
|
419
|
+
parameters: InitVar[Mapping[str, Any]]
|
|
420
|
+
documents: Optional[Mapping[str, str]] = None
|
|
421
|
+
page_size: Optional[int] = None
|
|
422
|
+
|
|
423
|
+
NODE_PLACEHOLDER = "__NODE_ID__"
|
|
424
|
+
|
|
425
|
+
# $parameters key holding the document that re-roots at each object type.
|
|
426
|
+
DOCUMENT_PARAMETERS = {
|
|
427
|
+
"PullRequest": "root_repository_document",
|
|
428
|
+
"PullRequestReview": "root_pull_request_document",
|
|
429
|
+
"PullRequestReviewComment": "root_review_document",
|
|
430
|
+
"Reaction": "root_comment_document",
|
|
431
|
+
}
|
|
432
|
+
|
|
433
|
+
def __post_init__(self, parameters: Mapping[str, Any]) -> None:
|
|
434
|
+
if self.documents is None:
|
|
435
|
+
# Same reason as NestedGraphQLPaginationStrategy: a custom component's fields are
|
|
436
|
+
# not interpolated, so the documents are read from $parameters.
|
|
437
|
+
self.documents = {typename: parameters[key] for typename, key in self.DOCUMENT_PARAMETERS.items() if parameters.get(key)}
|
|
438
|
+
missing = set(_REACTION_TRAVERSAL_PRIORITY) - set(self.documents or {})
|
|
439
|
+
if missing:
|
|
440
|
+
raise ValueError(f"DeepNestedGraphQLPaginationStrategy requires a document for every object type; missing {sorted(missing)}")
|
|
441
|
+
|
|
442
|
+
@property
|
|
443
|
+
def initial_token(self) -> Optional[Any]:
|
|
444
|
+
return None
|
|
445
|
+
|
|
446
|
+
def get_page_size(self) -> Optional[int]:
|
|
447
|
+
return _resolve_page_size(self.page_size, self.config)
|
|
448
|
+
|
|
449
|
+
def next_page_token(
|
|
450
|
+
self,
|
|
451
|
+
response: requests.Response,
|
|
452
|
+
last_page_size: int,
|
|
453
|
+
last_record: Optional[Any],
|
|
454
|
+
last_page_token_value: Optional[Any] = None,
|
|
455
|
+
page_size_override: Optional[int] = None,
|
|
456
|
+
) -> Optional[Mapping[str, Any]]:
|
|
457
|
+
previous = last_page_token_value if isinstance(last_page_token_value, Mapping) else {}
|
|
458
|
+
pending = [list(item) for item in previous.get("pending", [])]
|
|
459
|
+
sequence = previous.get("sequence", 0)
|
|
460
|
+
|
|
461
|
+
data = response.json().get("data") or {}
|
|
462
|
+
|
|
463
|
+
repository = data.get("repository")
|
|
464
|
+
if repository:
|
|
465
|
+
sequence = self._queue(repository, "PullRequest", pending, sequence)
|
|
466
|
+
for pull_request in self._nodes(repository, "pullRequests"):
|
|
467
|
+
sequence = self._walk_pull_request(pull_request, pending, sequence)
|
|
468
|
+
|
|
469
|
+
node = data.get("node")
|
|
470
|
+
if node:
|
|
471
|
+
typename = node.get("__typename")
|
|
472
|
+
if typename == "PullRequest":
|
|
473
|
+
sequence = self._walk_pull_request(node, pending, sequence)
|
|
474
|
+
elif typename == "PullRequestReview":
|
|
475
|
+
sequence = self._walk_review(node, pending, sequence)
|
|
476
|
+
elif typename == "PullRequestReviewComment":
|
|
477
|
+
sequence = self._queue(node, "Reaction", pending, sequence)
|
|
478
|
+
|
|
479
|
+
if not pending:
|
|
480
|
+
return None
|
|
481
|
+
|
|
482
|
+
# Deepest first, and first-queued first within a depth -- the ordering the heap gave.
|
|
483
|
+
pending.sort(key=lambda item: (item[0], item[1]))
|
|
484
|
+
_, _, typename, cursor, node_id = pending.pop(0)
|
|
485
|
+
document = self.documents[typename] # type: ignore[index]
|
|
486
|
+
if node_id is not None:
|
|
487
|
+
document = document.replace(self.NODE_PLACEHOLDER, str(node_id))
|
|
488
|
+
return {"document": document, "after": cursor, "typename": typename, "pending": pending, "sequence": sequence}
|
|
489
|
+
|
|
490
|
+
def _walk_pull_request(self, pull_request: Mapping[str, Any], pending: list, sequence: int) -> int:
|
|
491
|
+
sequence = self._queue(pull_request, "PullRequestReview", pending, sequence)
|
|
492
|
+
for review in self._nodes(pull_request, "reviews"):
|
|
493
|
+
sequence = self._walk_review(review, pending, sequence)
|
|
494
|
+
return sequence
|
|
495
|
+
|
|
496
|
+
def _walk_review(self, review: Mapping[str, Any], pending: list, sequence: int) -> int:
|
|
497
|
+
sequence = self._queue(review, "PullRequestReviewComment", pending, sequence)
|
|
498
|
+
for comment in self._nodes(review, "comments"):
|
|
499
|
+
sequence = self._queue(comment, "Reaction", pending, sequence)
|
|
500
|
+
return sequence
|
|
501
|
+
|
|
502
|
+
def _queue(self, node: Mapping[str, Any], typename: str, pending: list, sequence: int) -> int:
|
|
503
|
+
"""Queue `node`'s child connection if it has another page."""
|
|
504
|
+
connection = node.get(_REACTION_CONNECTION_OF[typename]) or {}
|
|
505
|
+
page_info = connection.get("pageInfo") or {}
|
|
506
|
+
if not page_info.get("hasNextPage"):
|
|
507
|
+
return sequence
|
|
508
|
+
# `PullRequest` re-roots at the repository listing, which needs no node id.
|
|
509
|
+
node_id = None if typename == "PullRequest" else node.get("node_id")
|
|
510
|
+
pending.append([_REACTION_TRAVERSAL_PRIORITY[typename], sequence, typename, page_info.get("endCursor"), node_id])
|
|
511
|
+
return sequence + 1
|
|
512
|
+
|
|
513
|
+
@staticmethod
|
|
514
|
+
def _nodes(node: Mapping[str, Any], connection: str) -> Iterable[Mapping[str, Any]]:
|
|
515
|
+
return ((node.get(connection) or {}).get("nodes")) or []
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
@dataclass
|
|
519
|
+
class DeepNestedGraphQLRecordExtractor(RecordExtractor):
|
|
520
|
+
"""Extract reaction records from any of the four roots the traversal can return.
|
|
521
|
+
|
|
522
|
+
Ported from `streams.PullRequestCommentReactions.parse_response`. Each reaction is stamped
|
|
523
|
+
with its repository and the id of the comment it belongs to; `user.type` is set because the
|
|
524
|
+
legacy record carried it and the GraphQL `user` field here is not a union.
|
|
525
|
+
"""
|
|
526
|
+
|
|
527
|
+
config: Config
|
|
528
|
+
parameters: InitVar[Mapping[str, Any]]
|
|
529
|
+
|
|
530
|
+
def __post_init__(self, parameters: Mapping[str, Any]) -> None:
|
|
531
|
+
pass
|
|
532
|
+
|
|
533
|
+
def extract_records(self, response: requests.Response) -> Iterable[MutableMapping[Any, Any]]:
|
|
534
|
+
data = response.json().get("data") or {}
|
|
535
|
+
|
|
536
|
+
repository = data.get("repository")
|
|
537
|
+
if repository:
|
|
538
|
+
for pull_request in self._nodes(repository, "pullRequests"):
|
|
539
|
+
yield from self._from_pull_request(pull_request, repository)
|
|
540
|
+
|
|
541
|
+
node = data.get("node")
|
|
542
|
+
if node:
|
|
543
|
+
# The drill-down documents select the repository alongside the node, so the record
|
|
544
|
+
# can still be stamped with it.
|
|
545
|
+
repository = node.get("repository") or {}
|
|
546
|
+
typename = node.get("__typename")
|
|
547
|
+
if typename == "PullRequest":
|
|
548
|
+
yield from self._from_pull_request(node, repository)
|
|
549
|
+
elif typename == "PullRequestReview":
|
|
550
|
+
yield from self._from_review(node, repository)
|
|
551
|
+
elif typename == "PullRequestReviewComment":
|
|
552
|
+
yield from self._from_comment(node, repository)
|
|
553
|
+
|
|
554
|
+
def _from_pull_request(self, pull_request: Mapping[str, Any], repository: Mapping[str, Any]):
|
|
555
|
+
for review in self._nodes(pull_request, "reviews"):
|
|
556
|
+
yield from self._from_review(review, repository)
|
|
557
|
+
|
|
558
|
+
def _from_review(self, review: Mapping[str, Any], repository: Mapping[str, Any]):
|
|
559
|
+
for comment in self._nodes(review, "comments"):
|
|
560
|
+
yield from self._from_comment(comment, repository)
|
|
561
|
+
|
|
562
|
+
def _from_comment(self, comment: Mapping[str, Any], repository: Mapping[str, Any]):
|
|
563
|
+
repository_name = f"{(repository.get('owner') or {}).get('login')}/{repository.get('name')}"
|
|
564
|
+
for reaction in self._nodes(comment, "reactions"):
|
|
565
|
+
reaction["repository"] = repository_name
|
|
566
|
+
reaction["comment_id"] = comment.get("id")
|
|
567
|
+
if reaction.get("user"):
|
|
568
|
+
reaction["user"]["type"] = "User"
|
|
569
|
+
yield reaction
|
|
570
|
+
|
|
571
|
+
@staticmethod
|
|
572
|
+
def _nodes(node: Mapping[str, Any], connection: str) -> Iterable[Mapping[str, Any]]:
|
|
573
|
+
return ((node.get(connection) or {}).get("nodes")) or []
|
|
18
574
|
|
|
19
575
|
|
|
20
576
|
@dataclass
|