airbyte-source-github 2.6.0__py3-none-any.whl → 2.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/METADATA +2 -3
  2. airbyte_source_github-2.7.0.dist-info/RECORD +11 -0
  3. source_github/components.py +558 -2
  4. source_github/manifest.yaml +9403 -695
  5. source_github/source.py +7 -37
  6. airbyte_source_github-2.6.0.dist-info/RECORD +0 -36
  7. source_github/backoff_strategies.py +0 -52
  8. source_github/errors_handlers.py +0 -223
  9. source_github/github_schema.py +0 -41035
  10. source_github/graphql.py +0 -375
  11. source_github/schemas/commit_comments.json +0 -68
  12. source_github/schemas/issue_reactions.json +0 -35
  13. source_github/schemas/issue_timeline_events.json +0 -1197
  14. source_github/schemas/projects.json +0 -64
  15. source_github/schemas/projects_v2.json +0 -103
  16. source_github/schemas/pull_request_comment_reactions.json +0 -35
  17. source_github/schemas/pull_request_stats.json +0 -105
  18. source_github/schemas/pull_requests.json +0 -432
  19. source_github/schemas/releases.json +0 -153
  20. source_github/schemas/reviews.json +0 -87
  21. source_github/schemas/shared/events/comment.json +0 -188
  22. source_github/schemas/shared/events/commented.json +0 -118
  23. source_github/schemas/shared/events/committed.json +0 -56
  24. source_github/schemas/shared/events/cross_referenced.json +0 -822
  25. source_github/schemas/shared/events/reviewed.json +0 -139
  26. source_github/schemas/shared/reaction.json +0 -27
  27. source_github/schemas/shared/reactions.json +0 -35
  28. source_github/schemas/shared/user.json +0 -59
  29. source_github/schemas/shared/user_graphql.json +0 -26
  30. source_github/streams.py +0 -909
  31. source_github/utils.py +0 -24
  32. {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/WHEEL +0 -0
  33. {airbyte_source_github-2.6.0.dist-info → airbyte_source_github-2.7.0.dist-info}/entry_points.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: airbyte-source-github
3
- Version: 2.6.0
3
+ Version: 2.7.0
4
4
  Summary: Source implementation for GitHub.
5
5
  Home-page: https://airbyte.com
6
6
  License: ELv2
@@ -13,8 +13,7 @@ Classifier: Programming Language :: Python :: 3.10
13
13
  Classifier: Programming Language :: Python :: 3.11
14
14
  Classifier: Programming Language :: Python :: 3.12
15
15
  Classifier: Programming Language :: Python :: 3.13
16
- Requires-Dist: airbyte-cdk (>=7.28.2,<8.0.0)
17
- Requires-Dist: sgqlc (==16.3)
16
+ Requires-Dist: airbyte-cdk (>=7.30.0,<8.0.0)
18
17
  Project-URL: Documentation, https://docs.airbyte.com/integrations/sources/github
19
18
  Project-URL: Repository, https://github.com/airbytehq/airbyte
20
19
  Description-Content-Type: text/markdown
@@ -0,0 +1,11 @@
1
+ source_github/__init__.py,sha256=8hLLjFC1wI1mMU442Y5qCcND9qIwyhaANQwT-GAIK0s,1135
2
+ source_github/components.py,sha256=iqhaoVq5Sao1fD86GyvCyMyM8-QlNCQYelZFC7BZzmw,35433
3
+ source_github/config_migrations.py,sha256=guUJAdNP-liUciVaJB4ackEEMCN4jd7bNd85QFmDNlU,3932
4
+ source_github/constants.py,sha256=Hj3Q4y7OoU-Iff4m9gEC2CjwmWJYXhNbHVNjg8EBLmQ,238
5
+ source_github/manifest.yaml,sha256=tohJ8Sx_yOPKeVeRCNMwGd-TD3IOib7zDpLf2QT_M0k,668630
6
+ source_github/run.py,sha256=gfWy8TcMHa0XRwaVUyq4KmslooST38fvfCLuP5D-5Fo,2391
7
+ source_github/source.py,sha256=5VvGvnuj2FhvtHkHDzjiGZIBfPgSg-LMqyw5liGGdP4,18139
8
+ airbyte_source_github-2.7.0.dist-info/METADATA,sha256=EjShSrKSQFQ0d9phj_VZJUgRviOZO3WR3Lszu2C-iss,1218
9
+ airbyte_source_github-2.7.0.dist-info/WHEEL,sha256=Nq82e9rUAnEjt98J6MlVmMCZb-t9cYE2Ir1kpBmnWfs,88
10
+ airbyte_source_github-2.7.0.dist-info/entry_points.txt,sha256=gYhqVrTAZvMwuYByg0b_-o115yUFLLcfNxMrLZmiW9k,55
11
+ airbyte_source_github-2.7.0.dist-info/RECORD,,
@@ -2,19 +2,575 @@
2
2
  # Copyright (c) 2023 Airbyte, Inc., all rights reserved.
3
3
  #
4
4
 
5
- from dataclasses import dataclass
5
+ """Custom low-code components for source-github.
6
+
7
+ Everything here exists because the GitHub GraphQL responses cannot be reshaped into the
8
+ connector's long-standing REST-compatible record shape with declarative transformations
9
+ alone. Pagination, page-size reduction, error handling and incremental behavior are all
10
+ handled by the manifest.
11
+ """
12
+
13
+ import base64
14
+ import binascii
15
+ import logging
16
+ import struct
17
+ from dataclasses import InitVar, dataclass
6
18
  from datetime import timedelta
7
19
  from itertools import groupby
8
20
  from typing import Any, Iterable, List, Mapping, MutableMapping, Optional
9
21
 
10
22
  import requests
11
23
 
24
+ from airbyte_cdk.models import FailureType
12
25
  from airbyte_cdk.sources.declarative.extractors.record_extractor import RecordExtractor
26
+ from airbyte_cdk.sources.declarative.interpolation.interpolated_string import InterpolatedString
13
27
  from airbyte_cdk.sources.declarative.migrations.state_migration import StateMigration
14
28
  from airbyte_cdk.sources.declarative.partition_routers.substream_partition_router import SubstreamPartitionRouter
15
29
  from airbyte_cdk.sources.declarative.requesters.paginators.strategies.cursor_pagination_strategy import CursorPaginationStrategy
16
- from airbyte_cdk.sources.types import Config, Record, StreamSlice
30
+ from airbyte_cdk.sources.declarative.requesters.paginators.strategies.pagination_strategy import (
31
+ PaginationStrategy,
32
+ )
33
+ from airbyte_cdk.sources.declarative.transformations import RecordTransformation
34
+ from airbyte_cdk.sources.types import Config, Record, StreamSlice, StreamState
17
35
  from airbyte_cdk.utils.datetime_helpers import ab_datetime_parse
36
+ from airbyte_cdk.utils.traced_exception import AirbyteTracedException
37
+
38
+
39
+ LOGGER = logging.getLogger("airbyte")
40
+
41
+ # GitHub's GraphQL reaction content enum, mapped to the field names the REST API used and
42
+ # therefore to the names already present in the `releases` schema and in user warehouses.
43
+ GRAPHQL_REACTION_TO_REST = {
44
+ "THUMBS_UP": "plus_one",
45
+ "THUMBS_DOWN": "minus_one",
46
+ "LAUGH": "laugh",
47
+ "HOORAY": "hooray",
48
+ "CONFUSED": "confused",
49
+ "HEART": "heart",
50
+ "ROCKET": "rocket",
51
+ "EYES": "eyes",
52
+ }
53
+
54
+
55
+ def _extract_database_id_from_node_id(node_id: Optional[str]) -> Optional[int]:
56
+ """Extract the numeric database ID from a GitHub GraphQL Node ID.
57
+
58
+ GitHub Node IDs with type prefixes (e.g. 'RA_...') are URL-safe base64 encodings of a
59
+ msgpack array: [type_flag, repo_database_id, entity_database_id]. The last 4 bytes encode
60
+ the entity's numeric database ID as a big-endian uint32.
61
+
62
+ Release assets are the only place this is needed: the GraphQL `ReleaseAsset` type exposes
63
+ no `databaseId`, but the REST-shaped schema has always carried a numeric `id`.
64
+ """
65
+ if not node_id or "_" not in node_id:
66
+ return None
67
+ try:
68
+ encoded = node_id.split("_", 1)[1]
69
+ decoded = base64.urlsafe_b64decode(encoded + "==")
70
+ if len(decoded) >= 4:
71
+ return struct.unpack(">I", decoded[-4:])[0]
72
+ except (ValueError, struct.error, binascii.Error):
73
+ return None
74
+ return None
75
+
76
+
77
+ def _resolve_page_size(page_size: Any, config: Config) -> Optional[int]:
78
+ """Interpolate and coerce a page size supplied to a custom component.
79
+
80
+ A custom component's fields are handed over uninterpolated, so a manifest value like
81
+ `"{{ config['page_size_for_large_streams'] }}"` arrives as that literal string. The CDK
82
+ reduces the page size arithmetically, so it has to be an int by the time it is returned
83
+ from `get_page_size`.
84
+
85
+ `page_size_for_large_streams` left the spec in 1.0.1 but is still honored, so a value that
86
+ is not a whole number can still arrive from the API, Terraform or an embedded config
87
+ without a form to validate it. Coercing it with a bare `int()` would surface as a
88
+ `ValueError` from the middle of a sync, so it is reported as the configuration error it is.
89
+ """
90
+ if page_size is None:
91
+ return None
92
+ if isinstance(page_size, str):
93
+ page_size = InterpolatedString.create(page_size, parameters={}).eval(config)
94
+ try:
95
+ if isinstance(page_size, bool) or (isinstance(page_size, float) and not page_size.is_integer()):
96
+ raise ValueError(page_size)
97
+ resolved = int(page_size)
98
+ except (TypeError, ValueError):
99
+ raise AirbyteTracedException(
100
+ internal_message=f"page_size_for_large_streams resolved to {page_size!r}, which is not a whole number",
101
+ message=f'"Page size for large streams" (page_size_for_large_streams) must be a whole number. ' f"Got {page_size!r}.",
102
+ failure_type=FailureType.config_error,
103
+ )
104
+ if resolved < 1:
105
+ raise AirbyteTracedException(
106
+ internal_message=f"page_size_for_large_streams resolved to {resolved}, which is not strictly positive",
107
+ message=f'"Page size for large streams" (page_size_for_large_streams) must be at least 1. ' f"Got {resolved}.",
108
+ failure_type=FailureType.config_error,
109
+ )
110
+ return resolved
111
+
112
+
113
+ @dataclass
114
+ class ReleasesRecordTransformation(RecordTransformation):
115
+ """Reshape a GraphQL `Release` node into the REST-compatible `releases` record.
116
+
117
+ Ported verbatim from the legacy `streams.Releases.parse_response`. Five separate
118
+ concerns, none of them expressible as AddFields/RemoveFields:
119
+
120
+ - `assets`: unwrap the connection, flatten `uploader` to `uploader_id`, and recover each
121
+ asset's numeric `id` from its node ID.
122
+ - `reactions`: collapse `reactionGroups` into the REST reaction-count object, including
123
+ the zero entries for reactions nobody used and the `total_count` sum.
124
+ - `mentions_count`: unwrap a `totalCount`-only connection.
125
+ - `target_commitish`: unwrap `tagCommit.oid`.
126
+ - `url`/`assets_url`/`upload_url`/`tarball_url`/`zipball_url`: GraphQL does not return
127
+ these, so they are synthesized from the repository, release ID and tag, exactly as the
128
+ REST payload had them.
129
+ """
130
+
131
+ def transform(
132
+ self,
133
+ record: MutableMapping[str, Any],
134
+ config: Optional[Config] = None,
135
+ stream_state: Optional[StreamState] = None,
136
+ stream_slice: Optional[StreamSlice] = None,
137
+ ) -> None:
138
+ repository = (stream_slice or {}).get("repository")
139
+ record["repository"] = repository
140
+
141
+ if record.get("author"):
142
+ record["author"]["type"] = record["author"].pop("__typename", "User")
143
+
144
+ record["assets"] = self._assets(record)
145
+ record["reactions"] = self._reactions(record)
146
+
147
+ mentions_connection = record.pop("mentions_connection", None)
148
+ if mentions_connection is not None:
149
+ record["mentions_count"] = mentions_connection.get("totalCount", 0)
150
+
151
+ tag_commit = record.pop("tagCommit", None)
152
+ record["target_commitish"] = tag_commit.get("target_commitish") if tag_commit else None
153
+
154
+ api_url = (config or {}).get("api_url") or "https://api.github.com"
155
+ record.update(
156
+ self._rest_urls(
157
+ api_url=api_url.rstrip("/"),
158
+ repository=repository,
159
+ release_id=record.get("id"),
160
+ tag_name=record.get("tag_name"),
161
+ )
162
+ )
163
+
164
+ def _assets(self, record: Mapping[str, Any]) -> list:
165
+ assets_data = record.get("assets") or {}
166
+ if (assets_data.get("pageInfo") or {}).get("hasNextPage"):
167
+ # The query asks for `releaseAssets(first: 100)` and the manifest paginates the
168
+ # releases connection only, so a release with more than 100 assets is truncated.
169
+ # Warn rather than fail, which is what the Python stream did.
170
+ LOGGER.warning(
171
+ "Release %s in %s has >100 assets; only the first 100 were synced. "
172
+ "Sub-pagination for release assets is not yet implemented.",
173
+ record.get("id"),
174
+ record.get("repository"),
175
+ )
176
+ assets = assets_data.get("nodes", [])
177
+ for asset in assets:
178
+ uploader = asset.pop("uploader", None)
179
+ asset["uploader_id"] = uploader.get("id") if uploader else None
180
+ asset["id"] = _extract_database_id_from_node_id(asset.get("node_id"))
181
+ return assets
182
+
183
+ def _reactions(self, record: MutableMapping[str, Any]) -> Optional[Mapping[str, Any]]:
184
+ reaction_groups = record.pop("reaction_groups", None)
185
+ if reaction_groups is None:
186
+ return None
187
+ reactions: MutableMapping[str, Any] = {key: 0 for key in GRAPHQL_REACTION_TO_REST.values()}
188
+ total = 0
189
+ for group in reaction_groups:
190
+ rest_key = GRAPHQL_REACTION_TO_REST.get(group.get("content"))
191
+ if rest_key:
192
+ count = (group.get("reactors") or {}).get("totalCount", 0)
193
+ reactions[rest_key] = count
194
+ total += count
195
+ reactions["total_count"] = total
196
+ return reactions
197
+
198
+ @staticmethod
199
+ def _rest_urls(api_url: str, repository: Optional[str], release_id: Optional[int], tag_name: Optional[str]) -> Mapping[str, Any]:
200
+ upload_url = api_url.replace("api.github.com", "uploads.github.com")
201
+ return {
202
+ "url": f"{api_url}/repos/{repository}/releases/{release_id}",
203
+ "assets_url": f"{api_url}/repos/{repository}/releases/{release_id}/assets",
204
+ "upload_url": f"{upload_url}/repos/{repository}/releases/{release_id}/assets{{?name,label}}",
205
+ "tarball_url": f"{api_url}/repos/{repository}/tarball/{tag_name}" if tag_name else None,
206
+ "zipball_url": f"{api_url}/repos/{repository}/zipball/{tag_name}" if tag_name else None,
207
+ }
208
+
209
+
210
+ @dataclass
211
+ class NestedGraphQLPaginationStrategy(PaginationStrategy):
212
+ """Two-level cursor traversal for `reviews` and `issue_reactions`.
213
+
214
+ Both streams walk a repository-level connection (`pullRequests` / `issues`) whose nodes
215
+ each carry a child connection (`reviews` / `reactions`). A child connection that has more
216
+ pages cannot be paginated in place, so the legacy streams switched the query to a
217
+ drill-down rooted at that single parent (`repository.pullRequest(number:)`) and came back
218
+ to the parent listing afterwards.
219
+
220
+ All traversal state lives in the page token rather than on this object. That is not a
221
+ stylistic choice: one `PaginationStrategy` instance is shared by every partition of a
222
+ stream, and the partitions are read concurrently, so the legacy `self.reviews_cursors` /
223
+ `self.pull_requests_cursor` dicts were keyed by repository precisely to work around state
224
+ that should never have been shared. A self-contained token removes the sharing instead.
225
+
226
+ The token is `{document, after, number, pending, list_after}`:
227
+
228
+ - `document` is the query to send next, so the retriever's `request_body` only has to
229
+ choose between "the token's document" and the root listing document.
230
+ - `pending` is the queue of `(parent_number, child_cursor)` pairs still to drill into,
231
+ popped LIFO to match the legacy `dict.popitem()`.
232
+ - `list_after` is where to resume the parent listing once `pending` drains.
233
+
234
+ `first` stays a GraphQL variable in both documents so `REDUCE_PAGE_SIZE` still works; only
235
+ the parent's `number` is inlined into the drill-down document, since it is not a page size.
236
+ """
237
+
238
+ config: Config
239
+ parameters: InitVar[Mapping[str, Any]]
240
+ list_connection: str = ""
241
+ drilldown_field: str = ""
242
+ child_connection: str = ""
243
+ page_size: Optional[int] = None
244
+
245
+ # The drill-down document carries the parent's number inline. A literal marker rather than
246
+ # `str.format`, because the GraphQL body is full of braces.
247
+ NUMBER_PLACEHOLDER = "__NUMBER__"
248
+
249
+ def __post_init__(self, parameters: Mapping[str, Any]) -> None:
250
+ for field in ("list_connection", "drilldown_field", "child_connection"):
251
+ if not getattr(self, field):
252
+ raise ValueError(f"NestedGraphQLPaginationStrategy requires `{field}`")
253
+ # Read from `$parameters` rather than from a manifest field: a custom component's
254
+ # string fields are not interpolated, so a `{{ parameters[...] }}` reference would
255
+ # arrive verbatim. The stream declares each document once and both the requester's
256
+ # `request_body` and this strategy read that one declaration.
257
+ for field in ("list_document", "drilldown_document"):
258
+ value = parameters.get(field)
259
+ if not value:
260
+ raise ValueError(f"NestedGraphQLPaginationStrategy requires `{field}` in $parameters")
261
+ setattr(self, field, value)
262
+
263
+ @property
264
+ def initial_token(self) -> Optional[Any]:
265
+ return None
266
+
267
+ def get_page_size(self) -> Optional[int]:
268
+ return _resolve_page_size(self.page_size, self.config)
269
+
270
+ def next_page_token(
271
+ self,
272
+ response: requests.Response,
273
+ last_page_size: int,
274
+ last_record: Optional[Any],
275
+ last_page_token_value: Optional[Any] = None,
276
+ page_size_override: Optional[int] = None,
277
+ ) -> Optional[Mapping[str, Any]]:
278
+ previous = last_page_token_value if isinstance(last_page_token_value, Mapping) else {}
279
+ pending = [list(item) for item in previous.get("pending", [])]
280
+ list_after = previous.get("list_after")
281
+
282
+ repository = (response.json().get("data") or {}).get("repository")
283
+ if repository:
284
+ if self.list_connection in repository:
285
+ connection = repository[self.list_connection] or {}
286
+ page_info = connection.get("pageInfo") or {}
287
+ if page_info.get("hasNextPage"):
288
+ list_after = page_info.get("endCursor")
289
+ for node in connection.get("nodes") or []:
290
+ self._queue_child(node, pending)
291
+ elif self.drilldown_field in repository:
292
+ self._queue_child(repository[self.drilldown_field] or {}, pending)
293
+
294
+ if pending:
295
+ number, after = pending.pop()
296
+ return {
297
+ "document": self.drilldown_document.replace(self.NUMBER_PLACEHOLDER, str(number)),
298
+ "after": after,
299
+ "number": number,
300
+ "pending": pending,
301
+ "list_after": list_after,
302
+ }
303
+ if list_after:
304
+ return {
305
+ "document": self.list_document,
306
+ "after": list_after,
307
+ "number": None,
308
+ "pending": [],
309
+ "list_after": None,
310
+ }
311
+ return None
312
+
313
+ def _queue_child(self, node: Mapping[str, Any], pending: list) -> None:
314
+ child = node.get(self.child_connection) or {}
315
+ if (child.get("pageInfo") or {}).get("hasNextPage"):
316
+ pending.append([node.get("number"), child["pageInfo"]["endCursor"]])
317
+
318
+
319
+ @dataclass
320
+ class NestedGraphQLRecordExtractor(RecordExtractor):
321
+ """Extract child records from either shape a two-level GraphQL traversal can return.
322
+
323
+ The listing query nests the child connection under every parent node:
324
+
325
+ data.repository.<list_connection>.nodes[*].<child_connection>.nodes[*]
326
+
327
+ the drill-down query returns a single parent:
328
+
329
+ data.repository.<drilldown_field>.<child_connection>.nodes[*]
330
+
331
+ A `DpathExtractor` can express either path but not both, and the records also need fields
332
+ that only exist on the parent node (`reviews.pull_request_url` comes from the pull
333
+ request's `url`), which a path-based extractor cannot reach at all. `parent_fields` maps a
334
+ field on the parent node to the field name to copy it into.
335
+ """
336
+
337
+ config: Config
338
+ parameters: InitVar[Mapping[str, Any]]
339
+ list_connection: str = ""
340
+ drilldown_field: str = ""
341
+ child_connection: str = ""
342
+ parent_fields: Optional[Mapping[str, str]] = None
343
+
344
+ def __post_init__(self, parameters: Mapping[str, Any]) -> None:
345
+ for field in ("list_connection", "drilldown_field", "child_connection"):
346
+ if not getattr(self, field):
347
+ raise ValueError(f"NestedGraphQLRecordExtractor requires `{field}`")
348
+
349
+ def extract_records(self, response: requests.Response) -> Iterable[MutableMapping[Any, Any]]:
350
+ repository = (response.json().get("data") or {}).get("repository")
351
+ if not repository:
352
+ # GitHub answers 200 with a null repository when the token cannot see it.
353
+ return
354
+ repository_name = f"{(repository.get('owner') or {}).get('login')}/{repository.get('name')}"
355
+ if self.list_connection in repository:
356
+ parents = ((repository[self.list_connection] or {}).get("nodes")) or []
357
+ else:
358
+ parent = repository.get(self.drilldown_field)
359
+ parents = [parent] if parent else []
360
+ for parent in parents:
361
+ children = ((parent.get(self.child_connection) or {}).get("nodes")) or []
362
+ for record in children:
363
+ record["repository"] = repository_name
364
+ for source, destination in (self.parent_fields or {}).items():
365
+ record[destination] = parent.get(source)
366
+ yield record
367
+
368
+
369
+ # Depth-first order for the four-level reaction traversal: the deepest pending connection is
370
+ # always drilled into first, so a comment's reactions are finished before the next pull
371
+ # request is opened. Ported from `graphql.CursorStorage`, which built the same ordering out of
372
+ # a heap keyed on this list reversed.
373
+ _REACTION_TRAVERSAL_PRIORITY = {
374
+ "Reaction": 0,
375
+ "PullRequestReviewComment": 1,
376
+ "PullRequestReview": 2,
377
+ "PullRequest": 3,
378
+ }
379
+
380
+ # Which child connection each queued object type needs paginated, and which document roots at
381
+ # it. `PullRequest` is the repository-level listing; the rest root at `node(id:)`.
382
+ _REACTION_CONNECTION_OF = {
383
+ "PullRequest": "pullRequests",
384
+ "PullRequestReview": "reviews",
385
+ "PullRequestReviewComment": "comments",
386
+ "Reaction": "reactions",
387
+ }
388
+
389
+
390
+ @dataclass
391
+ class DeepNestedGraphQLPaginationStrategy(PaginationStrategy):
392
+ """Four-level depth-first traversal for `pull_request_comment_reactions`.
393
+
394
+ `repository.pullRequests -> reviews -> comments -> reactions`. Every level can have more
395
+ pages than the query asked for, and none of them can be paginated in place, so each
396
+ overflowing connection is queued and later re-rooted with its own document:
397
+
398
+ - `PullRequest` -> the repository listing, paginated by `pullRequests`
399
+ - `PullRequestReview`-> `node(id: <pull request>)`, paginated by `reviews`
400
+ - `PullRequestReviewComment` -> `node(id: <review>)`, paginated by `comments`
401
+ - `Reaction` -> `node(id: <comment>)`, paginated by `reactions`
402
+
403
+ The queue is ordered deepest-first, which is what makes the traversal depth-first: the
404
+ reactions of a comment are drained before the next review is opened.
405
+
406
+ Like `NestedGraphQLPaginationStrategy`, the queue lives in the page token rather than on
407
+ this object, because one strategy instance is shared by every partition of the stream and
408
+ the partitions are read concurrently. The legacy `self.cursor_storage` was a single heap
409
+ shared across repositories.
410
+
411
+ One legacy behavior is deliberately not carried over: the legacy `request_body_json` used to send
412
+ `first = min(page_size, total_count)` to avoid paying for pages larger than what remained.
413
+ `first` has to stay a GraphQL variable for `REDUCE_PAGE_SIZE` to be able to shrink it, and
414
+ a variable cannot be per-token, so the connector may now over-ask on the last page of a
415
+ connection. GitHub returns fewer records; the cost is a slightly higher query score.
416
+ """
417
+
418
+ config: Config
419
+ parameters: InitVar[Mapping[str, Any]]
420
+ documents: Optional[Mapping[str, str]] = None
421
+ page_size: Optional[int] = None
422
+
423
+ NODE_PLACEHOLDER = "__NODE_ID__"
424
+
425
+ # $parameters key holding the document that re-roots at each object type.
426
+ DOCUMENT_PARAMETERS = {
427
+ "PullRequest": "root_repository_document",
428
+ "PullRequestReview": "root_pull_request_document",
429
+ "PullRequestReviewComment": "root_review_document",
430
+ "Reaction": "root_comment_document",
431
+ }
432
+
433
+ def __post_init__(self, parameters: Mapping[str, Any]) -> None:
434
+ if self.documents is None:
435
+ # Same reason as NestedGraphQLPaginationStrategy: a custom component's fields are
436
+ # not interpolated, so the documents are read from $parameters.
437
+ self.documents = {typename: parameters[key] for typename, key in self.DOCUMENT_PARAMETERS.items() if parameters.get(key)}
438
+ missing = set(_REACTION_TRAVERSAL_PRIORITY) - set(self.documents or {})
439
+ if missing:
440
+ raise ValueError(f"DeepNestedGraphQLPaginationStrategy requires a document for every object type; missing {sorted(missing)}")
441
+
442
+ @property
443
+ def initial_token(self) -> Optional[Any]:
444
+ return None
445
+
446
+ def get_page_size(self) -> Optional[int]:
447
+ return _resolve_page_size(self.page_size, self.config)
448
+
449
+ def next_page_token(
450
+ self,
451
+ response: requests.Response,
452
+ last_page_size: int,
453
+ last_record: Optional[Any],
454
+ last_page_token_value: Optional[Any] = None,
455
+ page_size_override: Optional[int] = None,
456
+ ) -> Optional[Mapping[str, Any]]:
457
+ previous = last_page_token_value if isinstance(last_page_token_value, Mapping) else {}
458
+ pending = [list(item) for item in previous.get("pending", [])]
459
+ sequence = previous.get("sequence", 0)
460
+
461
+ data = response.json().get("data") or {}
462
+
463
+ repository = data.get("repository")
464
+ if repository:
465
+ sequence = self._queue(repository, "PullRequest", pending, sequence)
466
+ for pull_request in self._nodes(repository, "pullRequests"):
467
+ sequence = self._walk_pull_request(pull_request, pending, sequence)
468
+
469
+ node = data.get("node")
470
+ if node:
471
+ typename = node.get("__typename")
472
+ if typename == "PullRequest":
473
+ sequence = self._walk_pull_request(node, pending, sequence)
474
+ elif typename == "PullRequestReview":
475
+ sequence = self._walk_review(node, pending, sequence)
476
+ elif typename == "PullRequestReviewComment":
477
+ sequence = self._queue(node, "Reaction", pending, sequence)
478
+
479
+ if not pending:
480
+ return None
481
+
482
+ # Deepest first, and first-queued first within a depth -- the ordering the heap gave.
483
+ pending.sort(key=lambda item: (item[0], item[1]))
484
+ _, _, typename, cursor, node_id = pending.pop(0)
485
+ document = self.documents[typename] # type: ignore[index]
486
+ if node_id is not None:
487
+ document = document.replace(self.NODE_PLACEHOLDER, str(node_id))
488
+ return {"document": document, "after": cursor, "typename": typename, "pending": pending, "sequence": sequence}
489
+
490
+ def _walk_pull_request(self, pull_request: Mapping[str, Any], pending: list, sequence: int) -> int:
491
+ sequence = self._queue(pull_request, "PullRequestReview", pending, sequence)
492
+ for review in self._nodes(pull_request, "reviews"):
493
+ sequence = self._walk_review(review, pending, sequence)
494
+ return sequence
495
+
496
+ def _walk_review(self, review: Mapping[str, Any], pending: list, sequence: int) -> int:
497
+ sequence = self._queue(review, "PullRequestReviewComment", pending, sequence)
498
+ for comment in self._nodes(review, "comments"):
499
+ sequence = self._queue(comment, "Reaction", pending, sequence)
500
+ return sequence
501
+
502
+ def _queue(self, node: Mapping[str, Any], typename: str, pending: list, sequence: int) -> int:
503
+ """Queue `node`'s child connection if it has another page."""
504
+ connection = node.get(_REACTION_CONNECTION_OF[typename]) or {}
505
+ page_info = connection.get("pageInfo") or {}
506
+ if not page_info.get("hasNextPage"):
507
+ return sequence
508
+ # `PullRequest` re-roots at the repository listing, which needs no node id.
509
+ node_id = None if typename == "PullRequest" else node.get("node_id")
510
+ pending.append([_REACTION_TRAVERSAL_PRIORITY[typename], sequence, typename, page_info.get("endCursor"), node_id])
511
+ return sequence + 1
512
+
513
+ @staticmethod
514
+ def _nodes(node: Mapping[str, Any], connection: str) -> Iterable[Mapping[str, Any]]:
515
+ return ((node.get(connection) or {}).get("nodes")) or []
516
+
517
+
518
+ @dataclass
519
+ class DeepNestedGraphQLRecordExtractor(RecordExtractor):
520
+ """Extract reaction records from any of the four roots the traversal can return.
521
+
522
+ Ported from `streams.PullRequestCommentReactions.parse_response`. Each reaction is stamped
523
+ with its repository and the id of the comment it belongs to; `user.type` is set because the
524
+ legacy record carried it and the GraphQL `user` field here is not a union.
525
+ """
526
+
527
+ config: Config
528
+ parameters: InitVar[Mapping[str, Any]]
529
+
530
+ def __post_init__(self, parameters: Mapping[str, Any]) -> None:
531
+ pass
532
+
533
+ def extract_records(self, response: requests.Response) -> Iterable[MutableMapping[Any, Any]]:
534
+ data = response.json().get("data") or {}
535
+
536
+ repository = data.get("repository")
537
+ if repository:
538
+ for pull_request in self._nodes(repository, "pullRequests"):
539
+ yield from self._from_pull_request(pull_request, repository)
540
+
541
+ node = data.get("node")
542
+ if node:
543
+ # The drill-down documents select the repository alongside the node, so the record
544
+ # can still be stamped with it.
545
+ repository = node.get("repository") or {}
546
+ typename = node.get("__typename")
547
+ if typename == "PullRequest":
548
+ yield from self._from_pull_request(node, repository)
549
+ elif typename == "PullRequestReview":
550
+ yield from self._from_review(node, repository)
551
+ elif typename == "PullRequestReviewComment":
552
+ yield from self._from_comment(node, repository)
553
+
554
+ def _from_pull_request(self, pull_request: Mapping[str, Any], repository: Mapping[str, Any]):
555
+ for review in self._nodes(pull_request, "reviews"):
556
+ yield from self._from_review(review, repository)
557
+
558
+ def _from_review(self, review: Mapping[str, Any], repository: Mapping[str, Any]):
559
+ for comment in self._nodes(review, "comments"):
560
+ yield from self._from_comment(comment, repository)
561
+
562
+ def _from_comment(self, comment: Mapping[str, Any], repository: Mapping[str, Any]):
563
+ repository_name = f"{(repository.get('owner') or {}).get('login')}/{repository.get('name')}"
564
+ for reaction in self._nodes(comment, "reactions"):
565
+ reaction["repository"] = repository_name
566
+ reaction["comment_id"] = comment.get("id")
567
+ if reaction.get("user"):
568
+ reaction["user"]["type"] = "User"
569
+ yield reaction
570
+
571
+ @staticmethod
572
+ def _nodes(node: Mapping[str, Any], connection: str) -> Iterable[Mapping[str, Any]]:
573
+ return ((node.get(connection) or {}).get("nodes")) or []
18
574
 
19
575
 
20
576
  @dataclass