airbyte-source-github 2.5.0__tar.gz → 2.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/PKG-INFO +1 -1
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/pyproject.toml +1 -1
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/backoff_strategies.py +0 -9
- airbyte_source_github-2.6.1/source_github/components.py +188 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/errors_handlers.py +1 -22
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/manifest.yaml +1837 -6
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/source.py +0 -10
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/streams.py +2 -357
- airbyte_source_github-2.5.0/source_github/components.py +0 -83
- airbyte_source_github-2.5.0/source_github/schemas/commits.json +0 -161
- airbyte_source_github-2.5.0/source_github/schemas/contributor_activity.json +0 -133
- airbyte_source_github-2.5.0/source_github/schemas/workflow_jobs.json +0 -139
- airbyte_source_github-2.5.0/source_github/schemas/workflow_runs.json +0 -585
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/.airbyte-pypi-readme.md +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/__init__.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/config_migrations.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/constants.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/github_schema.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/graphql.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/run.py +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/commit_comments.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/issue_reactions.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/issue_timeline_events.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/projects.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/projects_v2.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/pull_request_comment_reactions.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/pull_request_stats.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/pull_requests.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/releases.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/reviews.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/events/comment.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/events/commented.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/events/committed.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/events/cross_referenced.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/events/reviewed.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/reaction.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/reactions.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/user.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/schemas/shared/user_graphql.json +0 -0
- {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/utils.py +0 -0
|
@@ -3,7 +3,7 @@ requires = [ "poetry-core>=1.0.0",]
|
|
|
3
3
|
build-backend = "poetry.core.masonry.api"
|
|
4
4
|
|
|
5
5
|
[tool.poetry]
|
|
6
|
-
version = "2.
|
|
6
|
+
version = "2.6.1"
|
|
7
7
|
name = "airbyte-source-github"
|
|
8
8
|
description = "Source implementation for GitHub."
|
|
9
9
|
authors = [ "Airbyte <contact@airbyte.io>",]
|
{airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/backoff_strategies.py
RENAMED
|
@@ -50,12 +50,3 @@ class GithubStreamABCBackoffStrategy(BackoffStrategy):
|
|
|
50
50
|
return None
|
|
51
51
|
return wait_time
|
|
52
52
|
return None
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
class ContributorActivityBackoffStrategy(BackoffStrategy):
|
|
56
|
-
def backoff_time(
|
|
57
|
-
self, response_or_exception: Optional[Union[requests.Response, requests.RequestException]], **kwargs: Any
|
|
58
|
-
) -> Optional[float]:
|
|
59
|
-
if isinstance(response_or_exception, requests.Response) and response_or_exception.status_code == requests.codes.ACCEPTED:
|
|
60
|
-
return 90
|
|
61
|
-
return None
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright (c) 2023 Airbyte, Inc., all rights reserved.
|
|
3
|
+
#
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from datetime import timedelta
|
|
7
|
+
from itertools import groupby
|
|
8
|
+
from typing import Any, Iterable, List, Mapping, MutableMapping, Optional
|
|
9
|
+
|
|
10
|
+
import requests
|
|
11
|
+
|
|
12
|
+
from airbyte_cdk.sources.declarative.extractors.record_extractor import RecordExtractor
|
|
13
|
+
from airbyte_cdk.sources.declarative.migrations.state_migration import StateMigration
|
|
14
|
+
from airbyte_cdk.sources.declarative.partition_routers.substream_partition_router import SubstreamPartitionRouter
|
|
15
|
+
from airbyte_cdk.sources.declarative.requesters.paginators.strategies.cursor_pagination_strategy import CursorPaginationStrategy
|
|
16
|
+
from airbyte_cdk.sources.types import Config, Record, StreamSlice
|
|
17
|
+
from airbyte_cdk.utils.datetime_helpers import ab_datetime_parse
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class IssueTimelineEventsExtractor(RecordExtractor):
|
|
22
|
+
"""Collapse an issue's timeline page into one record keyed by event type.
|
|
23
|
+
|
|
24
|
+
The legacy `IssueTimelineEvents.parse_response` yielded `{<event type>: <event>, ...}` per page,
|
|
25
|
+
the last event of each type winning, and a bare record when the body was not a list.
|
|
26
|
+
`repository` and `issue_number` are added by the stream's transformations.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
config: Config
|
|
30
|
+
parameters: Mapping[str, Any]
|
|
31
|
+
|
|
32
|
+
def extract_records(self, response: requests.Response) -> Iterable[Mapping[str, Any]]:
|
|
33
|
+
try:
|
|
34
|
+
body = response.json()
|
|
35
|
+
except ValueError:
|
|
36
|
+
body = None
|
|
37
|
+
record: MutableMapping[str, Any] = {}
|
|
38
|
+
if isinstance(body, list):
|
|
39
|
+
for event in body:
|
|
40
|
+
if isinstance(event, Mapping) and "event" in event:
|
|
41
|
+
record[event["event"]] = event
|
|
42
|
+
yield record
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass
|
|
46
|
+
class NestedLegacyToPerPartitionStateMigration(StateMigration):
|
|
47
|
+
"""Migrate the nested state the Python parent-child streams wrote to per-partition state.
|
|
48
|
+
|
|
49
|
+
Legacy shape: `{repository: {<parent id>: {cursor_field: value}}}`, one more level of ids per
|
|
50
|
+
parent (`project_cards` nests `project_id` then `column_id`). `partition_fields` lists those id
|
|
51
|
+
levels, outermost first. Ids were stored as strings; the declarative partitions carry them as
|
|
52
|
+
the integers GitHub returns.
|
|
53
|
+
"""
|
|
54
|
+
|
|
55
|
+
config: Config
|
|
56
|
+
parameters: Mapping[str, Any]
|
|
57
|
+
cursor_field: str
|
|
58
|
+
partition_fields: List[str]
|
|
59
|
+
# GitHub ids were stored as strings and the routers carry them as integers; branch names stay strings.
|
|
60
|
+
integer_ids: bool = True
|
|
61
|
+
|
|
62
|
+
def should_migrate(self, stream_state: Mapping[str, Any]) -> bool:
|
|
63
|
+
if not stream_state or "states" in stream_state or "state" in stream_state:
|
|
64
|
+
return False
|
|
65
|
+
return all(self._is_legacy_repository_state(value) for value in stream_state.values())
|
|
66
|
+
|
|
67
|
+
def _is_legacy_repository_state(self, node: Any, depth: int = 0) -> bool:
|
|
68
|
+
if not isinstance(node, Mapping) or not node:
|
|
69
|
+
return False
|
|
70
|
+
if depth == len(self.partition_fields):
|
|
71
|
+
return set(node) == {self.cursor_field}
|
|
72
|
+
return all(self._is_legacy_repository_state(child, depth + 1) for child in node.values())
|
|
73
|
+
|
|
74
|
+
def migrate(self, stream_state: Mapping[str, Any]) -> Mapping[str, Any]:
|
|
75
|
+
states = []
|
|
76
|
+
for repository, node in stream_state.items():
|
|
77
|
+
self._collect(node, {"repository": repository}, 0, states)
|
|
78
|
+
return {"states": states}
|
|
79
|
+
|
|
80
|
+
def _collect(self, node: Mapping[str, Any], partition: Mapping[str, Any], depth: int, states: List[Mapping[str, Any]]) -> None:
|
|
81
|
+
if depth == len(self.partition_fields):
|
|
82
|
+
states.append({"partition": partition, "cursor": {self.cursor_field: node[self.cursor_field]}})
|
|
83
|
+
return
|
|
84
|
+
for key, child in node.items():
|
|
85
|
+
child_partition = {self.partition_fields[depth]: self._to_id(key), "parent_slice": partition}
|
|
86
|
+
self._collect(child, child_partition, depth + 1, states)
|
|
87
|
+
|
|
88
|
+
def _to_id(self, key: str) -> Any:
|
|
89
|
+
return int(key) if self.integer_ids and isinstance(key, str) and key.isdigit() else key
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass
|
|
93
|
+
class WorkflowJobsLegacyStateMigration(StateMigration):
|
|
94
|
+
"""Migrate the legacy `{repository: {completed_at: value}}` state of `workflow_jobs`.
|
|
95
|
+
|
|
96
|
+
The declarative stream keeps one global `completed_at` cursor and lets its parent
|
|
97
|
+
(`workflow_runs`) resume per repository, which is what the Python class did by handing its own
|
|
98
|
+
cursor to the parent as `updated_at`. The lowest repository cursor becomes the global one so no
|
|
99
|
+
repository skips jobs; the parent state keeps the per-repository values.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
config: Config
|
|
103
|
+
parameters: Mapping[str, Any]
|
|
104
|
+
|
|
105
|
+
def should_migrate(self, stream_state: Mapping[str, Any]) -> bool:
|
|
106
|
+
if not stream_state or "state" in stream_state or "states" in stream_state or "parent_state" in stream_state:
|
|
107
|
+
return False
|
|
108
|
+
return all(isinstance(value, Mapping) and set(value) == {"completed_at"} for value in stream_state.values())
|
|
109
|
+
|
|
110
|
+
def migrate(self, stream_state: Mapping[str, Any]) -> Mapping[str, Any]:
|
|
111
|
+
cursors = {repository: value["completed_at"] for repository, value in stream_state.items()}
|
|
112
|
+
return {
|
|
113
|
+
"use_global_cursor": True,
|
|
114
|
+
"state": {"completed_at": min(cursors.values(), key=ab_datetime_parse)},
|
|
115
|
+
"parent_state": {"workflow_runs": {repository: {"updated_at": cursor} for repository, cursor in cursors.items()}},
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
@dataclass
|
|
120
|
+
class CommitsBranchPartitionRouter(SubstreamPartitionRouter):
|
|
121
|
+
"""One partition per branch to pull commits from, resolved the way `Commits.stream_slices` did.
|
|
122
|
+
|
|
123
|
+
The parent lists every branch of every repository. For a repository, the configured `branches`
|
|
124
|
+
entries (`owner/repo/branch`) that exist are used; when none is configured, or none of the
|
|
125
|
+
configured ones exists, the repository's default branch is used instead.
|
|
126
|
+
"""
|
|
127
|
+
|
|
128
|
+
def stream_slices(self) -> Iterable[StreamSlice]:
|
|
129
|
+
configured = set(self.config.get("branches") or [])
|
|
130
|
+
# groupby groups only adjacent items. Branch slices arrive repo-by-repo because
|
|
131
|
+
# SubstreamPartitionRouter iterates parent partitions outer / parent records inner, and
|
|
132
|
+
# repository_partition_router emits each repository once -- UnionPartitionRouter dedupes
|
|
133
|
+
# partition values (union_partition_router.py:55-70), so a repository matched by both an
|
|
134
|
+
# explicit entry and a wildcard is not visited twice. Without that, each duplicate group
|
|
135
|
+
# would re-emit the same branch partitions.
|
|
136
|
+
for repository, branch_slices in groupby(super().stream_slices(), key=lambda s: s.partition["parent_slice"]["repository"]):
|
|
137
|
+
branch_slices = list(branch_slices)
|
|
138
|
+
wanted = [s for s in branch_slices if f"{repository}/{s.partition['branch']}" in configured]
|
|
139
|
+
if not wanted:
|
|
140
|
+
default_branch = next((s.extra_fields.get("default_branch") for s in branch_slices), None)
|
|
141
|
+
wanted = [s for s in branch_slices if s.partition["branch"] == default_branch]
|
|
142
|
+
if not wanted and default_branch:
|
|
143
|
+
wanted = [
|
|
144
|
+
StreamSlice(partition={"branch": default_branch, "parent_slice": {"repository": repository}}, cursor_slice={})
|
|
145
|
+
]
|
|
146
|
+
yield from wanted
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@dataclass
|
|
150
|
+
class WorkflowRunsPaginationStrategy(CursorPaginationStrategy):
|
|
151
|
+
"""Stop paging once the page's oldest run was created more than 32 days before the slice start.
|
|
152
|
+
|
|
153
|
+
Runs are listed newest-created first and can be re-run for 32 days, so nothing older can still
|
|
154
|
+
change: the legacy `WorkflowRuns.read_records` broke out of the page loop there.
|
|
155
|
+
|
|
156
|
+
This is a workaround for a CDK gap, not a GitHub quirk. `CursorPaginationStrategy` already
|
|
157
|
+
exposes the decoded page to `stop_condition`, but the paginator's interpolation context has no
|
|
158
|
+
`stream_slice` (only `config`, `response`, `headers`, `last_record`, `last_page_size`), so the
|
|
159
|
+
slice start cannot be compared against the page from YAML. Until that lands
|
|
160
|
+
(https://github.com/airbytehq/airbyte-python-cdk/issues/1166), the requester injects the slice
|
|
161
|
+
start as a request header GitHub ignores and this strategy reads it back from
|
|
162
|
+
`response.request`. Once `stream_slice` is available, delete this class and the header and use
|
|
163
|
+
a `stop_condition` on the built-in strategy instead.
|
|
164
|
+
"""
|
|
165
|
+
|
|
166
|
+
window_header: str = "X-Airbyte-Window-Start"
|
|
167
|
+
re_run_period_days: int = 32
|
|
168
|
+
|
|
169
|
+
def next_page_token(
|
|
170
|
+
self,
|
|
171
|
+
response: requests.Response,
|
|
172
|
+
last_page_size: int,
|
|
173
|
+
last_record: Optional[Record],
|
|
174
|
+
last_page_token_value: Optional[Any] = None,
|
|
175
|
+
) -> Optional[Any]:
|
|
176
|
+
window_start = response.request.headers.get(self.window_header) if response.request else None
|
|
177
|
+
try:
|
|
178
|
+
runs = (response.json() or {}).get("workflow_runs") or []
|
|
179
|
+
except ValueError:
|
|
180
|
+
runs = []
|
|
181
|
+
oldest = runs[-1].get("created_at") if runs and isinstance(runs[-1], Mapping) else None
|
|
182
|
+
if (
|
|
183
|
+
window_start
|
|
184
|
+
and oldest
|
|
185
|
+
and ab_datetime_parse(oldest) < ab_datetime_parse(window_start) - timedelta(days=self.re_run_period_days)
|
|
186
|
+
):
|
|
187
|
+
return None
|
|
188
|
+
return super().next_page_token(response, last_page_size, last_record, last_page_token_value)
|
{airbyte_source_github-2.5.0 → airbyte_source_github-2.6.1}/source_github/errors_handlers.py
RENAMED
|
@@ -9,7 +9,7 @@ import requests
|
|
|
9
9
|
|
|
10
10
|
from airbyte_cdk.models import FailureType
|
|
11
11
|
from airbyte_cdk.sources.streams.http import HttpStream
|
|
12
|
-
from airbyte_cdk.sources.streams.http.error_handlers import
|
|
12
|
+
from airbyte_cdk.sources.streams.http.error_handlers import ErrorResolution, HttpStatusErrorHandler, ResponseAction
|
|
13
13
|
from airbyte_cdk.sources.streams.http.error_handlers.default_error_mapping import DEFAULT_ERROR_MAPPING
|
|
14
14
|
|
|
15
15
|
from . import constants
|
|
@@ -174,27 +174,6 @@ class GithubStreamABCErrorHandler(HttpStatusErrorHandler):
|
|
|
174
174
|
return super().interpret_response(response_or_exception)
|
|
175
175
|
|
|
176
176
|
|
|
177
|
-
class ContributorActivityErrorHandler(GithubStreamABCErrorHandler):
|
|
178
|
-
"""
|
|
179
|
-
This custom error handler is needed for streams based on repository statistics endpoints like ContributorActivity because
|
|
180
|
-
when requesting data that hasn't been cached yet when the request is made, you'll receive a 202 response. And these requests
|
|
181
|
-
need to retried to get the actual results.
|
|
182
|
-
|
|
183
|
-
See the docs for more info:
|
|
184
|
-
https://docs.github.com/en/rest/metrics/statistics?apiVersion=2022-11-28#a-word-about-caching
|
|
185
|
-
"""
|
|
186
|
-
|
|
187
|
-
def interpret_response(self, response_or_exception: Optional[Union[requests.Response, Exception]] = None) -> ErrorResolution:
|
|
188
|
-
if isinstance(response_or_exception, requests.Response) and response_or_exception.status_code == requests.codes.ACCEPTED:
|
|
189
|
-
return ErrorResolution(
|
|
190
|
-
response_action=ResponseAction.RETRY,
|
|
191
|
-
failure_type=FailureType.transient_error,
|
|
192
|
-
error_message=f"Response status code: {response_or_exception.status_code}. Retrying...",
|
|
193
|
-
)
|
|
194
|
-
|
|
195
|
-
return super().interpret_response(response_or_exception)
|
|
196
|
-
|
|
197
|
-
|
|
198
177
|
class GitHubGraphQLErrorHandler(GithubStreamABCErrorHandler):
|
|
199
178
|
def _safe_json_get_errors(self, response: requests.Response) -> bool:
|
|
200
179
|
try:
|