airbyte-source-github 2.5.0__tar.gz → 2.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/PKG-INFO +1 -1
  2. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/pyproject.toml +1 -1
  3. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/backoff_strategies.py +0 -9
  4. airbyte_source_github-2.6.0/source_github/components.py +188 -0
  5. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/errors_handlers.py +1 -22
  6. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/manifest.yaml +1827 -0
  7. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/source.py +0 -10
  8. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/streams.py +2 -357
  9. airbyte_source_github-2.5.0/source_github/components.py +0 -83
  10. airbyte_source_github-2.5.0/source_github/schemas/commits.json +0 -161
  11. airbyte_source_github-2.5.0/source_github/schemas/contributor_activity.json +0 -133
  12. airbyte_source_github-2.5.0/source_github/schemas/workflow_jobs.json +0 -139
  13. airbyte_source_github-2.5.0/source_github/schemas/workflow_runs.json +0 -585
  14. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/.airbyte-pypi-readme.md +0 -0
  15. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/__init__.py +0 -0
  16. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/config_migrations.py +0 -0
  17. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/constants.py +0 -0
  18. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/github_schema.py +0 -0
  19. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/graphql.py +0 -0
  20. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/run.py +0 -0
  21. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/commit_comments.json +0 -0
  22. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/issue_reactions.json +0 -0
  23. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/issue_timeline_events.json +0 -0
  24. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/projects.json +0 -0
  25. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/projects_v2.json +0 -0
  26. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/pull_request_comment_reactions.json +0 -0
  27. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/pull_request_stats.json +0 -0
  28. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/pull_requests.json +0 -0
  29. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/releases.json +0 -0
  30. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/reviews.json +0 -0
  31. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/events/comment.json +0 -0
  32. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/events/commented.json +0 -0
  33. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/events/committed.json +0 -0
  34. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/events/cross_referenced.json +0 -0
  35. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/events/reviewed.json +0 -0
  36. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/reaction.json +0 -0
  37. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/reactions.json +0 -0
  38. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/user.json +0 -0
  39. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/schemas/shared/user_graphql.json +0 -0
  40. {airbyte_source_github-2.5.0 → airbyte_source_github-2.6.0}/source_github/utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: airbyte-source-github
3
- Version: 2.5.0
3
+ Version: 2.6.0
4
4
  Summary: Source implementation for GitHub.
5
5
  Home-page: https://airbyte.com
6
6
  License: ELv2
@@ -3,7 +3,7 @@ requires = [ "poetry-core>=1.0.0",]
3
3
  build-backend = "poetry.core.masonry.api"
4
4
 
5
5
  [tool.poetry]
6
- version = "2.5.0"
6
+ version = "2.6.0"
7
7
  name = "airbyte-source-github"
8
8
  description = "Source implementation for GitHub."
9
9
  authors = [ "Airbyte <contact@airbyte.io>",]
@@ -50,12 +50,3 @@ class GithubStreamABCBackoffStrategy(BackoffStrategy):
50
50
  return None
51
51
  return wait_time
52
52
  return None
53
-
54
-
55
- class ContributorActivityBackoffStrategy(BackoffStrategy):
56
- def backoff_time(
57
- self, response_or_exception: Optional[Union[requests.Response, requests.RequestException]], **kwargs: Any
58
- ) -> Optional[float]:
59
- if isinstance(response_or_exception, requests.Response) and response_or_exception.status_code == requests.codes.ACCEPTED:
60
- return 90
61
- return None
@@ -0,0 +1,188 @@
1
+ #
2
+ # Copyright (c) 2023 Airbyte, Inc., all rights reserved.
3
+ #
4
+
5
+ from dataclasses import dataclass
6
+ from datetime import timedelta
7
+ from itertools import groupby
8
+ from typing import Any, Iterable, List, Mapping, MutableMapping, Optional
9
+
10
+ import requests
11
+
12
+ from airbyte_cdk.sources.declarative.extractors.record_extractor import RecordExtractor
13
+ from airbyte_cdk.sources.declarative.migrations.state_migration import StateMigration
14
+ from airbyte_cdk.sources.declarative.partition_routers.substream_partition_router import SubstreamPartitionRouter
15
+ from airbyte_cdk.sources.declarative.requesters.paginators.strategies.cursor_pagination_strategy import CursorPaginationStrategy
16
+ from airbyte_cdk.sources.types import Config, Record, StreamSlice
17
+ from airbyte_cdk.utils.datetime_helpers import ab_datetime_parse
18
+
19
+
20
+ @dataclass
21
+ class IssueTimelineEventsExtractor(RecordExtractor):
22
+ """Collapse an issue's timeline page into one record keyed by event type.
23
+
24
+ The legacy `IssueTimelineEvents.parse_response` yielded `{<event type>: <event>, ...}` per page,
25
+ the last event of each type winning, and a bare record when the body was not a list.
26
+ `repository` and `issue_number` are added by the stream's transformations.
27
+ """
28
+
29
+ config: Config
30
+ parameters: Mapping[str, Any]
31
+
32
+ def extract_records(self, response: requests.Response) -> Iterable[Mapping[str, Any]]:
33
+ try:
34
+ body = response.json()
35
+ except ValueError:
36
+ body = None
37
+ record: MutableMapping[str, Any] = {}
38
+ if isinstance(body, list):
39
+ for event in body:
40
+ if isinstance(event, Mapping) and "event" in event:
41
+ record[event["event"]] = event
42
+ yield record
43
+
44
+
45
+ @dataclass
46
+ class NestedLegacyToPerPartitionStateMigration(StateMigration):
47
+ """Migrate the nested state the Python parent-child streams wrote to per-partition state.
48
+
49
+ Legacy shape: `{repository: {<parent id>: {cursor_field: value}}}`, one more level of ids per
50
+ parent (`project_cards` nests `project_id` then `column_id`). `partition_fields` lists those id
51
+ levels, outermost first. Ids were stored as strings; the declarative partitions carry them as
52
+ the integers GitHub returns.
53
+ """
54
+
55
+ config: Config
56
+ parameters: Mapping[str, Any]
57
+ cursor_field: str
58
+ partition_fields: List[str]
59
+ # GitHub ids were stored as strings and the routers carry them as integers; branch names stay strings.
60
+ integer_ids: bool = True
61
+
62
+ def should_migrate(self, stream_state: Mapping[str, Any]) -> bool:
63
+ if not stream_state or "states" in stream_state or "state" in stream_state:
64
+ return False
65
+ return all(self._is_legacy_repository_state(value) for value in stream_state.values())
66
+
67
+ def _is_legacy_repository_state(self, node: Any, depth: int = 0) -> bool:
68
+ if not isinstance(node, Mapping) or not node:
69
+ return False
70
+ if depth == len(self.partition_fields):
71
+ return set(node) == {self.cursor_field}
72
+ return all(self._is_legacy_repository_state(child, depth + 1) for child in node.values())
73
+
74
+ def migrate(self, stream_state: Mapping[str, Any]) -> Mapping[str, Any]:
75
+ states = []
76
+ for repository, node in stream_state.items():
77
+ self._collect(node, {"repository": repository}, 0, states)
78
+ return {"states": states}
79
+
80
+ def _collect(self, node: Mapping[str, Any], partition: Mapping[str, Any], depth: int, states: List[Mapping[str, Any]]) -> None:
81
+ if depth == len(self.partition_fields):
82
+ states.append({"partition": partition, "cursor": {self.cursor_field: node[self.cursor_field]}})
83
+ return
84
+ for key, child in node.items():
85
+ child_partition = {self.partition_fields[depth]: self._to_id(key), "parent_slice": partition}
86
+ self._collect(child, child_partition, depth + 1, states)
87
+
88
+ def _to_id(self, key: str) -> Any:
89
+ return int(key) if self.integer_ids and isinstance(key, str) and key.isdigit() else key
90
+
91
+
92
+ @dataclass
93
+ class WorkflowJobsLegacyStateMigration(StateMigration):
94
+ """Migrate the legacy `{repository: {completed_at: value}}` state of `workflow_jobs`.
95
+
96
+ The declarative stream keeps one global `completed_at` cursor and lets its parent
97
+ (`workflow_runs`) resume per repository, which is what the Python class did by handing its own
98
+ cursor to the parent as `updated_at`. The lowest repository cursor becomes the global one so no
99
+ repository skips jobs; the parent state keeps the per-repository values.
100
+ """
101
+
102
+ config: Config
103
+ parameters: Mapping[str, Any]
104
+
105
+ def should_migrate(self, stream_state: Mapping[str, Any]) -> bool:
106
+ if not stream_state or "state" in stream_state or "states" in stream_state or "parent_state" in stream_state:
107
+ return False
108
+ return all(isinstance(value, Mapping) and set(value) == {"completed_at"} for value in stream_state.values())
109
+
110
+ def migrate(self, stream_state: Mapping[str, Any]) -> Mapping[str, Any]:
111
+ cursors = {repository: value["completed_at"] for repository, value in stream_state.items()}
112
+ return {
113
+ "use_global_cursor": True,
114
+ "state": {"completed_at": min(cursors.values(), key=ab_datetime_parse)},
115
+ "parent_state": {"workflow_runs": {repository: {"updated_at": cursor} for repository, cursor in cursors.items()}},
116
+ }
117
+
118
+
119
+ @dataclass
120
+ class CommitsBranchPartitionRouter(SubstreamPartitionRouter):
121
+ """One partition per branch to pull commits from, resolved the way `Commits.stream_slices` did.
122
+
123
+ The parent lists every branch of every repository. For a repository, the configured `branches`
124
+ entries (`owner/repo/branch`) that exist are used; when none is configured, or none of the
125
+ configured ones exists, the repository's default branch is used instead.
126
+ """
127
+
128
+ def stream_slices(self) -> Iterable[StreamSlice]:
129
+ configured = set(self.config.get("branches") or [])
130
+ # groupby groups only adjacent items. Branch slices arrive repo-by-repo because
131
+ # SubstreamPartitionRouter iterates parent partitions outer / parent records inner, and
132
+ # repository_partition_router emits each repository once -- UnionPartitionRouter dedupes
133
+ # partition values (union_partition_router.py:55-70), so a repository matched by both an
134
+ # explicit entry and a wildcard is not visited twice. Without that, each duplicate group
135
+ # would re-emit the same branch partitions.
136
+ for repository, branch_slices in groupby(super().stream_slices(), key=lambda s: s.partition["parent_slice"]["repository"]):
137
+ branch_slices = list(branch_slices)
138
+ wanted = [s for s in branch_slices if f"{repository}/{s.partition['branch']}" in configured]
139
+ if not wanted:
140
+ default_branch = next((s.extra_fields.get("default_branch") for s in branch_slices), None)
141
+ wanted = [s for s in branch_slices if s.partition["branch"] == default_branch]
142
+ if not wanted and default_branch:
143
+ wanted = [
144
+ StreamSlice(partition={"branch": default_branch, "parent_slice": {"repository": repository}}, cursor_slice={})
145
+ ]
146
+ yield from wanted
147
+
148
+
149
+ @dataclass
150
+ class WorkflowRunsPaginationStrategy(CursorPaginationStrategy):
151
+ """Stop paging once the page's oldest run was created more than 32 days before the slice start.
152
+
153
+ Runs are listed newest-created first and can be re-run for 32 days, so nothing older can still
154
+ change: the legacy `WorkflowRuns.read_records` broke out of the page loop there.
155
+
156
+ This is a workaround for a CDK gap, not a GitHub quirk. `CursorPaginationStrategy` already
157
+ exposes the decoded page to `stop_condition`, but the paginator's interpolation context has no
158
+ `stream_slice` (only `config`, `response`, `headers`, `last_record`, `last_page_size`), so the
159
+ slice start cannot be compared against the page from YAML. Until that lands
160
+ (https://github.com/airbytehq/airbyte-python-cdk/issues/1166), the requester injects the slice
161
+ start as a request header GitHub ignores and this strategy reads it back from
162
+ `response.request`. Once `stream_slice` is available, delete this class and the header and use
163
+ a `stop_condition` on the built-in strategy instead.
164
+ """
165
+
166
+ window_header: str = "X-Airbyte-Window-Start"
167
+ re_run_period_days: int = 32
168
+
169
+ def next_page_token(
170
+ self,
171
+ response: requests.Response,
172
+ last_page_size: int,
173
+ last_record: Optional[Record],
174
+ last_page_token_value: Optional[Any] = None,
175
+ ) -> Optional[Any]:
176
+ window_start = response.request.headers.get(self.window_header) if response.request else None
177
+ try:
178
+ runs = (response.json() or {}).get("workflow_runs") or []
179
+ except ValueError:
180
+ runs = []
181
+ oldest = runs[-1].get("created_at") if runs and isinstance(runs[-1], Mapping) else None
182
+ if (
183
+ window_start
184
+ and oldest
185
+ and ab_datetime_parse(oldest) < ab_datetime_parse(window_start) - timedelta(days=self.re_run_period_days)
186
+ ):
187
+ return None
188
+ return super().next_page_token(response, last_page_size, last_record, last_page_token_value)
@@ -9,7 +9,7 @@ import requests
9
9
 
10
10
  from airbyte_cdk.models import FailureType
11
11
  from airbyte_cdk.sources.streams.http import HttpStream
12
- from airbyte_cdk.sources.streams.http.error_handlers import ErrorHandler, ErrorResolution, HttpStatusErrorHandler, ResponseAction
12
+ from airbyte_cdk.sources.streams.http.error_handlers import ErrorResolution, HttpStatusErrorHandler, ResponseAction
13
13
  from airbyte_cdk.sources.streams.http.error_handlers.default_error_mapping import DEFAULT_ERROR_MAPPING
14
14
 
15
15
  from . import constants
@@ -174,27 +174,6 @@ class GithubStreamABCErrorHandler(HttpStatusErrorHandler):
174
174
  return super().interpret_response(response_or_exception)
175
175
 
176
176
 
177
- class ContributorActivityErrorHandler(GithubStreamABCErrorHandler):
178
- """
179
- This custom error handler is needed for streams based on repository statistics endpoints like ContributorActivity because
180
- when requesting data that hasn't been cached yet when the request is made, you'll receive a 202 response. And these requests
181
- need to retried to get the actual results.
182
-
183
- See the docs for more info:
184
- https://docs.github.com/en/rest/metrics/statistics?apiVersion=2022-11-28#a-word-about-caching
185
- """
186
-
187
- def interpret_response(self, response_or_exception: Optional[Union[requests.Response, Exception]] = None) -> ErrorResolution:
188
- if isinstance(response_or_exception, requests.Response) and response_or_exception.status_code == requests.codes.ACCEPTED:
189
- return ErrorResolution(
190
- response_action=ResponseAction.RETRY,
191
- failure_type=FailureType.transient_error,
192
- error_message=f"Response status code: {response_or_exception.status_code}. Retrying...",
193
- )
194
-
195
- return super().interpret_response(response_or_exception)
196
-
197
-
198
177
  class GitHubGraphQLErrorHandler(GithubStreamABCErrorHandler):
199
178
  def _safe_json_get_errors(self, response: requests.Response) -> bool:
200
179
  try: