afly 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. afly/__init__.py +5 -0
  2. afly/alerting/__init__.py +3 -0
  3. afly/alerting/webhook.py +92 -0
  4. afly/appsflyer/__init__.py +1 -0
  5. afly/appsflyer/client.py +83 -0
  6. afly/appsflyer/errors.py +204 -0
  7. afly/appsflyer/mng_api.py +98 -0
  8. afly/appsflyer/pull_api.py +158 -0
  9. afly/appsflyer/retry.py +95 -0
  10. afly/cli/__init__.py +1 -0
  11. afly/cli/_output.py +78 -0
  12. afly/cli/_project.py +114 -0
  13. afly/cli/assets/claude/CLAUDE.section.md +86 -0
  14. afly/cli/assets/claude/rules/cli.md +246 -0
  15. afly/cli/assets/claude/rules/extracts.md +130 -0
  16. afly/cli/assets/claude/rules/formats.md +116 -0
  17. afly/cli/assets/claude/rules/idempotency.md +200 -0
  18. afly/cli/assets/claude/rules/overview.md +76 -0
  19. afly/cli/assets/claude/rules/project.md +147 -0
  20. afly/cli/assets/claude/rules/quotas.md +179 -0
  21. afly/cli/assets/claude/skills/afly-backfill/SKILL.md +68 -0
  22. afly/cli/assets/claude/skills/afly-debug-run/SKILL.md +70 -0
  23. afly/cli/assets/claude/skills/afly-new-extract/SKILL.md +75 -0
  24. afly/cli/assets/claude/skills/afly-setup-project/SKILL.md +62 -0
  25. afly/cli/commands/__init__.py +8 -0
  26. afly/cli/commands/apps.py +84 -0
  27. afly/cli/commands/debug.py +279 -0
  28. afly/cli/commands/init.py +228 -0
  29. afly/cli/commands/init_claude.py +191 -0
  30. afly/cli/commands/ls.py +67 -0
  31. afly/cli/commands/run.py +92 -0
  32. afly/cli/commands/unlock.py +74 -0
  33. afly/cli/commands/validate.py +73 -0
  34. afly/cli/main.py +249 -0
  35. afly/config/__init__.py +20 -0
  36. afly/config/discovery.py +86 -0
  37. afly/config/extract_config.py +286 -0
  38. afly/config/profile.py +235 -0
  39. afly/config/project_config.py +168 -0
  40. afly/config/selectors.py +65 -0
  41. afly/csvmap/__init__.py +1 -0
  42. afly/csvmap/currency.py +58 -0
  43. afly/csvmap/headers.py +163 -0
  44. afly/csvmap/parser.py +212 -0
  45. afly/csvmap/values.py +81 -0
  46. afly/database/__init__.py +1 -0
  47. afly/database/_http_client.py +186 -0
  48. afly/database/checks.py +257 -0
  49. afly/database/clickhouse.py +251 -0
  50. afly/database/ddl.py +197 -0
  51. afly/database/loads.py +159 -0
  52. afly/database/locks.py +157 -0
  53. afly/database/tables.py +87 -0
  54. afly/database/writer.py +205 -0
  55. afly/py.typed +0 -0
  56. afly/run/__init__.py +5 -0
  57. afly/run/_alert.py +66 -0
  58. afly/run/_apps.py +61 -0
  59. afly/run/_deps.py +113 -0
  60. afly/run/_echo.py +34 -0
  61. afly/run/_fetch.py +157 -0
  62. afly/run/_job_result.py +52 -0
  63. afly/run/_protocols.py +54 -0
  64. afly/run/_rebuild.py +139 -0
  65. afly/run/_render.py +80 -0
  66. afly/run/_setup.py +82 -0
  67. afly/run/_wave_window.py +71 -0
  68. afly/run/executor.py +413 -0
  69. afly/run/options.py +30 -0
  70. afly/run/planner.py +230 -0
  71. afly/run/runner.py +318 -0
  72. afly/run/scheduler.py +367 -0
  73. afly/run/summary.py +98 -0
  74. afly/run/windows.py +125 -0
  75. afly/schema.py +109 -0
  76. afly/utils/__init__.py +1 -0
  77. afly/utils/datetime_utils.py +67 -0
  78. afly/utils/env_interpolation.py +81 -0
  79. afly/utils/naming.py +49 -0
  80. afly-0.1.0.dist-info/METADATA +177 -0
  81. afly-0.1.0.dist-info/RECORD +85 -0
  82. afly-0.1.0.dist-info/WHEEL +5 -0
  83. afly-0.1.0.dist-info/entry_points.txt +2 -0
  84. afly-0.1.0.dist-info/licenses/LICENSE +21 -0
  85. afly-0.1.0.dist-info/top_level.txt +1 -0
afly/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """afly — idempotent AppsFlyer → ClickHouse extractor."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ __all__ = ["__version__"]
@@ -0,0 +1,3 @@
1
+ """afly.alerting — run-failure notifications (project-level, not data-quality)."""
2
+
3
+ from __future__ import annotations
@@ -0,0 +1,92 @@
1
+ """Failure-alert delivery over a webhook (Mattermost/Slack "attachments", or plain JSON).
2
+
3
+ This only ever fires for a *run-level* failure (the pipeline itself broke —
4
+ auth, ClickHouse, an aborted run) — see
5
+ ``afly.config.project_config.ErrorAlertingConfig`` and
6
+ ``afly.run.runner._send_alerts``. It is not a data-quality alert.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import logging
12
+ from collections.abc import Callable, Sequence
13
+ from typing import Any
14
+
15
+ import requests
16
+
17
+ from afly import __version__
18
+ from afly.config.profile import AlertChannelConfig
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+ _ATTACHMENT_COLOR = "#D63232"
23
+
24
+
25
+ def send_failure_alert(
26
+ channel: AlertChannelConfig,
27
+ *,
28
+ project: str,
29
+ profile: str,
30
+ run_id: str,
31
+ title: str,
32
+ lines: Sequence[str],
33
+ mentions: Sequence[str] = (),
34
+ post: Callable[..., Any] = requests.post,
35
+ ) -> bool:
36
+ """POST a failure alert to *channel*. Never raises — returns ``False`` and
37
+ logs a warning on any failure (network error, non-2xx, bad channel config).
38
+
39
+ The webhook URL (which may itself embed a token, per Mattermost/Slack
40
+ convention) is only ever used as the POST target — nothing here echoes
41
+ any credential back into the message body.
42
+ """
43
+ try:
44
+ payload = _build_payload(
45
+ channel, project=project, run_id=run_id, title=title, lines=lines, mentions=mentions
46
+ )
47
+ response = post(channel.webhook_url, json=payload, timeout=channel.timeout)
48
+ response.raise_for_status()
49
+ return True
50
+ except Exception:
51
+ logger.warning(
52
+ "failed to send failure alert via %s channel for run %s",
53
+ channel.type,
54
+ run_id,
55
+ exc_info=True,
56
+ )
57
+ return False
58
+
59
+
60
+ def _build_payload(
61
+ channel: AlertChannelConfig,
62
+ *,
63
+ project: str,
64
+ run_id: str,
65
+ title: str,
66
+ lines: Sequence[str],
67
+ mentions: Sequence[str],
68
+ ) -> dict[str, Any]:
69
+ if channel.type == "webhook":
70
+ return {"title": title, "text": "\n".join(lines), "project": project, "run_id": run_id}
71
+
72
+ # mattermost / slack: both speak the same incoming-webhook "attachments" shape.
73
+ payload: dict[str, Any] = {
74
+ "username": channel.username,
75
+ "text": " ".join(mentions),
76
+ "attachments": [
77
+ {
78
+ "color": _ATTACHMENT_COLOR,
79
+ "title": title,
80
+ "text": "\n".join(lines),
81
+ "footer": f"afly {__version__}",
82
+ }
83
+ ],
84
+ }
85
+ if channel.icon_emoji:
86
+ payload["icon_emoji"] = channel.icon_emoji
87
+ if channel.channel:
88
+ payload["channel"] = channel.channel
89
+ return payload
90
+
91
+
92
+ __all__ = ["send_failure_alert"]
@@ -0,0 +1 @@
1
+ """AppsFlyer HTTP client: errors, retry policy, and the Pull/management APIs."""
@@ -0,0 +1,83 @@
1
+ """Thin HTTP wrapper around AppsFlyer's REST APIs.
2
+
3
+ Deliberately does *not* interpret response status codes — that's
4
+ ``afly.appsflyer.errors.classify_response``'s job, applied by the caller
5
+ (``pull_api``/``mng_api``) once it knows which endpoint's error semantics
6
+ apply. This class only owns: auth headers, timeouts, redirect-following, and
7
+ turning a network-level failure (not an HTTP error response) into a
8
+ :class:`TransientError` the retry policy already knows how to handle.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Mapping
14
+ from typing import Any
15
+
16
+ import requests
17
+
18
+ from afly import __version__
19
+ from afly.appsflyer.errors import TransientError
20
+
21
+ _DEFAULT_BASE_URL = "https://hq1.appsflyer.com"
22
+
23
+
24
+ class AppsFlyerClient:
25
+ """Authenticated HTTP client for one AppsFlyer account (one token)."""
26
+
27
+ def __init__(
28
+ self,
29
+ token: str,
30
+ base_url: str = _DEFAULT_BASE_URL,
31
+ timeout_seconds: int = 120,
32
+ user_agent: str | None = None,
33
+ session: requests.Session | None = None,
34
+ ) -> None:
35
+ self._token = token
36
+ self.base_url = base_url.rstrip("/")
37
+ self.timeout_seconds = timeout_seconds
38
+ self.user_agent = user_agent or f"afly/{__version__}"
39
+ self._session = session or requests.Session()
40
+ self.api_calls = 0
41
+
42
+ def reset_counter(self) -> None:
43
+ """Zero the ``api_calls`` counter (e.g. between extracts in a run)."""
44
+ self.api_calls = 0
45
+
46
+ def get(
47
+ self,
48
+ path: str,
49
+ params: Mapping[str, Any] | None = None,
50
+ accept: str = "application/json",
51
+ ) -> requests.Response:
52
+ """GET ``base_url + path``, following redirects, with afly's standard headers.
53
+
54
+ Increments ``api_calls`` exactly once per call — including a call
55
+ that raises, since the network round-trip still happened and counts
56
+ against AppsFlyer's rate limit either way.
57
+ """
58
+ url = f"{self.base_url}{path}"
59
+ headers = {
60
+ "authorization": f"Bearer {self._token}",
61
+ "accept": accept,
62
+ "user-agent": self.user_agent,
63
+ }
64
+ try:
65
+ response = self._session.get(
66
+ url,
67
+ params=params,
68
+ headers=headers,
69
+ timeout=self.timeout_seconds,
70
+ allow_redirects=True,
71
+ )
72
+ except requests.RequestException as exc:
73
+ self.api_calls += 1
74
+ raise TransientError(
75
+ f"network error calling AppsFlyer: {exc}", status=None, body=str(exc), url=url
76
+ ) from exc
77
+
78
+ self.api_calls += 1
79
+ return response
80
+
81
+ def __repr__(self) -> str:
82
+ masked = f"{self._token[:4]}…" if self._token else "(empty)"
83
+ return f"AppsFlyerClient(base_url={self.base_url!r}, token={masked!r})"
@@ -0,0 +1,204 @@
1
+ """Exception hierarchy for the AppsFlyer HTTP client, and response classification.
2
+
3
+ Everything afly's retry/backoff logic (``afly.appsflyer.retry.RetryPolicy``)
4
+ and callers need to decide is captured in *which exception type* a call
5
+ raised — never a bare status-code check scattered across the codebase. This
6
+ module is the single place an HTTP response becomes one of these types, so
7
+ that decision is made once and made consistently.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from datetime import datetime, timezone
13
+ from email.utils import parsedate_to_datetime
14
+
15
+ _LIMIT_MARKER = "limit reached for"
16
+
17
+
18
+ class AppsFlyerError(Exception):
19
+ """Base for every error raised talking to the AppsFlyer API.
20
+
21
+ ``body_excerpt`` is capped at 300 chars — enough to see the error
22
+ message AppsFlyer sent without risking a multi-KB HTML error page ending
23
+ up in a log line or a CLI's stderr.
24
+ """
25
+
26
+ def __init__(
27
+ self,
28
+ message: str,
29
+ *,
30
+ status: int | None = None,
31
+ body: str = "",
32
+ url: str | None = None,
33
+ ) -> None:
34
+ super().__init__(message)
35
+ self.status = status
36
+ self.body_excerpt = (body or "")[:300]
37
+ self.url = url
38
+
39
+
40
+ class AuthError(AppsFlyerError):
41
+ """401, or a 403 that never turned into a real quota response after retries."""
42
+
43
+
44
+ class RateLimitError(AppsFlyerError):
45
+ """403 with the ``"Limit reached for"`` marker, or a plain 429."""
46
+
47
+ def __init__(
48
+ self,
49
+ message: str,
50
+ *,
51
+ status: int | None = None,
52
+ body: str = "",
53
+ url: str | None = None,
54
+ retry_after: float | None = None,
55
+ scope_hint: str | None = None,
56
+ ) -> None:
57
+ super().__init__(message, status=status, body=body, url=url)
58
+ self.retry_after = retry_after
59
+ self.scope_hint = scope_hint
60
+
61
+
62
+ class TransientError(AppsFlyerError):
63
+ """5xx, a network/timeout failure, or a 403 without the quota marker.
64
+
65
+ AppsFlyer sometimes fronts a real quota/abuse block with a bare 403 that
66
+ carries no ``"Limit reached for"`` text — indistinguishable, at the HTTP
67
+ layer, from a transient abuse-protection hiccup. Treating it as
68
+ transient (retry with backoff) is the safe default; see
69
+ ``RetryPolicy.execute`` for what happens once retries are exhausted.
70
+ """
71
+
72
+
73
+ class PermanentError(AppsFlyerError):
74
+ """Any other 4xx: bad app id (404), bad params/date-range (400), etc."""
75
+
76
+
77
+ class EmptyBodyError(AppsFlyerError):
78
+ """A 200 response whose body has no CSV header line at all."""
79
+
80
+
81
+ def classify_response(status: int, body: str, headers: object) -> type[AppsFlyerError] | None:
82
+ """Map an HTTP status + body to an error class, or ``None`` for success.
83
+
84
+ ``headers`` isn't used to *classify* (only the status and, for 403, the
85
+ body marker matter) — it's accepted here so callers can pass the same
86
+ three values they'd pass to :func:`build_error` without re-shaping them.
87
+ """
88
+ if 200 <= status < 300:
89
+ return None
90
+ if status == 401:
91
+ return AuthError
92
+ if status == 403:
93
+ return RateLimitError if _LIMIT_MARKER in body.lower() else TransientError
94
+ if status == 429:
95
+ return RateLimitError
96
+ if 500 <= status < 600:
97
+ return TransientError
98
+ return PermanentError
99
+
100
+
101
+ def parse_retry_after(value: str | None, now: datetime | None = None) -> float | None:
102
+ """Parse a ``Retry-After`` header value into seconds from *now*.
103
+
104
+ Accepts both forms RFC 7231 allows: an integer/float number of seconds,
105
+ or an HTTP-date. Returns ``None`` for a missing/unparseable value, and
106
+ never a negative number (a date already in the past clamps to 0).
107
+ """
108
+ if value is None:
109
+ return None
110
+ stripped = value.strip()
111
+ if not stripped:
112
+ return None
113
+
114
+ try:
115
+ return max(float(stripped), 0.0)
116
+ except ValueError:
117
+ pass
118
+
119
+ try:
120
+ target = parsedate_to_datetime(stripped)
121
+ except (TypeError, ValueError, IndexError):
122
+ return None
123
+ if target.tzinfo is None:
124
+ target = target.replace(tzinfo=timezone.utc)
125
+
126
+ reference = now if now is not None else datetime.now(timezone.utc)
127
+ if reference.tzinfo is None:
128
+ reference = reference.replace(tzinfo=timezone.utc)
129
+
130
+ return max((target - reference).total_seconds(), 0.0)
131
+
132
+
133
+ def _extract_scope_hint(body: str) -> str | None:
134
+ """Best-effort snippet of *what* quota was hit, for logging/diagnostics.
135
+
136
+ AppsFlyer's message shape after ``"Limit reached for "`` isn't
137
+ documented/stable, so this is deliberately a short, best-effort excerpt
138
+ rather than a parsed enum — good enough for a human reading a retry log,
139
+ not meant to be branched on.
140
+ """
141
+ lowered = body.lower()
142
+ idx = lowered.find(_LIMIT_MARKER)
143
+ if idx == -1:
144
+ return None
145
+ tail = body[idx + len(_LIMIT_MARKER) : idx + len(_LIMIT_MARKER) + 120]
146
+ for sep in (".", "\n", '"'):
147
+ pos = tail.find(sep)
148
+ if pos != -1:
149
+ tail = tail[:pos]
150
+ tail = tail.strip()
151
+ return tail or None
152
+
153
+
154
+ def build_error(
155
+ status: int,
156
+ body: str,
157
+ *,
158
+ url: str | None = None,
159
+ headers: object = None,
160
+ ) -> AppsFlyerError:
161
+ """Build the right :class:`AppsFlyerError` subclass instance for *status*/*body*.
162
+
163
+ Raises :class:`ValueError` if *status* isn't actually an error status —
164
+ callers should check :func:`classify_response` (or just call this only
165
+ after confirming the response failed) rather than relying on this as the
166
+ classification step itself.
167
+ """
168
+ headers = headers or {}
169
+ error_cls = classify_response(status, body, headers)
170
+ if error_cls is None:
171
+ raise ValueError(f"status {status} is not an AppsFlyer error status")
172
+
173
+ if error_cls is RateLimitError:
174
+ retry_after_raw = _get_header(headers, "Retry-After")
175
+ return RateLimitError(
176
+ f"AppsFlyer rate limit hit (status {status})",
177
+ status=status,
178
+ body=body,
179
+ url=url,
180
+ retry_after=parse_retry_after(retry_after_raw),
181
+ scope_hint=_extract_scope_hint(body),
182
+ )
183
+
184
+ messages = {
185
+ AuthError: f"AppsFlyer authentication failed (status {status})",
186
+ TransientError: f"AppsFlyer transient error (status {status})",
187
+ PermanentError: f"AppsFlyer request rejected (status {status})",
188
+ }
189
+ return error_cls(messages[error_cls], status=status, body=body, url=url)
190
+
191
+
192
+ def _get_header(headers: object, name: str) -> str | None:
193
+ """Case-insensitively fetch *name* from a mapping-like ``headers`` object."""
194
+ if headers is None:
195
+ return None
196
+ getter = getattr(headers, "get", None)
197
+ if getter is None:
198
+ return None
199
+ value = getter(name)
200
+ if value is None:
201
+ value = getter(name.lower())
202
+ if value is None:
203
+ value = getter(name.upper())
204
+ return value
@@ -0,0 +1,98 @@
1
+ """AppsFlyer management API — the account's app list.
2
+
3
+ Used by ``afly apps`` (a standalone lookup) and by ``afly debug``/``afly run``
4
+ to validate an app id before spending a Pull API call on it.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from dataclasses import dataclass
10
+ from typing import Any
11
+
12
+ from afly.appsflyer.client import AppsFlyerClient
13
+ from afly.appsflyer.errors import build_error, classify_response
14
+ from afly.appsflyer.retry import RetryPolicy
15
+
16
+ _APPS_PATH = "/api/mng/apps"
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class AppInfo:
21
+ """One app from the AppsFlyer account, as much as afly needs of it."""
22
+
23
+ id: str
24
+ name: str
25
+ platform: str
26
+ currency: str | None
27
+ time_zone: str | None
28
+
29
+
30
+ def list_apps(client: AppsFlyerClient, policy: RetryPolicy, limit: int = 1000) -> list[AppInfo]:
31
+ """Fetch every app visible to *client*'s token, paginating by offset.
32
+
33
+ Stops once a page returns fewer than *limit* rows, or once ``offset``
34
+ reaches ``meta.total_items`` — whichever comes first, so a server that
35
+ omits ``total_items`` (or gets it wrong) still terminates correctly off
36
+ the short-page signal alone.
37
+ """
38
+ apps: list[AppInfo] = []
39
+ offset = 0
40
+ total_items: int | None = None
41
+
42
+ while True:
43
+ page = _fetch_page(client, policy, limit, offset)
44
+ data = page.get("data") or []
45
+ apps.extend(_to_app_info(item) for item in data)
46
+
47
+ meta = page.get("meta") or {}
48
+ if total_items is None:
49
+ total_items = meta.get("total_items")
50
+
51
+ offset += limit
52
+ if len(data) < limit:
53
+ break
54
+ if total_items is not None and offset >= total_items:
55
+ break
56
+
57
+ apps.sort(key=lambda a: a.id)
58
+ return apps
59
+
60
+
61
+ def _fetch_page(
62
+ client: AppsFlyerClient, policy: RetryPolicy, limit: int, offset: int
63
+ ) -> dict[str, Any]:
64
+ def _attempt() -> dict[str, Any]:
65
+ response = client.get(
66
+ _APPS_PATH, params={"limit": limit, "offset": offset}, accept="application/json"
67
+ )
68
+ error_cls = classify_response(response.status_code, response.text, response.headers)
69
+ if error_cls is not None:
70
+ raise build_error(
71
+ response.status_code, response.text, url=response.url, headers=response.headers
72
+ )
73
+ result: dict[str, Any] = response.json()
74
+ return result
75
+
76
+ return policy.execute(_attempt)
77
+
78
+
79
+ def _to_app_info(item: dict[str, Any]) -> AppInfo:
80
+ attrs = item.get("attributes") or {}
81
+ return AppInfo(
82
+ id=item.get("id", ""),
83
+ name=attrs.get("name", ""),
84
+ platform=attrs.get("platform", ""),
85
+ currency=attrs.get("currency"),
86
+ time_zone=attrs.get("time_zone"),
87
+ )
88
+
89
+
90
+ def filter_apps(apps: list[AppInfo], platforms: list[str] | None) -> list[AppInfo]:
91
+ """Keep only apps whose platform matches one of *platforms* (case-insensitive).
92
+
93
+ ``platforms=None`` (or empty) means "no filter" — returns a copy of *apps*.
94
+ """
95
+ if not platforms:
96
+ return list(apps)
97
+ wanted = {p.lower() for p in platforms}
98
+ return [a for a in apps if (a.platform or "").lower() in wanted]
@@ -0,0 +1,158 @@
1
+ """AppsFlyer aggregate Pull API v5 — request building and report fetching.
2
+
3
+ Split into a pure request-builder (:func:`build_pull_request`, easy to unit
4
+ test param-by-param) and the actual network call (:func:`fetch_report`, which
5
+ drives it through a :class:`~afly.appsflyer.retry.RetryPolicy`) — so every
6
+ ``category``/``reattr``/``extra_params`` edge case can be asserted on the
7
+ URL/params alone, without mocking HTTP for each one.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import time
13
+ from collections.abc import Mapping
14
+ from dataclasses import dataclass, field
15
+ from datetime import date
16
+ from typing import Any
17
+
18
+ from afly.appsflyer.client import AppsFlyerClient
19
+ from afly.appsflyer.errors import EmptyBodyError, build_error, classify_response
20
+ from afly.appsflyer.retry import RetryPolicy
21
+
22
+ _RESERVED_PARAMS = ("from", "to")
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class PullRequestSpec:
27
+ """What to pull — the AppsFlyer-facing knobs of one extract, decoupled
28
+ from afly's own config model (see :meth:`from_extract`)."""
29
+
30
+ report_type: str
31
+ category: str = "standard"
32
+ media_source: str | None = None
33
+ reattr: bool = False
34
+ attribution_touch_type: str | None = None
35
+ timezone: str | None = None
36
+ currency: str | None = None
37
+ extra_params: Mapping[str, str] = field(default_factory=dict)
38
+
39
+ @classmethod
40
+ def from_extract(cls, extract: Any) -> PullRequestSpec:
41
+ """Build a spec from an ``ExtractConfig``-shaped object, via ``getattr``.
42
+
43
+ Reads by attribute name rather than importing ``afly.config``
44
+ (owned by a different milestone) — keeps this module usable and
45
+ testable independently of that package's exact type, both before it
46
+ exists and after.
47
+ """
48
+ return cls(
49
+ report_type=extract.report_type,
50
+ category=getattr(extract, "category", "standard") or "standard",
51
+ media_source=getattr(extract, "media_source", None),
52
+ reattr=bool(getattr(extract, "reattr", False)),
53
+ attribution_touch_type=getattr(extract, "attribution_touch_type", None),
54
+ timezone=getattr(extract, "timezone", None),
55
+ currency=getattr(extract, "currency", None),
56
+ extra_params=dict(getattr(extract, "extra_params", None) or {}),
57
+ )
58
+
59
+
60
+ def build_pull_request(
61
+ spec: PullRequestSpec,
62
+ app_id: str,
63
+ from_date: date,
64
+ to_date: date,
65
+ base_url: str,
66
+ ) -> tuple[str, dict[str, str]]:
67
+ """Build the ``(path, params)`` for one Pull API v5 call.
68
+
69
+ ``base_url`` is accepted (not used in the returned path, which is always
70
+ the same ``/api/agg-data/export/...`` shape regardless of region) purely
71
+ so callers can pass everything they know about the target in one place
72
+ without the function silently depending on a client's private state.
73
+ """
74
+ path = f"/api/agg-data/export/app/{app_id}/{spec.report_type}/v5"
75
+ params: dict[str, str] = {"from": from_date.isoformat(), "to": to_date.isoformat()}
76
+
77
+ if spec.media_source:
78
+ params["media_source"] = spec.media_source
79
+ if spec.category and spec.category != "standard":
80
+ params["category"] = spec.category
81
+ if spec.reattr:
82
+ params["reattr"] = "true"
83
+ if spec.attribution_touch_type == "impression":
84
+ params["attribution_touch_type"] = "impression"
85
+ if spec.timezone:
86
+ params["timezone"] = spec.timezone
87
+ if spec.currency and spec.currency != "preferred":
88
+ params["currency"] = spec.currency
89
+
90
+ for key, value in spec.extra_params.items():
91
+ if key in _RESERVED_PARAMS:
92
+ raise ValueError(
93
+ f"extra_params may not override {key!r} — it's controlled by the pull window"
94
+ )
95
+ params[key] = value
96
+
97
+ return path, params
98
+
99
+
100
+ @dataclass
101
+ class RawReport:
102
+ """The raw fetched CSV plus enough metadata to log/debug the call."""
103
+
104
+ text: str
105
+ status: int
106
+ url: str
107
+ params: dict[str, str]
108
+ api_calls: int
109
+ elapsed_ms: int
110
+
111
+
112
+ def fetch_report(
113
+ client: AppsFlyerClient,
114
+ spec: PullRequestSpec,
115
+ app_id: str,
116
+ from_date: date,
117
+ to_date: date,
118
+ policy: RetryPolicy,
119
+ ) -> RawReport:
120
+ """Fetch one report, retrying per *policy*. Returns the decoded CSV body.
121
+
122
+ ``api_calls``/``elapsed_ms`` on the result cover the *whole* fetch
123
+ (every retried attempt), not just the final successful one — that's what
124
+ a quota scheduler (M4) or a debug summary actually wants to know: what
125
+ did this fetch cost, not what did its last attempt cost.
126
+ """
127
+ path, params = build_pull_request(spec, app_id, from_date, to_date, client.base_url)
128
+ start_calls = client.api_calls
129
+ start_time = time.monotonic()
130
+
131
+ def _attempt() -> tuple[str, int, str]:
132
+ response = client.get(path, params=params, accept="text/csv")
133
+ error_cls = classify_response(response.status_code, response.text, response.headers)
134
+ if error_cls is not None:
135
+ raise build_error(
136
+ response.status_code, response.text, url=response.url, headers=response.headers
137
+ )
138
+
139
+ text = response.content.decode("utf-8-sig")
140
+ if not text.strip():
141
+ raise EmptyBodyError(
142
+ "AppsFlyer returned an empty report body",
143
+ status=response.status_code,
144
+ body="",
145
+ url=response.url,
146
+ )
147
+ return text, response.status_code, response.url
148
+
149
+ text, status, url = policy.execute(_attempt)
150
+
151
+ return RawReport(
152
+ text=text,
153
+ status=status,
154
+ url=url,
155
+ params=dict(params),
156
+ api_calls=client.api_calls - start_calls,
157
+ elapsed_ms=int((time.monotonic() - start_time) * 1000),
158
+ )