dtex 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (142) hide show
  1. {dtex-0.2.2 → dtex-0.2.4}/PKG-INFO +20 -8
  2. {dtex-0.2.2 → dtex-0.2.4}/README.md +20 -8
  3. dtex-0.2.4/dtex/destinations/duckdb/README.md +98 -0
  4. dtex-0.2.4/dtex/sources/revenuecat/README.md +160 -0
  5. dtex-0.2.4/dtex/sources/revenuecat/__init__.py +1 -0
  6. dtex-0.2.4/dtex/sources/revenuecat/client.py +126 -0
  7. dtex-0.2.4/dtex/sources/revenuecat/register.yaml +148 -0
  8. dtex-0.2.4/dtex/sources/revenuecat/source.py +312 -0
  9. {dtex-0.2.2 → dtex-0.2.4}/pyproject.toml +1 -1
  10. dtex-0.2.4/tests/connectors/test_revenuecat.py +499 -0
  11. {dtex-0.2.2 → dtex-0.2.4}/.gitignore +0 -0
  12. {dtex-0.2.2 → dtex-0.2.4}/CONTRIBUTING.md +0 -0
  13. {dtex-0.2.2 → dtex-0.2.4}/LICENSE +0 -0
  14. {dtex-0.2.2 → dtex-0.2.4}/docs/00-vision-and-naming.md +0 -0
  15. {dtex-0.2.2 → dtex-0.2.4}/docs/01-landscape-and-comparison.md +0 -0
  16. {dtex-0.2.2 → dtex-0.2.4}/docs/02-architecture.md +0 -0
  17. {dtex-0.2.2 → dtex-0.2.4}/docs/03-connector-contract.md +0 -0
  18. {dtex-0.2.2 → dtex-0.2.4}/docs/04-connector-body.md +0 -0
  19. {dtex-0.2.2 → dtex-0.2.4}/docs/05-destinations-and-state.md +0 -0
  20. {dtex-0.2.2 → dtex-0.2.4}/docs/06-project-anatomy.md +0 -0
  21. {dtex-0.2.2 → dtex-0.2.4}/docs/07-cli-and-library-api.md +0 -0
  22. {dtex-0.2.2 → dtex-0.2.4}/docs/08-security.md +0 -0
  23. {dtex-0.2.2 → dtex-0.2.4}/docs/09-logging-and-observability.md +0 -0
  24. {dtex-0.2.2 → dtex-0.2.4}/docs/10-roadmap-and-scope.md +0 -0
  25. {dtex-0.2.2 → dtex-0.2.4}/docs/11-open-questions.md +0 -0
  26. {dtex-0.2.2 → dtex-0.2.4}/docs/12-configs.md +0 -0
  27. {dtex-0.2.2 → dtex-0.2.4}/docs/README.md +0 -0
  28. {dtex-0.2.2 → dtex-0.2.4}/docs/_archive/stripe-research-2026-05.md +0 -0
  29. {dtex-0.2.2 → dtex-0.2.4}/docs/_internal/release.md +0 -0
  30. {dtex-0.2.2 → dtex-0.2.4}/docs/_internal/streams-redesign-plan.md +0 -0
  31. {dtex-0.2.2 → dtex-0.2.4}/dtex/__init__.py +0 -0
  32. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/__init__.py +0 -0
  33. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_discovery.py +0 -0
  34. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_format.py +0 -0
  35. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_runs.py +0 -0
  36. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_scaffold.py +0 -0
  37. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_secrets.py +0 -0
  38. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_skills.py +0 -0
  39. {dtex-0.2.2 → dtex-0.2.4}/dtex/cli/_state.py +0 -0
  40. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/__init__.py +0 -0
  41. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/README.md +0 -0
  42. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/__init__.py +0 -0
  43. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/client.py +0 -0
  44. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/ddl.py +0 -0
  45. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/destination.py +0 -0
  46. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/bigquery/register.yaml +0 -0
  47. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/duckdb/__init__.py +0 -0
  48. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/duckdb/ddl.py +0 -0
  49. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/duckdb/destination.py +0 -0
  50. {dtex-0.2.2 → dtex-0.2.4}/dtex/destinations/duckdb/register.yaml +0 -0
  51. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/__init__.py +0 -0
  52. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/config.py +0 -0
  53. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/configs.py +0 -0
  54. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/discovery.py +0 -0
  55. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/logger.py +0 -0
  56. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/normalize.py +0 -0
  57. {dtex-0.2.2 → dtex-0.2.4}/dtex/engine/runner.py +0 -0
  58. {dtex-0.2.2 → dtex-0.2.4}/dtex/py.typed +0 -0
  59. {dtex-0.2.2 → dtex-0.2.4}/dtex/registry.py +0 -0
  60. {dtex-0.2.2 → dtex-0.2.4}/dtex/secrets/__init__.py +0 -0
  61. {dtex-0.2.2 → dtex-0.2.4}/dtex/secrets/_aws.py +0 -0
  62. {dtex-0.2.2 → dtex-0.2.4}/dtex/secrets/_gcp.py +0 -0
  63. {dtex-0.2.2 → dtex-0.2.4}/dtex/secrets/_vault.py +0 -0
  64. {dtex-0.2.2 → dtex-0.2.4}/dtex/secrets/resolvers.py +0 -0
  65. {dtex-0.2.2 → dtex-0.2.4}/dtex/skills/dtex-debug.md +0 -0
  66. {dtex-0.2.2 → dtex-0.2.4}/dtex/skills/dtex-write-config.md +0 -0
  67. {dtex-0.2.2 → dtex-0.2.4}/dtex/skills/dtex-write-connector.md +0 -0
  68. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/__init__.py +0 -0
  69. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/README.md +0 -0
  70. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/__init__.py +0 -0
  71. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/backends.py +0 -0
  72. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/readers.py +0 -0
  73. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/register.yaml +0 -0
  74. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/filesystem/source.py +0 -0
  75. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/README.md +0 -0
  76. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/__init__.py +0 -0
  77. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/client.py +0 -0
  78. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/register.yaml +0 -0
  79. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/source.py +0 -0
  80. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/postgres/type_mapping.py +0 -0
  81. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/README.md +0 -0
  82. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/__init__.py +0 -0
  83. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/client.py +0 -0
  84. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/extractors.py +0 -0
  85. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/pagination.py +0 -0
  86. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/register.yaml +0 -0
  87. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/rest/source.py +0 -0
  88. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/README.md +0 -0
  89. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/__init__.py +0 -0
  90. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/client.py +0 -0
  91. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/pagination.py +0 -0
  92. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/queries.py +0 -0
  93. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/register.yaml +0 -0
  94. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/source.py +0 -0
  95. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/shiphero/windows.py +0 -0
  96. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/README.md +0 -0
  97. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/__init__.py +0 -0
  98. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/client.py +0 -0
  99. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/pagination.py +0 -0
  100. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/queries/charges_daily.sql +0 -0
  101. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/queries/invoices_paid.sql +0 -0
  102. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/queries/subscriptions_active.sql +0 -0
  103. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/register.yaml +0 -0
  104. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/sigma_client.py +0 -0
  105. {dtex-0.2.2 → dtex-0.2.4}/dtex/sources/stripe/source.py +0 -0
  106. {dtex-0.2.2 → dtex-0.2.4}/dtex/types.py +0 -0
  107. {dtex-0.2.2 → dtex-0.2.4}/tests/__init__.py +0 -0
  108. {dtex-0.2.2 → dtex-0.2.4}/tests/conftest.py +0 -0
  109. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/__init__.py +0 -0
  110. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/test_filesystem.py +0 -0
  111. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/test_postgres.py +0 -0
  112. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/test_rest.py +0 -0
  113. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/test_shiphero.py +0 -0
  114. {dtex-0.2.2 → dtex-0.2.4}/tests/connectors/test_stripe.py +0 -0
  115. {dtex-0.2.2 → dtex-0.2.4}/tests/destinations/__init__.py +0 -0
  116. {dtex-0.2.2 → dtex-0.2.4}/tests/destinations/test_bigquery.py +0 -0
  117. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/configs/echo.yml +0 -0
  118. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/destinations/lockedfake/destination.py +0 -0
  119. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/destinations/lockedfake/register.yaml +0 -0
  120. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/dtex_project.yml +0 -0
  121. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/profiles.yml +0 -0
  122. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/echo/register.yaml +0 -0
  123. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/echo/source.py +0 -0
  124. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/multifile_echo/__init__.py +0 -0
  125. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/multifile_echo/helper.py +0 -0
  126. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/multifile_echo/register.yaml +0 -0
  127. {dtex-0.2.2 → dtex-0.2.4}/tests/fixtures/sources/multifile_echo/source.py +0 -0
  128. {dtex-0.2.2 → dtex-0.2.4}/tests/test_cli.py +0 -0
  129. {dtex-0.2.2 → dtex-0.2.4}/tests/test_configs.py +0 -0
  130. {dtex-0.2.2 → dtex-0.2.4}/tests/test_duckdb_destination.py +0 -0
  131. {dtex-0.2.2 → dtex-0.2.4}/tests/test_engine.py +0 -0
  132. {dtex-0.2.2 → dtex-0.2.4}/tests/test_normalize.py +0 -0
  133. {dtex-0.2.2 → dtex-0.2.4}/tests/test_registry.py +0 -0
  134. {dtex-0.2.2 → dtex-0.2.4}/tests/test_run_records.py +0 -0
  135. {dtex-0.2.2 → dtex-0.2.4}/tests/test_secret_resolver_aws.py +0 -0
  136. {dtex-0.2.2 → dtex-0.2.4}/tests/test_secret_resolver_gcp.py +0 -0
  137. {dtex-0.2.2 → dtex-0.2.4}/tests/test_secret_resolver_vault.py +0 -0
  138. {dtex-0.2.2 → dtex-0.2.4}/tests/test_secret_resolvers.py +0 -0
  139. {dtex-0.2.2 → dtex-0.2.4}/tests/test_skeleton.py +0 -0
  140. {dtex-0.2.2 → dtex-0.2.4}/tests/test_skills_install.py +0 -0
  141. {dtex-0.2.2 → dtex-0.2.4}/tests/test_smoke.py +0 -0
  142. {dtex-0.2.2 → dtex-0.2.4}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dtex
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: dtex (data extraction tool) — an open-source Python EL tool: pipelines as configs, connectors as folders, CLI-first.
5
5
  Project-URL: Homepage, https://github.com/vej-ai/dtex
6
6
  Project-URL: Documentation, https://github.com/vej-ai/dtex/tree/main/docs
@@ -101,14 +101,26 @@ params. Run it with `dtex run -p <config>`. The library equivalent is
101
101
 
102
102
  ## Pre-baked connectors
103
103
 
104
- **Sources:** `filesystem` (CSV/JSONL/Parquet from local, GCS, or S3),
105
- `rest` (paginated REST APIs — 4 pagination strategies, 4 auth modes),
106
- `postgres` (keyset pagination, no `OFFSET`), `shiphero` (GraphQL),
107
- `stripe` (REST resource-as-stream + opt-in Sigma SQL-as-stream).
104
+ Each connector ships with its own README — click through for the
105
+ auth checklist, scope requirements, schema, and known limitations.
108
106
 
109
- **Destinations:** `duckdb` (zero-config dev default, all 5 capabilities) and
110
- `bigquery` (production warehouse — Parquet-staged via GCS + LOAD jobs,
111
- MERGE upserts, cursor-based partitioning).
107
+ **Sources:**
108
+
109
+ | Connector | What it does |
110
+ |---|---|
111
+ | [`filesystem`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/filesystem/README.md) | CSV / JSONL / Parquet from local, GCS, or S3 |
112
+ | [`rest`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/rest/README.md) | Paginated REST APIs — 4 pagination strategies, 4 auth modes |
113
+ | [`postgres`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/postgres/README.md) | Keyset pagination, no `OFFSET` |
114
+ | [`shiphero`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/shiphero/README.md) | GraphQL |
115
+ | [`stripe`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/stripe/README.md) | REST resource-as-stream + opt-in Sigma SQL-as-stream |
116
+ | [`revenuecat`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/revenuecat/README.md) | v2 API — customers + subscriptions + daily chart metrics |
117
+
118
+ **Destinations:**
119
+
120
+ | Connector | What it does |
121
+ |---|---|
122
+ | [`duckdb`](https://github.com/vej-ai/dtex/blob/main/dtex/destinations/duckdb/README.md) | Zero-config dev default, all 5 capabilities |
123
+ | [`bigquery`](https://github.com/vej-ai/dtex/blob/main/dtex/destinations/bigquery/README.md) | Production warehouse — Parquet-staged via GCS + LOAD jobs, MERGE upserts, cursor-based partitioning |
112
124
 
113
125
  **Engine:** per-stream commit + atomic transactions (rollback on failure),
114
126
  state in the destination's `_dtex_state` table, run records in `_dtex_runs`,
@@ -51,14 +51,26 @@ params. Run it with `dtex run -p <config>`. The library equivalent is
51
51
 
52
52
  ## Pre-baked connectors
53
53
 
54
- **Sources:** `filesystem` (CSV/JSONL/Parquet from local, GCS, or S3),
55
- `rest` (paginated REST APIs — 4 pagination strategies, 4 auth modes),
56
- `postgres` (keyset pagination, no `OFFSET`), `shiphero` (GraphQL),
57
- `stripe` (REST resource-as-stream + opt-in Sigma SQL-as-stream).
58
-
59
- **Destinations:** `duckdb` (zero-config dev default, all 5 capabilities) and
60
- `bigquery` (production warehouse — Parquet-staged via GCS + LOAD jobs,
61
- MERGE upserts, cursor-based partitioning).
54
+ Each connector ships with its own README — click through for the
55
+ auth checklist, scope requirements, schema, and known limitations.
56
+
57
+ **Sources:**
58
+
59
+ | Connector | What it does |
60
+ |---|---|
61
+ | [`filesystem`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/filesystem/README.md) | CSV / JSONL / Parquet from local, GCS, or S3 |
62
+ | [`rest`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/rest/README.md) | Paginated REST APIs — 4 pagination strategies, 4 auth modes |
63
+ | [`postgres`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/postgres/README.md) | Keyset pagination, no `OFFSET` |
64
+ | [`shiphero`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/shiphero/README.md) | GraphQL |
65
+ | [`stripe`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/stripe/README.md) | REST resource-as-stream + opt-in Sigma SQL-as-stream |
66
+ | [`revenuecat`](https://github.com/vej-ai/dtex/blob/main/dtex/sources/revenuecat/README.md) | v2 API — customers + subscriptions + daily chart metrics |
67
+
68
+ **Destinations:**
69
+
70
+ | Connector | What it does |
71
+ |---|---|
72
+ | [`duckdb`](https://github.com/vej-ai/dtex/blob/main/dtex/destinations/duckdb/README.md) | Zero-config dev default, all 5 capabilities |
73
+ | [`bigquery`](https://github.com/vej-ai/dtex/blob/main/dtex/destinations/bigquery/README.md) | Production warehouse — Parquet-staged via GCS + LOAD jobs, MERGE upserts, cursor-based partitioning |
62
74
 
63
75
  **Engine:** per-stream commit + atomic transactions (rollback on failure),
64
76
  state in the destination's `_dtex_state` table, run records in `_dtex_runs`,
@@ -0,0 +1,98 @@
1
+ # DuckDB destination — dtex v1
2
+
3
+ The **zero-config dev default**. Writes batches to a local DuckDB file
4
+ (`append` / `merge` / `replace`) — no external service to set up, no
5
+ credentials, no network. Tier A: DuckDB hosts its own state and audit
6
+ tables (`_dtex_state`, `_dtex_runs`) alongside the loaded data, so a
7
+ fresh checkout of a project resumes correctly with no extra files.
8
+
9
+ See [docs/05 — Destinations & State](../../../docs/05-destinations-and-state.md)
10
+ for the destination contract; this README is the operator's quick reference.
11
+
12
+ ## Install
13
+
14
+ ```bash
15
+ pip install dtex
16
+ ```
17
+
18
+ DuckDB ships in dtex's base dependencies — no extras needed.
19
+
20
+ ## Config
21
+
22
+ Default profile is fine for most projects:
23
+
24
+ ```yaml
25
+ # profiles.yml
26
+ duckdb:
27
+ default_target: dev
28
+ targets:
29
+ dev:
30
+ path: ".dtex/warehouse.duckdb" # the default — gitignored by `dtex init`
31
+ prod:
32
+ path: "/var/data/dtex/warehouse.duckdb"
33
+ ```
34
+
35
+ The `.duckdb` file is created on first run. `dtex init` scaffolds the
36
+ above + a `.gitignore` rule for `.dtex/`. For ephemeral / test runs use
37
+ `path: ":memory:"`.
38
+
39
+ ### Per-pipeline schema override
40
+
41
+ A config can land each pipeline's tables under a DuckDB schema by setting
42
+ `destination_params.dataset`:
43
+
44
+ ```yaml
45
+ # configs/my_pipeline.yml
46
+ destination_params:
47
+ dataset: my_pipeline # creates schema if absent; all tables land here
48
+ ```
49
+
50
+ Without `dataset` the destination uses DuckDB's default schema.
51
+
52
+ ## Capabilities
53
+
54
+ DuckDB implements all five destination capabilities:
55
+
56
+ | Capability | What it does |
57
+ |---|---|
58
+ | `append` | INSERT rows as-is. |
59
+ | `merge` | UPSERT on `primary_key` (requires non-empty PK). |
60
+ | `replace` | TRUNCATE then INSERT (full-refresh streams). |
61
+ | `state` | Stores `_dtex_state` rows for cursor resume. |
62
+ | `run_records` | Writes `_dtex_runs` per pipeline invocation. |
63
+
64
+ ## State + run records
65
+
66
+ Two tables live alongside the loaded data:
67
+
68
+ - **`_dtex_state`** — one row per (connector, stream) with the committed
69
+ cursor value, last run id, and cumulative `rows_total`. Read at every
70
+ run to resume incremental streams.
71
+ - **`_dtex_runs`** — one row per pipeline invocation with start/end time,
72
+ status, rows loaded, and (for failures) the error type + message.
73
+ Query history with `dtex runs list` or directly.
74
+
75
+ ## Operational notes
76
+
77
+ - **Concurrent writes are NOT supported.** DuckDB locks the database file
78
+ for the duration of a write transaction; two `dtex run` processes
79
+ targeting the same file will serialize at best, error at worst. For
80
+ production multi-pipeline use, target BigQuery instead.
81
+ - **Single-machine.** A DuckDB file lives on one disk. dtex doesn't
82
+ replicate it; if you want a queryable copy on another box, copy the
83
+ file or switch to BigQuery.
84
+ - **Memory.** DuckDB streams query results, so MERGE operations work fine
85
+ on large tables — but VERY large in-memory ANALYZE / aggregation jobs
86
+ can OOM. Adjust `PRAGMA memory_limit` via a connection callback if you
87
+ hit it.
88
+
89
+ ## What's not in v1
90
+
91
+ - No clustering / partitioning at the storage layer (DuckDB doesn't
92
+ expose physical partitioning the way warehouses do; `partition_by`
93
+ declarations on streams are recorded in metadata but don't influence
94
+ the physical layout). Use BigQuery if cursor-based partitioning is a
95
+ hard requirement.
96
+ - No cross-database joins from the warehouse file (a pipeline can read
97
+ via the `filesystem` source's DuckDB-backed Parquet paths, but the
98
+ destination itself doesn't host federation).
@@ -0,0 +1,160 @@
1
+ # RevenueCat — baked source connector
2
+
3
+ Extracts data from [RevenueCat](https://www.revenuecat.com)'s v2 API at
4
+ `https://api.revenuecat.com/v2`. Three streams covering RC's three
5
+ distinct extraction surfaces:
6
+
7
+ * **`customers`** — every customer in a project. NON-incremental.
8
+ * **`subscriptions`** — per-customer subscriptions. NON-incremental,
9
+ fan-out across the customer list.
10
+ * **`metrics_daily`** — daily chart metrics (revenue, MRR, churn, etc.).
11
+ REAL incremental on cohort date, long-format output.
12
+
13
+ The connector targets v2 specifically. v2 keys are **separate** from v1
14
+ keys and the two APIs are not interchangeable.
15
+
16
+ ## Authentication
17
+
18
+ The connector reads a **v2 RevenueCat secret key** (`sk_...`) from the
19
+ `REVENUECAT_API_KEY` environment variable by default:
20
+
21
+ ```sh
22
+ export REVENUECAT_API_KEY="sk_..."
23
+ ```
24
+
25
+ Required scopes (set on the key in the RC dashboard):
26
+
27
+ | Stream | Scope |
28
+ |---|---|
29
+ | `customers` | `customer_information:customers:read` |
30
+ | `subscriptions` | `customer_information:subscriptions:read` |
31
+ | `metrics_daily` | `charts_metrics:overview:read` |
32
+
33
+ For production, override the secret ref per profile (`profiles.yml`)
34
+ with any resolver-backed `secret://` URL — GCP Secret Manager, AWS
35
+ Secrets Manager, or HashiCorp Vault:
36
+
37
+ ```yaml
38
+ # profiles.yml
39
+ revenuecat:
40
+ default_target: prod
41
+ targets:
42
+ prod:
43
+ api_key: secret://gcp-secret-manager/projects/<proj>/secrets/<name>/versions/latest
44
+ ```
45
+
46
+ (Requires the matching extra: `pip install 'dtex[gcp-secrets]'` /
47
+ `[aws-secrets]` / `[vault]`.)
48
+
49
+ The key never appears in log output — it's set on the `requests.Session`
50
+ header once and the connector's logging emits no header contents.
51
+
52
+ ## The three streams
53
+
54
+ ### `customers` — non-incremental
55
+
56
+ RC v2's `/customers` endpoint has **NO server-side date filter**
57
+ (verified against the docs + the airbyte issue 70315 + the RC community
58
+ forum, June 2026). Every run paginates the full customer list. The
59
+ prior v1 design that filtered client-side had a subtle data-loss bug
60
+ because RC does not guarantee response ordering; the baked connector
61
+ deliberately drops the `incremental:` block to avoid the same trap.
62
+ `write_disposition: merge` on `id` makes re-pulls upsert idempotently;
63
+ the cost is HTTP calls, not duplicate rows.
64
+
65
+ For a 195k-customer account, expect ~70 minutes wall time at the
66
+ default page_size of 100 (RC's customer-domain rate limit is 480
67
+ req/min; the practical floor is ~3 min/100k rows).
68
+
69
+ ### `subscriptions` — non-incremental, per-customer fan-out
70
+
71
+ RC has no project-level `/subscriptions` endpoint, only
72
+ `/customers/{id}/subscriptions`. This stream iterates the customer list
73
+ and fetches per-customer subscriptions. Operationally expensive:
74
+ O(N+1) HTTP calls per run for N customers. For Sintra-scale accounts
75
+ this is the heaviest stream by far.
76
+
77
+ ### `metrics_daily` — real incremental
78
+
79
+ RC v2's `/charts/{chart_name}` endpoint DOES take server-side
80
+ `start_date`/`end_date` filters and tags each per-day value with
81
+ `incomplete=true` when the day is still finalizing. This stream:
82
+
83
+ 1. Computes `start_date = max(cursor.start_value(), today - lookback_days)`.
84
+ 2. For each chart in `metrics_charts`, fetches the date range with
85
+ `resolution=day`.
86
+ 3. Flattens the response's `measures[]` × `values[]` arrays into
87
+ long format: one row per `(cohort_date, chart_name, measure_name)`.
88
+ 4. Observes the cursor only against rows where `incomplete=false`, so
89
+ today's still-incomplete value gets re-pulled on the next run with
90
+ the corrected (final) value.
91
+
92
+ Default charts: `revenue`, `mrr`, `actives`, `trials`. Override via
93
+ the `metrics_charts` param (comma-separated). Adding a new chart is
94
+ zero schema migration — it lands as new rows in the same long-format
95
+ table.
96
+
97
+ The `cohort_explorer` and `prediction_explorer` charts return a
98
+ different shape and would break the flattener; exclude them unless
99
+ you fork the connector.
100
+
101
+ ## Config
102
+
103
+ A minimum config looks like this:
104
+
105
+ ```yaml
106
+ # configs/revenuecat_bq.yml
107
+ name: revenuecat_bq
108
+ source: revenuecat
109
+ destination: bigquery
110
+ target: prod
111
+ params:
112
+ project_id: "proj1ab2c3d4" # required, no default
113
+ destination_params:
114
+ dataset: revenuecat
115
+ streams: all # or list specific streams
116
+ ```
117
+
118
+ To narrow to just the cheap stream:
119
+
120
+ ```yaml
121
+ streams:
122
+ metrics_daily:
123
+ ```
124
+
125
+ ## Rate limits and timeouts
126
+
127
+ * `/customers` and `/subscriptions` share the **Customer Information**
128
+ domain with a 480 req/min default cap.
129
+ * `/charts/*` is in the **Charts & Metrics** domain with a 15 req/min
130
+ cap — much tighter; 4 charts × ~7 days per run hits this fine.
131
+ * Client uses a `(10s connect, 60s read)` timeout to prevent the
132
+ stale-socket hangs the prior version exhibited.
133
+ * Retries cover: HTTP 5xx (capped, exponential backoff), HTTP 429
134
+ (honors `Retry-After`, also capped), and the network-exception
135
+ family (Timeout, ConnectionError, ChunkedEncodingError). All capped
136
+ by `max_retries`; the prior version had two uncapped retry paths
137
+ that wedged on persistent rate-limits or dead sockets.
138
+
139
+ ## Operational notes
140
+
141
+ - **First run on a populated account is long.** Plan for ~70 min wall
142
+ time on ~200k-customer accounts. Subsequent runs are not faster
143
+ unless you switch to `metrics_daily` only.
144
+ - **No incremental shortcut exists for customers.** RC's official
145
+ guidance for "I need fresh data" is their S3/GCS export (a separate
146
+ paid feature). The connector documents this honestly rather than
147
+ pretending a client-side filter is incremental.
148
+ - **Long backfills need dtex 0.2.2+.** Earlier dtex BigQuery destinations
149
+ could fail mid-run on a stale TCP socket; 0.2.2 fixed it.
150
+
151
+ ## What's not in v1 of this connector
152
+
153
+ - `cohort_explorer` and `prediction_explorer` chart types (different
154
+ response shape).
155
+ - Webhooks (no v2 endpoint at the time of writing).
156
+ - Transactions / activities / payments lists (RC doesn't expose them
157
+ at the project level in v2 yet; staff has the feature request
158
+ open).
159
+ - `non-subscription_purchases` chart (currently excluded from the
160
+ defaults but should work — try it).
@@ -0,0 +1 @@
1
+ # Marker file — makes this folder a Python package so source.py / destination.py can use relative imports for sibling helpers (e.g. `from .client import X`).
@@ -0,0 +1,126 @@
1
+ """RevenueCat v2 HTTP client — thin wrapper over `requests`.
2
+
3
+ Two surfaces:
4
+
5
+ * `paginate(path, params)` — for list endpoints (/customers, per-customer
6
+ /subscriptions). RC paginates with `starting_after` cursor tokens; the
7
+ response's `next_page` field is the absolute URL of the next page (with
8
+ the cursor + original query params baked in), so subsequent calls pass
9
+ NO params — using the URL as-is is correct per docs.
10
+ * `get(path, params)` — for one-shot endpoints (/charts/{chart_name},
11
+ /metrics/*). Returns parsed JSON.
12
+
13
+ Both surfaces share retry + rate-limit handling. Three failure classes
14
+ get bounded, capped retry — the previous version had two real hangs:
15
+
16
+ * **No socket timeout.** Default `requests.get` waits forever for both
17
+ connect and read. One dead RC connection mid-walk → infinite sleep
18
+ inside `_session.get`. Fixed with `timeout=(connect, read)`.
19
+ * **Network errors weren't caught.** `Timeout`, `ConnectionError`,
20
+ `ChunkedEncodingError`, `ProtocolError` all raise BEFORE a `resp`
21
+ object exists, so neither status-code branch saw them — they
22
+ propagated up and failed the stream on the first blip.
23
+ * **429 retry was uncapped + did not increment `_attempt`.** A
24
+ sustained rate-limit (very real: ~480 req/min on the customer domain
25
+ vs. tens of thousands of customer pages per run) wedged the run in a
26
+ permanent sleep-retry loop. Now bounded by `max_retries`, same as
27
+ the other retry paths.
28
+
29
+ The Bearer token is set on the session header once; it never appears
30
+ in log output or error messages.
31
+ """
32
+
33
+ from __future__ import annotations
34
+
35
+ import time
36
+ from dataclasses import dataclass, field
37
+ from typing import Any, Iterator
38
+
39
+ import requests
40
+
41
+ # (connect_timeout, read_timeout) seconds. The connect leg should be fast
42
+ # — RC's edge is near every common region. The read leg is the dangerous
43
+ # one: a single hung response will block the whole run. 60s is generous
44
+ # enough for slow chart computations without trapping a dead socket.
45
+ _TIMEOUT: tuple[float, float] = (10.0, 60.0)
46
+
47
+
48
+ @dataclass
49
+ class RevenueCatClient:
50
+ api_key: str
51
+ project_id: str
52
+ base_url: str = "https://api.revenuecat.com/v2"
53
+ max_retries: int = 5
54
+ _session: requests.Session = field(
55
+ default_factory=requests.Session, init=False, repr=False
56
+ )
57
+
58
+ def __post_init__(self) -> None:
59
+ self._session.headers.update(
60
+ {
61
+ "Authorization": f"Bearer {self.api_key}",
62
+ "Accept": "application/json",
63
+ }
64
+ )
65
+
66
+ def paginate(
67
+ self, path: str, params: dict[str, Any] | None = None
68
+ ) -> Iterator[dict]:
69
+ """Yield every item from an RC v2 list endpoint.
70
+
71
+ Params apply ONLY to the first request — RC's `next_page` is an
72
+ absolute URL with the original query string + the `starting_after`
73
+ cursor token baked in, so subsequent fetches use it verbatim.
74
+ """
75
+ next_url: str | None = f"{self.base_url}{path}"
76
+ first = True
77
+ while next_url:
78
+ data = self._get(next_url, params=params if first else None)
79
+ first = False
80
+ for item in data.get("items", []):
81
+ yield item
82
+ next_url = data.get("next_page")
83
+
84
+ def get(self, path: str, params: dict[str, Any] | None = None) -> dict:
85
+ """One-shot GET on a non-list endpoint. Returns parsed JSON."""
86
+ url = f"{self.base_url}{path}"
87
+ return self._get(url, params=params)
88
+
89
+ def _get(
90
+ self,
91
+ url: str,
92
+ params: dict | None = None,
93
+ _attempt: int = 0,
94
+ ) -> dict:
95
+ # Network-level failures (timeout, connection reset, broken
96
+ # chunked encoding) raise BEFORE a `resp` object exists, so we
97
+ # have to catch them around the .get call and route through the
98
+ # same capped-backoff path as a 5xx.
99
+ try:
100
+ resp = self._session.get(url, params=params, timeout=_TIMEOUT)
101
+ except requests.exceptions.RequestException as exc:
102
+ if _attempt < self.max_retries:
103
+ time.sleep(min(2**_attempt, 60))
104
+ return self._get(url, params, _attempt + 1)
105
+ raise RuntimeError(
106
+ f"revenuecat: network failure after {self.max_retries} retries on {url}: {exc}"
107
+ ) from exc
108
+
109
+ if resp.status_code == 429:
110
+ # RC enforces a per-domain RPM cap. `Retry-After` is the
111
+ # server's directive — honor it. But ALSO bound the loop:
112
+ # the previous version did not increment `_attempt` and a
113
+ # sustained rate-limit wedged the run forever.
114
+ if _attempt >= self.max_retries:
115
+ raise RuntimeError(
116
+ f"revenuecat: rate-limited after {self.max_retries} retries on {url}; "
117
+ f"Retry-After={resp.headers.get('Retry-After')}"
118
+ )
119
+ wait = int(resp.headers.get("Retry-After", 60))
120
+ time.sleep(wait)
121
+ return self._get(url, params, _attempt + 1)
122
+ if resp.status_code in (500, 502, 503, 504) and _attempt < self.max_retries:
123
+ time.sleep(min(2**_attempt, 60))
124
+ return self._get(url, params, _attempt + 1)
125
+ resp.raise_for_status()
126
+ return resp.json()
@@ -0,0 +1,148 @@
1
+ # RevenueCat — baked source connector for the v2 API
2
+ # (https://api.revenuecat.com/v2). Three streams covering RC's three
3
+ # distinct extraction surfaces:
4
+ #
5
+ # * `customers` — every customer in a project. NON-incremental: RC v2 has
6
+ # no server-side date filter on /customers (verified June 2026 against
7
+ # the docs, the airbyte issue 70315, and the RC community forum), so
8
+ # every run paginates the full customer list. `write_disposition: merge`
9
+ # on `id` makes re-pulls upsert idempotently.
10
+ #
11
+ # * `subscriptions` — per-customer fan-out. RC has no project-level
12
+ # /subscriptions endpoint, only /customers/{id}/subscriptions, so this
13
+ # stream iterates the customer list and fetches per-customer
14
+ # subscriptions. NON-incremental; merge-on-id like customers.
15
+ # Operationally expensive: O(N+1) HTTP calls per run for N customers.
16
+ #
17
+ # * `metrics_daily` — the third surface. RC v2's Charts API
18
+ # (/charts/{chart_name}) DOES take server-side start_date/end_date
19
+ # filters AND tags per-day values with `incomplete=true` when the day
20
+ # is still finalizing. Long format (one row per cohort_date × chart ×
21
+ # measure) so adding a chart needs zero schema migration. Real
22
+ # incremental; the cursor advances only past days RC marked complete,
23
+ # so partial-today re-pulls cleanly on the next run.
24
+ #
25
+ # Auth: a v2 RC secret key (`sk_...`) with at minimum
26
+ # `customer_information:customers:read`,
27
+ # `customer_information:subscriptions:read`, and
28
+ # `charts_metrics:overview:read`. v2 keys are SEPARATE from v1 keys
29
+ # ("API v1 keys will not work with REST API v2" — RC docs). The default
30
+ # reads it from `${env.REVENUECAT_API_KEY}`; profile-level secret://
31
+ # refs (GCP Secret Manager / AWS / Vault) override per environment.
32
+
33
+ name: revenuecat
34
+ kind: source
35
+ version: "1.0.0"
36
+ summary: RevenueCat v2 API — customers, subscriptions, and daily chart metrics.
37
+ tags: [revenuecat, saas, subscriptions, analytics, mobile]
38
+
39
+ requires:
40
+ - "requests>=2.31"
41
+
42
+ params:
43
+ project_id:
44
+ type: string
45
+ required: true
46
+ description: RevenueCat project ID (e.g. proj1ab2c3d4).
47
+ base_url:
48
+ type: string
49
+ default: "https://api.revenuecat.com/v2"
50
+ description: RevenueCat v2 API base URL — overridable for stubbed tests.
51
+ page_size:
52
+ type: int
53
+ default: 100
54
+ description: Records per page on list endpoints (RC default is 20).
55
+
56
+ # --- metrics_daily-only ---------------------------------------------------
57
+ # Curated charts the metrics_daily stream pulls. Comma-separated chart_name
58
+ # values from RC's docs. Each chart can return multiple measures
59
+ # (e.g. churn = Actives + Churned Actives + Churn Rate); the connector
60
+ # flattens them all into the long-format output.
61
+ #
62
+ # The four defaults are charts that return the standard
63
+ # {cohort, incomplete, measure, value} shape. cohort_explorer and
64
+ # prediction_explorer return different shapes and will break the flattener
65
+ # — exclude unless you custom-handle them.
66
+ metrics_charts:
67
+ type: string
68
+ default: "revenue,mrr,actives,trials"
69
+ description: Comma-separated chart_name values for the metrics_daily stream.
70
+
71
+ # Earliest cohort date for the FIRST metrics_daily run. Once state exists
72
+ # this is ignored. ISO date, no time component.
73
+ metrics_initial_since_date:
74
+ type: string
75
+ default: "2024-01-01"
76
+ description: Earliest start_date for metrics_daily's first run.
77
+
78
+ # Re-pull window for metrics_daily: every run pulls from
79
+ # max(cursor, today - lookback_days) up to today. Wider window = more
80
+ # rate-limit churn but better safety for RC's "most recent day may be
81
+ # partial" caveat. 7 is safely past RC's same-day finalization window
82
+ # without burning many calls.
83
+ metrics_lookback_days:
84
+ type: int
85
+ default: 7
86
+ description: Days of overlap on every metrics_daily run (handles partial-today + late corrections).
87
+
88
+ secrets:
89
+ # The RC v2 secret key. The env-var default lets `pip install dtex`
90
+ # users get going without any extras; profile-level overrides
91
+ # (`ref: secret://gcp-secret-manager/...`) work for production.
92
+ - name: api_key
93
+ ref: ${env.REVENUECAT_API_KEY}
94
+
95
+ streams:
96
+ - name: customers
97
+ table: customers
98
+ primary_key: [id]
99
+ write_disposition: merge
100
+ schema:
101
+ - {name: id, type: STRING, mode: REQUIRED}
102
+ - {name: first_seen_at, type: TIMESTAMP}
103
+ - {name: last_seen_at, type: TIMESTAMP}
104
+ - {name: last_seen_app_version, type: STRING}
105
+ - {name: last_seen_country, type: STRING}
106
+ - {name: last_seen_platform, type: STRING}
107
+
108
+ - name: subscriptions
109
+ table: subscriptions
110
+ primary_key: [id]
111
+ write_disposition: merge
112
+ schema:
113
+ - {name: id, type: STRING, mode: REQUIRED}
114
+ - {name: customer_id, type: STRING}
115
+ - {name: customer_last_seen_at, type: TIMESTAMP}
116
+ - {name: product_id, type: STRING}
117
+ - {name: status, type: STRING}
118
+ - {name: gives_access, type: BOOLEAN}
119
+ - {name: auto_renewal_status, type: STRING}
120
+ - {name: store, type: STRING}
121
+ - {name: store_subscription_identifier, type: STRING}
122
+ - {name: current_period_starts_at, type: TIMESTAMP}
123
+ - {name: current_period_ends_at, type: TIMESTAMP}
124
+ - {name: total_revenue_gross_usd, type: FLOAT}
125
+ - {name: total_revenue_proceeds_usd, type: FLOAT}
126
+
127
+ # metrics_daily — long format (one row per cohort_date × chart × measure).
128
+ # Adding a new chart to `metrics_charts` requires no schema change.
129
+ # Cursor on `cohort_date`; the connector observes only complete days
130
+ # (incomplete=false in the API response) so the next run re-pulls
131
+ # today AND any day that was still partial last run.
132
+ - name: metrics_daily
133
+ table: metrics_daily
134
+ primary_key: [cohort_date, chart_name, measure_name]
135
+ write_disposition: merge
136
+ incremental:
137
+ cursor_field: cohort_date
138
+ cursor_type: date
139
+ initial_value: "2024-01-01"
140
+ schema:
141
+ - {name: cohort_date, type: DATE, mode: REQUIRED}
142
+ - {name: chart_name, type: STRING, mode: REQUIRED}
143
+ - {name: measure_name, type: STRING, mode: REQUIRED}
144
+ - {name: value, type: FLOAT}
145
+ - {name: incomplete, type: BOOLEAN}
146
+ - {name: pulled_at, type: TIMESTAMP, mode: REQUIRED}
147
+
148
+ schedule: "0 */6 * * *"