streamwright 0.1.4__tar.gz → 0.1.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {streamwright-0.1.4/src/streamwright.egg-info → streamwright-0.1.5}/PKG-INFO +5 -1
  2. {streamwright-0.1.4 → streamwright-0.1.5}/README.md +4 -0
  3. {streamwright-0.1.4 → streamwright-0.1.5}/pyproject.toml +2 -2
  4. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/catalog.py +44 -12
  5. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/cli.py +3 -1
  6. {streamwright-0.1.4 → streamwright-0.1.5/src/streamwright.egg-info}/PKG-INFO +5 -1
  7. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_catalog.py +77 -0
  8. {streamwright-0.1.4 → streamwright-0.1.5}/LICENSE +0 -0
  9. {streamwright-0.1.4 → streamwright-0.1.5}/setup.cfg +0 -0
  10. {streamwright-0.1.4 → streamwright-0.1.5}/setup.py +0 -0
  11. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/__init__.py +0 -0
  12. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/config/__init__.py +0 -0
  13. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/config/inputs.py +0 -0
  14. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/config/loader.py +0 -0
  15. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/config/reader.py +0 -0
  16. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/connectors.json +0 -0
  17. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/engine/__init__.py +0 -0
  18. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/engine/queries.py +0 -0
  19. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/engine/runner.py +0 -0
  20. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/engine/sql.py +0 -0
  21. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/net/__init__.py +0 -0
  22. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/net/downloads.py +0 -0
  23. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/net/http.py +0 -0
  24. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/__init__.py +0 -0
  25. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/dlt_output.py +0 -0
  26. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/exporter.py +0 -0
  27. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/output.py +0 -0
  28. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/staging.py +0 -0
  29. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/outputs/warehouse.py +0 -0
  30. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/runtime/__init__.py +0 -0
  31. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/runtime/components.py +0 -0
  32. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/runtime/logs.py +0 -0
  33. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/runtime/templates.py +0 -0
  34. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/runtime/testing.py +0 -0
  35. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/validation/__init__.py +0 -0
  36. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/validation/engine.py +0 -0
  37. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/validation/schema.py +0 -0
  38. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright/core/validation/source.py +0 -0
  39. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright.egg-info/SOURCES.txt +0 -0
  40. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright.egg-info/dependency_links.txt +0 -0
  41. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright.egg-info/entry_points.txt +0 -0
  42. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright.egg-info/requires.txt +0 -0
  43. {streamwright-0.1.4 → streamwright-0.1.5}/src/streamwright.egg-info/top_level.txt +0 -0
  44. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_connectors.py +0 -0
  45. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_dlt_output.py +0 -0
  46. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_exact_decimals.py +0 -0
  47. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_exporter.py +0 -0
  48. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_logging.py +0 -0
  49. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_output_formats.py +0 -0
  50. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_runner.py +0 -0
  51. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_sql.py +0 -0
  52. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_transform.py +0 -0
  53. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_validation.py +0 -0
  54. {streamwright-0.1.4 → streamwright-0.1.5}/tests/test_warehouse_output.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: streamwright
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Declarative data connectors — pull from APIs, files and databases with YAML, shape with SQL, write anywhere.
5
5
  Author: Karthick Jaganathan
6
6
  License-Expression: Apache-2.0
@@ -57,6 +57,10 @@ streamwright connectors install google_ads # an API connector (pulls in its SD
57
57
  streamwright connectors install google_ads meta_ads microsoft_ads # several at once (one pip install)
58
58
  ```
59
59
 
60
+ The catalog comes from the live [StreamWright hub](https://github.com/karthick-jaganathan/streamwright-hub). When it is
61
+ unreachable, the CLI warns and uses the last copy it fetched, else the one bundled with this release;
62
+ `--hub-url bundled` uses the bundled copy only, with no network (reproducible builds).
63
+
60
64
  ## Quickstart
61
65
 
62
66
  ```bash
@@ -29,6 +29,10 @@ streamwright connectors install google_ads # an API connector (pulls in its SD
29
29
  streamwright connectors install google_ads meta_ads microsoft_ads # several at once (one pip install)
30
30
  ```
31
31
 
32
+ The catalog comes from the live [StreamWright hub](https://github.com/karthick-jaganathan/streamwright-hub). When it is
33
+ unreachable, the CLI warns and uses the last copy it fetched, else the one bundled with this release;
34
+ `--hub-url bundled` uses the bundled copy only, with no network (reproducible builds).
35
+
32
36
  ## Quickstart
33
37
 
34
38
  ```bash
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "streamwright"
7
- version = "0.1.4"
7
+ version = "0.1.5"
8
8
  description = "Declarative data connectors — pull from APIs, files and databases with YAML, shape with SQL, write anywhere."
9
9
  readme = "README.md"
10
10
  license = "Apache-2.0"
@@ -58,5 +58,5 @@ packages = [
58
58
  ]
59
59
 
60
60
  [tool.setuptools.package-data]
61
- # the bundled connector catalog (default for `streamwright connectors`, refreshed from $STREAMWRIGHT_HUB_URL)
61
+ # the bundled connector catalog (the fallback of `streamwright connectors` when the live hub is unreachable)
62
62
  "streamwright.core" = ["connectors.json"]
@@ -1,9 +1,12 @@
1
1
  #!/usr/bin/env python
2
2
  """The connector catalog: discover and install connectors from the StreamWright hub.
3
3
 
4
- A catalog is a static index (bundled with streamwright, optionally refreshed from a
5
- remote hub URL) that maps a connector key to how it's installed - a PyPI
6
- requirement, a pinned git URL, or an image. The key is the same name used in
4
+ A catalog is a static index that maps a connector key to how it's installed - a
5
+ PyPI requirement, a pinned git URL, or an image. It is read from the live hub
6
+ (DEFAULT_HUB, or $STREAMWRIGHT_HUB_URL / --hub-url); when the hub is unreachable,
7
+ from the last copy fetched (~/.streamwright/connectors.json), else from the copy
8
+ bundled with this release - with a warning either way. The hub URL "bundled" uses
9
+ the bundled copy only (no network): for reproducible builds. The key is the same name used in
7
10
  source YAML (`provider`/`sdk`); the install source is just a delivery detail, so
8
11
  a package can move from git to PyPI without changing what users type.
9
12
 
@@ -24,7 +27,10 @@ except ImportError: # pragma: no cover
24
27
  from streamwright.core.runtime import components
25
28
 
26
29
  HUB_ENV = "STREAMWRIGHT_HUB_URL"
30
+ DEFAULT_HUB = "https://karthick-jaganathan.github.io/streamwright-hub/index.json"
31
+ BUNDLED = "bundled" # a hub URL that means: the catalog bundled with this release, no network
27
32
  CACHE = Path(os.path.expanduser("~/.streamwright/connectors.json"))
33
+ TIMEOUT_S = 5
28
34
 
29
35
 
30
36
  def _bundled():
@@ -32,12 +38,18 @@ def _bundled():
32
38
  return json.loads(text)
33
39
 
34
40
 
41
+ def _valid(data):
42
+ return isinstance(data, dict) and isinstance(data.get("connectors"), dict)
43
+
44
+
35
45
  def _remote(url):
36
46
  import requests
37
47
 
38
- response = requests.get(url, timeout=10)
48
+ response = requests.get(url, timeout=TIMEOUT_S)
39
49
  response.raise_for_status()
40
50
  data = response.json()
51
+ if not _valid(data):
52
+ raise ValueError("not a connector catalog (no `connectors` mapping)")
41
53
  try:
42
54
  CACHE.parent.mkdir(parents=True, exist_ok=True)
43
55
  CACHE.write_text(json.dumps(data))
@@ -47,21 +59,41 @@ def _remote(url):
47
59
 
48
60
 
49
61
  def load(hub_url=None):
50
- """The catalog: the remote hub if configured (cached), else the bundled default."""
51
- url = hub_url or os.environ.get(HUB_ENV)
52
- if url:
53
- try:
54
- return _remote(url)
55
- except Exception: # offline or hub down: fall back
56
- pass
62
+ """
63
+ The catalog: the live hub (`hub_url`, else $STREAMWRIGHT_HUB_URL, else DEFAULT_HUB; each fetch is cached). When the
64
+ hub cannot be read, the cached copy, else the bundled one, with a warning on stderr. `bundled`: the bundled copy only.
65
+ """
66
+ url = hub_url or os.environ.get(HUB_ENV) or DEFAULT_HUB
67
+ if url == BUNDLED:
68
+ return _bundled()
69
+ try:
70
+ return _remote(url)
71
+ except Exception as exc: # offline, hub down or a bad publish: fall back, visibly
72
+ reason = type(exc).__name__ # network errors: the type is enough; an HTTP status or a bad index says why
73
+ status = getattr(getattr(exc, "response", None), "status_code", None)
74
+ if status:
75
+ reason = "HTTP %s" % status
76
+ elif isinstance(exc, ValueError):
77
+ reason += ": %s" % str(exc)[:100]
57
78
  if CACHE.exists():
58
79
  try:
59
- return json.loads(CACHE.read_text())
80
+ data = json.loads(CACHE.read_text())
81
+ if _valid(data):
82
+ sys.stderr.write("streamwright: the connector hub %s is unreachable (%s); using the copy fetched on %s\n"
83
+ % (url, reason, _mtime(CACHE)))
84
+ return data
60
85
  except Exception:
61
86
  pass
87
+ sys.stderr.write("streamwright: the connector hub %s is unreachable (%s); using the catalog bundled with this "
88
+ "release\n" % (url, reason))
62
89
  return _bundled()
63
90
 
64
91
 
92
+ def _mtime(path):
93
+ import datetime
94
+ return datetime.datetime.fromtimestamp(path.stat().st_mtime).strftime("%Y-%m-%d %H:%M")
95
+
96
+
65
97
  def connectors(hub_url=None):
66
98
  return load(hub_url).get("connectors", {})
67
99
 
@@ -247,7 +247,9 @@ def _parser():
247
247
  help="connector keys (install: one or more) or words (search)")
248
248
  conn.add_argument("--yes", action="store_true", help="skip the confirmation prompt when installing")
249
249
  conn.add_argument("--hub-url", metavar="URL",
250
- help="catalog URL to use (default: the bundled catalog, or $STREAMWRIGHT_HUB_URL)")
250
+ help="the connector hub (default: $STREAMWRIGHT_HUB_URL, else the live StreamWright hub; when it is "
251
+ "unreachable, the last fetched or the bundled catalog, with a warning). 'bundled': the "
252
+ "catalog bundled with this release, no network")
251
253
 
252
254
  validate = commands.add_parser("validate", help="check configuration files: streamwright-validate, plus the checks of "
253
255
  "the installed connectors and query builders", add_help=False)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: streamwright
3
- Version: 0.1.4
3
+ Version: 0.1.5
4
4
  Summary: Declarative data connectors — pull from APIs, files and databases with YAML, shape with SQL, write anywhere.
5
5
  Author: Karthick Jaganathan
6
6
  License-Expression: Apache-2.0
@@ -57,6 +57,10 @@ streamwright connectors install google_ads # an API connector (pulls in its SD
57
57
  streamwright connectors install google_ads meta_ads microsoft_ads # several at once (one pip install)
58
58
  ```
59
59
 
60
+ The catalog comes from the live [StreamWright hub](https://github.com/karthick-jaganathan/streamwright-hub). When it is
61
+ unreachable, the CLI warns and uses the last copy it fetched, else the one bundled with this release;
62
+ `--hub-url bundled` uses the bundled copy only, with no network (reproducible builds).
63
+
60
64
  ## Quickstart
61
65
 
62
66
  ```bash
@@ -1,6 +1,7 @@
1
1
  """`streamwright connectors install KEY [KEY ...]` and search: resolution, consent and the single pip command."""
2
2
 
3
3
  import io
4
+ import json
4
5
  import sys
5
6
 
6
7
  import pytest
@@ -125,3 +126,79 @@ def test_cli_search_joins_its_words(pip, capsys):
125
126
  assert cli.main(["connectors", "search", "meta", "insights"]) == 0
126
127
  out = capsys.readouterr().out
127
128
  assert "meta_ads" in out and "google_ads" not in out
129
+
130
+
131
+ # --- where the catalog comes from: the live hub, else the last fetched copy, else the bundled one ---
132
+
133
+ GOOD = {"schema_version": 1, "connectors": {"files": CATALOG["files"]}}
134
+
135
+
136
+ class Response:
137
+ def __init__(self, data, status=200):
138
+ self.data, self.status = data, status
139
+
140
+ def raise_for_status(self):
141
+ if self.status >= 400:
142
+ raise RuntimeError("HTTP %d" % self.status)
143
+
144
+ def json(self):
145
+ return self.data
146
+
147
+
148
+ @pytest.fixture
149
+ def hub(monkeypatch, tmp_path):
150
+ """No $STREAMWRIGHT_HUB_URL, a private cache file, and requests.get answering from `responses` (url -> data)."""
151
+ monkeypatch.delenv(catalog.HUB_ENV, raising=False)
152
+ monkeypatch.setattr(catalog, "CACHE", tmp_path / "connectors.json")
153
+ state = {"responses": {}, "urls": []}
154
+
155
+ def get(url, timeout):
156
+ state["urls"].append(url)
157
+ if url not in state["responses"]:
158
+ raise ConnectionError("cannot reach %s" % url)
159
+ return Response(state["responses"][url])
160
+ monkeypatch.setattr("requests.get", get)
161
+ return state
162
+
163
+
164
+ def test_the_live_hub_is_the_default_and_is_cached(hub, capsys):
165
+ hub["responses"][catalog.DEFAULT_HUB] = GOOD
166
+ assert catalog.load() == GOOD
167
+ assert hub["urls"] == [catalog.DEFAULT_HUB]
168
+ assert json.loads(catalog.CACHE.read_text()) == GOOD
169
+ assert capsys.readouterr().err == ""
170
+
171
+
172
+ def test_the_hub_url_can_be_set(hub, monkeypatch):
173
+ hub["responses"]["https://hub.example/a.json"] = GOOD
174
+ hub["responses"]["https://hub.example/b.json"] = GOOD
175
+ monkeypatch.setenv(catalog.HUB_ENV, "https://hub.example/a.json")
176
+ catalog.load()
177
+ catalog.load("https://hub.example/b.json") # --hub-url wins over the environment
178
+ assert hub["urls"] == ["https://hub.example/a.json", "https://hub.example/b.json"]
179
+
180
+
181
+ def test_bundled_uses_no_network(hub):
182
+ data = catalog.load(catalog.BUNDLED)
183
+ assert hub["urls"] == [] and data == catalog._bundled()
184
+
185
+
186
+ def test_an_unreachable_hub_falls_back_to_the_last_fetched_copy_with_a_warning(hub, capsys):
187
+ catalog.CACHE.write_text(json.dumps(GOOD))
188
+ assert catalog.load() == GOOD
189
+ err = capsys.readouterr().err
190
+ assert "the connector hub %s is unreachable (ConnectionError" % catalog.DEFAULT_HUB in err
191
+ assert "using the copy fetched on" in err
192
+
193
+
194
+ def test_with_no_fetched_copy_the_bundled_catalog_is_used_with_a_warning(hub, capsys):
195
+ assert catalog.load() == catalog._bundled()
196
+ assert "using the catalog bundled with this release" in capsys.readouterr().err
197
+
198
+
199
+ def test_a_malformed_index_is_not_used_nor_cached(hub, capsys):
200
+ catalog.CACHE.write_text(json.dumps(GOOD))
201
+ hub["responses"][catalog.DEFAULT_HUB] = {"oops": True}
202
+ assert catalog.load() == GOOD # the good copy, not the bad publish
203
+ assert json.loads(catalog.CACHE.read_text()) == GOOD
204
+ assert "not a connector catalog" in capsys.readouterr().err
File without changes
File without changes
File without changes