marpledata 3.5.0__tar.gz → 3.6.0.dev1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/AGENTS.md +3 -1
  2. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/CHANGELOG.md +14 -0
  3. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/CONTRIBUTING.md +2 -3
  4. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/PKG-INFO +2 -1
  5. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/README.md +1 -0
  6. marpledata-3.6.0.dev1/RELEASING.md +13 -0
  7. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/getting-started.rst +7 -0
  8. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/tutorials.rst +3 -0
  9. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/pyproject.toml +6 -1
  10. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/__init__.py +1 -1
  11. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/__init__.py +14 -0
  12. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/dataset.py +64 -17
  13. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/signal.py +13 -0
  14. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/signal_upload.py +1 -1
  15. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/utils.py +8 -13
  16. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/conftest.py +6 -10
  17. marpledata-3.6.0.dev1/tests/support.py +60 -0
  18. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/test_db.py +89 -83
  19. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/test_insight.py +3 -1
  20. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/test_signal_upload.py +148 -108
  21. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/test_sql.py +3 -0
  22. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/uv.lock +1 -1
  23. marpledata-3.5.0/RELEASING.md +0 -11
  24. marpledata-3.5.0/tests/support.py +0 -24
  25. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.flake8 +0 -0
  26. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.gitignore +0 -0
  27. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.python-version +0 -0
  28. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.uv-cache/.gitignore +0 -0
  29. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.uv-cache/.lock +0 -0
  30. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.uv-cache/CACHEDIR.TAG +0 -0
  31. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
  32. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/.uv-cache/sdists-v9/.gitignore +0 -0
  33. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/LICENSE +0 -0
  34. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/DEPLOYMENT.md +0 -0
  35. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/_static/custom.css +0 -0
  36. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/_static/favicon.png +0 -0
  37. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/_static/logo.png +0 -0
  38. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/db.rst +0 -0
  39. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/insight.rst +0 -0
  40. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.Insight.rst +0 -0
  41. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.DB.rst +0 -0
  42. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.DataStream.rst +0 -0
  43. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.Dataset.rst +0 -0
  44. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.DatasetList.rst +0 -0
  45. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.LAKE_ARROW_SCHEMA.rst +0 -0
  46. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.SCHEMA.rst +0 -0
  47. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.Signal.rst +0 -0
  48. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.SignalUpload.rst +0 -0
  49. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api/marple.db.SignalsAlreadyExistError.rst +0 -0
  50. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/api.rst +0 -0
  51. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/conf.py +0 -0
  52. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/docs/index.rst +0 -0
  53. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/example.py +0 -0
  54. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/examples_race.csv +0 -0
  55. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/pytest.xml +0 -0
  56. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/constants.py +0 -0
  57. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/datastream.py +0 -0
  58. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/db/sql.py +0 -0
  59. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/insight.py +0 -0
  60. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/src/marple/py.typed +0 -0
  61. {marpledata-3.5.0 → marpledata-3.6.0.dev1}/tests/test_reliability.py +0 -0
@@ -15,6 +15,7 @@ This directory contains the Python SDK package published as `marpledata`.
15
15
  ## Commands
16
16
 
17
17
  - Run tests with output: `uv run pytest -vs`
18
+ - Unit tests only: `uv run pytest -m "not integration"`
18
19
  - Build docs: `uv run --group docs sphinx-build -b html docs docs/_build/html`
19
20
  - Build package: `uv build`
20
21
  - Fix formatting: `uv run isort src tests && uv run black src tests`
@@ -29,5 +30,6 @@ Run commands from `python/` unless a command explicitly says otherwise.
29
30
  - Keep Python SDK tests in `python/tests/`.
30
31
  - Integration tests run against live Marple services and may require
31
32
  `MDB_TOKEN`, `MDB_URL`, `INSIGHT_TOKEN`, and `INSIGHT_URL`. Tests should skip
32
- or fail clearly when required credentials are missing.
33
+ clearly when required credentials are missing. Mark live tests with
34
+ `pytest.mark.integration`.
33
35
  - Release steps: `RELEASING.md`.
@@ -5,6 +5,20 @@ All notable changes to the Python SDK package `marpledata` will be documented in
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [Unreleased]
9
+
10
+ ### Changed
11
+
12
+ - Exact signal names in `get_data` / `get_signal` / `get_signals` resolve via the dataset cache or `GET /datapool/{pool}/signal/{name}/id` instead of downloading the full datapool `signal_map`. Regex patterns still use the map.
13
+
14
+ ### Added
15
+
16
+ - `DB.delete_signals`, `Dataset.delete_signal` / `Dataset.delete_signals`, and `Signal.delete` to remove signals from a dataset.
17
+
18
+ ### Fixed
19
+
20
+ - Allow `add_signal` in post-processing
21
+
8
22
  ## [3.5.0] - 2026-08-20
9
23
 
10
24
  ### Added
@@ -22,11 +22,10 @@ uv run mypy --install-types --non-interactive
22
22
 
23
23
  ### Testing
24
24
 
25
- The testing suite runs against a real linked DB & Insight deployment on our SaaS (e.g. Castro Comrades).
26
- Among other things it will create a stream, ingest a dataset, and export that dataset from Insight
27
- Upload tests also exercise server and multipart upload flows, and create/delete temporary streams with a test prefix.
25
+ The testing suite includes local unit tests and integration tests against a live Marple DB / Insight deployment. Integration tests share one imported example dataset per session, use a small CSV for extra ingestions, and mark themselves with `pytest.mark.integration`.
28
26
 
29
27
  ```bash
28
+ uv run pytest -m "not integration" # local unit tests only
30
29
  export MDB_TOKEN=...
31
30
  export INSIGHT_TOKEN=...
32
31
  # Optional, defaults to SaaS URLs:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: marpledata
3
- Version: 3.5.0
3
+ Version: 3.6.0.dev1
4
4
  Summary: Marple SDK for Python
5
5
  Project-URL: Homepage, https://www.marpledata.com/
6
6
  Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
@@ -220,6 +220,7 @@ if len(datasets) > 0:
220
220
  - **Get a resampled df of multiple signals**: `dataset.get_data(signals=[...], resample_rule="1s")`
221
221
  - **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
222
222
  - **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
223
+ - **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
223
224
 
224
225
  For live streams:
225
226
 
@@ -185,6 +185,7 @@ if len(datasets) > 0:
185
185
  - **Get a resampled df of multiple signals**: `dataset.get_data(signals=[...], resample_rule="1s")`
186
186
  - **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
187
187
  - **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
188
+ - **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
188
189
 
189
190
  For live streams:
190
191
 
@@ -0,0 +1,13 @@
1
+ # Releasing `marpledata`
2
+
3
+ Tokens: 1Password (`username: __token__`). Run from `python/`.
4
+
5
+ - **Bump version**
6
+ - `uv version --bump minor` / `uv version x.y.z`
7
+ - Manual update `__init__.py`, `CHANGELOG.md`
8
+ - **TestPyPI** — `rm -rf dist && uv build && uv publish --index testpypi`
9
+ Smoke: `pip install -i https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata==<version>`
10
+ - **PyPI** — `rm -rf dist && uv build && uv publish`
11
+ - **Docs** — run the GitLab `pages` job (Sphinx → https://marpledata.gitlab.io/marple-sdk/)
12
+
13
+ No GitHub tag/release; `CHANGELOG.md` is the release history.
@@ -50,6 +50,13 @@ longer ``wait_for_import`` timeout and optionally increase upload concurrency:
50
50
  dataset = stream.push_file("large_export.csv", concurrency=8)
51
51
  dataset = dataset.wait_for_import(timeout=180)
52
52
 
53
+ To replace an existing dataset with the same name, pass ``overwrite=True``:
54
+
55
+ .. code-block:: python
56
+
57
+ dataset = stream.push_file("examples_race.csv", overwrite=True)
58
+ dataset = dataset.wait_for_import(timeout=10)
59
+
53
60
  If direct storage uploads are blocked by your network or proxy, force upload
54
61
  through the Marple DB API server:
55
62
 
@@ -31,8 +31,11 @@ Import a file and wait for import
31
31
  "examples_race.csv",
32
32
  metadata={"source": "testbench"},
33
33
  concurrency=8,
34
+ overwrite=False,
34
35
  ).wait_for_import(timeout=180)
35
36
 
37
+ Pass ``overwrite=True`` to replace an existing dataset with the same name.
38
+
36
39
  Add signals to a dataset
37
40
  ------------------------
38
41
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "marpledata"
3
- version = "3.5.0"
3
+ version = "3.6.0.dev1"
4
4
  description = "Marple SDK for Python"
5
5
  authors = [
6
6
  { name = "Matthias Baert", email = "support@marpledata.com" },
@@ -72,6 +72,11 @@ docs = [
72
72
  "sphinx>=7.2.6",
73
73
  ]
74
74
 
75
+ [tool.pytest.ini_options]
76
+ markers = [
77
+ "integration: tests that talk to a live Marple DB / Insight instance",
78
+ ]
79
+
75
80
  [tool.black]
76
81
  line-length = 113
77
82
  target-version = ["py311"]
@@ -3,4 +3,4 @@ from .db import DB
3
3
  from .insight import Insight
4
4
 
5
5
  __all__ = ["DB", "Insight", "db"]
6
- __version__ = "3.5.0"
6
+ __version__ = "3.6.0.dev1"
@@ -432,6 +432,20 @@ class DB:
432
432
  r = self.post(f"/stream/{dataset.datastream_id}/dataset/{dataset.id}/delete")
433
433
  validate_response(r, "Delete dataset failed")
434
434
 
435
+ def delete_signals(
436
+ self,
437
+ dataset_id: int | None,
438
+ dataset_path: str | None,
439
+ signal_ids: Sequence[int],
440
+ ) -> None:
441
+ """
442
+ Delete signals from a dataset by ID or path.
443
+
444
+ Warning:
445
+ This is a destructive operation that cannot be undone.
446
+ """
447
+ self.get_dataset(dataset_id, dataset_path).delete_signals(signal_ids)
448
+
435
449
  @deprecated
436
450
  def update_metadata(
437
451
  self,
@@ -131,7 +131,7 @@ class Dataset(BaseModel):
131
131
  """Get a specific signal in this dataset by its name or ID.
132
132
 
133
133
  Args:
134
- name: Signal name (resolved via the datapool signal map).
134
+ name: Signal name.
135
135
  id: Signal ID.
136
136
  refresh: If True, refetch from the API even when cached.
137
137
  """
@@ -141,7 +141,7 @@ class Dataset(BaseModel):
141
141
  raise ValueError("Only one of name or id can be provided.")
142
142
 
143
143
  if name is not None:
144
- id = self._client.get_signal_map().get(name)
144
+ id = self._client.get_signal_id(name)
145
145
 
146
146
  if id is None:
147
147
  raise ValueError(f"Signal with name {name} not found in dataset with id {self.id}.")
@@ -192,19 +192,39 @@ class Dataset(BaseModel):
192
192
  """
193
193
  if signal_names is not None and signal_ids is not None:
194
194
  raise ValueError("Provide only one of signal_names or signal_ids")
195
-
196
- if signal_ids is not None:
197
- return self._get_signals_by_ids(list(signal_ids), refresh=refresh)
198
-
199
- if signal_names is None:
195
+ if signal_ids is None and signal_names is None:
200
196
  return self._get_all_signals()
201
-
202
- matched = self._client.find_matching_signals(signal_names)
203
- if not matched:
204
- return []
205
- return self._get_signals_by_ids(list(matched.values()), refresh=refresh)
206
-
207
- def _get_signals_by_ids(self, signal_ids: Sequence[int], *, refresh: bool = False) -> list["Signal"]:
197
+ if signal_ids is None:
198
+ assert signal_names is not None
199
+ signal_ids = self._find_signal_ids(signal_names).values()
200
+ return self._get_signals_by_ids(list(signal_ids), refresh=refresh)
201
+
202
+ def _find_signal_ids(self, signals: Iterable[str | re.Pattern]) -> dict[str, int]:
203
+ found: dict[str, int] = {}
204
+ exact: list[str] = []
205
+ patterns: list[re.Pattern] = []
206
+ for signal in signals:
207
+ (exact if isinstance(signal, str) else patterns).append(signal) # type: ignore [arg-type]
208
+
209
+ if patterns:
210
+ # Only look up the signal map in the regex case.
211
+ for name, sig_id in self._client.get_signal_map().items():
212
+ if any(pattern.search(name) for pattern in patterns):
213
+ found[name] = sig_id
214
+
215
+ if self.n_signals is not None and self.n_signals < 100:
216
+ self._get_all_signals()
217
+
218
+ cached = {s.name: s.id for s in self._signals.values()}
219
+ for name in exact:
220
+ if (signal_id := cached.get(name)) is not None:
221
+ found[name] = signal_id
222
+ elif (signal_id := self._client.get_signal_id(name)) is not None:
223
+ found[name] = signal_id
224
+
225
+ return found
226
+
227
+ def _get_signals_by_ids(self, signal_ids: list[int], *, refresh: bool = False) -> list["Signal"]:
208
228
  if not signal_ids:
209
229
  return []
210
230
 
@@ -254,7 +274,7 @@ class Dataset(BaseModel):
254
274
  A pandas DataFrame containing one column per signal, aligned on time.
255
275
  """
256
276
  return self._get_signals_dataframe(
257
- self._client.find_matching_signals(signals).items(),
277
+ self._find_signal_ids(signals).items(),
258
278
  resample_rule,
259
279
  resample_aggregate,
260
280
  dtype,
@@ -390,7 +410,7 @@ class Dataset(BaseModel):
390
410
  return []
391
411
  if len(signals) > MAX_SIGNALS_PER_ADD:
392
412
  raise ValueError(f"Provide at most {MAX_SIGNALS_PER_ADD} signals per call")
393
- if self.import_status != ImportStatus.FINISHED:
413
+ if self.import_status not in (ImportStatus.FINISHED, ImportStatus.POSTPROCESSING):
394
414
  raise ValueError(f"Dataset {self.id} is not in a writable state (status: {self.import_status})")
395
415
 
396
416
  signal_ids = run_signal_uploads(self, signals, overwrite=overwrite, concurrency=concurrency)
@@ -488,6 +508,33 @@ class Dataset(BaseModel):
488
508
  warnings.warn(f"Import did not finish after {timeout} seconds")
489
509
  return self.fetch(self._client, self.id)
490
510
 
511
+ def delete_signal(self, signal_id: int) -> None:
512
+ """
513
+ Delete a single signal from this dataset by ID.
514
+
515
+ Warning:
516
+ This is a destructive action that cannot be undone.
517
+ """
518
+ self.delete_signals([signal_id])
519
+
520
+ def delete_signals(self, signal_ids: Sequence[int]) -> None:
521
+ """
522
+ Delete one or more signals from this dataset by ID.
523
+
524
+ Warning:
525
+ This is a destructive action that cannot be undone.
526
+ """
527
+ to_delete = list(set(signal_ids))
528
+ if not to_delete:
529
+ return
530
+ r = self._client.post(
531
+ f"/stream/{self.datastream_id}/dataset/{self.id}/signals/delete",
532
+ json={"signal_ids": to_delete},
533
+ )
534
+ validate_response(r, "Delete signals failed")
535
+ for signal_id in to_delete:
536
+ self._signals.pop(signal_id, None)
537
+
491
538
  def delete(self) -> None:
492
539
  """
493
540
  Delete the dataset.
@@ -691,7 +738,7 @@ class DatasetList(UserList[Dataset]):
691
738
  if len(self.data) == 0:
692
739
  return
693
740
  # Avoid having to search signals for every individual dataset
694
- signal_pairs = list(self.data[0]._client.find_matching_signals(signals).items())
741
+ signal_pairs = list(self.data[0]._find_signal_ids(signals).items())
695
742
  for dataset in self.data:
696
743
  yield dataset, dataset._get_signals_dataframe(
697
744
  signals=signal_pairs,
@@ -137,3 +137,16 @@ class Signal(BaseModel):
137
137
  warnings.warn(f"Signal {self.name} did not reach a queryable state after {timeout} seconds")
138
138
  return fresh
139
139
  time.sleep(0.5)
140
+
141
+ def delete(self) -> None:
142
+ """
143
+ Delete this signal from its dataset.
144
+
145
+ Warning:
146
+ This is a destructive action that cannot be undone.
147
+ """
148
+ r = self._client.post(
149
+ f"/stream/{self.datastream_id}/dataset/{self.dataset_id}/signals/delete",
150
+ json={"signal_ids": [self.id]},
151
+ )
152
+ validate_response(r, "Delete signal failed")
@@ -286,7 +286,7 @@ def _presign_signals(
286
286
  body = {}
287
287
  raise SignalsAlreadyExistError(
288
288
  body.get("signals") or [], # type: ignore[arg-type]
289
- message=f"Signal upload failed: {body.get('error', 'signals_already_exist')}",
289
+ message=f"Signal upload failed (use overwrite=True to replace existing signals): {body.get('error', 'signals_already_exist')}",
290
290
  )
291
291
  signals = {
292
292
  item["name"]: PresignedSignal.model_validate(item)
@@ -1,6 +1,5 @@
1
- import re
2
1
  from pathlib import Path
3
- from typing import Any, Iterable, Literal
2
+ from typing import Any, Literal
4
3
  from urllib import parse, request
5
4
 
6
5
  import pandas as pd
@@ -130,17 +129,13 @@ class DBClient:
130
129
  self._signal_map = validate_response(r, "Get signals failed")
131
130
  return self._signal_map
132
131
 
133
- def find_matching_signals(self, signals: Iterable[str | re.Pattern]) -> dict[str, int]:
134
- all_signals = self.get_signal_map()
135
- matching = dict()
136
- for pattern in signals:
137
- if isinstance(pattern, str) and pattern in all_signals:
138
- matching[pattern] = all_signals[pattern]
139
- elif isinstance(pattern, re.Pattern):
140
- for name, id in all_signals.items():
141
- if pattern.search(name):
142
- matching[name] = id
143
- return matching
132
+ def get_signal_id(self, name: str) -> int | None:
133
+ if self._signal_map is not None:
134
+ return self._signal_map.get(name)
135
+ r = self.get(f"/datapool/{self.datapool}/signal/{parse.quote(name, safe='')}/id")
136
+ if r.status_code == 404:
137
+ return None
138
+ return validate_response(r, "Get signal id failed")["id"]
144
139
 
145
140
  def cache_parquet(self, dataset_id: int, signal_id: int, refresh_cache: bool = False) -> Path:
146
141
  """
@@ -16,11 +16,11 @@ dotenv.load_dotenv()
16
16
  def _required_env(name: str) -> str:
17
17
  value = os.getenv(name)
18
18
  if value is None:
19
- pytest.fail(f"Missing env var {name}; skipping integration test.")
19
+ pytest.skip(f"Missing env var {name}; skipping integration test.")
20
20
  return value
21
21
 
22
22
 
23
- @pytest.fixture()
23
+ @pytest.fixture(scope="session")
24
24
  def db() -> DB:
25
25
  url = os.getenv("MDB_URL", marple.db.SAAS_URL)
26
26
  assert url is not None
@@ -35,17 +35,13 @@ def insight() -> Insight:
35
35
 
36
36
 
37
37
  @pytest.fixture(scope="session")
38
- def example_stream() -> Generator[DataStream, None, None]:
39
- url = os.getenv("MDB_URL", marple.db.SAAS_URL)
40
- assert url is not None
41
- session_db = DB(_required_env("MDB_TOKEN"), url)
42
-
38
+ def example_stream(db: DB) -> Generator[DataStream, None, None]:
43
39
  name = "Salty Compulsory Pytest " + datetime.now().isoformat()
44
- yield session_db.create_stream(name)
40
+ yield db.create_stream(name)
45
41
  print("Cleaning up stream...")
46
- session_db.delete_stream(name)
42
+ db.delete_stream(name)
47
43
 
48
44
 
49
- @pytest.fixture()
45
+ @pytest.fixture(scope="session")
50
46
  def example_dataset(example_stream: DataStream) -> Dataset:
51
47
  return ingest_dataset(example_stream, metadata={"A": 1, "B": 1})
@@ -0,0 +1,60 @@
1
+ from collections.abc import Iterator
2
+ from contextlib import contextmanager
3
+ from datetime import datetime
4
+ from pathlib import Path
5
+ from uuid import uuid4
6
+
7
+ import marple
8
+ from marple import DB
9
+ from marple.db import Dataset, DataStream
10
+
11
+ EXAMPLE_CSV = Path(__file__).parents[2] / "test_data" / "examples_race.csv"
12
+
13
+ _swept_prefixes: set[str] = set()
14
+
15
+
16
+ def unique_name(prefix: str, suffix: str = "") -> str:
17
+ """Datapool-unique name so parallel Python/Rust CI jobs cannot collide."""
18
+ return f"{prefix}-{uuid4().hex}{suffix}"
19
+
20
+
21
+ def sweep_streams_by_prefix(db: DB, prefix: str) -> None:
22
+ """Delete leftover streams with this prefix once per process."""
23
+ if prefix in _swept_prefixes:
24
+ return
25
+ _swept_prefixes.add(prefix)
26
+ for stream in db.get_streams():
27
+ if stream.name.startswith(prefix):
28
+ try:
29
+ db.delete_stream(stream.id)
30
+ except Exception:
31
+ pass
32
+
33
+
34
+ @contextmanager
35
+ def isolated_stream(db: DB, prefix: str, suffix: str, **create_kwargs) -> Iterator[DataStream]:
36
+ sweep_streams_by_prefix(db, prefix)
37
+ stream = db.create_stream(f"{prefix} {suffix} {datetime.now().isoformat()}", **create_kwargs)
38
+ try:
39
+ yield stream
40
+ finally:
41
+ try:
42
+ db.delete_stream(stream.id)
43
+ except Exception:
44
+ pass
45
+
46
+
47
+ def ingest_dataset(stream: DataStream, metadata: dict | None = None) -> Dataset:
48
+ dataset = stream.push_file(
49
+ str(EXAMPLE_CSV),
50
+ metadata={
51
+ "source": "pytest:test_db.py",
52
+ "sdk_version": marple.__version__,
53
+ }
54
+ | (metadata or {}),
55
+ file_name=unique_name("py-sdk", ".csv"),
56
+ ).wait_for_import(timeout=180)
57
+ assert (
58
+ dataset.import_status == "FINISHED"
59
+ ), f"Dataset {dataset.id} did not finish importing (status: {dataset.import_status})"
60
+ return dataset