marpledata 3.4.0.dev1__tar.gz → 3.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.gitignore +4 -1
  2. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/AGENTS.md +1 -0
  3. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/CHANGELOG.md +12 -1
  4. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/CONTRIBUTING.md +2 -30
  5. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/PKG-INFO +25 -18
  6. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/README.md +23 -16
  7. marpledata-3.5.0/RELEASING.md +11 -0
  8. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.Signal.rst +1 -1
  9. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SignalUpload.rst +0 -1
  10. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/getting-started.rst +3 -2
  11. marpledata-3.5.0/docs/tutorials.rst +125 -0
  12. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/example.py +3 -1
  13. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/pyproject.toml +1 -1
  14. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/__init__.py +1 -1
  15. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/__init__.py +19 -10
  16. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/dataset.py +8 -5
  17. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/datastream.py +21 -8
  18. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/signal.py +4 -5
  19. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/signal_upload.py +32 -3
  20. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/utils.py +2 -1
  21. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_db.py +24 -6
  22. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_signal_upload.py +112 -8
  23. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/uv.lock +1 -1
  24. marpledata-3.4.0.dev1/.env.local +0 -3
  25. marpledata-3.4.0.dev1/.env.nightly +0 -2
  26. marpledata-3.4.0.dev1/.env.staging +0 -4
  27. marpledata-3.4.0.dev1/augment-dataset.py +0 -681
  28. marpledata-3.4.0.dev1/docs/tutorials.rst +0 -106
  29. marpledata-3.4.0.dev1/ingest-ibiza.py +0 -538
  30. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.flake8 +0 -0
  31. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.python-version +0 -0
  32. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/.gitignore +0 -0
  33. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/.lock +0 -0
  34. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/CACHEDIR.TAG +0 -0
  35. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
  36. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/sdists-v9/.gitignore +0 -0
  37. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/LICENSE +0 -0
  38. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/DEPLOYMENT.md +0 -0
  39. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/custom.css +0 -0
  40. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/favicon.png +0 -0
  41. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/logo.png +0 -0
  42. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/db.rst +0 -0
  43. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/insight.rst +0 -0
  44. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.Insight.rst +0 -0
  45. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DB.rst +0 -0
  46. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DataStream.rst +0 -0
  47. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.Dataset.rst +0 -0
  48. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DatasetList.rst +0 -0
  49. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.LAKE_ARROW_SCHEMA.rst +0 -0
  50. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SCHEMA.rst +0 -0
  51. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SignalsAlreadyExistError.rst +0 -0
  52. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api.rst +0 -0
  53. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/conf.py +0 -0
  54. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/index.rst +0 -0
  55. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/examples_race.csv +0 -0
  56. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/pytest.xml +0 -0
  57. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/constants.py +0 -0
  58. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/sql.py +0 -0
  59. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/insight.py +0 -0
  60. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/py.typed +0 -0
  61. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/conftest.py +0 -0
  62. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/support.py +0 -0
  63. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_insight.py +0 -0
  64. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_reliability.py +0 -0
  65. {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_sql.py +0 -0
@@ -169,4 +169,7 @@ cython_debug/
169
169
  # and can be added to the global gitignore or merged into this file. For a more nuclear
170
170
  # option (not recommended) you can uncomment the following to ignore the entire idea folder.
171
171
  .idea/
172
- *.env
172
+ .env*
173
+
174
+ # Client handoff pack (may contain a real API token)
175
+ /marple-matlab/
@@ -30,3 +30,4 @@ Run commands from `python/` unless a command explicitly says otherwise.
30
30
  - Integration tests run against live Marple services and may require
31
31
  `MDB_TOKEN`, `MDB_URL`, `INSIGHT_TOKEN`, and `INSIGHT_URL`. Tests should skip
32
32
  or fail clearly when required credentials are missing.
33
+ - Release steps: `RELEASING.md`.
@@ -5,7 +5,18 @@ All notable changes to the Python SDK package `marpledata` will be documented in
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
- ## [3.4.0] - 2026-07-16
8
+ ## [3.5.0] - 2026-08-20
9
+
10
+ ### Added
11
+
12
+ - `Dataset.add_signal` and `Dataset.add_signals` now accept a time-indexed pandas Series or DataFrame (DatetimeIndex / TimedeltaIndex when there is no `time` column).
13
+ - `DataStream.push_file(..., overwrite=False)` to replace an existing dataset with the same name (also on the deprecated `DB.push_file`).
14
+
15
+ ### Fixed
16
+
17
+ - An empty local parquet cache folder is treated as a cache miss and re-downloaded, instead of being assumed to mean the signal has no data.
18
+
19
+ ## [3.4.0] - 2026-07-28
9
20
 
10
21
  ### Added
11
22
 
@@ -55,34 +55,6 @@ Open `docs/_build/html/index.html` in your browser to view the site.
55
55
  - `marple.__version__`
56
56
  - `from marple import Insight, DB`
57
57
 
58
- ### Publishing (Test Pypi)
58
+ ### Publishing
59
59
 
60
- Only once on your laptop
61
-
62
- - `uv build`
63
- - `uv publish --index testpypi`
64
- - `username: __token__`
65
- - `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
66
-
67
- ### Publishing Production
68
-
69
- - Make sure you don't have any local changes
70
- - Delete `./dist` folder
71
- - `uv version x.y.z(.devi)`
72
- - bump version in `__version__` variable
73
- - `uv build`
74
- - `uv publish` ⚠ **Impacts users, be careful**
75
- - `username: __token__`
76
- - `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
77
- - Run the GitLab pipeline `pages` to build & release the docs
78
-
79
- **Versioning**
80
-
81
- To ensure pip correctly updates our package, correct versioning is important. Use `major.minor.patch`
82
-
83
- **Recommended test workflow**
84
-
85
- - Publish to test Pypi
86
- - Open a docker container with the desired python version: `docker run -it python:3.11-alpine sh`
87
- - Install the test pypi package `pip install --upgrade -i --pre https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata`
88
- - If you want to publish again, change the version to one that has not been used before. Even if you delete a build via [the UI](https://test.pypi.org/manage/project/marpledata/releases/) you cannot publish that version again. For testing, you could use something like `x.y.z.dev1`, `x.y.z.dev2`, `x.y.z.dev3`, ...
60
+ See [`RELEASING.md`](RELEASING.md).
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: marpledata
3
- Version: 3.4.0.dev1
3
+ Version: 3.5.0
4
4
  Summary: Marple SDK for Python
5
5
  Project-URL: Homepage, https://www.marpledata.com/
6
6
  Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
@@ -52,7 +52,7 @@ from marple import DB # Marple DB
52
52
  from marple import Insight # Marple Insight
53
53
  ```
54
54
 
55
- For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
55
+ For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
56
56
 
57
57
  ## Marple DB
58
58
 
@@ -90,34 +90,40 @@ dataset = dataset.wait_for_import(timeout=10)
90
90
  #### Add signals to an imported dataset
91
91
 
92
92
  Upload additional signals (for example derived channels) on an existing dataset.
93
- Each signal is a DataFrame, Arrow table, or on-disk parquet matching `LAKE_ARROW_SCHEMA`:
94
- columns `time` (int64 nanoseconds) plus `value` and/or `value_text`.
93
+ DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
94
+ `time` (int64 nanoseconds) plus `value` and/or `value_text`.
95
95
  Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
96
96
  a conflict without overwrite raises `SignalsAlreadyExistError`.
97
97
 
98
- ```python
99
- import pandas as pd
100
- from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
98
+ A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
99
+ `TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
101
100
 
102
- speed = dataset.get_signal("car.speed").get_data()
103
- derived = pd.DataFrame({
104
- "time": speed.index.asi8,
105
- "value": speed["value"] * 3.6,
106
- })
101
+ ```python
107
102
  # Single signal: wait until available before reading
103
+ speed = dataset.get_signal("car.speed").get_data()
108
104
  signal = dataset.add_signal(
109
105
  "car.speed_kmh",
110
- derived,
106
+ speed * 3.6,
111
107
  metadata={"unit": "km/h"},
112
108
  ).wait_until_available()
113
109
 
114
110
  # Batch: returns IDs as soon as upload completes (no wait)
115
111
  ids = dataset.add_signals([
116
- {"name": "car.speed_kmh", "data": derived, "metadata": {"unit": "km/h"}},
112
+ {"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
117
113
  ], overwrite=True, concurrency=4)
118
114
  signals = dataset.get_signals(signal_ids=ids)
119
115
  ```
120
116
 
117
+ Data assembled from scratch uses the explicit schema columns instead:
118
+
119
+ ```python
120
+ import pandas as pd
121
+ from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
122
+
123
+ samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
124
+ dataset.add_signal("car.custom", samples)
125
+ ```
126
+
121
127
  #### Upload large files
122
128
 
123
129
  `stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
@@ -203,7 +209,8 @@ if len(datasets) > 0:
203
209
 
204
210
  - **List streams**: `db.get_streams()`
205
211
  - **List datasets in a stream**: `stream.get_datasets()`
206
- - **Upload a file to a file-stream**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
212
+ - **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
213
+ - **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
207
214
  - **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
208
215
  - **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
209
216
  - **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
@@ -214,12 +221,12 @@ if len(datasets) > 0:
214
221
  - **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
215
222
  - **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
216
223
 
217
- For live/realtime streams (creating and appending data):
224
+ For live streams:
218
225
 
219
226
  - **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
220
227
  - **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
221
228
  - **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
222
- - **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
229
+ - **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
223
230
 
224
231
  ### Calling endpoints directly
225
232
 
@@ -17,7 +17,7 @@ from marple import DB # Marple DB
17
17
  from marple import Insight # Marple Insight
18
18
  ```
19
19
 
20
- For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
20
+ For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
21
21
 
22
22
  ## Marple DB
23
23
 
@@ -55,34 +55,40 @@ dataset = dataset.wait_for_import(timeout=10)
55
55
  #### Add signals to an imported dataset
56
56
 
57
57
  Upload additional signals (for example derived channels) on an existing dataset.
58
- Each signal is a DataFrame, Arrow table, or on-disk parquet matching `LAKE_ARROW_SCHEMA`:
59
- columns `time` (int64 nanoseconds) plus `value` and/or `value_text`.
58
+ DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
59
+ `time` (int64 nanoseconds) plus `value` and/or `value_text`.
60
60
  Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
61
61
  a conflict without overwrite raises `SignalsAlreadyExistError`.
62
62
 
63
- ```python
64
- import pandas as pd
65
- from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
63
+ A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
64
+ `TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
66
65
 
67
- speed = dataset.get_signal("car.speed").get_data()
68
- derived = pd.DataFrame({
69
- "time": speed.index.asi8,
70
- "value": speed["value"] * 3.6,
71
- })
66
+ ```python
72
67
  # Single signal: wait until available before reading
68
+ speed = dataset.get_signal("car.speed").get_data()
73
69
  signal = dataset.add_signal(
74
70
  "car.speed_kmh",
75
- derived,
71
+ speed * 3.6,
76
72
  metadata={"unit": "km/h"},
77
73
  ).wait_until_available()
78
74
 
79
75
  # Batch: returns IDs as soon as upload completes (no wait)
80
76
  ids = dataset.add_signals([
81
- {"name": "car.speed_kmh", "data": derived, "metadata": {"unit": "km/h"}},
77
+ {"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
82
78
  ], overwrite=True, concurrency=4)
83
79
  signals = dataset.get_signals(signal_ids=ids)
84
80
  ```
85
81
 
82
+ Data assembled from scratch uses the explicit schema columns instead:
83
+
84
+ ```python
85
+ import pandas as pd
86
+ from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
87
+
88
+ samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
89
+ dataset.add_signal("car.custom", samples)
90
+ ```
91
+
86
92
  #### Upload large files
87
93
 
88
94
  `stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
@@ -168,7 +174,8 @@ if len(datasets) > 0:
168
174
 
169
175
  - **List streams**: `db.get_streams()`
170
176
  - **List datasets in a stream**: `stream.get_datasets()`
171
- - **Upload a file to a file-stream**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
177
+ - **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
178
+ - **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
172
179
  - **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
173
180
  - **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
174
181
  - **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
@@ -179,12 +186,12 @@ if len(datasets) > 0:
179
186
  - **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
180
187
  - **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
181
188
 
182
- For live/realtime streams (creating and appending data):
189
+ For live streams:
183
190
 
184
191
  - **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
185
192
  - **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
186
193
  - **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
187
- - **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
194
+ - **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
188
195
 
189
196
  ### Calling endpoints directly
190
197
 
@@ -0,0 +1,11 @@
1
+ # Releasing `marpledata`
2
+
3
+ Bump `pyproject.toml`, `__version__`, and `CHANGELOG.md` on `main` first.
4
+ Tokens: 1Password (`username: __token__`). Run from `python/`.
5
+
6
+ 1. **TestPyPI** — `rm -rf dist && uv build && uv publish --index testpypi`
7
+ Smoke: `pip install -i https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata==<version>`
8
+ 2. **PyPI** — `rm -rf dist && uv build && uv publish`
9
+ 3. **Docs** — run the GitLab `pages` job (Sphinx → https://marpledata.gitlab.io/marple-sdk/)
10
+
11
+ No GitHub tag/release; `CHANGELOG.md` is the release history.
@@ -16,7 +16,7 @@
16
16
  ~Signal.cache_parquet
17
17
  ~Signal.get_data
18
18
  ~Signal.list_parquet_files
19
- ~Signal.wait_until_cold
19
+ ~Signal.wait_until_available
20
20
 
21
21
 
22
22
 
@@ -26,7 +26,6 @@
26
26
  ~SignalUpload.name
27
27
  ~SignalUpload.data
28
28
  ~SignalUpload.metadata
29
- ~SignalUpload.overwrite
30
29
  ~SignalUpload.priority
31
30
 
32
31
 
@@ -37,8 +37,9 @@ Marple DB quickstart
37
37
  dataset = dataset.wait_for_import(timeout=10)
38
38
 
39
39
  After import, you can add derived signals with
40
- ``dataset.add_signal(...)`` / ``dataset.add_signals([...])``. See
41
- :doc:`tutorials` for a full example.
40
+ ``dataset.add_signal(...)`` / ``dataset.add_signals([...])``. For custom
41
+ ingest without file parsing, use ``stream.add_dataset(...)`` then
42
+ ``add_signal``. See :doc:`tutorials` for a full example.
42
43
 
43
44
  ``stream.push_file(...)`` starts an ingestion and lets the Marple DB API choose
44
45
  the best upload mode for the deployment and file size. For large files, use a
@@ -0,0 +1,125 @@
1
+ Tutorials
2
+ =========
3
+
4
+ These examples use the high-level ``DataStream``, ``Dataset``, and ``Signal``
5
+ APIs. Create a stream and API token in Marple DB first.
6
+
7
+ Setup
8
+ -----
9
+
10
+ .. code-block:: python
11
+
12
+ import os
13
+ import re
14
+
15
+ import pandas as pd
16
+
17
+ from marple import DB
18
+
19
+ db = DB(os.environ["MDB_TOKEN"])
20
+ stream = db.get_stream("Car data")
21
+
22
+ For VPC or self-hosted deployments, pass ``os.environ["MDB_URL"]`` as the
23
+ second argument to ``DB``.
24
+
25
+ Import a file and wait for import
26
+ ---------------------------------
27
+
28
+ .. code-block:: python
29
+
30
+ dataset = stream.push_file(
31
+ "examples_race.csv",
32
+ metadata={"source": "testbench"},
33
+ concurrency=8,
34
+ ).wait_for_import(timeout=180)
35
+
36
+ Add signals to a dataset
37
+ ------------------------
38
+
39
+ Add signals after import, or start with an empty dataset using
40
+ ``stream.add_dataset``. Input follows :data:`marple.db.LAKE_ARROW_SCHEMA`:
41
+ ``time`` (int64 nanoseconds) plus ``value`` and/or ``value_text``.
42
+
43
+ A Series, or a DataFrame without a ``time`` column, takes its times from a
44
+ ``DatetimeIndex`` or ``TimedeltaIndex``.
45
+
46
+ .. code-block:: python
47
+
48
+ speed = dataset.get_signal("car.speed").get_data()
49
+ signal = dataset.add_signal(
50
+ "car.speed_kmh",
51
+ speed * 3.6,
52
+ metadata={"unit": "km/h"},
53
+ ).wait_until_available()
54
+
55
+ Data assembled from scratch uses the explicit schema columns instead:
56
+
57
+ .. code-block:: python
58
+
59
+ derived = pd.DataFrame({
60
+ "time": [t0, t0 + 1_000_000_000],
61
+ "value": [1.0, 2.0],
62
+ })
63
+ dataset.add_signal("car.custom", derived)
64
+
65
+ For batches, ``add_signals`` returns signal IDs without waiting:
66
+
67
+ .. code-block:: python
68
+
69
+ ids = dataset.add_signals(
70
+ [{"name": "car.speed_kmh", "data": speed * 3.6}],
71
+ overwrite=True,
72
+ )
73
+ signals = [
74
+ signal.wait_until_available()
75
+ for signal in dataset.get_signals(signal_ids=ids, refresh=True)
76
+ ]
77
+
78
+ Filter datasets and get resampled data
79
+ --------------------------------------
80
+
81
+ .. code-block:: python
82
+
83
+ datasets = (
84
+ stream.get_datasets()
85
+ .where_metadata({"car_id": [1, 2], "track": "track_1"})
86
+ .wait_for_import()
87
+ .where_imported()
88
+ .where_signal("car.speed", "max", greater_than=75)
89
+ )
90
+ for dataset, data in datasets.get_data(
91
+ signals=["car.speed", re.compile(r"car\.wheel\..*\.speed")],
92
+ resample_rule="0.17s",
93
+ ):
94
+ print(dataset.path, data.shape)
95
+
96
+ Ingest realtime data
97
+ --------------------
98
+
99
+ Use ``append`` with a realtime stream, then ``cool`` the dataset to cold
100
+ storage when ingestion is complete.
101
+
102
+ .. code-block:: python
103
+
104
+ realtime = db.get_stream("Live car data")
105
+ dataset = realtime.add_dataset("race-1", metadata={"driver": "Alice"})
106
+ dataset.upsert_signals([{"signal": "car.speed", "unit": "m/s"}])
107
+ dataset.append(pd.DataFrame({
108
+ "time": [1_700_000_000_000_000_000, 1_700_000_001_000_000_000],
109
+ "car.speed": [10.0, 12.0],
110
+ }))
111
+ dataset = dataset.cool().wait_for_import(timeout=180)
112
+
113
+ Query with SQL
114
+ --------------
115
+
116
+ Trino querying is available on VPC and self-hosted deployments, not SaaS.
117
+
118
+ .. code-block:: python
119
+
120
+ info = db.trino_info
121
+ table = f"{info['cold_catalog']}.{info['datapool']}.data"
122
+ data = db.query(
123
+ f"SELECT time, signal, value FROM {table} WHERE dataset = ? LIMIT 1000",
124
+ params=[dataset.id],
125
+ )
@@ -20,7 +20,9 @@ if stream_name not in [stream.name for stream in db.get_streams()]:
20
20
 
21
21
  stream = db.get_stream(stream_name)
22
22
  stream.push_file("examples_race.csv", metadata={"car_id": 1, "track": "track_1", "weather": "sunny"})
23
- stream.push_file("examples_race.csv", metadata={"car_id": 2, "track": "track_1", "weather": "cloudy"})
23
+ stream.push_file(
24
+ "examples_race.csv", metadata={"car_id": 2, "track": "track_1", "weather": "cloudy"}, overwrite=True
25
+ )
24
26
  stream.push_file("examples_race.csv", metadata={"car_id": 3, "track": "track_1", "weather": "rainy"})
25
27
  stream.push_file("examples_race.csv", metadata={"car_id": 1, "track": "track_2", "weather": "sunny"})
26
28
  stream.push_file("examples_race.csv", metadata={"car_id": 2, "track": "track_2", "weather": "sunny"})
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "marpledata"
3
- version = "3.4.0.dev1"
3
+ version = "3.5.0"
4
4
  description = "Marple SDK for Python"
5
5
  authors = [
6
6
  { name = "Matthias Baert", email = "support@marpledata.com" },
@@ -3,4 +3,4 @@ from .db import DB
3
3
  from .insight import Insight
4
4
 
5
5
  __all__ = ["DB", "Insight", "db"]
6
- __version__ = "3.4.0.dev1"
6
+ __version__ = "3.5.0"
@@ -353,19 +353,21 @@ class DB:
353
353
  file_path: str,
354
354
  metadata: dict | None = None,
355
355
  file_name: str | None = None,
356
+ overwrite: bool = False,
356
357
  ) -> int:
357
358
  """
358
359
  Push a file to a datastream.
359
360
  - `stream_key`: The name or ID of the stream to push the file to.
360
361
  - `file_path`: The path to the file to be pushed.
361
362
  - `metadata`: (optional) A dictionary of metadata to be associated with the file.
362
- - `file_name`: (optional) The name of the file to be stored in the
363
+ - `file_name`: (optional) The name of the file to be stored in the stream.
364
+ - `overwrite`: (optional) If true, existing dataset with the same name will be overwritten.
363
365
 
364
366
  Note:
365
367
  This function is deprecated and it is encouraged to use the `push_file` method in the `DataStream` class directly.
366
368
  """
367
369
  stream = self.get_stream(stream_key)
368
- return stream.push_file(file_path, metadata, file_name).id
370
+ return stream.push_file(file_path, metadata, file_name, overwrite=overwrite).id
369
371
 
370
372
  @deprecated
371
373
  def get_status(self, stream_key: str | int, dataset_id: int) -> dict:
@@ -451,17 +453,24 @@ class DB:
451
453
  metadata = {}
452
454
  self.get_dataset(dataset_id, dataset_path).update_metadata(metadata, overwrite)
453
455
 
454
- # Realtime functions #
456
+ # Dataset creation and realtime ingest #
455
457
 
456
458
  def add_dataset(self, stream_key: str | int, dataset_name: str, metadata: dict | None = None) -> int:
457
459
  """
458
- Create a new empty dataset in the specified live stream.
459
- Returns the ID of the newly created dataset.
460
-
461
- Use `dataset_append` to add data to the dataset and `upsert_signals` to define signals.
462
- Use `cool` to finalize the dataset when done.
463
-
464
- To add datasets from a file to a file stream, use `push_file` instead.
460
+ Create a new empty dataset in the specified stream.
461
+
462
+ **Live (realtime) stream**
463
+ - Define signals with :meth:`~marple.db.dataset.Dataset.upsert_signals`
464
+ - Push data with :meth:`~marple.db.dataset.Dataset.append`
465
+ - Call :meth:`~marple.db.dataset.Dataset.cool` when finished to move the
466
+ dataset to cold Parquet/Iceberg storage, then
467
+ :meth:`~marple.db.dataset.Dataset.wait_for_import` until it is `FINISHED`.
468
+ **File (non-live) stream**
469
+ - Default path: use :meth:`push_file` instead of `add_dataset` — upload a
470
+ file (e.g. CSV); Marple parses it and writes it to the data lake.
471
+ - Custom ingestion: create an empty dataset with `add_dataset`, then upload
472
+ signal data with :meth:`~marple.db.dataset.Dataset.add_signal` /
473
+ :meth:`~marple.db.dataset.Dataset.add_signals` straight into the lake.
465
474
  """
466
475
  return self.get_stream(stream_key).add_dataset(dataset_name, metadata).id
467
476
 
@@ -333,24 +333,27 @@ class Dataset(BaseModel):
333
333
  def add_signal(
334
334
  self,
335
335
  name: str,
336
- data: pa.Table | pd.DataFrame | Path | str,
336
+ data: pa.Table | pd.DataFrame | pd.Series | Path | str,
337
337
  metadata: dict[str, Any] | None = None,
338
338
  overwrite: bool = False,
339
339
  priority: Literal["default", "high"] = "default",
340
340
  ) -> Signal:
341
341
  """
342
- Upload one signal with data into this dataset.
342
+ Upload one signal with data into this dataset (enrichment after import,
343
+ or custom ingest after :meth:`~marple.db.datastream.DataStream.add_dataset`).
343
344
 
344
- ``data`` must be a DataFrame, Arrow table, or parquet path matching
345
+ ``data`` must be a DataFrame, Series, Arrow table, or parquet path matching
345
346
  :data:`~marple.db.LAKE_ARROW_SCHEMA` (``time`` plus ``value`` and/or
346
- ``value_text``). Times must overlap the dataset time range.
347
+ ``value_text``). A Series, or a DataFrame without a ``time`` column, takes its
348
+ sample times from a ``DatetimeIndex`` or ``TimedeltaIndex``.
349
+ It must also contain ``value`` and/or ``value_text`` columns.
347
350
 
348
351
  Returns the new signal immediately after upload completes. Call
349
352
  :meth:`Signal.wait_until_available` to wait until the signal is available.
350
353
 
351
354
  Args:
352
355
  name: Signal name.
353
- data: Signal samples (DataFrame, Arrow table, or parquet path).
356
+ data: Signal samples (DataFrame, Series, Arrow table, or parquet path).
354
357
  metadata: Optional signal metadata (for example ``unit``).
355
358
  overwrite: If True, replace an existing signal with the same name.
356
359
  priority: Import priority (``default`` or ``high``).
@@ -79,12 +79,20 @@ class DataStream(BaseModel):
79
79
 
80
80
  def add_dataset(self, dataset_name: str, metadata: dict | None = None) -> Dataset:
81
81
  """
82
- Create a new empty dataset in this realtime stream.
83
-
84
- Use :meth:`~marple.db.dataset.Dataset.append` to add data and
85
- :meth:`~marple.db.dataset.Dataset.cool` to finalize the dataset.
86
-
87
- To add datasets from a file to a file stream, use :meth:`push_file` instead.
82
+ Create a new empty dataset in the specified stream.
83
+
84
+ **Live (realtime) stream**
85
+ - Define signals with :meth:`~marple.db.dataset.Dataset.upsert_signals`
86
+ - Push data with :meth:`~marple.db.dataset.Dataset.append`
87
+ - Call :meth:`~marple.db.dataset.Dataset.cool` when finished to move the
88
+ dataset to cold Parquet/Iceberg storage, then
89
+ :meth:`~marple.db.dataset.Dataset.wait_for_import` until it is `FINISHED`.
90
+ **File (non-live) stream**
91
+ - Default path: use :meth:`push_file` instead of `add_dataset` — upload a
92
+ file (e.g. CSV); Marple parses it and writes it to the data lake.
93
+ - Custom ingestion: create an empty dataset with `add_dataset`, then upload
94
+ signal data with :meth:`~marple.db.dataset.Dataset.add_signal` /
95
+ :meth:`~marple.db.dataset.Dataset.add_signals` straight into the lake.
88
96
  """
89
97
  r = self._client.post(
90
98
  f"/stream/{self.id}/dataset/add",
@@ -100,6 +108,7 @@ class DataStream(BaseModel):
100
108
  file_name: str | None = None,
101
109
  concurrency: int = 4,
102
110
  upload_mode: Literal["auto", "server"] = "auto",
111
+ overwrite: bool = False,
103
112
  ) -> Dataset:
104
113
  """
105
114
  Push a file to the datastream. The file will be ingested as a new dataset.
@@ -110,13 +119,14 @@ class DataStream(BaseModel):
110
119
  file_name: Optional name for the dataset. If not provided, the file name will be used.
111
120
  concurrency: Maximum number of concurrent part uploads for multipart uploads.
112
121
  upload_mode: Upload mode override. Use "server" to force upload through the Marple DB API server.
122
+ overwrite: If true, existing dataset with the same name will be overwritten.
113
123
  """
114
124
  if upload_mode not in ("auto", "server"):
115
125
  raise ValueError("upload_mode must be either 'auto' or 'server'")
116
126
 
117
127
  path = Path(file_path)
118
128
  file_size = path.stat().st_size
119
- init = self._init_ingestion(file_name or path.name, file_size, metadata or {})
129
+ init = self._init_ingestion(file_name or path.name, file_size, metadata or {}, overwrite)
120
130
 
121
131
  try:
122
132
  if upload_mode == "server" or init.mode == "server":
@@ -137,7 +147,9 @@ class DataStream(BaseModel):
137
147
 
138
148
  return self.get_dataset(init.dataset_id)
139
149
 
140
- def _init_ingestion(self, dataset_name: str, file_size: int, metadata: dict) -> IngestionInit:
150
+ def _init_ingestion(
151
+ self, dataset_name: str, file_size: int, metadata: dict, overwrite: bool = False
152
+ ) -> IngestionInit:
141
153
  r = self._client.post(
142
154
  "/ingestion",
143
155
  json={
@@ -145,6 +157,7 @@ class DataStream(BaseModel):
145
157
  "dataset_name": dataset_name,
146
158
  "file_size": file_size,
147
159
  "metadata": metadata,
160
+ "overwrite": overwrite,
148
161
  },
149
162
  )
150
163
  return IngestionInit(**validate_response(r, "Initialize ingestion failed"))
@@ -76,6 +76,7 @@ class Signal(BaseModel):
76
76
  Download the parquet files for this signal to a local cache folder and return the folder path.
77
77
  Args:
78
78
  refresh_cache: If True, re-download the parquet files even if they already exist in the cache.
79
+ An empty cache folder is always treated as a miss.
79
80
 
80
81
  Returns:
81
82
  The path to the local cache folder.
@@ -111,13 +112,11 @@ class Signal(BaseModel):
111
112
 
112
113
  def wait_until_available(self, timeout: float = 60) -> "Signal":
113
114
  """
114
- Poll until this signal is available for querying (storage status is not
115
- :attr:`StorageStatus.FROZEN_TO_COLD`). Updates the parent dataset's signal
116
- cache when a parent dataset is known.
115
+ Poll until this signal is available for querying.
117
116
  Args:
118
- timeout: Maximum time to wait for the signal to be available, in seconds.
117
+ timeout: Maximum time to wait for the signal to be available in seconds.
119
118
  Returns:
120
- The signal object.
119
+ The signal object when it is available for querying, or a the last state before the timeout.
121
120
  """
122
121
  deadline_s = time.monotonic() + max(timeout, 0.1)
123
122
  while True: