marpledata 3.4.0.dev1__tar.gz → 3.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.gitignore +4 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/AGENTS.md +1 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/CHANGELOG.md +12 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/CONTRIBUTING.md +2 -30
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/PKG-INFO +25 -18
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/README.md +23 -16
- marpledata-3.5.0/RELEASING.md +11 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.Signal.rst +1 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SignalUpload.rst +0 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/getting-started.rst +3 -2
- marpledata-3.5.0/docs/tutorials.rst +125 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/example.py +3 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/pyproject.toml +1 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/__init__.py +1 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/__init__.py +19 -10
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/dataset.py +8 -5
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/datastream.py +21 -8
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/signal.py +4 -5
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/signal_upload.py +32 -3
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/utils.py +2 -1
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_db.py +24 -6
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_signal_upload.py +112 -8
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/uv.lock +1 -1
- marpledata-3.4.0.dev1/.env.local +0 -3
- marpledata-3.4.0.dev1/.env.nightly +0 -2
- marpledata-3.4.0.dev1/.env.staging +0 -4
- marpledata-3.4.0.dev1/augment-dataset.py +0 -681
- marpledata-3.4.0.dev1/docs/tutorials.rst +0 -106
- marpledata-3.4.0.dev1/ingest-ibiza.py +0 -538
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.flake8 +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.python-version +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/.gitignore +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/.lock +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/CACHEDIR.TAG +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/.uv-cache/sdists-v9/.gitignore +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/LICENSE +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/DEPLOYMENT.md +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/custom.css +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/favicon.png +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/_static/logo.png +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/db.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/insight.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.Insight.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DB.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DataStream.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.Dataset.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.DatasetList.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.LAKE_ARROW_SCHEMA.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SCHEMA.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api/marple.db.SignalsAlreadyExistError.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/api.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/conf.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/docs/index.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/examples_race.csv +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/pytest.xml +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/constants.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/db/sql.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/insight.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/src/marple/py.typed +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/conftest.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/support.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_insight.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_reliability.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.5.0}/tests/test_sql.py +0 -0
|
@@ -169,4 +169,7 @@ cython_debug/
|
|
|
169
169
|
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
|
170
170
|
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
|
171
171
|
.idea/
|
|
172
|
-
|
|
172
|
+
.env*
|
|
173
|
+
|
|
174
|
+
# Client handoff pack (may contain a real API token)
|
|
175
|
+
/marple-matlab/
|
|
@@ -30,3 +30,4 @@ Run commands from `python/` unless a command explicitly says otherwise.
|
|
|
30
30
|
- Integration tests run against live Marple services and may require
|
|
31
31
|
`MDB_TOKEN`, `MDB_URL`, `INSIGHT_TOKEN`, and `INSIGHT_URL`. Tests should skip
|
|
32
32
|
or fail clearly when required credentials are missing.
|
|
33
|
+
- Release steps: `RELEASING.md`.
|
|
@@ -5,7 +5,18 @@ All notable changes to the Python SDK package `marpledata` will be documented in
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
-
## [3.
|
|
8
|
+
## [3.5.0] - 2026-08-20
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- `Dataset.add_signal` and `Dataset.add_signals` now accept a time-indexed pandas Series or DataFrame (DatetimeIndex / TimedeltaIndex when there is no `time` column).
|
|
13
|
+
- `DataStream.push_file(..., overwrite=False)` to replace an existing dataset with the same name (also on the deprecated `DB.push_file`).
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
|
|
17
|
+
- An empty local parquet cache folder is treated as a cache miss and re-downloaded, instead of being assumed to mean the signal has no data.
|
|
18
|
+
|
|
19
|
+
## [3.4.0] - 2026-07-28
|
|
9
20
|
|
|
10
21
|
### Added
|
|
11
22
|
|
|
@@ -55,34 +55,6 @@ Open `docs/_build/html/index.html` in your browser to view the site.
|
|
|
55
55
|
- `marple.__version__`
|
|
56
56
|
- `from marple import Insight, DB`
|
|
57
57
|
|
|
58
|
-
### Publishing
|
|
58
|
+
### Publishing
|
|
59
59
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
- `uv build`
|
|
63
|
-
- `uv publish --index testpypi`
|
|
64
|
-
- `username: __token__`
|
|
65
|
-
- `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
|
|
66
|
-
|
|
67
|
-
### Publishing Production
|
|
68
|
-
|
|
69
|
-
- Make sure you don't have any local changes
|
|
70
|
-
- Delete `./dist` folder
|
|
71
|
-
- `uv version x.y.z(.devi)`
|
|
72
|
-
- bump version in `__version__` variable
|
|
73
|
-
- `uv build`
|
|
74
|
-
- `uv publish` ⚠ **Impacts users, be careful**
|
|
75
|
-
- `username: __token__`
|
|
76
|
-
- `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
|
|
77
|
-
- Run the GitLab pipeline `pages` to build & release the docs
|
|
78
|
-
|
|
79
|
-
**Versioning**
|
|
80
|
-
|
|
81
|
-
To ensure pip correctly updates our package, correct versioning is important. Use `major.minor.patch`
|
|
82
|
-
|
|
83
|
-
**Recommended test workflow**
|
|
84
|
-
|
|
85
|
-
- Publish to test Pypi
|
|
86
|
-
- Open a docker container with the desired python version: `docker run -it python:3.11-alpine sh`
|
|
87
|
-
- Install the test pypi package `pip install --upgrade -i --pre https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata`
|
|
88
|
-
- If you want to publish again, change the version to one that has not been used before. Even if you delete a build via [the UI](https://test.pypi.org/manage/project/marpledata/releases/) you cannot publish that version again. For testing, you could use something like `x.y.z.dev1`, `x.y.z.dev2`, `x.y.z.dev3`, ...
|
|
60
|
+
See [`RELEASING.md`](RELEASING.md).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: marpledata
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.5.0
|
|
4
4
|
Summary: Marple SDK for Python
|
|
5
5
|
Project-URL: Homepage, https://www.marpledata.com/
|
|
6
6
|
Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
|
|
@@ -52,7 +52,7 @@ from marple import DB # Marple DB
|
|
|
52
52
|
from marple import Insight # Marple Insight
|
|
53
53
|
```
|
|
54
54
|
|
|
55
|
-
For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
|
|
55
|
+
For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
|
|
56
56
|
|
|
57
57
|
## Marple DB
|
|
58
58
|
|
|
@@ -90,34 +90,40 @@ dataset = dataset.wait_for_import(timeout=10)
|
|
|
90
90
|
#### Add signals to an imported dataset
|
|
91
91
|
|
|
92
92
|
Upload additional signals (for example derived channels) on an existing dataset.
|
|
93
|
-
|
|
94
|
-
|
|
93
|
+
DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
|
|
94
|
+
`time` (int64 nanoseconds) plus `value` and/or `value_text`.
|
|
95
95
|
Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
|
|
96
96
|
a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
97
97
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
98
|
+
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
99
|
+
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
101
100
|
|
|
102
|
-
|
|
103
|
-
derived = pd.DataFrame({
|
|
104
|
-
"time": speed.index.asi8,
|
|
105
|
-
"value": speed["value"] * 3.6,
|
|
106
|
-
})
|
|
101
|
+
```python
|
|
107
102
|
# Single signal: wait until available before reading
|
|
103
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
108
104
|
signal = dataset.add_signal(
|
|
109
105
|
"car.speed_kmh",
|
|
110
|
-
|
|
106
|
+
speed * 3.6,
|
|
111
107
|
metadata={"unit": "km/h"},
|
|
112
108
|
).wait_until_available()
|
|
113
109
|
|
|
114
110
|
# Batch: returns IDs as soon as upload completes (no wait)
|
|
115
111
|
ids = dataset.add_signals([
|
|
116
|
-
{"name": "car.speed_kmh", "data":
|
|
112
|
+
{"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
|
|
117
113
|
], overwrite=True, concurrency=4)
|
|
118
114
|
signals = dataset.get_signals(signal_ids=ids)
|
|
119
115
|
```
|
|
120
116
|
|
|
117
|
+
Data assembled from scratch uses the explicit schema columns instead:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
import pandas as pd
|
|
121
|
+
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
122
|
+
|
|
123
|
+
samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
124
|
+
dataset.add_signal("car.custom", samples)
|
|
125
|
+
```
|
|
126
|
+
|
|
121
127
|
#### Upload large files
|
|
122
128
|
|
|
123
129
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -203,7 +209,8 @@ if len(datasets) > 0:
|
|
|
203
209
|
|
|
204
210
|
- **List streams**: `db.get_streams()`
|
|
205
211
|
- **List datasets in a stream**: `stream.get_datasets()`
|
|
206
|
-
- **Upload a file
|
|
212
|
+
- **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
|
|
213
|
+
- **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
|
|
207
214
|
- **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
|
|
208
215
|
- **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
|
|
209
216
|
- **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
|
|
@@ -214,12 +221,12 @@ if len(datasets) > 0:
|
|
|
214
221
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
215
222
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
216
223
|
|
|
217
|
-
For live
|
|
224
|
+
For live streams:
|
|
218
225
|
|
|
219
226
|
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
220
227
|
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
221
228
|
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
222
|
-
- **Cool
|
|
229
|
+
- **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
|
|
223
230
|
|
|
224
231
|
### Calling endpoints directly
|
|
225
232
|
|
|
@@ -17,7 +17,7 @@ from marple import DB # Marple DB
|
|
|
17
17
|
from marple import Insight # Marple Insight
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
-
For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
|
|
20
|
+
For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
|
|
21
21
|
|
|
22
22
|
## Marple DB
|
|
23
23
|
|
|
@@ -55,34 +55,40 @@ dataset = dataset.wait_for_import(timeout=10)
|
|
|
55
55
|
#### Add signals to an imported dataset
|
|
56
56
|
|
|
57
57
|
Upload additional signals (for example derived channels) on an existing dataset.
|
|
58
|
-
|
|
59
|
-
|
|
58
|
+
DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
|
|
59
|
+
`time` (int64 nanoseconds) plus `value` and/or `value_text`.
|
|
60
60
|
Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
|
|
61
61
|
a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
62
62
|
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
63
|
+
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
64
|
+
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
66
65
|
|
|
67
|
-
|
|
68
|
-
derived = pd.DataFrame({
|
|
69
|
-
"time": speed.index.asi8,
|
|
70
|
-
"value": speed["value"] * 3.6,
|
|
71
|
-
})
|
|
66
|
+
```python
|
|
72
67
|
# Single signal: wait until available before reading
|
|
68
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
73
69
|
signal = dataset.add_signal(
|
|
74
70
|
"car.speed_kmh",
|
|
75
|
-
|
|
71
|
+
speed * 3.6,
|
|
76
72
|
metadata={"unit": "km/h"},
|
|
77
73
|
).wait_until_available()
|
|
78
74
|
|
|
79
75
|
# Batch: returns IDs as soon as upload completes (no wait)
|
|
80
76
|
ids = dataset.add_signals([
|
|
81
|
-
{"name": "car.speed_kmh", "data":
|
|
77
|
+
{"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
|
|
82
78
|
], overwrite=True, concurrency=4)
|
|
83
79
|
signals = dataset.get_signals(signal_ids=ids)
|
|
84
80
|
```
|
|
85
81
|
|
|
82
|
+
Data assembled from scratch uses the explicit schema columns instead:
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
import pandas as pd
|
|
86
|
+
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
87
|
+
|
|
88
|
+
samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
89
|
+
dataset.add_signal("car.custom", samples)
|
|
90
|
+
```
|
|
91
|
+
|
|
86
92
|
#### Upload large files
|
|
87
93
|
|
|
88
94
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -168,7 +174,8 @@ if len(datasets) > 0:
|
|
|
168
174
|
|
|
169
175
|
- **List streams**: `db.get_streams()`
|
|
170
176
|
- **List datasets in a stream**: `stream.get_datasets()`
|
|
171
|
-
- **Upload a file
|
|
177
|
+
- **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
|
|
178
|
+
- **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
|
|
172
179
|
- **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
|
|
173
180
|
- **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
|
|
174
181
|
- **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
|
|
@@ -179,12 +186,12 @@ if len(datasets) > 0:
|
|
|
179
186
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
180
187
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
181
188
|
|
|
182
|
-
For live
|
|
189
|
+
For live streams:
|
|
183
190
|
|
|
184
191
|
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
185
192
|
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
186
193
|
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
187
|
-
- **Cool
|
|
194
|
+
- **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
|
|
188
195
|
|
|
189
196
|
### Calling endpoints directly
|
|
190
197
|
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Releasing `marpledata`
|
|
2
|
+
|
|
3
|
+
Bump `pyproject.toml`, `__version__`, and `CHANGELOG.md` on `main` first.
|
|
4
|
+
Tokens: 1Password (`username: __token__`). Run from `python/`.
|
|
5
|
+
|
|
6
|
+
1. **TestPyPI** — `rm -rf dist && uv build && uv publish --index testpypi`
|
|
7
|
+
Smoke: `pip install -i https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata==<version>`
|
|
8
|
+
2. **PyPI** — `rm -rf dist && uv build && uv publish`
|
|
9
|
+
3. **Docs** — run the GitLab `pages` job (Sphinx → https://marpledata.gitlab.io/marple-sdk/)
|
|
10
|
+
|
|
11
|
+
No GitHub tag/release; `CHANGELOG.md` is the release history.
|
|
@@ -37,8 +37,9 @@ Marple DB quickstart
|
|
|
37
37
|
dataset = dataset.wait_for_import(timeout=10)
|
|
38
38
|
|
|
39
39
|
After import, you can add derived signals with
|
|
40
|
-
``dataset.add_signal(...)`` / ``dataset.add_signals([...])``.
|
|
41
|
-
|
|
40
|
+
``dataset.add_signal(...)`` / ``dataset.add_signals([...])``. For custom
|
|
41
|
+
ingest without file parsing, use ``stream.add_dataset(...)`` then
|
|
42
|
+
``add_signal``. See :doc:`tutorials` for a full example.
|
|
42
43
|
|
|
43
44
|
``stream.push_file(...)`` starts an ingestion and lets the Marple DB API choose
|
|
44
45
|
the best upload mode for the deployment and file size. For large files, use a
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Tutorials
|
|
2
|
+
=========
|
|
3
|
+
|
|
4
|
+
These examples use the high-level ``DataStream``, ``Dataset``, and ``Signal``
|
|
5
|
+
APIs. Create a stream and API token in Marple DB first.
|
|
6
|
+
|
|
7
|
+
Setup
|
|
8
|
+
-----
|
|
9
|
+
|
|
10
|
+
.. code-block:: python
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
|
|
15
|
+
import pandas as pd
|
|
16
|
+
|
|
17
|
+
from marple import DB
|
|
18
|
+
|
|
19
|
+
db = DB(os.environ["MDB_TOKEN"])
|
|
20
|
+
stream = db.get_stream("Car data")
|
|
21
|
+
|
|
22
|
+
For VPC or self-hosted deployments, pass ``os.environ["MDB_URL"]`` as the
|
|
23
|
+
second argument to ``DB``.
|
|
24
|
+
|
|
25
|
+
Import a file and wait for import
|
|
26
|
+
---------------------------------
|
|
27
|
+
|
|
28
|
+
.. code-block:: python
|
|
29
|
+
|
|
30
|
+
dataset = stream.push_file(
|
|
31
|
+
"examples_race.csv",
|
|
32
|
+
metadata={"source": "testbench"},
|
|
33
|
+
concurrency=8,
|
|
34
|
+
).wait_for_import(timeout=180)
|
|
35
|
+
|
|
36
|
+
Add signals to a dataset
|
|
37
|
+
------------------------
|
|
38
|
+
|
|
39
|
+
Add signals after import, or start with an empty dataset using
|
|
40
|
+
``stream.add_dataset``. Input follows :data:`marple.db.LAKE_ARROW_SCHEMA`:
|
|
41
|
+
``time`` (int64 nanoseconds) plus ``value`` and/or ``value_text``.
|
|
42
|
+
|
|
43
|
+
A Series, or a DataFrame without a ``time`` column, takes its times from a
|
|
44
|
+
``DatetimeIndex`` or ``TimedeltaIndex``.
|
|
45
|
+
|
|
46
|
+
.. code-block:: python
|
|
47
|
+
|
|
48
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
49
|
+
signal = dataset.add_signal(
|
|
50
|
+
"car.speed_kmh",
|
|
51
|
+
speed * 3.6,
|
|
52
|
+
metadata={"unit": "km/h"},
|
|
53
|
+
).wait_until_available()
|
|
54
|
+
|
|
55
|
+
Data assembled from scratch uses the explicit schema columns instead:
|
|
56
|
+
|
|
57
|
+
.. code-block:: python
|
|
58
|
+
|
|
59
|
+
derived = pd.DataFrame({
|
|
60
|
+
"time": [t0, t0 + 1_000_000_000],
|
|
61
|
+
"value": [1.0, 2.0],
|
|
62
|
+
})
|
|
63
|
+
dataset.add_signal("car.custom", derived)
|
|
64
|
+
|
|
65
|
+
For batches, ``add_signals`` returns signal IDs without waiting:
|
|
66
|
+
|
|
67
|
+
.. code-block:: python
|
|
68
|
+
|
|
69
|
+
ids = dataset.add_signals(
|
|
70
|
+
[{"name": "car.speed_kmh", "data": speed * 3.6}],
|
|
71
|
+
overwrite=True,
|
|
72
|
+
)
|
|
73
|
+
signals = [
|
|
74
|
+
signal.wait_until_available()
|
|
75
|
+
for signal in dataset.get_signals(signal_ids=ids, refresh=True)
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
Filter datasets and get resampled data
|
|
79
|
+
--------------------------------------
|
|
80
|
+
|
|
81
|
+
.. code-block:: python
|
|
82
|
+
|
|
83
|
+
datasets = (
|
|
84
|
+
stream.get_datasets()
|
|
85
|
+
.where_metadata({"car_id": [1, 2], "track": "track_1"})
|
|
86
|
+
.wait_for_import()
|
|
87
|
+
.where_imported()
|
|
88
|
+
.where_signal("car.speed", "max", greater_than=75)
|
|
89
|
+
)
|
|
90
|
+
for dataset, data in datasets.get_data(
|
|
91
|
+
signals=["car.speed", re.compile(r"car\.wheel\..*\.speed")],
|
|
92
|
+
resample_rule="0.17s",
|
|
93
|
+
):
|
|
94
|
+
print(dataset.path, data.shape)
|
|
95
|
+
|
|
96
|
+
Ingest realtime data
|
|
97
|
+
--------------------
|
|
98
|
+
|
|
99
|
+
Use ``append`` with a realtime stream, then ``cool`` the dataset to cold
|
|
100
|
+
storage when ingestion is complete.
|
|
101
|
+
|
|
102
|
+
.. code-block:: python
|
|
103
|
+
|
|
104
|
+
realtime = db.get_stream("Live car data")
|
|
105
|
+
dataset = realtime.add_dataset("race-1", metadata={"driver": "Alice"})
|
|
106
|
+
dataset.upsert_signals([{"signal": "car.speed", "unit": "m/s"}])
|
|
107
|
+
dataset.append(pd.DataFrame({
|
|
108
|
+
"time": [1_700_000_000_000_000_000, 1_700_000_001_000_000_000],
|
|
109
|
+
"car.speed": [10.0, 12.0],
|
|
110
|
+
}))
|
|
111
|
+
dataset = dataset.cool().wait_for_import(timeout=180)
|
|
112
|
+
|
|
113
|
+
Query with SQL
|
|
114
|
+
--------------
|
|
115
|
+
|
|
116
|
+
Trino querying is available on VPC and self-hosted deployments, not SaaS.
|
|
117
|
+
|
|
118
|
+
.. code-block:: python
|
|
119
|
+
|
|
120
|
+
info = db.trino_info
|
|
121
|
+
table = f"{info['cold_catalog']}.{info['datapool']}.data"
|
|
122
|
+
data = db.query(
|
|
123
|
+
f"SELECT time, signal, value FROM {table} WHERE dataset = ? LIMIT 1000",
|
|
124
|
+
params=[dataset.id],
|
|
125
|
+
)
|
|
@@ -20,7 +20,9 @@ if stream_name not in [stream.name for stream in db.get_streams()]:
|
|
|
20
20
|
|
|
21
21
|
stream = db.get_stream(stream_name)
|
|
22
22
|
stream.push_file("examples_race.csv", metadata={"car_id": 1, "track": "track_1", "weather": "sunny"})
|
|
23
|
-
stream.push_file(
|
|
23
|
+
stream.push_file(
|
|
24
|
+
"examples_race.csv", metadata={"car_id": 2, "track": "track_1", "weather": "cloudy"}, overwrite=True
|
|
25
|
+
)
|
|
24
26
|
stream.push_file("examples_race.csv", metadata={"car_id": 3, "track": "track_1", "weather": "rainy"})
|
|
25
27
|
stream.push_file("examples_race.csv", metadata={"car_id": 1, "track": "track_2", "weather": "sunny"})
|
|
26
28
|
stream.push_file("examples_race.csv", metadata={"car_id": 2, "track": "track_2", "weather": "sunny"})
|
|
@@ -353,19 +353,21 @@ class DB:
|
|
|
353
353
|
file_path: str,
|
|
354
354
|
metadata: dict | None = None,
|
|
355
355
|
file_name: str | None = None,
|
|
356
|
+
overwrite: bool = False,
|
|
356
357
|
) -> int:
|
|
357
358
|
"""
|
|
358
359
|
Push a file to a datastream.
|
|
359
360
|
- `stream_key`: The name or ID of the stream to push the file to.
|
|
360
361
|
- `file_path`: The path to the file to be pushed.
|
|
361
362
|
- `metadata`: (optional) A dictionary of metadata to be associated with the file.
|
|
362
|
-
- `file_name`: (optional) The name of the file to be stored in the
|
|
363
|
+
- `file_name`: (optional) The name of the file to be stored in the stream.
|
|
364
|
+
- `overwrite`: (optional) If true, existing dataset with the same name will be overwritten.
|
|
363
365
|
|
|
364
366
|
Note:
|
|
365
367
|
This function is deprecated and it is encouraged to use the `push_file` method in the `DataStream` class directly.
|
|
366
368
|
"""
|
|
367
369
|
stream = self.get_stream(stream_key)
|
|
368
|
-
return stream.push_file(file_path, metadata, file_name).id
|
|
370
|
+
return stream.push_file(file_path, metadata, file_name, overwrite=overwrite).id
|
|
369
371
|
|
|
370
372
|
@deprecated
|
|
371
373
|
def get_status(self, stream_key: str | int, dataset_id: int) -> dict:
|
|
@@ -451,17 +453,24 @@ class DB:
|
|
|
451
453
|
metadata = {}
|
|
452
454
|
self.get_dataset(dataset_id, dataset_path).update_metadata(metadata, overwrite)
|
|
453
455
|
|
|
454
|
-
#
|
|
456
|
+
# Dataset creation and realtime ingest #
|
|
455
457
|
|
|
456
458
|
def add_dataset(self, stream_key: str | int, dataset_name: str, metadata: dict | None = None) -> int:
|
|
457
459
|
"""
|
|
458
|
-
Create a new empty dataset in the specified
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
460
|
+
Create a new empty dataset in the specified stream.
|
|
461
|
+
|
|
462
|
+
**Live (realtime) stream**
|
|
463
|
+
- Define signals with :meth:`~marple.db.dataset.Dataset.upsert_signals`
|
|
464
|
+
- Push data with :meth:`~marple.db.dataset.Dataset.append`
|
|
465
|
+
- Call :meth:`~marple.db.dataset.Dataset.cool` when finished to move the
|
|
466
|
+
dataset to cold Parquet/Iceberg storage, then
|
|
467
|
+
:meth:`~marple.db.dataset.Dataset.wait_for_import` until it is `FINISHED`.
|
|
468
|
+
**File (non-live) stream**
|
|
469
|
+
- Default path: use :meth:`push_file` instead of `add_dataset` — upload a
|
|
470
|
+
file (e.g. CSV); Marple parses it and writes it to the data lake.
|
|
471
|
+
- Custom ingestion: create an empty dataset with `add_dataset`, then upload
|
|
472
|
+
signal data with :meth:`~marple.db.dataset.Dataset.add_signal` /
|
|
473
|
+
:meth:`~marple.db.dataset.Dataset.add_signals` straight into the lake.
|
|
465
474
|
"""
|
|
466
475
|
return self.get_stream(stream_key).add_dataset(dataset_name, metadata).id
|
|
467
476
|
|
|
@@ -333,24 +333,27 @@ class Dataset(BaseModel):
|
|
|
333
333
|
def add_signal(
|
|
334
334
|
self,
|
|
335
335
|
name: str,
|
|
336
|
-
data: pa.Table | pd.DataFrame | Path | str,
|
|
336
|
+
data: pa.Table | pd.DataFrame | pd.Series | Path | str,
|
|
337
337
|
metadata: dict[str, Any] | None = None,
|
|
338
338
|
overwrite: bool = False,
|
|
339
339
|
priority: Literal["default", "high"] = "default",
|
|
340
340
|
) -> Signal:
|
|
341
341
|
"""
|
|
342
|
-
Upload one signal with data into this dataset
|
|
342
|
+
Upload one signal with data into this dataset (enrichment after import,
|
|
343
|
+
or custom ingest after :meth:`~marple.db.datastream.DataStream.add_dataset`).
|
|
343
344
|
|
|
344
|
-
``data`` must be a DataFrame, Arrow table, or parquet path matching
|
|
345
|
+
``data`` must be a DataFrame, Series, Arrow table, or parquet path matching
|
|
345
346
|
:data:`~marple.db.LAKE_ARROW_SCHEMA` (``time`` plus ``value`` and/or
|
|
346
|
-
``value_text``).
|
|
347
|
+
``value_text``). A Series, or a DataFrame without a ``time`` column, takes its
|
|
348
|
+
sample times from a ``DatetimeIndex`` or ``TimedeltaIndex``.
|
|
349
|
+
It must also contain ``value`` and/or ``value_text`` columns.
|
|
347
350
|
|
|
348
351
|
Returns the new signal immediately after upload completes. Call
|
|
349
352
|
:meth:`Signal.wait_until_available` to wait until the signal is available.
|
|
350
353
|
|
|
351
354
|
Args:
|
|
352
355
|
name: Signal name.
|
|
353
|
-
data: Signal samples (DataFrame, Arrow table, or parquet path).
|
|
356
|
+
data: Signal samples (DataFrame, Series, Arrow table, or parquet path).
|
|
354
357
|
metadata: Optional signal metadata (for example ``unit``).
|
|
355
358
|
overwrite: If True, replace an existing signal with the same name.
|
|
356
359
|
priority: Import priority (``default`` or ``high``).
|
|
@@ -79,12 +79,20 @@ class DataStream(BaseModel):
|
|
|
79
79
|
|
|
80
80
|
def add_dataset(self, dataset_name: str, metadata: dict | None = None) -> Dataset:
|
|
81
81
|
"""
|
|
82
|
-
Create a new empty dataset in
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
82
|
+
Create a new empty dataset in the specified stream.
|
|
83
|
+
|
|
84
|
+
**Live (realtime) stream**
|
|
85
|
+
- Define signals with :meth:`~marple.db.dataset.Dataset.upsert_signals`
|
|
86
|
+
- Push data with :meth:`~marple.db.dataset.Dataset.append`
|
|
87
|
+
- Call :meth:`~marple.db.dataset.Dataset.cool` when finished to move the
|
|
88
|
+
dataset to cold Parquet/Iceberg storage, then
|
|
89
|
+
:meth:`~marple.db.dataset.Dataset.wait_for_import` until it is `FINISHED`.
|
|
90
|
+
**File (non-live) stream**
|
|
91
|
+
- Default path: use :meth:`push_file` instead of `add_dataset` — upload a
|
|
92
|
+
file (e.g. CSV); Marple parses it and writes it to the data lake.
|
|
93
|
+
- Custom ingestion: create an empty dataset with `add_dataset`, then upload
|
|
94
|
+
signal data with :meth:`~marple.db.dataset.Dataset.add_signal` /
|
|
95
|
+
:meth:`~marple.db.dataset.Dataset.add_signals` straight into the lake.
|
|
88
96
|
"""
|
|
89
97
|
r = self._client.post(
|
|
90
98
|
f"/stream/{self.id}/dataset/add",
|
|
@@ -100,6 +108,7 @@ class DataStream(BaseModel):
|
|
|
100
108
|
file_name: str | None = None,
|
|
101
109
|
concurrency: int = 4,
|
|
102
110
|
upload_mode: Literal["auto", "server"] = "auto",
|
|
111
|
+
overwrite: bool = False,
|
|
103
112
|
) -> Dataset:
|
|
104
113
|
"""
|
|
105
114
|
Push a file to the datastream. The file will be ingested as a new dataset.
|
|
@@ -110,13 +119,14 @@ class DataStream(BaseModel):
|
|
|
110
119
|
file_name: Optional name for the dataset. If not provided, the file name will be used.
|
|
111
120
|
concurrency: Maximum number of concurrent part uploads for multipart uploads.
|
|
112
121
|
upload_mode: Upload mode override. Use "server" to force upload through the Marple DB API server.
|
|
122
|
+
overwrite: If true, existing dataset with the same name will be overwritten.
|
|
113
123
|
"""
|
|
114
124
|
if upload_mode not in ("auto", "server"):
|
|
115
125
|
raise ValueError("upload_mode must be either 'auto' or 'server'")
|
|
116
126
|
|
|
117
127
|
path = Path(file_path)
|
|
118
128
|
file_size = path.stat().st_size
|
|
119
|
-
init = self._init_ingestion(file_name or path.name, file_size, metadata or {})
|
|
129
|
+
init = self._init_ingestion(file_name or path.name, file_size, metadata or {}, overwrite)
|
|
120
130
|
|
|
121
131
|
try:
|
|
122
132
|
if upload_mode == "server" or init.mode == "server":
|
|
@@ -137,7 +147,9 @@ class DataStream(BaseModel):
|
|
|
137
147
|
|
|
138
148
|
return self.get_dataset(init.dataset_id)
|
|
139
149
|
|
|
140
|
-
def _init_ingestion(
|
|
150
|
+
def _init_ingestion(
|
|
151
|
+
self, dataset_name: str, file_size: int, metadata: dict, overwrite: bool = False
|
|
152
|
+
) -> IngestionInit:
|
|
141
153
|
r = self._client.post(
|
|
142
154
|
"/ingestion",
|
|
143
155
|
json={
|
|
@@ -145,6 +157,7 @@ class DataStream(BaseModel):
|
|
|
145
157
|
"dataset_name": dataset_name,
|
|
146
158
|
"file_size": file_size,
|
|
147
159
|
"metadata": metadata,
|
|
160
|
+
"overwrite": overwrite,
|
|
148
161
|
},
|
|
149
162
|
)
|
|
150
163
|
return IngestionInit(**validate_response(r, "Initialize ingestion failed"))
|
|
@@ -76,6 +76,7 @@ class Signal(BaseModel):
|
|
|
76
76
|
Download the parquet files for this signal to a local cache folder and return the folder path.
|
|
77
77
|
Args:
|
|
78
78
|
refresh_cache: If True, re-download the parquet files even if they already exist in the cache.
|
|
79
|
+
An empty cache folder is always treated as a miss.
|
|
79
80
|
|
|
80
81
|
Returns:
|
|
81
82
|
The path to the local cache folder.
|
|
@@ -111,13 +112,11 @@ class Signal(BaseModel):
|
|
|
111
112
|
|
|
112
113
|
def wait_until_available(self, timeout: float = 60) -> "Signal":
|
|
113
114
|
"""
|
|
114
|
-
Poll until this signal is available for querying
|
|
115
|
-
:attr:`StorageStatus.FROZEN_TO_COLD`). Updates the parent dataset's signal
|
|
116
|
-
cache when a parent dataset is known.
|
|
115
|
+
Poll until this signal is available for querying.
|
|
117
116
|
Args:
|
|
118
|
-
timeout: Maximum time to wait for the signal to be available
|
|
117
|
+
timeout: Maximum time to wait for the signal to be available in seconds.
|
|
119
118
|
Returns:
|
|
120
|
-
The signal object.
|
|
119
|
+
The signal object when it is available for querying, or a the last state before the timeout.
|
|
121
120
|
"""
|
|
122
121
|
deadline_s = time.monotonic() + max(timeout, 0.1)
|
|
123
122
|
while True:
|