marpledata 3.4.0.dev1__tar.gz → 3.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/.gitignore +4 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/AGENTS.md +4 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/CHANGELOG.md +35 -1
- marpledata-3.6.0/CONTRIBUTING.md +59 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/PKG-INFO +67 -18
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/README.md +65 -16
- marpledata-3.6.0/RELEASING.md +13 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/db.rst +4 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.db.DB.rst +0 -4
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.db.DataStream.rst +0 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.db.Dataset.rst +0 -5
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.db.Signal.rst +3 -3
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/getting-started.rst +10 -2
- marpledata-3.6.0/docs/tutorials.rst +166 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/example.py +5 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/pyproject.toml +6 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/__init__.py +1 -1
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/__init__.py +133 -12
- marpledata-3.6.0/src/marple/db/activity.py +28 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/dataset.py +187 -26
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/datastream.py +137 -21
- marpledata-3.6.0/src/marple/db/script.py +188 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/signal.py +19 -5
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/signal_upload.py +33 -4
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/utils.py +28 -16
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/tests/conftest.py +6 -10
- marpledata-3.6.0/tests/support.py +60 -0
- marpledata-3.6.0/tests/test_activity.py +52 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/tests/test_db.py +135 -85
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/tests/test_insight.py +3 -1
- marpledata-3.6.0/tests/test_scripts.py +266 -0
- marpledata-3.6.0/tests/test_signal_upload.py +477 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/tests/test_sql.py +3 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/uv.lock +1 -1
- marpledata-3.4.0.dev1/.env.local +0 -3
- marpledata-3.4.0.dev1/.env.nightly +0 -2
- marpledata-3.4.0.dev1/.env.staging +0 -4
- marpledata-3.4.0.dev1/.uv-cache/.gitignore +0 -1
- marpledata-3.4.0.dev1/.uv-cache/.lock +0 -0
- marpledata-3.4.0.dev1/.uv-cache/CACHEDIR.TAG +0 -1
- marpledata-3.4.0.dev1/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
- marpledata-3.4.0.dev1/.uv-cache/sdists-v9/.gitignore +0 -0
- marpledata-3.4.0.dev1/CONTRIBUTING.md +0 -88
- marpledata-3.4.0.dev1/augment-dataset.py +0 -681
- marpledata-3.4.0.dev1/docs/api/marple.db.LAKE_ARROW_SCHEMA.rst +0 -6
- marpledata-3.4.0.dev1/docs/api/marple.db.SCHEMA.rst +0 -6
- marpledata-3.4.0.dev1/docs/api/marple.db.SignalUpload.rst +0 -32
- marpledata-3.4.0.dev1/docs/api/marple.db.SignalsAlreadyExistError.rst +0 -6
- marpledata-3.4.0.dev1/docs/tutorials.rst +0 -106
- marpledata-3.4.0.dev1/ingest-ibiza.py +0 -538
- marpledata-3.4.0.dev1/tests/support.py +0 -24
- marpledata-3.4.0.dev1/tests/test_signal_upload.py +0 -332
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/.flake8 +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/.python-version +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/LICENSE +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/DEPLOYMENT.md +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/_static/custom.css +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/_static/favicon.png +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/_static/logo.png +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/insight.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.Insight.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api/marple.db.DatasetList.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/api.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/conf.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/docs/index.rst +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/examples_race.csv +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/pytest.xml +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/constants.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/db/sql.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/insight.py +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/src/marple/py.typed +0 -0
- {marpledata-3.4.0.dev1 → marpledata-3.6.0}/tests/test_reliability.py +0 -0
|
@@ -169,4 +169,7 @@ cython_debug/
|
|
|
169
169
|
# and can be added to the global gitignore or merged into this file. For a more nuclear
|
|
170
170
|
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
|
|
171
171
|
.idea/
|
|
172
|
-
|
|
172
|
+
.env*
|
|
173
|
+
|
|
174
|
+
# Client handoff pack (may contain a real API token)
|
|
175
|
+
/marple-matlab/
|
|
@@ -15,6 +15,7 @@ This directory contains the Python SDK package published as `marpledata`.
|
|
|
15
15
|
## Commands
|
|
16
16
|
|
|
17
17
|
- Run tests with output: `uv run pytest -vs`
|
|
18
|
+
- Unit tests only: `uv run pytest -m "not integration"`
|
|
18
19
|
- Build docs: `uv run --group docs sphinx-build -b html docs docs/_build/html`
|
|
19
20
|
- Build package: `uv build`
|
|
20
21
|
- Fix formatting: `uv run isort src tests && uv run black src tests`
|
|
@@ -29,4 +30,6 @@ Run commands from `python/` unless a command explicitly says otherwise.
|
|
|
29
30
|
- Keep Python SDK tests in `python/tests/`.
|
|
30
31
|
- Integration tests run against live Marple services and may require
|
|
31
32
|
`MDB_TOKEN`, `MDB_URL`, `INSIGHT_TOKEN`, and `INSIGHT_URL`. Tests should skip
|
|
32
|
-
|
|
33
|
+
clearly when required credentials are missing. Mark live tests with
|
|
34
|
+
`pytest.mark.integration`.
|
|
35
|
+
- Release steps: `RELEASING.md`.
|
|
@@ -5,7 +5,41 @@ All notable changes to the Python SDK package `marpledata` will be documented in
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
-
## [3.
|
|
8
|
+
## [3.6.0] - 2026-09-04
|
|
9
|
+
|
|
10
|
+
### Changed
|
|
11
|
+
|
|
12
|
+
- Exact signal names in `get_data` / `get_signal` / `get_signals` resolve via the dataset cache or `GET /datapool/{pool}/signal/{name}/id` instead of downloading the full datapool `signal_map`. Regex patterns still use the map.
|
|
13
|
+
- Optional `plugin_args` on `DataStream.push_file` (and the deprecated `DB.push_file` wrapper).
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
|
|
17
|
+
- Processing Pipeline
|
|
18
|
+
- `DB.create_script` / `get_scripts` / `get_script` / `delete_script`
|
|
19
|
+
- `Script.update` / `duplicate` / `delete`, to manage stored `process(dataset)` scripts.
|
|
20
|
+
- `Dataset.run` / `DB.run_script` to execute a stored processing script against a dataset via a server-side sandbox job.
|
|
21
|
+
- `DataStream.scripts`, `DataStream.update`, and `DataStream.rerun_processing` (`DB.rerun_processing`) to edit a stream and set the script pipeline
|
|
22
|
+
- `Dataset.rerun_processing` and `Dataset.get_debug_messages`.
|
|
23
|
+
- `DB.delete_signals`, `Dataset.delete_signal` / `Dataset.delete_signals`, and `Signal.delete` to remove signals from a dataset.
|
|
24
|
+
- `Dataset.reingest` to reingest a dataset from its original uploaded file, with optional `plugin_args`.
|
|
25
|
+
- `logger.debug` lines on SDK mutations (`add_signal`, `update_metadata`, …), off by default. Call `db.verbose()` to print them.
|
|
26
|
+
|
|
27
|
+
### Fixed
|
|
28
|
+
|
|
29
|
+
- Allow `add_signal` in post-processing
|
|
30
|
+
|
|
31
|
+
## [3.5.0] - 2026-08-20
|
|
32
|
+
|
|
33
|
+
### Added
|
|
34
|
+
|
|
35
|
+
- `Dataset.add_signal` and `Dataset.add_signals` now accept a time-indexed pandas Series or DataFrame (DatetimeIndex / TimedeltaIndex when there is no `time` column).
|
|
36
|
+
- `DataStream.push_file(..., overwrite=False)` to replace an existing dataset with the same name (also on the deprecated `DB.push_file`).
|
|
37
|
+
|
|
38
|
+
### Fixed
|
|
39
|
+
|
|
40
|
+
- An empty local parquet cache folder is treated as a cache miss and re-downloaded, instead of being assumed to mean the signal has no data.
|
|
41
|
+
|
|
42
|
+
## [3.4.0] - 2026-07-28
|
|
9
43
|
|
|
10
44
|
### Added
|
|
11
45
|
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
## Development (Python)
|
|
2
|
+
|
|
3
|
+
### Formatting, linting, and typing
|
|
4
|
+
|
|
5
|
+
Checks run on `src/` and `tests/` in GitLab CI (`python:lint`, `python:typing`).
|
|
6
|
+
|
|
7
|
+
Fix formatting locally:
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
uv run isort src tests
|
|
11
|
+
uv run black src tests
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Verify (mirror CI):
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
uv run isort src tests --check --diff
|
|
18
|
+
uv run flake8 --config .flake8 src tests
|
|
19
|
+
uv run black --check src tests
|
|
20
|
+
uv run mypy --install-types --non-interactive
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
### Testing
|
|
24
|
+
|
|
25
|
+
The testing suite includes local unit tests and integration tests against a live Marple DB / Insight deployment. Integration tests share one imported example dataset per session, use a small CSV for extra ingestions, and mark themselves with `pytest.mark.integration`.
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
uv run pytest -m "not integration" # local unit tests only
|
|
29
|
+
export MDB_TOKEN=...
|
|
30
|
+
export INSIGHT_TOKEN=...
|
|
31
|
+
# Optional, defaults to SaaS URLs:
|
|
32
|
+
export MDB_URL=https://db.marpledata.com/api/v1
|
|
33
|
+
export INSIGHT_URL=https://insight.marpledata.com/api/v1
|
|
34
|
+
uv run pytest -vs
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
### Documentation
|
|
38
|
+
|
|
39
|
+
Build the Sphinx docs locally:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
uv sync --group docs
|
|
43
|
+
uv run sphinx-build -b html docs docs/_build/html
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Open `docs/_build/html/index.html` in your browser to view the site.
|
|
47
|
+
|
|
48
|
+
### Local build
|
|
49
|
+
|
|
50
|
+
- `uv build`
|
|
51
|
+
- `uv run pip install dist/*.whl` (Install in your local .venv)
|
|
52
|
+
- `uv run python`
|
|
53
|
+
- `import marple`
|
|
54
|
+
- `marple.__version__`
|
|
55
|
+
- `from marple import Insight, DB`
|
|
56
|
+
|
|
57
|
+
### Publishing
|
|
58
|
+
|
|
59
|
+
See [`RELEASING.md`](RELEASING.md).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: marpledata
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.6.0
|
|
4
4
|
Summary: Marple SDK for Python
|
|
5
5
|
Project-URL: Homepage, https://www.marpledata.com/
|
|
6
6
|
Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
|
|
@@ -52,7 +52,7 @@ from marple import DB # Marple DB
|
|
|
52
52
|
from marple import Insight # Marple Insight
|
|
53
53
|
```
|
|
54
54
|
|
|
55
|
-
For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
|
|
55
|
+
For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
|
|
56
56
|
|
|
57
57
|
## Marple DB
|
|
58
58
|
|
|
@@ -78,6 +78,7 @@ API_TOKEN = "<your api token>"
|
|
|
78
78
|
API_URL = "https://db.marpledata.com/api/v1" # optional if using the default SaaS
|
|
79
79
|
|
|
80
80
|
db = DB(API_TOKEN, API_URL)
|
|
81
|
+
db.verbose() # print SDK mutation activity (push_file, add_signal, …) to stdout
|
|
81
82
|
|
|
82
83
|
db.check_connection()
|
|
83
84
|
|
|
@@ -90,34 +91,75 @@ dataset = dataset.wait_for_import(timeout=10)
|
|
|
90
91
|
#### Add signals to an imported dataset
|
|
91
92
|
|
|
92
93
|
Upload additional signals (for example derived channels) on an existing dataset.
|
|
93
|
-
|
|
94
|
-
|
|
94
|
+
DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
|
|
95
|
+
`time` (int64 nanoseconds) plus `value` and/or `value_text`.
|
|
95
96
|
Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
|
|
96
97
|
a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
97
98
|
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
99
|
+
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
100
|
+
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
101
101
|
|
|
102
|
-
|
|
103
|
-
derived = pd.DataFrame({
|
|
104
|
-
"time": speed.index.asi8,
|
|
105
|
-
"value": speed["value"] * 3.6,
|
|
106
|
-
})
|
|
102
|
+
```python
|
|
107
103
|
# Single signal: wait until available before reading
|
|
104
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
108
105
|
signal = dataset.add_signal(
|
|
109
106
|
"car.speed_kmh",
|
|
110
|
-
|
|
107
|
+
speed * 3.6,
|
|
111
108
|
metadata={"unit": "km/h"},
|
|
112
109
|
).wait_until_available()
|
|
113
110
|
|
|
114
111
|
# Batch: returns IDs as soon as upload completes (no wait)
|
|
115
112
|
ids = dataset.add_signals([
|
|
116
|
-
{"name": "car.speed_kmh", "data":
|
|
113
|
+
{"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
|
|
117
114
|
], overwrite=True, concurrency=4)
|
|
118
115
|
signals = dataset.get_signals(signal_ids=ids)
|
|
119
116
|
```
|
|
120
117
|
|
|
118
|
+
Data assembled from scratch uses the explicit schema columns instead:
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
import pandas as pd
|
|
122
|
+
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
123
|
+
|
|
124
|
+
samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
125
|
+
dataset.add_signal("car.custom", samples)
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
#### Processing scripts
|
|
129
|
+
|
|
130
|
+
Write a `process(dataset)` function, store it, and try it on any imported dataset. This runs on the server and writes to that dataset.
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
source = """
|
|
134
|
+
from marple.db import Dataset
|
|
135
|
+
|
|
136
|
+
def process(dataset: Dataset) -> None:
|
|
137
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
138
|
+
dataset.add_signal("car.speed_kmh", speed * 3.6, metadata={"unit": "km/h"})
|
|
139
|
+
"""
|
|
140
|
+
|
|
141
|
+
script = db.create_script("speed_kmh", source)
|
|
142
|
+
dataset = stream.get_dataset(path="lap.csv")
|
|
143
|
+
# or: dataset = stream.push_file("lap.csv").wait_for_import()
|
|
144
|
+
dataset.run(script)
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Pass a `.py` path or `pathlib.Path` to `create_script` / `script.update` instead of source text. Iterate with `script.update(script=...)` then `dataset.run(script)` again (or `dataset.run(script, source=...)` to save and run in one step). New signals written by the script may need `signal.wait_until_available(...)` before you read them.
|
|
148
|
+
|
|
149
|
+
When the script looks right, attach it to the stream with `stream.update(scripts=...)`. That **replaces** the pipeline (pass `[]` to detach all). New uploads then run those scripts after ingest.
|
|
150
|
+
|
|
151
|
+
```python
|
|
152
|
+
stream = stream.update(scripts=[script.id])
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
For files already imported, rerun aliasing and the stream's script pipeline:
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
dataset = dataset.rerun_processing().wait_for_import()
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Use `script.update(...)` to change source or metadata, and `dataset.get_debug_messages()` for the latest ingestion's debug log (not the sandbox job log from `dataset.run`).
|
|
162
|
+
|
|
121
163
|
#### Upload large files
|
|
122
164
|
|
|
123
165
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -203,7 +245,8 @@ if len(datasets) > 0:
|
|
|
203
245
|
|
|
204
246
|
- **List streams**: `db.get_streams()`
|
|
205
247
|
- **List datasets in a stream**: `stream.get_datasets()`
|
|
206
|
-
- **Upload a file
|
|
248
|
+
- **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
|
|
249
|
+
- **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
|
|
207
250
|
- **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
|
|
208
251
|
- **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
|
|
209
252
|
- **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
|
|
@@ -213,13 +256,19 @@ if len(datasets) > 0:
|
|
|
213
256
|
- **Get a resampled df of multiple signals**: `dataset.get_data(signals=[...], resample_rule="1s")`
|
|
214
257
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
215
258
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
259
|
+
- **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
|
|
260
|
+
- **Run a processing script**: `dataset.run(script)` or `db.run_script(dataset_id, script)`
|
|
261
|
+
- **Create a processing script**: `db.create_script(name, script)`
|
|
262
|
+
- **Set script pipeline** (replaces the full list): `stream.update(scripts=[script.id])`
|
|
263
|
+
- **Rerun aliasing + scripts**: `dataset.rerun_processing().wait_for_import()` or `stream.rerun_processing([dataset.id])`
|
|
264
|
+
- **Read ingest debug logs**: `dataset.get_debug_messages()`
|
|
216
265
|
|
|
217
|
-
For live
|
|
266
|
+
For live streams:
|
|
218
267
|
|
|
219
268
|
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
220
269
|
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
221
270
|
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
222
|
-
- **Cool
|
|
271
|
+
- **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
|
|
223
272
|
|
|
224
273
|
### Calling endpoints directly
|
|
225
274
|
|
|
@@ -17,7 +17,7 @@ from marple import DB # Marple DB
|
|
|
17
17
|
from marple import Insight # Marple Insight
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
-
For release notes, see [`CHANGELOG.md`](CHANGELOG.md).
|
|
20
|
+
For release notes, see [`CHANGELOG.md`](CHANGELOG.md). To publish a release, see [`RELEASING.md`](RELEASING.md).
|
|
21
21
|
|
|
22
22
|
## Marple DB
|
|
23
23
|
|
|
@@ -43,6 +43,7 @@ API_TOKEN = "<your api token>"
|
|
|
43
43
|
API_URL = "https://db.marpledata.com/api/v1" # optional if using the default SaaS
|
|
44
44
|
|
|
45
45
|
db = DB(API_TOKEN, API_URL)
|
|
46
|
+
db.verbose() # print SDK mutation activity (push_file, add_signal, …) to stdout
|
|
46
47
|
|
|
47
48
|
db.check_connection()
|
|
48
49
|
|
|
@@ -55,34 +56,75 @@ dataset = dataset.wait_for_import(timeout=10)
|
|
|
55
56
|
#### Add signals to an imported dataset
|
|
56
57
|
|
|
57
58
|
Upload additional signals (for example derived channels) on an existing dataset.
|
|
58
|
-
|
|
59
|
-
|
|
59
|
+
DataFrames, Arrow tables, and on-disk parquet use `LAKE_ARROW_SCHEMA`: columns
|
|
60
|
+
`time` (int64 nanoseconds) plus `value` and/or `value_text`.
|
|
60
61
|
Times must overlap the dataset time range. Use `overwrite=True` to replace an existing signal name;
|
|
61
62
|
a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
62
63
|
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
64
|
+
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
65
|
+
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
66
66
|
|
|
67
|
-
|
|
68
|
-
derived = pd.DataFrame({
|
|
69
|
-
"time": speed.index.asi8,
|
|
70
|
-
"value": speed["value"] * 3.6,
|
|
71
|
-
})
|
|
67
|
+
```python
|
|
72
68
|
# Single signal: wait until available before reading
|
|
69
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
73
70
|
signal = dataset.add_signal(
|
|
74
71
|
"car.speed_kmh",
|
|
75
|
-
|
|
72
|
+
speed * 3.6,
|
|
76
73
|
metadata={"unit": "km/h"},
|
|
77
74
|
).wait_until_available()
|
|
78
75
|
|
|
79
76
|
# Batch: returns IDs as soon as upload completes (no wait)
|
|
80
77
|
ids = dataset.add_signals([
|
|
81
|
-
{"name": "car.speed_kmh", "data":
|
|
78
|
+
{"name": "car.speed_kmh", "data": speed * 3.6, "metadata": {"unit": "km/h"}},
|
|
82
79
|
], overwrite=True, concurrency=4)
|
|
83
80
|
signals = dataset.get_signals(signal_ids=ids)
|
|
84
81
|
```
|
|
85
82
|
|
|
83
|
+
Data assembled from scratch uses the explicit schema columns instead:
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
import pandas as pd
|
|
87
|
+
from marple.db import LAKE_ARROW_SCHEMA # time + value and/or value_text
|
|
88
|
+
|
|
89
|
+
samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
90
|
+
dataset.add_signal("car.custom", samples)
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
#### Processing scripts
|
|
94
|
+
|
|
95
|
+
Write a `process(dataset)` function, store it, and try it on any imported dataset. This runs on the server and writes to that dataset.
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
source = """
|
|
99
|
+
from marple.db import Dataset
|
|
100
|
+
|
|
101
|
+
def process(dataset: Dataset) -> None:
|
|
102
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
103
|
+
dataset.add_signal("car.speed_kmh", speed * 3.6, metadata={"unit": "km/h"})
|
|
104
|
+
"""
|
|
105
|
+
|
|
106
|
+
script = db.create_script("speed_kmh", source)
|
|
107
|
+
dataset = stream.get_dataset(path="lap.csv")
|
|
108
|
+
# or: dataset = stream.push_file("lap.csv").wait_for_import()
|
|
109
|
+
dataset.run(script)
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Pass a `.py` path or `pathlib.Path` to `create_script` / `script.update` instead of source text. Iterate with `script.update(script=...)` then `dataset.run(script)` again (or `dataset.run(script, source=...)` to save and run in one step). New signals written by the script may need `signal.wait_until_available(...)` before you read them.
|
|
113
|
+
|
|
114
|
+
When the script looks right, attach it to the stream with `stream.update(scripts=...)`. That **replaces** the pipeline (pass `[]` to detach all). New uploads then run those scripts after ingest.
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
stream = stream.update(scripts=[script.id])
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
For files already imported, rerun aliasing and the stream's script pipeline:
|
|
121
|
+
|
|
122
|
+
```python
|
|
123
|
+
dataset = dataset.rerun_processing().wait_for_import()
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
Use `script.update(...)` to change source or metadata, and `dataset.get_debug_messages()` for the latest ingestion's debug log (not the sandbox job log from `dataset.run`).
|
|
127
|
+
|
|
86
128
|
#### Upload large files
|
|
87
129
|
|
|
88
130
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -168,7 +210,8 @@ if len(datasets) > 0:
|
|
|
168
210
|
|
|
169
211
|
- **List streams**: `db.get_streams()`
|
|
170
212
|
- **List datasets in a stream**: `stream.get_datasets()`
|
|
171
|
-
- **Upload a file
|
|
213
|
+
- **Upload a file (file stream, default)**: `stream.push_file(file_path, metadata={...}, concurrency=4)`
|
|
214
|
+
- **Custom lake ingest (file stream)**: `stream.add_dataset(...)` then `dataset.add_signal(...)` / `add_signals([...])`
|
|
172
215
|
- **Add signals to a dataset**: `dataset.add_signal(name, data, ...)` / `dataset.add_signals([...])` — data matches `LAKE_ARROW_SCHEMA` (`time` + `value` and/or `value_text`)
|
|
173
216
|
- **Wait until a signal is available**: `signal.wait_until_available(timeout=60)`
|
|
174
217
|
- **Fetch signals by ID**: `dataset.get_signals(signal_ids=[...], refresh=False)`
|
|
@@ -178,13 +221,19 @@ if len(datasets) > 0:
|
|
|
178
221
|
- **Get a resampled df of multiple signals**: `dataset.get_data(signals=[...], resample_rule="1s")`
|
|
179
222
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
180
223
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
224
|
+
- **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
|
|
225
|
+
- **Run a processing script**: `dataset.run(script)` or `db.run_script(dataset_id, script)`
|
|
226
|
+
- **Create a processing script**: `db.create_script(name, script)`
|
|
227
|
+
- **Set script pipeline** (replaces the full list): `stream.update(scripts=[script.id])`
|
|
228
|
+
- **Rerun aliasing + scripts**: `dataset.rerun_processing().wait_for_import()` or `stream.rerun_processing([dataset.id])`
|
|
229
|
+
- **Read ingest debug logs**: `dataset.get_debug_messages()`
|
|
181
230
|
|
|
182
|
-
For live
|
|
231
|
+
For live streams:
|
|
183
232
|
|
|
184
233
|
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
185
234
|
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
186
235
|
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
187
|
-
- **Cool
|
|
236
|
+
- **Cool to cold storage**: `dataset.cool()` then `dataset.wait_for_import()` until `FINISHED`
|
|
188
237
|
|
|
189
238
|
### Calling endpoints directly
|
|
190
239
|
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Releasing `marpledata`
|
|
2
|
+
|
|
3
|
+
Tokens: 1Password (`username: __token__`). Run from `python/`.
|
|
4
|
+
|
|
5
|
+
- **Bump version**
|
|
6
|
+
- `uv version --bump minor` / `uv version x.y.z`
|
|
7
|
+
- Manual update `__init__.py`, `CHANGELOG.md`
|
|
8
|
+
- **TestPyPI** — `rm -rf dist && uv build && uv publish --index testpypi`
|
|
9
|
+
Smoke: `pip install -i https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata==<version>`
|
|
10
|
+
- **PyPI** — `rm -rf dist && uv build && uv publish`
|
|
11
|
+
- **Docs** — run the GitLab `pages` job (Sphinx → https://marpledata.gitlab.io/marple-sdk/)
|
|
12
|
+
|
|
13
|
+
No GitHub tag/release; `CHANGELOG.md` is the release history.
|
|
@@ -14,6 +14,10 @@ See :doc:`../tutorials` for upload examples.
|
|
|
14
14
|
marple.db.DataStream
|
|
15
15
|
marple.db.Dataset
|
|
16
16
|
marple.db.DatasetList
|
|
17
|
+
marple.db.SandboxJob
|
|
18
|
+
marple.db.SandboxJobStatus
|
|
19
|
+
marple.db.Script
|
|
20
|
+
marple.db.ScriptVersion
|
|
17
21
|
marple.db.Signal
|
|
18
22
|
marple.db.SignalUpload
|
|
19
23
|
marple.db.SignalsAlreadyExistError
|
|
@@ -15,10 +15,8 @@
|
|
|
15
15
|
|
|
16
16
|
~DB.add_dataset
|
|
17
17
|
~DB.check_connection
|
|
18
|
-
~DB.connect_trino
|
|
19
18
|
~DB.create_stream
|
|
20
19
|
~DB.dataset_append
|
|
21
|
-
~DB.dataset_cool
|
|
22
20
|
~DB.delete
|
|
23
21
|
~DB.delete_dataset
|
|
24
22
|
~DB.delete_stream
|
|
@@ -35,7 +33,6 @@
|
|
|
35
33
|
~DB.patch
|
|
36
34
|
~DB.post
|
|
37
35
|
~DB.push_file
|
|
38
|
-
~DB.query
|
|
39
36
|
~DB.update_metadata
|
|
40
37
|
~DB.upsert_signals
|
|
41
38
|
|
|
@@ -47,7 +44,6 @@
|
|
|
47
44
|
|
|
48
45
|
.. autosummary::
|
|
49
46
|
|
|
50
|
-
~DB.trino_info
|
|
51
47
|
~DB.client
|
|
52
48
|
|
|
53
49
|
|
|
@@ -13,10 +13,6 @@
|
|
|
13
13
|
|
|
14
14
|
.. autosummary::
|
|
15
15
|
|
|
16
|
-
~Dataset.add_signal
|
|
17
|
-
~Dataset.add_signals
|
|
18
|
-
~Dataset.append
|
|
19
|
-
~Dataset.cool
|
|
20
16
|
~Dataset.delete
|
|
21
17
|
~Dataset.download
|
|
22
18
|
~Dataset.fetch
|
|
@@ -24,7 +20,6 @@
|
|
|
24
20
|
~Dataset.get_signal
|
|
25
21
|
~Dataset.get_signals
|
|
26
22
|
~Dataset.update_metadata
|
|
27
|
-
~Dataset.upsert_signals
|
|
28
23
|
~Dataset.wait_for_import
|
|
29
24
|
|
|
30
25
|
|
|
@@ -37,8 +37,9 @@ Marple DB quickstart
|
|
|
37
37
|
dataset = dataset.wait_for_import(timeout=10)
|
|
38
38
|
|
|
39
39
|
After import, you can add derived signals with
|
|
40
|
-
``dataset.add_signal(...)`` / ``dataset.add_signals([...])``.
|
|
41
|
-
|
|
40
|
+
``dataset.add_signal(...)`` / ``dataset.add_signals([...])``. For custom
|
|
41
|
+
ingest without file parsing, use ``stream.add_dataset(...)`` then
|
|
42
|
+
``add_signal``. See :doc:`tutorials` for a full example.
|
|
42
43
|
|
|
43
44
|
``stream.push_file(...)`` starts an ingestion and lets the Marple DB API choose
|
|
44
45
|
the best upload mode for the deployment and file size. For large files, use a
|
|
@@ -49,6 +50,13 @@ longer ``wait_for_import`` timeout and optionally increase upload concurrency:
|
|
|
49
50
|
dataset = stream.push_file("large_export.csv", concurrency=8)
|
|
50
51
|
dataset = dataset.wait_for_import(timeout=180)
|
|
51
52
|
|
|
53
|
+
To replace an existing dataset with the same name, pass ``overwrite=True``:
|
|
54
|
+
|
|
55
|
+
.. code-block:: python
|
|
56
|
+
|
|
57
|
+
dataset = stream.push_file("examples_race.csv", overwrite=True)
|
|
58
|
+
dataset = dataset.wait_for_import(timeout=10)
|
|
59
|
+
|
|
52
60
|
If direct storage uploads are blocked by your network or proxy, force upload
|
|
53
61
|
through the Marple DB API server:
|
|
54
62
|
|