marpledata 3.2.2__tar.gz → 3.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- marpledata-3.2.4/.flake8 +5 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/AGENTS.md +3 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/CHANGELOG.md +18 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/CONTRIBUTING.md +28 -16
- {marpledata-3.2.2 → marpledata-3.2.4}/PKG-INFO +5 -4
- {marpledata-3.2.2 → marpledata-3.2.4}/README.md +4 -3
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.Signal.rst +3 -2
- {marpledata-3.2.2 → marpledata-3.2.4}/pyproject.toml +22 -1
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/__init__.py +1 -1
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/__init__.py +31 -90
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/dataset.py +138 -17
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/datastream.py +16 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/signal.py +5 -4
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/insight.py +4 -40
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/utils.py +12 -5
- marpledata-3.2.4/tests/conftest.py +51 -0
- marpledata-3.2.4/tests/support.py +24 -0
- marpledata-3.2.2/tests/test_sdk.py → marpledata-3.2.4/tests/test_db.py +59 -94
- marpledata-3.2.4/tests/test_insight.py +36 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/tests/test_reliability.py +5 -3
- {marpledata-3.2.2 → marpledata-3.2.4}/uv.lock +351 -3
- marpledata-3.2.2/.uv-cache/.gitignore +0 -1
- marpledata-3.2.2/.uv-cache/.lock +0 -0
- marpledata-3.2.2/.uv-cache/CACHEDIR.TAG +0 -1
- marpledata-3.2.2/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
- marpledata-3.2.2/.uv-cache/sdists-v9/.gitignore +0 -0
- marpledata-3.2.2/src/marple/profile.json +0 -95
- {marpledata-3.2.2 → marpledata-3.2.4}/.gitignore +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/.python-version +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/LICENSE +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/DEPLOYMENT.md +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/custom.css +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/favicon.png +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/logo.png +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/db.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/insight.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.Insight.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DB.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DataStream.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.Dataset.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DatasetList.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/api.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/conf.py +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/getting-started.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/index.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/docs/tutorials.rst +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/example.py +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/examples_race.csv +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/pytest.xml +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/constants.py +0 -0
- {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/py.typed +0 -0
marpledata-3.2.4/.flake8
ADDED
|
@@ -17,6 +17,9 @@ This directory contains the Python SDK package published as `marpledata`.
|
|
|
17
17
|
- Run tests with output: `uv run pytest -vs`
|
|
18
18
|
- Build docs: `uv run --group docs sphinx-build -b html docs docs/_build/html`
|
|
19
19
|
- Build package: `uv build`
|
|
20
|
+
- Fix formatting: `uv run isort src tests && uv run black src tests`
|
|
21
|
+
- Verify lint/format/types (mirror CI):
|
|
22
|
+
`uv run isort src tests --check --diff && uv run flake8 --config .flake8 src tests && uv run black --check src tests && uv run mypy --install-types --non-interactive`
|
|
20
23
|
|
|
21
24
|
Run commands from `python/` unless a command explicitly says otherwise.
|
|
22
25
|
|
|
@@ -5,6 +5,24 @@ All notable changes to the Python SDK package `marpledata` will be documented in
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
+
## [3.2.4] - 2026-06-19
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Added `DataStream.add_dataset`, `Dataset.upsert_signals`, `Dataset.append`, and `Dataset.cool` for realtime ingest.
|
|
13
|
+
- Extended `Dataset.wait_for_import` to recognize `COOLING` as a busy status.
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- `DB.add_dataset`, `DB.upsert_signals`, and `DB.dataset_append` now delegate to the DataStream and Dataset methods.
|
|
18
|
+
- Fixed broken `DB.download_signal` compatibility path (`get_parquet_files` → `list_parquet_files`).
|
|
19
|
+
|
|
20
|
+
## [3.2.3] - 2026-06-08
|
|
21
|
+
|
|
22
|
+
### Changed
|
|
23
|
+
|
|
24
|
+
- Bugfix build artifact `insight.py`
|
|
25
|
+
|
|
8
26
|
## [3.2.2] - 2026-06-01
|
|
9
27
|
|
|
10
28
|
### Changed
|
|
@@ -1,5 +1,25 @@
|
|
|
1
1
|
## Development (Python)
|
|
2
2
|
|
|
3
|
+
### Formatting, linting, and typing
|
|
4
|
+
|
|
5
|
+
Checks run on `src/` and `tests/` in GitLab CI (`python:lint`, `python:typing`).
|
|
6
|
+
|
|
7
|
+
Fix formatting locally:
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
uv run isort src tests
|
|
11
|
+
uv run black src tests
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
Verify (mirror CI):
|
|
15
|
+
|
|
16
|
+
```bash
|
|
17
|
+
uv run isort src tests --check --diff
|
|
18
|
+
uv run flake8 --config .flake8 src tests
|
|
19
|
+
uv run black --check src tests
|
|
20
|
+
uv run mypy --install-types --non-interactive
|
|
21
|
+
```
|
|
22
|
+
|
|
3
23
|
### Testing
|
|
4
24
|
|
|
5
25
|
The testing suite runs against a real linked DB & Insight deployment on our SaaS (e.g. Castro Comrades).
|
|
@@ -44,33 +64,25 @@ Only once on your laptop
|
|
|
44
64
|
- `username: __token__`
|
|
45
65
|
- `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
|
|
46
66
|
|
|
47
|
-
|
|
67
|
+
### Publishing Production
|
|
48
68
|
|
|
49
|
-
-
|
|
69
|
+
- Make sure you don't have any local changes
|
|
70
|
+
- Delete `./dist` folder
|
|
71
|
+
- `uv version x.y.z(.devi)`
|
|
50
72
|
- bump version in `__version__` variable
|
|
51
73
|
- `uv build`
|
|
52
|
-
- `uv publish
|
|
74
|
+
- `uv publish` ⚠ **Impacts users, be careful**
|
|
75
|
+
- `username: __token__`
|
|
76
|
+
- `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
|
|
77
|
+
- Run the GitLab pipeline `pages` to build & release the docs
|
|
53
78
|
|
|
54
79
|
**Versioning**
|
|
55
80
|
|
|
56
81
|
To ensure pip correctly updates our package, correct versioning is important. Use `major.minor.patch`
|
|
57
82
|
|
|
58
|
-
- `major` when they make incompatible API changes
|
|
59
|
-
- `minor` when they add functionality in a backwards-compatible manner
|
|
60
|
-
- `patch` when they make backwards-compatible bug fixes
|
|
61
|
-
|
|
62
83
|
**Recommended test workflow**
|
|
63
84
|
|
|
64
85
|
- Publish to test Pypi
|
|
65
86
|
- Open a docker container with the desired python version: `docker run -it python:3.11-alpine sh`
|
|
66
87
|
- Install the test pypi package `pip install --upgrade -i --pre https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata`
|
|
67
88
|
- If you want to publish again, change the version to one that has not been used before. Even if you delete a build via [the UI](https://test.pypi.org/manage/project/marpledata/releases/) you cannot publish that version again. For testing, you could use something like `x.y.z.dev1`, `x.y.z.dev2`, `x.y.z.dev3`, ...
|
|
68
|
-
|
|
69
|
-
### Publishing (real Pypi)
|
|
70
|
-
|
|
71
|
-
⚠ **Impacts users, be careful**
|
|
72
|
-
|
|
73
|
-
- check version: `pyproject.toml:version`, `__init__.py:__version__`
|
|
74
|
-
- `uv build`
|
|
75
|
-
- `uv publish --token pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
|
|
76
|
-
- Run the GitLab pipeline `pages` to build & release the docs
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: marpledata
|
|
3
|
-
Version: 3.2.
|
|
3
|
+
Version: 3.2.4
|
|
4
4
|
Summary: Marple SDK for Python
|
|
5
5
|
Project-URL: Homepage, https://www.marpledata.com/
|
|
6
6
|
Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
|
|
@@ -181,9 +181,10 @@ if len(datasets) > 0:
|
|
|
181
181
|
|
|
182
182
|
For live/realtime streams (creating and appending data):
|
|
183
183
|
|
|
184
|
-
- **Create an empty dataset**: `
|
|
185
|
-
- **Upsert signal definitions**: `
|
|
186
|
-
- **Append timeseries data**: `
|
|
184
|
+
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
185
|
+
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
186
|
+
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
187
|
+
- **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
|
|
187
188
|
|
|
188
189
|
### Calling endpoints directly
|
|
189
190
|
|
|
@@ -147,9 +147,10 @@ if len(datasets) > 0:
|
|
|
147
147
|
|
|
148
148
|
For live/realtime streams (creating and appending data):
|
|
149
149
|
|
|
150
|
-
- **Create an empty dataset**: `
|
|
151
|
-
- **Upsert signal definitions**: `
|
|
152
|
-
- **Append timeseries data**: `
|
|
150
|
+
- **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
|
|
151
|
+
- **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
|
|
152
|
+
- **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
|
|
153
|
+
- **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
|
|
153
154
|
|
|
154
155
|
### Calling endpoints directly
|
|
155
156
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "marpledata"
|
|
3
|
-
version = "3.2.
|
|
3
|
+
version = "3.2.4"
|
|
4
4
|
description = "Marple SDK for Python"
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Matthias Baert", email = "support@marpledata.com" },
|
|
@@ -56,12 +56,33 @@ explicit = true
|
|
|
56
56
|
|
|
57
57
|
[dependency-groups]
|
|
58
58
|
dev = [
|
|
59
|
+
"black>=24.2.0",
|
|
60
|
+
"flake8>=7.0.0",
|
|
59
61
|
"h5py>=3.15.1",
|
|
62
|
+
"isort>=5.13.2",
|
|
63
|
+
"mypy>=1.8.0",
|
|
60
64
|
"pytest>=9.0.2",
|
|
61
65
|
"pytest-cov>=7.0.0",
|
|
66
|
+
"types-requests>=2.32.0",
|
|
62
67
|
]
|
|
63
68
|
docs = [
|
|
64
69
|
"autodoc-pydantic>=2.2.0",
|
|
65
70
|
"pydata-sphinx-theme>=0.15.4",
|
|
66
71
|
"sphinx>=7.2.6",
|
|
67
72
|
]
|
|
73
|
+
|
|
74
|
+
[tool.black]
|
|
75
|
+
line-length = 113
|
|
76
|
+
target-version = ["py311"]
|
|
77
|
+
|
|
78
|
+
[tool.isort]
|
|
79
|
+
profile = "black"
|
|
80
|
+
src_paths = ["src", "tests"]
|
|
81
|
+
|
|
82
|
+
[tool.mypy]
|
|
83
|
+
mypy_path = "src"
|
|
84
|
+
files = ["src/marple", "tests"]
|
|
85
|
+
ignore_missing_imports = true
|
|
86
|
+
allow_untyped_globals = true
|
|
87
|
+
explicit_package_bases = true
|
|
88
|
+
|
|
@@ -1,29 +1,18 @@
|
|
|
1
|
-
import logging
|
|
2
1
|
import warnings
|
|
3
2
|
from functools import wraps
|
|
4
|
-
from io import BytesIO
|
|
5
3
|
from pathlib import Path
|
|
6
4
|
from typing import Literal, Optional
|
|
7
5
|
|
|
8
|
-
import numpy as np
|
|
9
6
|
import pandas as pd
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
from
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
COL_VAL,
|
|
16
|
-
COL_VAL_TEXT,
|
|
17
|
-
SAAS_URL,
|
|
18
|
-
SCHEMA,
|
|
19
|
-
)
|
|
7
|
+
from pydantic import ValidationError
|
|
8
|
+
from requests import Response
|
|
9
|
+
from requests.exceptions import ConnectionError
|
|
10
|
+
|
|
11
|
+
from marple.db.constants import SAAS_URL, SCHEMA
|
|
20
12
|
from marple.db.dataset import Dataset, DatasetList
|
|
21
13
|
from marple.db.datastream import DataStream
|
|
22
14
|
from marple.db.signal import Signal
|
|
23
15
|
from marple.utils import DBClient, validate_response
|
|
24
|
-
from pydantic import ValidationError
|
|
25
|
-
from requests import Response
|
|
26
|
-
from requests.exceptions import ConnectionError
|
|
27
16
|
|
|
28
17
|
__all__ = ["DB", "DataStream", "Dataset", "DatasetList", "Signal", "SCHEMA"]
|
|
29
18
|
|
|
@@ -191,7 +180,11 @@ class DB:
|
|
|
191
180
|
if isinstance(stream_key, int):
|
|
192
181
|
return self._streams.get(stream_key)
|
|
193
182
|
return next(
|
|
194
|
-
(
|
|
183
|
+
(
|
|
184
|
+
s
|
|
185
|
+
for s in self._streams.values()
|
|
186
|
+
if s.name.lower() == stream_key.lower() or str(s.id) == stream_key
|
|
187
|
+
),
|
|
195
188
|
None,
|
|
196
189
|
)
|
|
197
190
|
|
|
@@ -229,8 +222,8 @@ class DB:
|
|
|
229
222
|
if stream_key is not None:
|
|
230
223
|
return self.get_stream(stream_key).get_datasets()
|
|
231
224
|
r = self.get(f"/datapool/{self.client.datapool}/datasets")
|
|
232
|
-
|
|
233
|
-
return DatasetList.from_dicts(self.client,
|
|
225
|
+
datasets = validate_response(r, f"Failed to get datasets for datapool {self.client.datapool}")
|
|
226
|
+
return DatasetList.from_dicts(self.client, datasets)
|
|
234
227
|
|
|
235
228
|
def get_dataset(self, dataset_id: int | None = None, dataset_path: str | None = None) -> Dataset:
|
|
236
229
|
"""
|
|
@@ -331,7 +324,7 @@ class DB:
|
|
|
331
324
|
)
|
|
332
325
|
if signal is None:
|
|
333
326
|
raise Exception("Signal not found")
|
|
334
|
-
return signal.
|
|
327
|
+
return signal.list_parquet_files(refresh_cache)
|
|
335
328
|
|
|
336
329
|
def delete_dataset(self, dataset_id: int | None, dataset_path: str | None):
|
|
337
330
|
"""
|
|
@@ -373,17 +366,11 @@ class DB:
|
|
|
373
366
|
Returns the ID of the newly created dataset.
|
|
374
367
|
|
|
375
368
|
Use `dataset_append` to add data to the dataset and `upsert_signals` to define signals.
|
|
369
|
+
Use `cool` to finalize the dataset when done.
|
|
376
370
|
|
|
377
371
|
To add datasets from a file to a file stream, use `push_file` instead.
|
|
378
372
|
"""
|
|
379
|
-
|
|
380
|
-
r = self.post(
|
|
381
|
-
f"/stream/{stream_id}/dataset/add",
|
|
382
|
-
json={"dataset_name": dataset_name, "metadata": metadata or {}},
|
|
383
|
-
)
|
|
384
|
-
r_json = validate_response(r, "Add dataset failed")
|
|
385
|
-
|
|
386
|
-
return r_json["dataset_id"]
|
|
373
|
+
return self.get_stream(stream_key).add_dataset(dataset_name, metadata).id
|
|
387
374
|
|
|
388
375
|
def upsert_signals(self, stream_key: str | int, dataset_id: int, signals: list[dict]) -> None:
|
|
389
376
|
"""
|
|
@@ -395,10 +382,7 @@ class DB:
|
|
|
395
382
|
- `description`: (optional) Description of the signal
|
|
396
383
|
- `[any metadata key]`: (optional) Any metadata value
|
|
397
384
|
"""
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
r = self.post(f"/stream/{stream_id}/dataset/{dataset_id}/signals", json=signals)
|
|
401
|
-
validate_response(r, "Upsert signals failed")
|
|
385
|
+
self.get_dataset(dataset_id=dataset_id).upsert_signals(signals)
|
|
402
386
|
|
|
403
387
|
def dataset_append(
|
|
404
388
|
self,
|
|
@@ -423,62 +407,19 @@ class DB:
|
|
|
423
407
|
- `"wide"` format: Each row represents a single time point with multiple signals as columns. Expects at least a `time` column.
|
|
424
408
|
|
|
425
409
|
"""
|
|
426
|
-
|
|
410
|
+
self.get_dataset(dataset_id=dataset_id).append(data, shape)
|
|
427
411
|
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
pq.write_table(table, parquet_buffer)
|
|
443
|
-
parquet_buffer.seek(0)
|
|
444
|
-
|
|
445
|
-
# Send as multipart/form-data
|
|
446
|
-
files = {"file": ("data.parquet", parquet_buffer, "application/octet-stream")}
|
|
447
|
-
|
|
448
|
-
r = self.post(f"/stream/{stream_id}/dataset/{dataset_id}/append", files=files)
|
|
449
|
-
validate_response(r, "Append data failed")
|
|
450
|
-
|
|
451
|
-
# Internal functions #
|
|
452
|
-
|
|
453
|
-
@staticmethod
|
|
454
|
-
def _detect_shape(shape: Optional[Literal["long", "wide"]], df: pd.DataFrame) -> Literal["long", "wide"]:
|
|
455
|
-
if shape is not None:
|
|
456
|
-
return shape
|
|
457
|
-
|
|
458
|
-
if "signal" in df.columns and ((COL_VAL in df.columns) or (COL_VAL_TEXT in df.columns)):
|
|
459
|
-
return "long"
|
|
460
|
-
else:
|
|
461
|
-
return "wide"
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
def _wide_to_long(df: pd.DataFrame) -> pa.Table:
|
|
465
|
-
signals = []
|
|
466
|
-
time = pa.array(df[COL_TIME], type=pa.int64())
|
|
467
|
-
for col in df.columns:
|
|
468
|
-
if col == COL_TIME:
|
|
469
|
-
continue
|
|
470
|
-
if (value := _to_numeric(df[col])) is not None:
|
|
471
|
-
(value, value_text) = (value.to_numpy().astype(np.float64), pa.nulls(len(time)))
|
|
472
|
-
else:
|
|
473
|
-
(value, value_text) = (pa.nulls(len(time)), df[col].fillna("").to_numpy().astype(str))
|
|
474
|
-
signals.append(pa.Table.from_arrays([time, [col] * len(time), value, value_text], schema=SCHEMA))
|
|
475
|
-
return pa.concat_tables(signals)
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
def _to_numeric(col: pd.Series) -> pd.Series | None:
|
|
479
|
-
if pd.api.types.is_numeric_dtype(col.dtype):
|
|
480
|
-
return col
|
|
481
|
-
null_count = col.isnull().sum()
|
|
482
|
-
numeric_col = pd.to_numeric(col, errors="coerce")
|
|
483
|
-
is_numeric = (numeric_col.isnull().sum() - null_count) / max(len(col), 1) < 0.2
|
|
484
|
-
return numeric_col if is_numeric else None
|
|
412
|
+
def dataset_cool(self, stream_key: str | int, dataset_id: int) -> Dataset:
|
|
413
|
+
"""
|
|
414
|
+
Move all realtime data to cold storage and finalize the specified realtime dataset.
|
|
415
|
+
|
|
416
|
+
Cooling is started asynchronously on the server. Poll completion with
|
|
417
|
+
`Dataset.wait_for_import`.
|
|
418
|
+
|
|
419
|
+
Only realtime datasets in LIVE or COOLING_FAILED status can be cooled. After
|
|
420
|
+
cooling completes, the dataset no longer accepts appends and
|
|
421
|
+
`import_status` becomes FINISHED.
|
|
422
|
+
|
|
423
|
+
Returns the current dataset state (typically `import_status == "COOLING"`).
|
|
424
|
+
"""
|
|
425
|
+
return self.get_dataset(dataset_id=dataset_id).cool()
|
|
@@ -2,17 +2,19 @@ import re
|
|
|
2
2
|
import time
|
|
3
3
|
import warnings
|
|
4
4
|
from collections import UserList
|
|
5
|
+
from io import BytesIO
|
|
5
6
|
from pathlib import Path
|
|
6
|
-
from typing import Callable, Iterable, Literal, Optional
|
|
7
|
+
from typing import Callable, Iterable, Literal, Optional
|
|
7
8
|
from urllib import parse, request
|
|
8
9
|
|
|
10
|
+
import numpy as np
|
|
9
11
|
import pandas as pd
|
|
10
|
-
|
|
11
|
-
from pydantic import BaseModel, ConfigDict, Field, PrivateAttr
|
|
12
|
-
from pydantic import ValidationError
|
|
12
|
+
import pyarrow as pa
|
|
13
13
|
import pyarrow.parquet as pq
|
|
14
|
-
from
|
|
14
|
+
from pandas._typing import AggFuncType, Frequency
|
|
15
|
+
from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, ValidationError
|
|
15
16
|
|
|
17
|
+
from marple.db.constants import COL_SIG, COL_TIME, COL_VAL, COL_VAL_TEXT, SCHEMA
|
|
16
18
|
from marple.db.signal import Signal
|
|
17
19
|
from marple.utils import DBClient, validate_response
|
|
18
20
|
|
|
@@ -21,6 +23,7 @@ BUSY_STATUSES = [
|
|
|
21
23
|
"IMPORTING",
|
|
22
24
|
"POST_PROCESSING",
|
|
23
25
|
"UPDATING_ICEBERG",
|
|
26
|
+
"COOLING",
|
|
24
27
|
]
|
|
25
28
|
|
|
26
29
|
|
|
@@ -67,7 +70,9 @@ class Dataset(BaseModel):
|
|
|
67
70
|
self._client = client
|
|
68
71
|
|
|
69
72
|
@classmethod
|
|
70
|
-
def fetch(
|
|
73
|
+
def fetch(
|
|
74
|
+
cls, client: DBClient, dataset_id: int | None = None, dataset_path: str | None = None
|
|
75
|
+
) -> "Dataset":
|
|
71
76
|
"""
|
|
72
77
|
Fetch a dataset by its ID or path.
|
|
73
78
|
|
|
@@ -130,24 +135,24 @@ class Dataset(BaseModel):
|
|
|
130
135
|
"""
|
|
131
136
|
if signal_names is None:
|
|
132
137
|
return self._get_all_signals()
|
|
133
|
-
|
|
134
|
-
if len(
|
|
138
|
+
signal_ids = self._client.find_matching_signals(signal_names)
|
|
139
|
+
if len(signal_ids) == 0:
|
|
135
140
|
return []
|
|
136
141
|
|
|
137
142
|
r = self._client.get(
|
|
138
143
|
f"/stream/{self.datastream_id}/dataset/{self.id}/signals",
|
|
139
|
-
params={"signal_ids": list(
|
|
144
|
+
params={"signal_ids": list(signal_ids.values())},
|
|
140
145
|
)
|
|
141
|
-
|
|
146
|
+
result: list[Signal] = []
|
|
142
147
|
for response in validate_response(r, "Failed to get signals by name"):
|
|
143
148
|
try:
|
|
144
149
|
signal = Signal(self._client, self.datastream_id, self.id, **response)
|
|
145
|
-
|
|
150
|
+
result.append(signal)
|
|
146
151
|
except ValidationError as e:
|
|
147
152
|
warnings.warn(f"Failed to create signal {response['name']} (id {response['id']}): {e}")
|
|
148
153
|
continue
|
|
149
154
|
self._signals[signal.id] = signal
|
|
150
|
-
return
|
|
155
|
+
return result
|
|
151
156
|
|
|
152
157
|
def get_data(
|
|
153
158
|
self,
|
|
@@ -236,12 +241,90 @@ class Dataset(BaseModel):
|
|
|
236
241
|
validate_response(r, "Update metadata failed")
|
|
237
242
|
return self.fetch(self._client, self.id)
|
|
238
243
|
|
|
244
|
+
def upsert_signals(self, signals: list[dict]) -> None:
|
|
245
|
+
"""
|
|
246
|
+
Add signals to this dataset or update existing ones.
|
|
247
|
+
|
|
248
|
+
Each signal in the `signals` list should be a dictionary with the following keys:
|
|
249
|
+
- `signal`: Name of the signal
|
|
250
|
+
- `unit`: (optional) Unit of the signal
|
|
251
|
+
- `description`: (optional) Description of the signal
|
|
252
|
+
- `[any metadata key]`: (optional) Any metadata value
|
|
253
|
+
"""
|
|
254
|
+
r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/signals", json=signals)
|
|
255
|
+
validate_response(r, "Upsert signals failed")
|
|
256
|
+
|
|
257
|
+
def append(
|
|
258
|
+
self,
|
|
259
|
+
data: pd.DataFrame,
|
|
260
|
+
shape: Optional[Literal["wide", "long"]] = None,
|
|
261
|
+
) -> None:
|
|
262
|
+
"""
|
|
263
|
+
Append new data to this realtime dataset.
|
|
264
|
+
|
|
265
|
+
`data` is a DataFrame with the following columns. It can be in either "long"
|
|
266
|
+
or "wide" format. If `shape` is not specified, the format is automatically
|
|
267
|
+
detected:
|
|
268
|
+
|
|
269
|
+
- `"long"` format: Each row represents a single measurement for a single signal
|
|
270
|
+
at a specific time. Expects `time`, `signal`, and at least one of `value` or
|
|
271
|
+
`value_text`.
|
|
272
|
+
- `"wide"` format: Each row represents a single time point with multiple signals
|
|
273
|
+
as columns. Expects at least a `time` column.
|
|
274
|
+
"""
|
|
275
|
+
if _detect_shape(shape, data) == "wide":
|
|
276
|
+
if COL_TIME not in data.columns:
|
|
277
|
+
raise ValueError("DataFrame must contain a time column")
|
|
278
|
+
table = _wide_to_long(data)
|
|
279
|
+
else:
|
|
280
|
+
if COL_TIME not in data.columns or COL_SIG not in data.columns:
|
|
281
|
+
raise ValueError(f"DataFrame must contain {COL_TIME} and {COL_SIG} columns")
|
|
282
|
+
if not (COL_VAL in data.columns or COL_VAL_TEXT in data.columns):
|
|
283
|
+
raise ValueError(f"DataFrame must contain at least one of {COL_VAL} or {COL_VAL_TEXT} columns")
|
|
284
|
+
value = (
|
|
285
|
+
pd.to_numeric(data[COL_VAL], errors="coerce") if COL_VAL in data.columns else pa.nulls(len(data))
|
|
286
|
+
)
|
|
287
|
+
value_text = data[COL_VAL_TEXT] if COL_VAL_TEXT in data.columns else pa.nulls(len(data))
|
|
288
|
+
table = pa.Table.from_arrays([data[COL_TIME], data[COL_SIG], value, value_text], schema=SCHEMA)
|
|
289
|
+
|
|
290
|
+
parquet_buffer = BytesIO()
|
|
291
|
+
pq.write_table(table, parquet_buffer)
|
|
292
|
+
parquet_buffer.seek(0)
|
|
293
|
+
|
|
294
|
+
files = {"file": ("data.parquet", parquet_buffer, "application/octet-stream")}
|
|
295
|
+
r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/append", files=files)
|
|
296
|
+
validate_response(r, "Append data failed")
|
|
297
|
+
|
|
298
|
+
def cool(self) -> "Dataset":
|
|
299
|
+
"""
|
|
300
|
+
Move all realtime data to cold storage and finalize this realtime dataset.
|
|
301
|
+
|
|
302
|
+
Cooling is started asynchronously on the server. Poll completion with
|
|
303
|
+
:meth:`wait_for_import`.
|
|
304
|
+
|
|
305
|
+
Only datasets in LIVE or COOLING_FAILED status can be cooled.
|
|
306
|
+
After cooling completes, the dataset no longer accepts appends and
|
|
307
|
+
``import_status`` becomes FINISHED.
|
|
308
|
+
|
|
309
|
+
Returns:
|
|
310
|
+
The current dataset state (typically ``import_status == "COOLING"``).
|
|
311
|
+
"""
|
|
312
|
+
if self.import_status not in ("LIVE", "COOLING_FAILED"):
|
|
313
|
+
raise ValueError(f"Dataset {self.id} cannot be cooled (status: {self.import_status})")
|
|
314
|
+
|
|
315
|
+
r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/cool")
|
|
316
|
+
validate_response(r, "Cool dataset failed")
|
|
317
|
+
return self.fetch(self._client, self.id)
|
|
318
|
+
|
|
239
319
|
def wait_for_import(self, timeout: float = 60, force_fetch: bool = False) -> "Dataset":
|
|
240
320
|
"""
|
|
241
|
-
Wait for the dataset import to complete.
|
|
321
|
+
Wait for the dataset import or cooling to complete.
|
|
242
322
|
|
|
243
|
-
If the dataset is still in a busy status (WAITING, IMPORTING, POST_PROCESSING,
|
|
244
|
-
|
|
323
|
+
If the dataset is still in a busy status (WAITING, IMPORTING, POST_PROCESSING,
|
|
324
|
+
UPDATING_ICEBERG, COOLING) after the timeout, a warning is issued and the current
|
|
325
|
+
dataset information is returned.
|
|
326
|
+
If `force_fetch` is True, the import status is fetched at least once even if the
|
|
327
|
+
dataset is not in a busy status, to ensure the latest status is returned.
|
|
245
328
|
"""
|
|
246
329
|
if not (force_fetch or self.import_status in BUSY_STATUSES):
|
|
247
330
|
return self
|
|
@@ -285,7 +368,9 @@ class DatasetList(UserList[Dataset]):
|
|
|
285
368
|
try:
|
|
286
369
|
dataset = Dataset(client=client, **value)
|
|
287
370
|
except ValidationError as e:
|
|
288
|
-
warnings.warn(
|
|
371
|
+
warnings.warn(
|
|
372
|
+
f"Failed to create dataset with id {value.get('id')} and path {value.get('path')}: {e}"
|
|
373
|
+
)
|
|
289
374
|
continue
|
|
290
375
|
datasets.append(dataset)
|
|
291
376
|
return cls(datasets)
|
|
@@ -296,7 +381,9 @@ class DatasetList(UserList[Dataset]):
|
|
|
296
381
|
"""
|
|
297
382
|
return self.where(lambda d: d.import_status == "FINISHED")
|
|
298
383
|
|
|
299
|
-
def where_metadata(
|
|
384
|
+
def where_metadata(
|
|
385
|
+
self, metadata: dict[str, int | str | Iterable[int | str]] | None = None
|
|
386
|
+
) -> "DatasetList":
|
|
300
387
|
"""
|
|
301
388
|
Filter datasets by their metadata fields.
|
|
302
389
|
|
|
@@ -510,3 +597,37 @@ class DatasetList(UserList[Dataset]):
|
|
|
510
597
|
index=False,
|
|
511
598
|
)
|
|
512
599
|
return f"{df_str}\nDatasetList with {len(self.data)} datasets and {len(df.columns) - 5} unique metadata fields."
|
|
600
|
+
|
|
601
|
+
|
|
602
|
+
def _detect_shape(shape: Optional[Literal["long", "wide"]], df: pd.DataFrame) -> Literal["long", "wide"]:
|
|
603
|
+
if shape is not None:
|
|
604
|
+
return shape
|
|
605
|
+
|
|
606
|
+
if "signal" in df.columns and ((COL_VAL in df.columns) or (COL_VAL_TEXT in df.columns)):
|
|
607
|
+
return "long"
|
|
608
|
+
return "wide"
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def _wide_to_long(df: pd.DataFrame) -> pa.Table:
|
|
612
|
+
signals = []
|
|
613
|
+
time = pa.array(df[COL_TIME], type=pa.int64())
|
|
614
|
+
for col in df.columns:
|
|
615
|
+
if col == COL_TIME:
|
|
616
|
+
continue
|
|
617
|
+
if (numeric_col := _to_numeric(df[col])) is not None:
|
|
618
|
+
value_arr = numeric_col.to_numpy().astype(np.float64)
|
|
619
|
+
value_text = pa.nulls(len(time))
|
|
620
|
+
else:
|
|
621
|
+
value_arr = pa.nulls(len(time))
|
|
622
|
+
value_text = df[col].fillna("").to_numpy().astype(str)
|
|
623
|
+
signals.append(pa.Table.from_arrays([time, [col] * len(time), value_arr, value_text], schema=SCHEMA))
|
|
624
|
+
return pa.concat_tables(signals)
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def _to_numeric(col: pd.Series) -> pd.Series | None:
|
|
628
|
+
if pd.api.types.is_numeric_dtype(col.dtype):
|
|
629
|
+
return col
|
|
630
|
+
null_count = col.isnull().sum()
|
|
631
|
+
numeric_col = pd.to_numeric(col, errors="coerce")
|
|
632
|
+
is_numeric = (numeric_col.isnull().sum() - null_count) / max(len(col), 1) < 0.2
|
|
633
|
+
return numeric_col if is_numeric else None
|
|
@@ -78,6 +78,22 @@ class DataStream(BaseModel):
|
|
|
78
78
|
r = self._client.get(f"/stream/{self.id}/datasets")
|
|
79
79
|
return DatasetList.from_dicts(self._client, validate_response(r, "Get datasets failed"))
|
|
80
80
|
|
|
81
|
+
def add_dataset(self, dataset_name: str, metadata: dict | None = None) -> Dataset:
|
|
82
|
+
"""
|
|
83
|
+
Create a new empty dataset in this realtime stream.
|
|
84
|
+
|
|
85
|
+
Use :meth:`~marple.db.dataset.Dataset.append` to add data and
|
|
86
|
+
:meth:`~marple.db.dataset.Dataset.cool` to finalize the dataset.
|
|
87
|
+
|
|
88
|
+
To add datasets from a file to a file stream, use :meth:`push_file` instead.
|
|
89
|
+
"""
|
|
90
|
+
r = self._client.post(
|
|
91
|
+
f"/stream/{self.id}/dataset/add",
|
|
92
|
+
json={"dataset_name": dataset_name, "metadata": metadata or {}},
|
|
93
|
+
)
|
|
94
|
+
r_json = validate_response(r, "Add dataset failed")
|
|
95
|
+
return self.get_dataset(r_json["dataset_id"])
|
|
96
|
+
|
|
81
97
|
def push_file(
|
|
82
98
|
self,
|
|
83
99
|
file_path: str,
|
|
@@ -1,12 +1,11 @@
|
|
|
1
|
+
from pathlib import Path
|
|
1
2
|
from typing import Literal
|
|
2
3
|
|
|
4
|
+
import pandas as pd
|
|
3
5
|
from pydantic import BaseModel, PrivateAttr
|
|
4
6
|
|
|
5
7
|
from marple.utils import DBClient
|
|
6
8
|
|
|
7
|
-
from pathlib import Path
|
|
8
|
-
import pandas as pd
|
|
9
|
-
|
|
10
9
|
|
|
11
10
|
class Signal(BaseModel):
|
|
12
11
|
"""
|
|
@@ -67,7 +66,9 @@ class Signal(BaseModel):
|
|
|
67
66
|
"""
|
|
68
67
|
return self._client.list_parquet_files(self.dataset_id, self.id, refresh_cache)
|
|
69
68
|
|
|
70
|
-
def get_data(
|
|
69
|
+
def get_data(
|
|
70
|
+
self, dtype: Literal["numeric", "text"] | None = None, refresh_cache: bool = False
|
|
71
|
+
) -> pd.DataFrame:
|
|
71
72
|
"""
|
|
72
73
|
Get this signal's raw data as a pandas DataFrame.
|
|
73
74
|
|