marpledata 3.2.2__tar.gz → 3.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. marpledata-3.2.4/.flake8 +5 -0
  2. {marpledata-3.2.2 → marpledata-3.2.4}/AGENTS.md +3 -0
  3. {marpledata-3.2.2 → marpledata-3.2.4}/CHANGELOG.md +18 -0
  4. {marpledata-3.2.2 → marpledata-3.2.4}/CONTRIBUTING.md +28 -16
  5. {marpledata-3.2.2 → marpledata-3.2.4}/PKG-INFO +5 -4
  6. {marpledata-3.2.2 → marpledata-3.2.4}/README.md +4 -3
  7. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.Signal.rst +3 -2
  8. {marpledata-3.2.2 → marpledata-3.2.4}/pyproject.toml +22 -1
  9. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/__init__.py +1 -1
  10. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/__init__.py +31 -90
  11. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/dataset.py +138 -17
  12. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/datastream.py +16 -0
  13. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/signal.py +5 -4
  14. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/insight.py +4 -40
  15. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/utils.py +12 -5
  16. marpledata-3.2.4/tests/conftest.py +51 -0
  17. marpledata-3.2.4/tests/support.py +24 -0
  18. marpledata-3.2.2/tests/test_sdk.py → marpledata-3.2.4/tests/test_db.py +59 -94
  19. marpledata-3.2.4/tests/test_insight.py +36 -0
  20. {marpledata-3.2.2 → marpledata-3.2.4}/tests/test_reliability.py +5 -3
  21. {marpledata-3.2.2 → marpledata-3.2.4}/uv.lock +351 -3
  22. marpledata-3.2.2/.uv-cache/.gitignore +0 -1
  23. marpledata-3.2.2/.uv-cache/.lock +0 -0
  24. marpledata-3.2.2/.uv-cache/CACHEDIR.TAG +0 -1
  25. marpledata-3.2.2/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
  26. marpledata-3.2.2/.uv-cache/sdists-v9/.gitignore +0 -0
  27. marpledata-3.2.2/src/marple/profile.json +0 -95
  28. {marpledata-3.2.2 → marpledata-3.2.4}/.gitignore +0 -0
  29. {marpledata-3.2.2 → marpledata-3.2.4}/.python-version +0 -0
  30. {marpledata-3.2.2 → marpledata-3.2.4}/LICENSE +0 -0
  31. {marpledata-3.2.2 → marpledata-3.2.4}/docs/DEPLOYMENT.md +0 -0
  32. {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/custom.css +0 -0
  33. {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/favicon.png +0 -0
  34. {marpledata-3.2.2 → marpledata-3.2.4}/docs/_static/logo.png +0 -0
  35. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/db.rst +0 -0
  36. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/insight.rst +0 -0
  37. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.Insight.rst +0 -0
  38. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DB.rst +0 -0
  39. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DataStream.rst +0 -0
  40. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.Dataset.rst +0 -0
  41. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api/marple.db.DatasetList.rst +0 -0
  42. {marpledata-3.2.2 → marpledata-3.2.4}/docs/api.rst +0 -0
  43. {marpledata-3.2.2 → marpledata-3.2.4}/docs/conf.py +0 -0
  44. {marpledata-3.2.2 → marpledata-3.2.4}/docs/getting-started.rst +0 -0
  45. {marpledata-3.2.2 → marpledata-3.2.4}/docs/index.rst +0 -0
  46. {marpledata-3.2.2 → marpledata-3.2.4}/docs/tutorials.rst +0 -0
  47. {marpledata-3.2.2 → marpledata-3.2.4}/example.py +0 -0
  48. {marpledata-3.2.2 → marpledata-3.2.4}/examples_race.csv +0 -0
  49. {marpledata-3.2.2 → marpledata-3.2.4}/pytest.xml +0 -0
  50. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/db/constants.py +0 -0
  51. {marpledata-3.2.2 → marpledata-3.2.4}/src/marple/py.typed +0 -0
@@ -0,0 +1,5 @@
1
+ [flake8]
2
+ max-line-length = 1000
3
+ extend-ignore = E203
4
+ exclude =
5
+ .venv
@@ -17,6 +17,9 @@ This directory contains the Python SDK package published as `marpledata`.
17
17
  - Run tests with output: `uv run pytest -vs`
18
18
  - Build docs: `uv run --group docs sphinx-build -b html docs docs/_build/html`
19
19
  - Build package: `uv build`
20
+ - Fix formatting: `uv run isort src tests && uv run black src tests`
21
+ - Verify lint/format/types (mirror CI):
22
+ `uv run isort src tests --check --diff && uv run flake8 --config .flake8 src tests && uv run black --check src tests && uv run mypy --install-types --non-interactive`
20
23
 
21
24
  Run commands from `python/` unless a command explicitly says otherwise.
22
25
 
@@ -5,6 +5,24 @@ All notable changes to the Python SDK package `marpledata` will be documented in
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [3.2.4] - 2026-06-19
9
+
10
+ ### Added
11
+
12
+ - Added `DataStream.add_dataset`, `Dataset.upsert_signals`, `Dataset.append`, and `Dataset.cool` for realtime ingest.
13
+ - Extended `Dataset.wait_for_import` to recognize `COOLING` as a busy status.
14
+
15
+ ### Changed
16
+
17
+ - `DB.add_dataset`, `DB.upsert_signals`, and `DB.dataset_append` now delegate to the DataStream and Dataset methods.
18
+ - Fixed broken `DB.download_signal` compatibility path (`get_parquet_files` → `list_parquet_files`).
19
+
20
+ ## [3.2.3] - 2026-06-08
21
+
22
+ ### Changed
23
+
24
+ - Bugfix build artifact `insight.py`
25
+
8
26
  ## [3.2.2] - 2026-06-01
9
27
 
10
28
  ### Changed
@@ -1,5 +1,25 @@
1
1
  ## Development (Python)
2
2
 
3
+ ### Formatting, linting, and typing
4
+
5
+ Checks run on `src/` and `tests/` in GitLab CI (`python:lint`, `python:typing`).
6
+
7
+ Fix formatting locally:
8
+
9
+ ```bash
10
+ uv run isort src tests
11
+ uv run black src tests
12
+ ```
13
+
14
+ Verify (mirror CI):
15
+
16
+ ```bash
17
+ uv run isort src tests --check --diff
18
+ uv run flake8 --config .flake8 src tests
19
+ uv run black --check src tests
20
+ uv run mypy --install-types --non-interactive
21
+ ```
22
+
3
23
  ### Testing
4
24
 
5
25
  The testing suite runs against a real linked DB & Insight deployment on our SaaS (e.g. Castro Comrades).
@@ -44,33 +64,25 @@ Only once on your laptop
44
64
  - `username: __token__`
45
65
  - `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
46
66
 
47
- For every build
67
+ ### Publishing Production
48
68
 
49
- - `uv version x.y.z.devi`
69
+ - Make sure you don't have any local changes
70
+ - Delete `./dist` folder
71
+ - `uv version x.y.z(.devi)`
50
72
  - bump version in `__version__` variable
51
73
  - `uv build`
52
- - `uv publish --index testpypi`
74
+ - `uv publish` ⚠ **Impacts users, be careful**
75
+ - `username: __token__`
76
+ - `password: pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
77
+ - Run the GitLab pipeline `pages` to build & release the docs
53
78
 
54
79
  **Versioning**
55
80
 
56
81
  To ensure pip correctly updates our package, correct versioning is important. Use `major.minor.patch`
57
82
 
58
- - `major` when they make incompatible API changes
59
- - `minor` when they add functionality in a backwards-compatible manner
60
- - `patch` when they make backwards-compatible bug fixes
61
-
62
83
  **Recommended test workflow**
63
84
 
64
85
  - Publish to test Pypi
65
86
  - Open a docker container with the desired python version: `docker run -it python:3.11-alpine sh`
66
87
  - Install the test pypi package `pip install --upgrade -i --pre https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple marpledata`
67
88
  - If you want to publish again, change the version to one that has not been used before. Even if you delete a build via [the UI](https://test.pypi.org/manage/project/marpledata/releases/) you cannot publish that version again. For testing, you could use something like `x.y.z.dev1`, `x.y.z.dev2`, `x.y.z.dev3`, ...
68
-
69
- ### Publishing (real Pypi)
70
-
71
- ⚠ **Impacts users, be careful**
72
-
73
- - check version: `pyproject.toml:version`, `__init__.py:__version__`
74
- - `uv build`
75
- - `uv publish --token pypi-XXXXXXXXXXXXXXXXXXXXXXXXXXXX` (see 1Password)
76
- - Run the GitLab pipeline `pages` to build & release the docs
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: marpledata
3
- Version: 3.2.2
3
+ Version: 3.2.4
4
4
  Summary: Marple SDK for Python
5
5
  Project-URL: Homepage, https://www.marpledata.com/
6
6
  Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
@@ -181,9 +181,10 @@ if len(datasets) > 0:
181
181
 
182
182
  For live/realtime streams (creating and appending data):
183
183
 
184
- - **Create an empty dataset**: `db.add_dataset(stream_key, dataset_name, metadata=None)`
185
- - **Upsert signal definitions**: `db.upsert_signals(stream_key, dataset_id, signals=[...])`
186
- - **Append timeseries data**: `db.dataset_append(stream_key, dataset_id, data=df, shape="long"|"wide"|None)`
184
+ - **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
185
+ - **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
186
+ - **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
187
+ - **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
187
188
 
188
189
  ### Calling endpoints directly
189
190
 
@@ -147,9 +147,10 @@ if len(datasets) > 0:
147
147
 
148
148
  For live/realtime streams (creating and appending data):
149
149
 
150
- - **Create an empty dataset**: `db.add_dataset(stream_key, dataset_name, metadata=None)`
151
- - **Upsert signal definitions**: `db.upsert_signals(stream_key, dataset_id, signals=[...])`
152
- - **Append timeseries data**: `db.dataset_append(stream_key, dataset_id, data=df, shape="long"|"wide"|None)`
150
+ - **Create an empty dataset**: `stream.add_dataset(dataset_name, metadata=None)`
151
+ - **Upsert signal definitions**: `dataset.upsert_signals(signals=[...])`
152
+ - **Append timeseries data**: `dataset.append(data=df, shape="long"|"wide"|None)`
153
+ - **Cool a realtime dataset**: `dataset.cool()` then `dataset.wait_for_import()` to wait until `FINISHED`
153
154
 
154
155
  ### Calling endpoints directly
155
156
 
@@ -13,9 +13,10 @@
13
13
 
14
14
  .. autosummary::
15
15
 
16
- ~Signal.cache_parquet
16
+ ~Signal.download
17
+ ~Signal.from_dict
17
18
  ~Signal.get_data
18
- ~Signal.list_parquet_files
19
+ ~Signal.get_parquet_files
19
20
 
20
21
 
21
22
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "marpledata"
3
- version = "3.2.2"
3
+ version = "3.2.4"
4
4
  description = "Marple SDK for Python"
5
5
  authors = [
6
6
  { name = "Matthias Baert", email = "support@marpledata.com" },
@@ -56,12 +56,33 @@ explicit = true
56
56
 
57
57
  [dependency-groups]
58
58
  dev = [
59
+ "black>=24.2.0",
60
+ "flake8>=7.0.0",
59
61
  "h5py>=3.15.1",
62
+ "isort>=5.13.2",
63
+ "mypy>=1.8.0",
60
64
  "pytest>=9.0.2",
61
65
  "pytest-cov>=7.0.0",
66
+ "types-requests>=2.32.0",
62
67
  ]
63
68
  docs = [
64
69
  "autodoc-pydantic>=2.2.0",
65
70
  "pydata-sphinx-theme>=0.15.4",
66
71
  "sphinx>=7.2.6",
67
72
  ]
73
+
74
+ [tool.black]
75
+ line-length = 113
76
+ target-version = ["py311"]
77
+
78
+ [tool.isort]
79
+ profile = "black"
80
+ src_paths = ["src", "tests"]
81
+
82
+ [tool.mypy]
83
+ mypy_path = "src"
84
+ files = ["src/marple", "tests"]
85
+ ignore_missing_imports = true
86
+ allow_untyped_globals = true
87
+ explicit_package_bases = true
88
+
@@ -3,4 +3,4 @@ from .db import DB
3
3
  from .insight import Insight
4
4
 
5
5
  __all__ = ["DB", "Insight", "db"]
6
- __version__ = "3.2.2"
6
+ __version__ = "3.2.4"
@@ -1,29 +1,18 @@
1
- import logging
2
1
  import warnings
3
2
  from functools import wraps
4
- from io import BytesIO
5
3
  from pathlib import Path
6
4
  from typing import Literal, Optional
7
5
 
8
- import numpy as np
9
6
  import pandas as pd
10
- import pyarrow as pa
11
- import pyarrow.parquet as pq
12
- from marple.db.constants import (
13
- COL_SIG,
14
- COL_TIME,
15
- COL_VAL,
16
- COL_VAL_TEXT,
17
- SAAS_URL,
18
- SCHEMA,
19
- )
7
+ from pydantic import ValidationError
8
+ from requests import Response
9
+ from requests.exceptions import ConnectionError
10
+
11
+ from marple.db.constants import SAAS_URL, SCHEMA
20
12
  from marple.db.dataset import Dataset, DatasetList
21
13
  from marple.db.datastream import DataStream
22
14
  from marple.db.signal import Signal
23
15
  from marple.utils import DBClient, validate_response
24
- from pydantic import ValidationError
25
- from requests import Response
26
- from requests.exceptions import ConnectionError
27
16
 
28
17
  __all__ = ["DB", "DataStream", "Dataset", "DatasetList", "Signal", "SCHEMA"]
29
18
 
@@ -191,7 +180,11 @@ class DB:
191
180
  if isinstance(stream_key, int):
192
181
  return self._streams.get(stream_key)
193
182
  return next(
194
- (s for s in self._streams.values() if s.name.lower() == stream_key.lower() or str(s.id) == stream_key),
183
+ (
184
+ s
185
+ for s in self._streams.values()
186
+ if s.name.lower() == stream_key.lower() or str(s.id) == stream_key
187
+ ),
195
188
  None,
196
189
  )
197
190
 
@@ -229,8 +222,8 @@ class DB:
229
222
  if stream_key is not None:
230
223
  return self.get_stream(stream_key).get_datasets()
231
224
  r = self.get(f"/datapool/{self.client.datapool}/datasets")
232
- r = validate_response(r, f"Failed to get datasets for datapool {self.client.datapool}")
233
- return DatasetList.from_dicts(self.client, r)
225
+ datasets = validate_response(r, f"Failed to get datasets for datapool {self.client.datapool}")
226
+ return DatasetList.from_dicts(self.client, datasets)
234
227
 
235
228
  def get_dataset(self, dataset_id: int | None = None, dataset_path: str | None = None) -> Dataset:
236
229
  """
@@ -331,7 +324,7 @@ class DB:
331
324
  )
332
325
  if signal is None:
333
326
  raise Exception("Signal not found")
334
- return signal.get_parquet_files(refresh_cache)
327
+ return signal.list_parquet_files(refresh_cache)
335
328
 
336
329
  def delete_dataset(self, dataset_id: int | None, dataset_path: str | None):
337
330
  """
@@ -373,17 +366,11 @@ class DB:
373
366
  Returns the ID of the newly created dataset.
374
367
 
375
368
  Use `dataset_append` to add data to the dataset and `upsert_signals` to define signals.
369
+ Use `cool` to finalize the dataset when done.
376
370
 
377
371
  To add datasets from a file to a file stream, use `push_file` instead.
378
372
  """
379
- stream_id = self._get_stream_id(stream_key)
380
- r = self.post(
381
- f"/stream/{stream_id}/dataset/add",
382
- json={"dataset_name": dataset_name, "metadata": metadata or {}},
383
- )
384
- r_json = validate_response(r, "Add dataset failed")
385
-
386
- return r_json["dataset_id"]
373
+ return self.get_stream(stream_key).add_dataset(dataset_name, metadata).id
387
374
 
388
375
  def upsert_signals(self, stream_key: str | int, dataset_id: int, signals: list[dict]) -> None:
389
376
  """
@@ -395,10 +382,7 @@ class DB:
395
382
  - `description`: (optional) Description of the signal
396
383
  - `[any metadata key]`: (optional) Any metadata value
397
384
  """
398
- stream_id = self._get_stream_id(stream_key)
399
-
400
- r = self.post(f"/stream/{stream_id}/dataset/{dataset_id}/signals", json=signals)
401
- validate_response(r, "Upsert signals failed")
385
+ self.get_dataset(dataset_id=dataset_id).upsert_signals(signals)
402
386
 
403
387
  def dataset_append(
404
388
  self,
@@ -423,62 +407,19 @@ class DB:
423
407
  - `"wide"` format: Each row represents a single time point with multiple signals as columns. Expects at least a `time` column.
424
408
 
425
409
  """
426
- stream_id = self._get_stream_id(stream_key)
410
+ self.get_dataset(dataset_id=dataset_id).append(data, shape)
427
411
 
428
- if self._detect_shape(shape, data) == "wide":
429
- if COL_TIME not in data.columns:
430
- raise ValueError("DataFrame must contain a time column")
431
- table = _wide_to_long(data)
432
- else:
433
- if COL_TIME not in data.columns or COL_SIG not in data.columns:
434
- raise Exception(f"DataFrame must contain {COL_TIME} and {COL_SIG} columns")
435
- if not (COL_VAL in data.columns or COL_VAL_TEXT in data.columns):
436
- raise Exception(f"DataFrame must contain at least one of {COL_VAL} or {COL_VAL_TEXT} columns")
437
- value = pd.to_numeric(data[COL_VAL], errors="coerce") if COL_VAL in data.columns else pa.nulls(len(data))
438
- value_text = data[COL_VAL_TEXT] if COL_VAL_TEXT in data.columns else pa.nulls(len(data))
439
- table = pa.Table.from_arrays([data[COL_TIME], data[COL_SIG], value, value_text], schema=SCHEMA)
440
-
441
- parquet_buffer = BytesIO()
442
- pq.write_table(table, parquet_buffer)
443
- parquet_buffer.seek(0)
444
-
445
- # Send as multipart/form-data
446
- files = {"file": ("data.parquet", parquet_buffer, "application/octet-stream")}
447
-
448
- r = self.post(f"/stream/{stream_id}/dataset/{dataset_id}/append", files=files)
449
- validate_response(r, "Append data failed")
450
-
451
- # Internal functions #
452
-
453
- @staticmethod
454
- def _detect_shape(shape: Optional[Literal["long", "wide"]], df: pd.DataFrame) -> Literal["long", "wide"]:
455
- if shape is not None:
456
- return shape
457
-
458
- if "signal" in df.columns and ((COL_VAL in df.columns) or (COL_VAL_TEXT in df.columns)):
459
- return "long"
460
- else:
461
- return "wide"
462
-
463
-
464
- def _wide_to_long(df: pd.DataFrame) -> pa.Table:
465
- signals = []
466
- time = pa.array(df[COL_TIME], type=pa.int64())
467
- for col in df.columns:
468
- if col == COL_TIME:
469
- continue
470
- if (value := _to_numeric(df[col])) is not None:
471
- (value, value_text) = (value.to_numpy().astype(np.float64), pa.nulls(len(time)))
472
- else:
473
- (value, value_text) = (pa.nulls(len(time)), df[col].fillna("").to_numpy().astype(str))
474
- signals.append(pa.Table.from_arrays([time, [col] * len(time), value, value_text], schema=SCHEMA))
475
- return pa.concat_tables(signals)
476
-
477
-
478
- def _to_numeric(col: pd.Series) -> pd.Series | None:
479
- if pd.api.types.is_numeric_dtype(col.dtype):
480
- return col
481
- null_count = col.isnull().sum()
482
- numeric_col = pd.to_numeric(col, errors="coerce")
483
- is_numeric = (numeric_col.isnull().sum() - null_count) / max(len(col), 1) < 0.2
484
- return numeric_col if is_numeric else None
412
+ def dataset_cool(self, stream_key: str | int, dataset_id: int) -> Dataset:
413
+ """
414
+ Move all realtime data to cold storage and finalize the specified realtime dataset.
415
+
416
+ Cooling is started asynchronously on the server. Poll completion with
417
+ `Dataset.wait_for_import`.
418
+
419
+ Only realtime datasets in LIVE or COOLING_FAILED status can be cooled. After
420
+ cooling completes, the dataset no longer accepts appends and
421
+ `import_status` becomes FINISHED.
422
+
423
+ Returns the current dataset state (typically `import_status == "COOLING"`).
424
+ """
425
+ return self.get_dataset(dataset_id=dataset_id).cool()
@@ -2,17 +2,19 @@ import re
2
2
  import time
3
3
  import warnings
4
4
  from collections import UserList
5
+ from io import BytesIO
5
6
  from pathlib import Path
6
- from typing import Callable, Iterable, Literal, Optional, Sequence
7
+ from typing import Callable, Iterable, Literal, Optional
7
8
  from urllib import parse, request
8
9
 
10
+ import numpy as np
9
11
  import pandas as pd
10
- from pandas._typing import AggFuncType, Frequency
11
- from pydantic import BaseModel, ConfigDict, Field, PrivateAttr
12
- from pydantic import ValidationError
12
+ import pyarrow as pa
13
13
  import pyarrow.parquet as pq
14
- from marple.db.constants import COL_TIME, COL_VAL
14
+ from pandas._typing import AggFuncType, Frequency
15
+ from pydantic import BaseModel, ConfigDict, Field, PrivateAttr, ValidationError
15
16
 
17
+ from marple.db.constants import COL_SIG, COL_TIME, COL_VAL, COL_VAL_TEXT, SCHEMA
16
18
  from marple.db.signal import Signal
17
19
  from marple.utils import DBClient, validate_response
18
20
 
@@ -21,6 +23,7 @@ BUSY_STATUSES = [
21
23
  "IMPORTING",
22
24
  "POST_PROCESSING",
23
25
  "UPDATING_ICEBERG",
26
+ "COOLING",
24
27
  ]
25
28
 
26
29
 
@@ -67,7 +70,9 @@ class Dataset(BaseModel):
67
70
  self._client = client
68
71
 
69
72
  @classmethod
70
- def fetch(cls, client: DBClient, dataset_id: int | None = None, dataset_path: str | None = None) -> "Dataset":
73
+ def fetch(
74
+ cls, client: DBClient, dataset_id: int | None = None, dataset_path: str | None = None
75
+ ) -> "Dataset":
71
76
  """
72
77
  Fetch a dataset by its ID or path.
73
78
 
@@ -130,24 +135,24 @@ class Dataset(BaseModel):
130
135
  """
131
136
  if signal_names is None:
132
137
  return self._get_all_signals()
133
- signals = self._client.find_matching_signals(signal_names)
134
- if len(signals) == 0:
138
+ signal_ids = self._client.find_matching_signals(signal_names)
139
+ if len(signal_ids) == 0:
135
140
  return []
136
141
 
137
142
  r = self._client.get(
138
143
  f"/stream/{self.datastream_id}/dataset/{self.id}/signals",
139
- params={"signal_ids": list(signals.values())},
144
+ params={"signal_ids": list(signal_ids.values())},
140
145
  )
141
- signals = []
146
+ result: list[Signal] = []
142
147
  for response in validate_response(r, "Failed to get signals by name"):
143
148
  try:
144
149
  signal = Signal(self._client, self.datastream_id, self.id, **response)
145
- signals.append(signal)
150
+ result.append(signal)
146
151
  except ValidationError as e:
147
152
  warnings.warn(f"Failed to create signal {response['name']} (id {response['id']}): {e}")
148
153
  continue
149
154
  self._signals[signal.id] = signal
150
- return signals
155
+ return result
151
156
 
152
157
  def get_data(
153
158
  self,
@@ -236,12 +241,90 @@ class Dataset(BaseModel):
236
241
  validate_response(r, "Update metadata failed")
237
242
  return self.fetch(self._client, self.id)
238
243
 
244
+ def upsert_signals(self, signals: list[dict]) -> None:
245
+ """
246
+ Add signals to this dataset or update existing ones.
247
+
248
+ Each signal in the `signals` list should be a dictionary with the following keys:
249
+ - `signal`: Name of the signal
250
+ - `unit`: (optional) Unit of the signal
251
+ - `description`: (optional) Description of the signal
252
+ - `[any metadata key]`: (optional) Any metadata value
253
+ """
254
+ r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/signals", json=signals)
255
+ validate_response(r, "Upsert signals failed")
256
+
257
+ def append(
258
+ self,
259
+ data: pd.DataFrame,
260
+ shape: Optional[Literal["wide", "long"]] = None,
261
+ ) -> None:
262
+ """
263
+ Append new data to this realtime dataset.
264
+
265
+ `data` is a DataFrame with the following columns. It can be in either "long"
266
+ or "wide" format. If `shape` is not specified, the format is automatically
267
+ detected:
268
+
269
+ - `"long"` format: Each row represents a single measurement for a single signal
270
+ at a specific time. Expects `time`, `signal`, and at least one of `value` or
271
+ `value_text`.
272
+ - `"wide"` format: Each row represents a single time point with multiple signals
273
+ as columns. Expects at least a `time` column.
274
+ """
275
+ if _detect_shape(shape, data) == "wide":
276
+ if COL_TIME not in data.columns:
277
+ raise ValueError("DataFrame must contain a time column")
278
+ table = _wide_to_long(data)
279
+ else:
280
+ if COL_TIME not in data.columns or COL_SIG not in data.columns:
281
+ raise ValueError(f"DataFrame must contain {COL_TIME} and {COL_SIG} columns")
282
+ if not (COL_VAL in data.columns or COL_VAL_TEXT in data.columns):
283
+ raise ValueError(f"DataFrame must contain at least one of {COL_VAL} or {COL_VAL_TEXT} columns")
284
+ value = (
285
+ pd.to_numeric(data[COL_VAL], errors="coerce") if COL_VAL in data.columns else pa.nulls(len(data))
286
+ )
287
+ value_text = data[COL_VAL_TEXT] if COL_VAL_TEXT in data.columns else pa.nulls(len(data))
288
+ table = pa.Table.from_arrays([data[COL_TIME], data[COL_SIG], value, value_text], schema=SCHEMA)
289
+
290
+ parquet_buffer = BytesIO()
291
+ pq.write_table(table, parquet_buffer)
292
+ parquet_buffer.seek(0)
293
+
294
+ files = {"file": ("data.parquet", parquet_buffer, "application/octet-stream")}
295
+ r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/append", files=files)
296
+ validate_response(r, "Append data failed")
297
+
298
+ def cool(self) -> "Dataset":
299
+ """
300
+ Move all realtime data to cold storage and finalize this realtime dataset.
301
+
302
+ Cooling is started asynchronously on the server. Poll completion with
303
+ :meth:`wait_for_import`.
304
+
305
+ Only datasets in LIVE or COOLING_FAILED status can be cooled.
306
+ After cooling completes, the dataset no longer accepts appends and
307
+ ``import_status`` becomes FINISHED.
308
+
309
+ Returns:
310
+ The current dataset state (typically ``import_status == "COOLING"``).
311
+ """
312
+ if self.import_status not in ("LIVE", "COOLING_FAILED"):
313
+ raise ValueError(f"Dataset {self.id} cannot be cooled (status: {self.import_status})")
314
+
315
+ r = self._client.post(f"/stream/{self.datastream_id}/dataset/{self.id}/cool")
316
+ validate_response(r, "Cool dataset failed")
317
+ return self.fetch(self._client, self.id)
318
+
239
319
  def wait_for_import(self, timeout: float = 60, force_fetch: bool = False) -> "Dataset":
240
320
  """
241
- Wait for the dataset import to complete.
321
+ Wait for the dataset import or cooling to complete.
242
322
 
243
- If the dataset is still in a busy status (WAITING, IMPORTING, POST_PROCESSING, UPDATING_ICEBERG) after the timeout, a warning is issued and the current dataset information is returned.
244
- If `force_fetch` is True, the import status is fetched at least once even if the dataset is not in a busy status, to ensure the latest status is returned.
323
+ If the dataset is still in a busy status (WAITING, IMPORTING, POST_PROCESSING,
324
+ UPDATING_ICEBERG, COOLING) after the timeout, a warning is issued and the current
325
+ dataset information is returned.
326
+ If `force_fetch` is True, the import status is fetched at least once even if the
327
+ dataset is not in a busy status, to ensure the latest status is returned.
245
328
  """
246
329
  if not (force_fetch or self.import_status in BUSY_STATUSES):
247
330
  return self
@@ -285,7 +368,9 @@ class DatasetList(UserList[Dataset]):
285
368
  try:
286
369
  dataset = Dataset(client=client, **value)
287
370
  except ValidationError as e:
288
- warnings.warn(f"Failed to create dataset with id {value.get('id')} and path {value.get('path')}: {e}")
371
+ warnings.warn(
372
+ f"Failed to create dataset with id {value.get('id')} and path {value.get('path')}: {e}"
373
+ )
289
374
  continue
290
375
  datasets.append(dataset)
291
376
  return cls(datasets)
@@ -296,7 +381,9 @@ class DatasetList(UserList[Dataset]):
296
381
  """
297
382
  return self.where(lambda d: d.import_status == "FINISHED")
298
383
 
299
- def where_metadata(self, metadata: dict[str, int | str | Iterable[int | str]] | None = None) -> "DatasetList":
384
+ def where_metadata(
385
+ self, metadata: dict[str, int | str | Iterable[int | str]] | None = None
386
+ ) -> "DatasetList":
300
387
  """
301
388
  Filter datasets by their metadata fields.
302
389
 
@@ -510,3 +597,37 @@ class DatasetList(UserList[Dataset]):
510
597
  index=False,
511
598
  )
512
599
  return f"{df_str}\nDatasetList with {len(self.data)} datasets and {len(df.columns) - 5} unique metadata fields."
600
+
601
+
602
+ def _detect_shape(shape: Optional[Literal["long", "wide"]], df: pd.DataFrame) -> Literal["long", "wide"]:
603
+ if shape is not None:
604
+ return shape
605
+
606
+ if "signal" in df.columns and ((COL_VAL in df.columns) or (COL_VAL_TEXT in df.columns)):
607
+ return "long"
608
+ return "wide"
609
+
610
+
611
+ def _wide_to_long(df: pd.DataFrame) -> pa.Table:
612
+ signals = []
613
+ time = pa.array(df[COL_TIME], type=pa.int64())
614
+ for col in df.columns:
615
+ if col == COL_TIME:
616
+ continue
617
+ if (numeric_col := _to_numeric(df[col])) is not None:
618
+ value_arr = numeric_col.to_numpy().astype(np.float64)
619
+ value_text = pa.nulls(len(time))
620
+ else:
621
+ value_arr = pa.nulls(len(time))
622
+ value_text = df[col].fillna("").to_numpy().astype(str)
623
+ signals.append(pa.Table.from_arrays([time, [col] * len(time), value_arr, value_text], schema=SCHEMA))
624
+ return pa.concat_tables(signals)
625
+
626
+
627
+ def _to_numeric(col: pd.Series) -> pd.Series | None:
628
+ if pd.api.types.is_numeric_dtype(col.dtype):
629
+ return col
630
+ null_count = col.isnull().sum()
631
+ numeric_col = pd.to_numeric(col, errors="coerce")
632
+ is_numeric = (numeric_col.isnull().sum() - null_count) / max(len(col), 1) < 0.2
633
+ return numeric_col if is_numeric else None
@@ -78,6 +78,22 @@ class DataStream(BaseModel):
78
78
  r = self._client.get(f"/stream/{self.id}/datasets")
79
79
  return DatasetList.from_dicts(self._client, validate_response(r, "Get datasets failed"))
80
80
 
81
+ def add_dataset(self, dataset_name: str, metadata: dict | None = None) -> Dataset:
82
+ """
83
+ Create a new empty dataset in this realtime stream.
84
+
85
+ Use :meth:`~marple.db.dataset.Dataset.append` to add data and
86
+ :meth:`~marple.db.dataset.Dataset.cool` to finalize the dataset.
87
+
88
+ To add datasets from a file to a file stream, use :meth:`push_file` instead.
89
+ """
90
+ r = self._client.post(
91
+ f"/stream/{self.id}/dataset/add",
92
+ json={"dataset_name": dataset_name, "metadata": metadata or {}},
93
+ )
94
+ r_json = validate_response(r, "Add dataset failed")
95
+ return self.get_dataset(r_json["dataset_id"])
96
+
81
97
  def push_file(
82
98
  self,
83
99
  file_path: str,
@@ -1,12 +1,11 @@
1
+ from pathlib import Path
1
2
  from typing import Literal
2
3
 
4
+ import pandas as pd
3
5
  from pydantic import BaseModel, PrivateAttr
4
6
 
5
7
  from marple.utils import DBClient
6
8
 
7
- from pathlib import Path
8
- import pandas as pd
9
-
10
9
 
11
10
  class Signal(BaseModel):
12
11
  """
@@ -67,7 +66,9 @@ class Signal(BaseModel):
67
66
  """
68
67
  return self._client.list_parquet_files(self.dataset_id, self.id, refresh_cache)
69
68
 
70
- def get_data(self, dtype: Literal["numeric", "text"] | None = None, refresh_cache: bool = False) -> pd.DataFrame:
69
+ def get_data(
70
+ self, dtype: Literal["numeric", "text"] | None = None, refresh_cache: bool = False
71
+ ) -> pd.DataFrame:
71
72
  """
72
73
  Get this signal's raw data as a pandas DataFrame.
73
74