marpledata 3.6.0.dev1__tar.gz → 3.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/CHANGELOG.md +18 -2
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/PKG-INFO +45 -1
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/README.md +44 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/api/db.rst +4 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/getting-started.rst +1 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/tutorials.rst +41 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/example.py +2 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/pyproject.toml +1 -1
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/__init__.py +1 -1
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/__init__.py +101 -3
- marpledata-3.7.0/src/marple/db/activity.py +28 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/dataset.py +126 -5
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/datastream.py +119 -16
- marpledata-3.7.0/src/marple/db/script.py +188 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/signal.py +2 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/utils.py +18 -2
- marpledata-3.7.0/tests/test_activity.py +52 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/test_db.py +27 -1
- marpledata-3.7.0/tests/test_scripts.py +266 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/test_signal_upload.py +15 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/uv.lock +1 -1
- marpledata-3.6.0.dev1/.uv-cache/.gitignore +0 -1
- marpledata-3.6.0.dev1/.uv-cache/.lock +0 -0
- marpledata-3.6.0.dev1/.uv-cache/CACHEDIR.TAG +0 -1
- marpledata-3.6.0.dev1/.uv-cache/interpreter-v4/c2c2931c8a99aae1/588e1e7c128e2872.msgpack +0 -0
- marpledata-3.6.0.dev1/.uv-cache/sdists-v9/.gitignore +0 -0
- marpledata-3.6.0.dev1/docs/api/marple.Insight.rst +0 -33
- marpledata-3.6.0.dev1/docs/api/marple.db.DB.rst +0 -53
- marpledata-3.6.0.dev1/docs/api/marple.db.DataStream.rst +0 -50
- marpledata-3.6.0.dev1/docs/api/marple.db.Dataset.rst +0 -63
- marpledata-3.6.0.dev1/docs/api/marple.db.DatasetList.rst +0 -40
- marpledata-3.6.0.dev1/docs/api/marple.db.LAKE_ARROW_SCHEMA.rst +0 -6
- marpledata-3.6.0.dev1/docs/api/marple.db.SCHEMA.rst +0 -6
- marpledata-3.6.0.dev1/docs/api/marple.db.Signal.rst +0 -47
- marpledata-3.6.0.dev1/docs/api/marple.db.SignalUpload.rst +0 -31
- marpledata-3.6.0.dev1/docs/api/marple.db.SignalsAlreadyExistError.rst +0 -6
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/.flake8 +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/.gitignore +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/.python-version +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/AGENTS.md +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/CONTRIBUTING.md +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/LICENSE +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/RELEASING.md +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/DEPLOYMENT.md +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/_static/custom.css +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/_static/favicon.png +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/_static/logo.png +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/api/insight.rst +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/api.rst +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/conf.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/docs/index.rst +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/examples_race.csv +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/pytest.xml +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/constants.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/signal_upload.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/db/sql.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/insight.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/src/marple/py.typed +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/conftest.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/support.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/test_insight.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/test_reliability.py +0 -0
- {marpledata-3.6.0.dev1 → marpledata-3.7.0}/tests/test_sql.py +0 -0
|
@@ -5,15 +5,31 @@ All notable changes to the Python SDK package `marpledata` will be documented in
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
-
## [
|
|
8
|
+
## [3.7.0] - 2026-09-28
|
|
9
|
+
|
|
10
|
+
### Changed
|
|
11
|
+
|
|
12
|
+
- Clarify docs to use add_signals() for bulk adding signals
|
|
13
|
+
- `Dataset.add_signal` and `Dataset.add_signals` can upload onto a dataset in `POSTPROCESSING_FAILED`
|
|
14
|
+
|
|
15
|
+
## [3.6.0] - 2026-09-04
|
|
9
16
|
|
|
10
17
|
### Changed
|
|
11
18
|
|
|
12
19
|
- Exact signal names in `get_data` / `get_signal` / `get_signals` resolve via the dataset cache or `GET /datapool/{pool}/signal/{name}/id` instead of downloading the full datapool `signal_map`. Regex patterns still use the map.
|
|
20
|
+
- Optional `plugin_args` on `DataStream.push_file` (and the deprecated `DB.push_file` wrapper).
|
|
13
21
|
|
|
14
22
|
### Added
|
|
15
23
|
|
|
16
|
-
-
|
|
24
|
+
- Processing Pipeline
|
|
25
|
+
- `DB.create_script` / `get_scripts` / `get_script` / `delete_script`
|
|
26
|
+
- `Script.update` / `duplicate` / `delete`, to manage stored `process(dataset)` scripts.
|
|
27
|
+
- `Dataset.run` / `DB.run_script` to execute a stored processing script against a dataset via a server-side sandbox job.
|
|
28
|
+
- `DataStream.scripts`, `DataStream.update`, and `DataStream.rerun_processing` (`DB.rerun_processing`) to edit a stream and set the script pipeline
|
|
29
|
+
- `Dataset.rerun_processing` and `Dataset.get_debug_messages`.
|
|
30
|
+
- `DB.delete_signals`, `Dataset.delete_signal` / `Dataset.delete_signals`, and `Signal.delete` to remove signals from a dataset.
|
|
31
|
+
- `Dataset.reingest` to reingest a dataset from its original uploaded file, with optional `plugin_args`.
|
|
32
|
+
- `logger.debug` lines on SDK mutations (`add_signal`, `update_metadata`, …), off by default. Call `db.verbose()` to print them.
|
|
17
33
|
|
|
18
34
|
### Fixed
|
|
19
35
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: marpledata
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.7.0
|
|
4
4
|
Summary: Marple SDK for Python
|
|
5
5
|
Project-URL: Homepage, https://www.marpledata.com/
|
|
6
6
|
Project-URL: Documentation, https://marpledata.gitlab.io/marple-sdk/
|
|
@@ -78,6 +78,7 @@ API_TOKEN = "<your api token>"
|
|
|
78
78
|
API_URL = "https://db.marpledata.com/api/v1" # optional if using the default SaaS
|
|
79
79
|
|
|
80
80
|
db = DB(API_TOKEN, API_URL)
|
|
81
|
+
db.verbose() # print SDK mutation activity (push_file, add_signal, …) to stdout
|
|
81
82
|
|
|
82
83
|
db.check_connection()
|
|
83
84
|
|
|
@@ -98,6 +99,8 @@ a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
|
98
99
|
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
99
100
|
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
100
101
|
|
|
102
|
+
Warning: for performance reasons, prefer adding signals in bulk using `add_signals` over multiple usages of `add_signal`.
|
|
103
|
+
|
|
101
104
|
```python
|
|
102
105
|
# Single signal: wait until available before reading
|
|
103
106
|
speed = dataset.get_signal("car.speed").get_data()
|
|
@@ -124,6 +127,42 @@ samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
|
124
127
|
dataset.add_signal("car.custom", samples)
|
|
125
128
|
```
|
|
126
129
|
|
|
130
|
+
#### Processing scripts
|
|
131
|
+
|
|
132
|
+
Write a `process(dataset)` function, store it, and try it on any imported dataset. This runs on the server and writes to that dataset.
|
|
133
|
+
Warning: for performance reasons, prefer adding signals in bulk using `add_signals` over multiple usages of `add_signal`.
|
|
134
|
+
|
|
135
|
+
```python
|
|
136
|
+
source = """
|
|
137
|
+
from marple.db import Dataset
|
|
138
|
+
|
|
139
|
+
def process(dataset: Dataset) -> None:
|
|
140
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
141
|
+
dataset.add_signal("car.speed_kmh", speed * 3.6, metadata={"unit": "km/h"})
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
script = db.create_script("speed_kmh", source)
|
|
145
|
+
dataset = stream.get_dataset(path="lap.csv")
|
|
146
|
+
# or: dataset = stream.push_file("lap.csv").wait_for_import()
|
|
147
|
+
dataset.run(script)
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
Pass a `.py` path or `pathlib.Path` to `create_script` / `script.update` instead of source text. Iterate with `script.update(script=...)` then `dataset.run(script)` again (or `dataset.run(script, source=...)` to save and run in one step). New signals written by the script may need `signal.wait_until_available(...)` before you read them.
|
|
151
|
+
|
|
152
|
+
When the script looks right, attach it to the stream with `stream.update(scripts=...)`. That **replaces** the pipeline (pass `[]` to detach all). New uploads then run those scripts after ingest.
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
stream = stream.update(scripts=[script.id])
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
For files already imported, rerun aliasing and the stream's script pipeline:
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
dataset = dataset.rerun_processing().wait_for_import()
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Use `script.update(...)` to change source or metadata, and `dataset.get_debug_messages()` for the latest ingestion's debug log (not the sandbox job log from `dataset.run`).
|
|
165
|
+
|
|
127
166
|
#### Upload large files
|
|
128
167
|
|
|
129
168
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -221,6 +260,11 @@ if len(datasets) > 0:
|
|
|
221
260
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
222
261
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
223
262
|
- **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
|
|
263
|
+
- **Run a processing script**: `dataset.run(script)` or `db.run_script(dataset_id, script)`
|
|
264
|
+
- **Create a processing script**: `db.create_script(name, script)`
|
|
265
|
+
- **Set script pipeline** (replaces the full list): `stream.update(scripts=[script.id])`
|
|
266
|
+
- **Rerun aliasing + scripts**: `dataset.rerun_processing().wait_for_import()` or `stream.rerun_processing([dataset.id])`
|
|
267
|
+
- **Read ingest debug logs**: `dataset.get_debug_messages()`
|
|
224
268
|
|
|
225
269
|
For live streams:
|
|
226
270
|
|
|
@@ -43,6 +43,7 @@ API_TOKEN = "<your api token>"
|
|
|
43
43
|
API_URL = "https://db.marpledata.com/api/v1" # optional if using the default SaaS
|
|
44
44
|
|
|
45
45
|
db = DB(API_TOKEN, API_URL)
|
|
46
|
+
db.verbose() # print SDK mutation activity (push_file, add_signal, …) to stdout
|
|
46
47
|
|
|
47
48
|
db.check_connection()
|
|
48
49
|
|
|
@@ -63,6 +64,8 @@ a conflict without overwrite raises `SignalsAlreadyExistError`.
|
|
|
63
64
|
A Series, or a DataFrame without a `time` column, takes its times from a `DatetimeIndex` or
|
|
64
65
|
`TimedeltaIndex`. Indexed DataFrames must still have a `value` and/or `value_text` column.
|
|
65
66
|
|
|
67
|
+
Warning: for performance reasons, prefer adding signals in bulk using `add_signals` over multiple usages of `add_signal`.
|
|
68
|
+
|
|
66
69
|
```python
|
|
67
70
|
# Single signal: wait until available before reading
|
|
68
71
|
speed = dataset.get_signal("car.speed").get_data()
|
|
@@ -89,6 +92,42 @@ samples = pd.DataFrame({"time": [t0, t0 + 1_000_000_000], "value": [1.0, 2.0]})
|
|
|
89
92
|
dataset.add_signal("car.custom", samples)
|
|
90
93
|
```
|
|
91
94
|
|
|
95
|
+
#### Processing scripts
|
|
96
|
+
|
|
97
|
+
Write a `process(dataset)` function, store it, and try it on any imported dataset. This runs on the server and writes to that dataset.
|
|
98
|
+
Warning: for performance reasons, prefer adding signals in bulk using `add_signals` over multiple usages of `add_signal`.
|
|
99
|
+
|
|
100
|
+
```python
|
|
101
|
+
source = """
|
|
102
|
+
from marple.db import Dataset
|
|
103
|
+
|
|
104
|
+
def process(dataset: Dataset) -> None:
|
|
105
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
106
|
+
dataset.add_signal("car.speed_kmh", speed * 3.6, metadata={"unit": "km/h"})
|
|
107
|
+
"""
|
|
108
|
+
|
|
109
|
+
script = db.create_script("speed_kmh", source)
|
|
110
|
+
dataset = stream.get_dataset(path="lap.csv")
|
|
111
|
+
# or: dataset = stream.push_file("lap.csv").wait_for_import()
|
|
112
|
+
dataset.run(script)
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
Pass a `.py` path or `pathlib.Path` to `create_script` / `script.update` instead of source text. Iterate with `script.update(script=...)` then `dataset.run(script)` again (or `dataset.run(script, source=...)` to save and run in one step). New signals written by the script may need `signal.wait_until_available(...)` before you read them.
|
|
116
|
+
|
|
117
|
+
When the script looks right, attach it to the stream with `stream.update(scripts=...)`. That **replaces** the pipeline (pass `[]` to detach all). New uploads then run those scripts after ingest.
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
stream = stream.update(scripts=[script.id])
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
For files already imported, rerun aliasing and the stream's script pipeline:
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
dataset = dataset.rerun_processing().wait_for_import()
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Use `script.update(...)` to change source or metadata, and `dataset.get_debug_messages()` for the latest ingestion's debug log (not the sandbox job log from `dataset.run`).
|
|
130
|
+
|
|
92
131
|
#### Upload large files
|
|
93
132
|
|
|
94
133
|
`stream.push_file(...)` starts an ingestion and lets the Marple DB API choose the best upload mode. Depending on the deployment and file size, the SDK can upload through the API server, upload directly to Azure Blob Storage, use a single presigned URL, or split the file into multipart uploads.
|
|
@@ -186,6 +225,11 @@ if len(datasets) > 0:
|
|
|
186
225
|
- **Delete a stream**: `stream.delete()` or `db.delete_stream(stream_key)`
|
|
187
226
|
- **Delete a dataset**: `dataset.delete()` or `db.delete_dataset(dataset_id, dataset_path)`
|
|
188
227
|
- **Delete signals**: `signal.delete()`, `dataset.delete_signal(signal_id)` / `dataset.delete_signals(signal_ids)`, or `db.delete_signals(dataset_id, dataset_path, signal_ids)`
|
|
228
|
+
- **Run a processing script**: `dataset.run(script)` or `db.run_script(dataset_id, script)`
|
|
229
|
+
- **Create a processing script**: `db.create_script(name, script)`
|
|
230
|
+
- **Set script pipeline** (replaces the full list): `stream.update(scripts=[script.id])`
|
|
231
|
+
- **Rerun aliasing + scripts**: `dataset.rerun_processing().wait_for_import()` or `stream.rerun_processing([dataset.id])`
|
|
232
|
+
- **Read ingest debug logs**: `dataset.get_debug_messages()`
|
|
189
233
|
|
|
190
234
|
For live streams:
|
|
191
235
|
|
|
@@ -14,6 +14,10 @@ See :doc:`../tutorials` for upload examples.
|
|
|
14
14
|
marple.db.DataStream
|
|
15
15
|
marple.db.Dataset
|
|
16
16
|
marple.db.DatasetList
|
|
17
|
+
marple.db.SandboxJob
|
|
18
|
+
marple.db.SandboxJobStatus
|
|
19
|
+
marple.db.Script
|
|
20
|
+
marple.db.ScriptVersion
|
|
17
21
|
marple.db.Signal
|
|
18
22
|
marple.db.SignalUpload
|
|
19
23
|
marple.db.SignalsAlreadyExistError
|
|
@@ -40,6 +40,7 @@ After import, you can add derived signals with
|
|
|
40
40
|
``dataset.add_signal(...)`` / ``dataset.add_signals([...])``. For custom
|
|
41
41
|
ingest without file parsing, use ``stream.add_dataset(...)`` then
|
|
42
42
|
``add_signal``. See :doc:`tutorials` for a full example.
|
|
43
|
+
Warning: for performance reasons, prefer adding signals in bulk using ``add_signals`` over multiple usages of ``add_signal``.
|
|
43
44
|
|
|
44
45
|
``stream.push_file(...)`` starts an ingestion and lets the Marple DB API choose
|
|
45
46
|
the best upload mode for the deployment and file size. For large files, use a
|
|
@@ -46,6 +46,8 @@ Add signals after import, or start with an empty dataset using
|
|
|
46
46
|
A Series, or a DataFrame without a ``time`` column, takes its times from a
|
|
47
47
|
``DatetimeIndex`` or ``TimedeltaIndex``.
|
|
48
48
|
|
|
49
|
+
Warning: for performance reasons, prefer adding signals in bulk using ``add_signals`` over multiple usages of ``add_signal``.
|
|
50
|
+
|
|
49
51
|
.. code-block:: python
|
|
50
52
|
|
|
51
53
|
speed = dataset.get_signal("car.speed").get_data()
|
|
@@ -78,6 +80,45 @@ For batches, ``add_signals`` returns signal IDs without waiting:
|
|
|
78
80
|
for signal in dataset.get_signals(signal_ids=ids, refresh=True)
|
|
79
81
|
]
|
|
80
82
|
|
|
83
|
+
Processing scripts
|
|
84
|
+
------------------
|
|
85
|
+
|
|
86
|
+
Write a ``process(dataset)`` function, store it, and try it on any imported dataset.
|
|
87
|
+
This runs on the server and writes to that dataset.
|
|
88
|
+
Warning: for performance reasons, prefer adding signals in bulk using ``add_signals`` over multiple usages of ``add_signal``.
|
|
89
|
+
|
|
90
|
+
.. code-block:: python
|
|
91
|
+
|
|
92
|
+
source = """
|
|
93
|
+
from marple.db import Dataset
|
|
94
|
+
|
|
95
|
+
def process(dataset: Dataset) -> None:
|
|
96
|
+
speed = dataset.get_signal("car.speed").get_data()
|
|
97
|
+
dataset.add_signal("car.speed_kmh", speed * 3.6, metadata={"unit": "km/h"})
|
|
98
|
+
"""
|
|
99
|
+
|
|
100
|
+
script = db.create_script("speed_kmh", source)
|
|
101
|
+
dataset = stream.get_dataset(path="lap.csv")
|
|
102
|
+
dataset.run(script)
|
|
103
|
+
|
|
104
|
+
Pass a ``.py`` path instead of source text. Iterate with
|
|
105
|
+
``script.update(script=...)`` then ``dataset.run(script)`` (or
|
|
106
|
+
``dataset.run(script, source=...)`` to save and run in one step).
|
|
107
|
+
|
|
108
|
+
When the script looks right, attach it to the stream with
|
|
109
|
+
``stream.update(scripts=...)``. That **replaces** the pipeline
|
|
110
|
+
(pass ``[]`` to detach all). New uploads then run those scripts after ingest.
|
|
111
|
+
|
|
112
|
+
.. code-block:: python
|
|
113
|
+
|
|
114
|
+
stream = stream.update(scripts=[script.id])
|
|
115
|
+
|
|
116
|
+
For files already imported, rerun aliasing and the stream's script pipeline:
|
|
117
|
+
|
|
118
|
+
.. code-block:: python
|
|
119
|
+
|
|
120
|
+
dataset = dataset.rerun_processing().wait_for_import()
|
|
121
|
+
|
|
81
122
|
Filter datasets and get resampled data
|
|
82
123
|
--------------------------------------
|
|
83
124
|
|
|
@@ -2,6 +2,7 @@ import os
|
|
|
2
2
|
import re
|
|
3
3
|
|
|
4
4
|
import dotenv
|
|
5
|
+
|
|
5
6
|
from src.marple.db import DB, Dataset
|
|
6
7
|
|
|
7
8
|
dotenv.load_dotenv()
|
|
@@ -11,6 +12,7 @@ if api_token is None:
|
|
|
11
12
|
|
|
12
13
|
url = os.getenv("MDB_URL")
|
|
13
14
|
db = DB(api_token, url)
|
|
15
|
+
db.verbose() # print SDK mutation activity (push_file, add_signal, …) to stdout
|
|
14
16
|
db.check_connection()
|
|
15
17
|
|
|
16
18
|
stream_name = "CSV Stream"
|
|
@@ -1,19 +1,21 @@
|
|
|
1
1
|
import warnings
|
|
2
2
|
from functools import wraps
|
|
3
3
|
from pathlib import Path
|
|
4
|
-
from typing import TYPE_CHECKING, Any, Literal, Optional, Sequence
|
|
4
|
+
from typing import TYPE_CHECKING, Any, Literal, Optional, Sequence, TextIO
|
|
5
5
|
|
|
6
6
|
import pandas as pd
|
|
7
7
|
from pydantic import ValidationError
|
|
8
8
|
from requests import Response
|
|
9
9
|
from requests.exceptions import ConnectionError
|
|
10
10
|
|
|
11
|
-
from marple.db import sql
|
|
11
|
+
from marple.db import activity, sql
|
|
12
|
+
from marple.db.activity import logger
|
|
12
13
|
from marple.db.constants import LAKE_ARROW_SCHEMA as _LAKE_ARROW_SCHEMA
|
|
13
14
|
from marple.db.constants import SAAS_URL
|
|
14
15
|
from marple.db.constants import SCHEMA as _SCHEMA
|
|
15
16
|
from marple.db.dataset import Dataset, DatasetList
|
|
16
17
|
from marple.db.datastream import DataStream
|
|
18
|
+
from marple.db.script import SandboxJob, SandboxJobStatus, Script, ScriptVersion
|
|
17
19
|
from marple.db.signal import Signal
|
|
18
20
|
from marple.db.signal_upload import SignalsAlreadyExistError, SignalUpload
|
|
19
21
|
from marple.utils import DBClient, validate_response
|
|
@@ -37,6 +39,10 @@ __all__ = [
|
|
|
37
39
|
"DataStream",
|
|
38
40
|
"Dataset",
|
|
39
41
|
"DatasetList",
|
|
42
|
+
"SandboxJob",
|
|
43
|
+
"SandboxJobStatus",
|
|
44
|
+
"Script",
|
|
45
|
+
"ScriptVersion",
|
|
40
46
|
"Signal",
|
|
41
47
|
"SignalUpload",
|
|
42
48
|
"SignalsAlreadyExistError",
|
|
@@ -151,6 +157,10 @@ class DB:
|
|
|
151
157
|
self._refresh_stream_cache(r)
|
|
152
158
|
return True
|
|
153
159
|
|
|
160
|
+
def verbose(self, enabled: bool = True, file: TextIO | None = None) -> None:
|
|
161
|
+
"""Turn SDK mutation activity logging on or off (stdout by default)."""
|
|
162
|
+
activity.verbose(enabled, file)
|
|
163
|
+
|
|
154
164
|
# Stream functions #
|
|
155
165
|
|
|
156
166
|
def create_stream(
|
|
@@ -237,6 +247,92 @@ class DB:
|
|
|
237
247
|
f"Stream with name or id {stream_key} not found, available streams: {', '.join([s.name for s in self._streams.values()])}"
|
|
238
248
|
)
|
|
239
249
|
|
|
250
|
+
def rerun_processing(self, stream_key: str | int, dataset_ids: Sequence[int] | None = None) -> None:
|
|
251
|
+
"""
|
|
252
|
+
Rerun aliasing and processing scripts for a stream or selected datasets.
|
|
253
|
+
|
|
254
|
+
See :meth:`~marple.db.datastream.DataStream.rerun_processing`.
|
|
255
|
+
"""
|
|
256
|
+
self.get_stream(stream_key).rerun_processing(dataset_ids)
|
|
257
|
+
|
|
258
|
+
def get_scripts(self) -> list[Script]:
|
|
259
|
+
"""List processing scripts in this workspace (without source / version history)."""
|
|
260
|
+
r = self.get("/scripts")
|
|
261
|
+
return [
|
|
262
|
+
Script(client=self.client, **script)
|
|
263
|
+
for script in validate_response(r, "Get scripts failed")["scripts"]
|
|
264
|
+
]
|
|
265
|
+
|
|
266
|
+
def get_script(self, script_id: int) -> Script:
|
|
267
|
+
"""Get a processing script by ID, including recent versions and source."""
|
|
268
|
+
return Script.fetch(self.client, script_id)
|
|
269
|
+
|
|
270
|
+
def run_script(
|
|
271
|
+
self,
|
|
272
|
+
dataset_id: int,
|
|
273
|
+
script: Script | int,
|
|
274
|
+
*,
|
|
275
|
+
source: str | Path | None = None,
|
|
276
|
+
version: int | None = None,
|
|
277
|
+
timeout: float = 180,
|
|
278
|
+
) -> Dataset:
|
|
279
|
+
"""Run a stored processing script on a dataset.
|
|
280
|
+
|
|
281
|
+
See :meth:`~marple.db.dataset.Dataset.run`.
|
|
282
|
+
|
|
283
|
+
Args:
|
|
284
|
+
dataset_id: The ID of the dataset to run the script on.
|
|
285
|
+
script: A stored script or its ID.
|
|
286
|
+
source: Optional source text or ``.py`` path to save before running.
|
|
287
|
+
version: Script version ID to run. Defaults to the latest version.
|
|
288
|
+
timeout: Seconds to wait for the sandbox job to finish.
|
|
289
|
+
|
|
290
|
+
Warning:
|
|
291
|
+
This is not a dry-run. The script will be executed and the dataset will be modified.
|
|
292
|
+
"""
|
|
293
|
+
return self.get_dataset(dataset_id).run(script, source=source, version=version, timeout=timeout)
|
|
294
|
+
|
|
295
|
+
def create_script(
|
|
296
|
+
self,
|
|
297
|
+
name: str,
|
|
298
|
+
script: str | Path,
|
|
299
|
+
*,
|
|
300
|
+
description: str | None = None,
|
|
301
|
+
) -> Script:
|
|
302
|
+
"""
|
|
303
|
+
Create a processing script.
|
|
304
|
+
|
|
305
|
+
Attach it to a stream with :meth:`~marple.db.datastream.DataStream.update`
|
|
306
|
+
(``scripts=`` replaces the full pipeline).
|
|
307
|
+
|
|
308
|
+
Args:
|
|
309
|
+
name: The name of the script.
|
|
310
|
+
script: Source text or a path to a file that must define ``process(dataset)``.
|
|
311
|
+
description: The description of the script.
|
|
312
|
+
"""
|
|
313
|
+
r = self.post(
|
|
314
|
+
"/script",
|
|
315
|
+
json={
|
|
316
|
+
"name": name,
|
|
317
|
+
"description": description,
|
|
318
|
+
"script": Script.resolve_source(script),
|
|
319
|
+
},
|
|
320
|
+
)
|
|
321
|
+
created = Script(client=self.client, **validate_response(r, "Create script failed"))
|
|
322
|
+
logger.debug(f"Created script {name}")
|
|
323
|
+
return created
|
|
324
|
+
|
|
325
|
+
def delete_script(self, script_id: int) -> None:
|
|
326
|
+
"""
|
|
327
|
+
Delete a processing script.
|
|
328
|
+
|
|
329
|
+
Warning:
|
|
330
|
+
This cannot be undone. The script is removed from every stream pipeline.
|
|
331
|
+
"""
|
|
332
|
+
r = self.delete(f"/script/{script_id}")
|
|
333
|
+
validate_response(r, "Delete script failed")
|
|
334
|
+
logger.debug(f"Deleted script {script_id}")
|
|
335
|
+
|
|
240
336
|
def _refresh_stream_cache(self, r: Response | None = None) -> None:
|
|
241
337
|
if r is None:
|
|
242
338
|
r = self.get("/streams")
|
|
@@ -354,6 +450,7 @@ class DB:
|
|
|
354
450
|
metadata: dict | None = None,
|
|
355
451
|
file_name: str | None = None,
|
|
356
452
|
overwrite: bool = False,
|
|
453
|
+
plugin_args: str | None = None,
|
|
357
454
|
) -> int:
|
|
358
455
|
"""
|
|
359
456
|
Push a file to a datastream.
|
|
@@ -362,12 +459,13 @@ class DB:
|
|
|
362
459
|
- `metadata`: (optional) A dictionary of metadata to be associated with the file.
|
|
363
460
|
- `file_name`: (optional) The name of the file to be stored in the stream.
|
|
364
461
|
- `overwrite`: (optional) If true, existing dataset with the same name will be overwritten.
|
|
462
|
+
- `plugin_args`: (optional) Plugin arguments for this ingest.
|
|
365
463
|
|
|
366
464
|
Note:
|
|
367
465
|
This function is deprecated and it is encouraged to use the `push_file` method in the `DataStream` class directly.
|
|
368
466
|
"""
|
|
369
467
|
stream = self.get_stream(stream_key)
|
|
370
|
-
return stream.push_file(file_path, metadata, file_name, overwrite=overwrite).id
|
|
468
|
+
return stream.push_file(file_path, metadata, file_name, overwrite=overwrite, plugin_args=plugin_args).id
|
|
371
469
|
|
|
372
470
|
@deprecated
|
|
373
471
|
def get_status(self, stream_key: str | int, dataset_id: int) -> dict:
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import sys
|
|
3
|
+
from typing import TextIO
|
|
4
|
+
|
|
5
|
+
logger = logging.getLogger("marple.sdk")
|
|
6
|
+
logger.addHandler(logging.NullHandler())
|
|
7
|
+
logger.propagate = False
|
|
8
|
+
|
|
9
|
+
_handler: logging.Handler | None = None
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def verbose(enabled: bool = True, file: TextIO | None = None) -> None:
|
|
13
|
+
"""Turn SDK mutation activity logging on or off (stdout by default)."""
|
|
14
|
+
global _handler
|
|
15
|
+
if enabled:
|
|
16
|
+
logger.setLevel(logging.DEBUG)
|
|
17
|
+
if _handler is not None:
|
|
18
|
+
return
|
|
19
|
+
_handler = logging.StreamHandler(file or sys.stdout)
|
|
20
|
+
_handler.setFormatter(logging.Formatter("%(message)s"))
|
|
21
|
+
logger.addHandler(_handler)
|
|
22
|
+
return
|
|
23
|
+
|
|
24
|
+
if _handler is not None:
|
|
25
|
+
logger.removeHandler(_handler)
|
|
26
|
+
_handler.close()
|
|
27
|
+
_handler = None
|
|
28
|
+
logger.setLevel(logging.NOTSET)
|