simudyne-pulse 0.6.0.dev1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,710 @@
1
+ """
2
+ Simulation Resource for the Pulse SDK.
3
+
4
+ This module provides methods for running agent-based market simulations,
5
+ tracking job status, and retrieving results.
6
+
7
+ Workflow:
8
+ 1. Submit a simulation with run() -> returns job_id and sim_ids
9
+ 2. Track progress with get_job_status(job_id)
10
+ 3. View all past jobs with get_jobs()
11
+ 4. Once complete, retrieve results with get_job_results() or get_sim_data()
12
+ """
13
+
14
+ import io
15
+
16
+ from simudyne.exceptions import PulseAPIError
17
+
18
+ RUN_PATH = "/simulation/run"
19
+ JOBS_PATH = "/simulation/jobs"
20
+ RESULTS_PATH = "/simulation/results"
21
+ CACHED_PATH = "/simulation/cached"
22
+ SAMPLE_PATH = "/simulation/sample"
23
+ CALIBRATE_PATH = "/calibrate"
24
+
25
+ # Available market scenarios
26
+ SCENARIOS = {
27
+ "normal": "No scenario injection - background agents only",
28
+ "flash_crash": "Large rapid SELL depleting bid-side liquidity",
29
+ "buy_panic": "Large rapid BUY depleting ask-side liquidity",
30
+ "gradual_selloff": "Slow sustained SELL over an extended period",
31
+ "trending_up": "Small steady BUY flow producing a persistent uptrend",
32
+ "trending_down": "Small steady SELL flow producing a persistent downtrend",
33
+ }
34
+
35
+ # Scenario parameter defaults
36
+ SCENARIO_DEFAULTS = {
37
+ "flash_crash": {
38
+ "impact_multiplier": 22.0,
39
+ "order_size_ratio": 0.19,
40
+ "order_freq": "500ms",
41
+ "start_time": "10:30:00",
42
+ },
43
+ "buy_panic": {
44
+ "impact_multiplier": 22.0,
45
+ "order_size_ratio": 0.19,
46
+ "order_freq": "500ms",
47
+ "start_time": "10:30:00",
48
+ },
49
+ "gradual_selloff": {
50
+ "impact_multiplier": 10.0,
51
+ "order_size_ratio": 0.05,
52
+ "order_freq": "5s",
53
+ "start_time": "10:30:00",
54
+ },
55
+ "trending_up": {
56
+ "impact_multiplier": 5.0,
57
+ "order_size_ratio": 0.03,
58
+ "order_freq": "30s",
59
+ "start_time": "10:30:00",
60
+ },
61
+ "trending_down": {
62
+ "impact_multiplier": 5.0,
63
+ "order_size_ratio": 0.03,
64
+ "order_freq": "30s",
65
+ "start_time": "10:30:00",
66
+ },
67
+ }
68
+
69
+
70
+ class SimulationResource:
71
+ """
72
+ Run agent-based market simulations and track job status.
73
+
74
+ Example workflow:
75
+ >>> # Submit a simulation
76
+ >>> result = client.simulation.run(
77
+ ... symbol="9999.HK",
78
+ ... cal_date="2025-09-01",
79
+ ... provider="omd",
80
+ ... exchange="hkex_securities",
81
+ ... n_runs=10,
82
+ ... scenario="flash_crash"
83
+ ... )
84
+ >>> job_id = result["job_id"]
85
+ >>> print(f"Submitted job: {job_id}")
86
+
87
+ >>> # Check status
88
+ >>> status = client.simulation.get_job_status(job_id)
89
+ >>> print(f"Progress: {status['status_summary']}")
90
+ >>> print(f"Complete: {status['is_complete']}")
91
+
92
+ >>> # View all jobs
93
+ >>> jobs = client.simulation.get_jobs()
94
+ >>> print(f"Total jobs: {jobs['total']}")
95
+ """
96
+
97
+ def __init__(self, client):
98
+ self._client = client
99
+
100
+ def _pro_request(self, method: str, endpoint: str, **kwargs):
101
+ """Wrapper for pro-only endpoints — passes errors through from the server."""
102
+ return self._client._request(method, endpoint, **kwargs)
103
+
104
+ def run(
105
+ self,
106
+ symbol: str,
107
+ cal_date: str,
108
+ provider: str,
109
+ exchange: str,
110
+ n_runs: int = 100,
111
+ seed: int = 42,
112
+ scenario: str = "normal",
113
+ scenario_params: dict = None,
114
+ exec_algos: list = None,
115
+ ):
116
+ """
117
+ Submit a simulation run to be executed asynchronously.
118
+
119
+ The simulation will run n_runs independent Monte Carlo samples. Each run
120
+ produces a unique sim_id that can be used to retrieve results once complete.
121
+ The user_id is automatically set from your API key.
122
+
123
+ A symbol is identified by four fields: ``provider``, ``exchange``,
124
+ ``symbol`` and ``cal_date``.
125
+
126
+ Args:
127
+ symbol: Trading symbol (e.g., "9999.HK", "0005.HK")
128
+ cal_date: Calibration date in YYYY-MM-DD format (e.g., "2025-09-01")
129
+ provider: Data provider the symbol is sourced from (e.g., "omd", "bmll")
130
+ exchange: Exchange protocol name (e.g., "hkex_securities", "hkex_derivatives")
131
+ n_runs: Number of independent Monte Carlo runs (default: 100)
132
+ seed: Master random seed for reproducibility (default: 42)
133
+ scenario: Market scenario to simulate. Options:
134
+ - "normal": No scenario injection (default)
135
+ - "flash_crash": Large rapid SELL depleting bid-side liquidity
136
+ - "buy_panic": Large rapid BUY depleting ask-side liquidity
137
+ - "gradual_selloff": Slow sustained SELL over extended period
138
+ - "trending_up": Small steady BUY producing uptrend
139
+ - "trending_down": Small steady SELL producing downtrend
140
+ scenario_params: Override scenario defaults. Keys:
141
+ - impact_multiplier (float): Total volume as multiple of resting liquidity
142
+ - order_size_ratio (float): Child order size as fraction of liquidity
143
+ - order_freq (str): Child order spacing (e.g., "500ms", "5s", "30s")
144
+ - start_time (str): Time to begin orders (e.g., "10:30:00")
145
+ exec_algos: List of execution algorithm configs. Each dict must have "type".
146
+ Supported types: "twap", "vwap", "css"
147
+
148
+ For TWAP/VWAP:
149
+ - type: "twap" or "vwap" (required)
150
+ - order_size: Total shares. Negative = BUY, positive = SELL (required)
151
+ - horizon: Execution window in SECONDS, e.g. 3600 for 1 hour (required)
152
+ - start_time: When to start, e.g. "09:30:00" (optional, defaults to market open)
153
+
154
+ For CSS (Custom Static Schedule):
155
+ - type: "css" (required)
156
+ - orders: Dict mapping timestamps to quantities (required)
157
+
158
+ Multiple exec_algos can be submitted in one simulation.
159
+
160
+ Returns:
161
+ dict: Submission result containing:
162
+ - job_id (str): Unique job identifier for tracking
163
+ - queued_sim_ids (list): List of simulation IDs that will be run
164
+ - run_offset (int): Starting run index
165
+ - n_runs (int): Number of runs queued
166
+
167
+ Example - Basic run:
168
+ >>> result = client.simulation.run(
169
+ ... symbol="9999.HK",
170
+ ... cal_date="2025-09-01",
171
+ ... provider="omd",
172
+ ... exchange="hkex_securities",
173
+ ... n_runs=10
174
+ ... )
175
+ >>> print(result["job_id"])
176
+
177
+ Example - Flash crash scenario:
178
+ >>> result = client.simulation.run(
179
+ ... symbol="9999.HK",
180
+ ... cal_date="2025-09-01",
181
+ ... provider="omd",
182
+ ... exchange="hkex_securities",
183
+ ... n_runs=50,
184
+ ... scenario="flash_crash",
185
+ ... scenario_params={
186
+ ... "start_time": "11:00:00",
187
+ ... "impact_multiplier": 15.0
188
+ ... }
189
+ ... )
190
+
191
+ Example - With TWAP execution algo (sell 50k shares over 1 hour):
192
+ >>> result = client.simulation.run(
193
+ ... symbol="9999.HK",
194
+ ... cal_date="2025-09-01",
195
+ ... provider="omd",
196
+ ... exchange="hkex_securities",
197
+ ... n_runs=20,
198
+ ... exec_algos=[{
199
+ ... "type": "twap",
200
+ ... "order_size": 50000, # positive = sell
201
+ ... "horizon": 3600, # seconds (1 hour)
202
+ ... "start_time": "09:30:00"
203
+ ... }]
204
+ ... )
205
+
206
+ Example - With TWAP buy order:
207
+ >>> result = client.simulation.run(
208
+ ... symbol="9999.HK",
209
+ ... cal_date="2025-09-01",
210
+ ... provider="omd",
211
+ ... exchange="hkex_securities",
212
+ ... n_runs=20,
213
+ ... exec_algos=[{
214
+ ... "type": "twap",
215
+ ... "order_size": -50000, # negative = buy
216
+ ... "horizon": 3600
217
+ ... }]
218
+ ... )
219
+ """
220
+ payload = {
221
+ "symbol": symbol,
222
+ "cal_date": cal_date,
223
+ "provider": provider,
224
+ "exchange": exchange,
225
+ "n_runs": n_runs,
226
+ "seed": seed,
227
+ "scenario": scenario,
228
+ }
229
+
230
+ if scenario_params:
231
+ payload["scenario_params"] = scenario_params
232
+ if exec_algos:
233
+ payload["exec_algos"] = self._serialize_exec_algos(exec_algos)
234
+
235
+ return self._pro_request("POST", RUN_PATH, json=payload)
236
+
237
+ @staticmethod
238
+ def _serialize_exec_algos(exec_algos: list) -> list:
239
+ """Convert pd.Series orders in CSS configs to JSON-serializable dicts."""
240
+ result = []
241
+ for algo in exec_algos:
242
+ algo = dict(algo)
243
+ if algo.get("type") == "css" and "orders" in algo:
244
+ orders = algo["orders"]
245
+ if hasattr(orders, "items"):
246
+ algo["orders"] = {str(k): int(v) for k, v in orders.items()}
247
+ result.append(algo)
248
+ return result
249
+
250
+ def calibrate(
251
+ self,
252
+ symbol: str,
253
+ cal_date: str,
254
+ provider: str,
255
+ exchange: str,
256
+ simulations: int = None,
257
+ batch_size: int = None,
258
+ optimize_adj_params: bool = True,
259
+ ):
260
+ """
261
+ Trigger model calibration for a symbol and date.
262
+
263
+ Calibration runs asynchronously to fit model parameters to observed
264
+ market data for the given symbol and date.
265
+
266
+ A symbol is identified by four fields: ``provider``, ``exchange``,
267
+ ``symbol`` and ``cal_date``.
268
+
269
+ Args:
270
+ symbol: Trading symbol (e.g., "9999.HK")
271
+ cal_date: Calibration date in YYYY-MM-DD format
272
+ provider: Data provider the symbol is sourced from (e.g., "omd", "bmll")
273
+ exchange: Exchange protocol name (e.g., "hkex_securities", "hkex_derivatives")
274
+ simulations: Number of simulations to run during calibration
275
+ batch_size: Batch size for calibration runs
276
+ optimize_adj_params: Whether to optimise adjustment parameters (default: True)
277
+
278
+ Returns:
279
+ dict: Calibration job submission result
280
+ """
281
+ payload: dict = {
282
+ "symbol": symbol,
283
+ "cal_date": cal_date,
284
+ "provider": provider,
285
+ "exchange": exchange,
286
+ "optimize_adj_params": optimize_adj_params,
287
+ }
288
+ if simulations is not None:
289
+ payload["simulations"] = simulations
290
+ if batch_size is not None:
291
+ payload["batch_size"] = batch_size
292
+ return self._pro_request("POST", CALIBRATE_PATH, json=payload)
293
+
294
+ def get_jobs(self):
295
+ """
296
+ Get all simulation jobs submitted by the authenticated user.
297
+
298
+ Returns a list of jobs with their associated simulation IDs. Use this
299
+ to find job IDs for past runs or to see what simulations are pending.
300
+
301
+ Returns:
302
+ dict: Contains:
303
+ - jobs (list): List of job objects, each with:
304
+ - job_id (str): Unique job identifier
305
+ - sim_ids (list): List of simulation IDs in this job
306
+ - created_at (str): Timestamp when job was submitted
307
+ - total (int): Total number of jobs
308
+
309
+ Example:
310
+ >>> result = client.simulation.get_jobs()
311
+ >>> print(f"You have {result['total']} jobs")
312
+ >>>
313
+ >>> for job in result["jobs"]:
314
+ ... print(f"Job {job['job_id']}: {len(job['sim_ids'])} simulations")
315
+ ... print(f" Created: {job['created_at']}")
316
+ """
317
+ return self._pro_request("GET", JOBS_PATH)
318
+
319
+ def get_job_status(self, job_id: str):
320
+ """
321
+ Get the status of all simulations in a job.
322
+
323
+ Use this to track simulation progress. Each simulation in the job goes
324
+ through states: queued -> running -> completed (or error).
325
+
326
+ Args:
327
+ job_id: The job ID from run() or get_jobs()
328
+
329
+ Returns:
330
+ dict: Job status containing:
331
+ - job_id (str): The job identifier
332
+ - total_simulations (int): Number of simulations in the job
333
+ - status_summary (dict): Count by status (e.g., {"running": 2, "completed": 8})
334
+ - is_complete (bool): True if all simulations are completed
335
+ - has_errors (bool): True if any simulation failed
336
+ - simulations (list): Individual simulation statuses with:
337
+ - sim_id (str): Simulation identifier
338
+ - status (str): Current status (queued/running/completed/error)
339
+ - error_message (str): Error details if failed
340
+ - symbol_id (str): Full symbol identifier
341
+ - timestamp (str): Last status update time
342
+
343
+ Example - Check if job is done:
344
+ >>> status = client.simulation.get_job_status("2103533f15ab0893")
345
+ >>> if status["is_complete"]:
346
+ ... print("All simulations finished!")
347
+ ... else:
348
+ ... print(f"Progress: {status['status_summary']}")
349
+
350
+ Example - Poll for completion:
351
+ >>> import time
352
+ >>>
353
+ >>> result = client.simulation.run(symbol="9999.HK", cal_date="2025-09-01", provider="omd", exchange="hkex_securities", n_runs=10)
354
+ >>> job_id = result["job_id"]
355
+ >>>
356
+ >>> while True:
357
+ ... status = client.simulation.get_job_status(job_id)
358
+ ... print(f"Status: {status['status_summary']}")
359
+ ...
360
+ ... if status["is_complete"]:
361
+ ... print("Done!")
362
+ ... break
363
+ ... elif status["has_errors"]:
364
+ ... print("Some simulations failed")
365
+ ... break
366
+ ...
367
+ ... time.sleep(30) # Check every 30 seconds
368
+ """
369
+ return self._pro_request("GET", f"{JOBS_PATH}/{job_id}/status")
370
+
371
+ @staticmethod
372
+ def list_scenarios():
373
+ """
374
+ List available market scenarios and their descriptions.
375
+
376
+ Returns:
377
+ dict: Scenario names mapped to descriptions
378
+
379
+ Example:
380
+ >>> scenarios = client.simulation.list_scenarios()
381
+ >>> for name, desc in scenarios.items():
382
+ ... print(f"{name}: {desc}")
383
+ """
384
+ return SCENARIOS.copy()
385
+
386
+ @staticmethod
387
+ def get_scenario_defaults(scenario: str):
388
+ """
389
+ Get default parameters for a scenario.
390
+
391
+ Args:
392
+ scenario: Scenario name (e.g., "flash_crash")
393
+
394
+ Returns:
395
+ dict: Default parameter values, or empty dict for "normal"
396
+
397
+ Example:
398
+ >>> defaults = client.simulation.get_scenario_defaults("flash_crash")
399
+ >>> print(defaults)
400
+ {'impact_multiplier': 22.0, 'order_size_ratio': 0.19, ...}
401
+ """
402
+ return SCENARIO_DEFAULTS.get(scenario, {}).copy()
403
+
404
+ def get_job_results(self, job_id: str):
405
+ """
406
+ Get aggregated results for all simulations in a job.
407
+
408
+ Returns params, metrics, and available files for each completed simulation.
409
+
410
+ Args:
411
+ job_id: The job ID from run() or get_jobs()
412
+
413
+ Returns:
414
+ dict: Contains:
415
+ - job_id (str): The job identifier
416
+ - total_simulations (int): Total simulations in job
417
+ - completed (int): Number of completed simulations
418
+ - simulations (list): Per-simulation results with:
419
+ - sim_id (str): Simulation identifier
420
+ - status (str): Completion status
421
+ - available_files (list): Files available for download
422
+ - params (dict): Simulation parameters
423
+ - metrics (dict): Result metrics
424
+
425
+ Example:
426
+ >>> results = client.simulation.get_job_results(job_id)
427
+ >>> for sim in results["simulations"]:
428
+ ... if sim["status"] == "completed":
429
+ ... print(f"{sim['sim_id']}: {sim['metrics']}")
430
+ """
431
+ return self._pro_request("GET", f"{JOBS_PATH}/{job_id}/results")
432
+
433
+ def list_sim_files(self, sim_id: str):
434
+ """
435
+ List available files for a specific simulation.
436
+
437
+ Args:
438
+ sim_id: The simulation ID
439
+
440
+ Returns:
441
+ dict: Contains:
442
+ - sim_id (str): The simulation identifier
443
+ - files (list): List of available filenames
444
+ - has_sim_data (bool): Whether sim_data.parquet exists
445
+ - has_params (bool): Whether params.json exists
446
+ - has_results (bool): Whether results.json exists
447
+ - has_mid_price (bool): Whether mid_price_by_min.parquet exists
448
+ - has_l2_by_second (bool): Whether l2_by_second.parquet exists
449
+ - has_exec_schedule (bool): Whether exec_schedule.parquet exists
450
+ - has_schedule_by_min (bool): Whether schedule_by_min.parquet exists
451
+
452
+ Example:
453
+ >>> files = client.simulation.list_sim_files(sim_id)
454
+ >>> print(f"Available: {files['files']}")
455
+ """
456
+ return self._client._request("GET", f"{RESULTS_PATH}/{sim_id}/files")
457
+
458
+ def get_sim_params(self, sim_id: str):
459
+ """
460
+ Get simulation parameters for a specific simulation.
461
+
462
+ Args:
463
+ sim_id: The simulation ID
464
+
465
+ Returns:
466
+ dict: Simulation parameters including:
467
+ - sim_id (str)
468
+ - calibration_params (dict)
469
+ - scenario_params (dict)
470
+ - exec_algo_params (dict)
471
+ - sim_params (dict)
472
+
473
+ Example:
474
+ >>> params = client.simulation.get_sim_params(sim_id)
475
+ >>> print(f"Scenario: {params['scenario_params']['scenario_name']}")
476
+ """
477
+ return self._client._request("GET", f"{RESULTS_PATH}/{sim_id}/params")
478
+
479
+ def get_sim_metrics(self, sim_id: str):
480
+ """
481
+ Get result metrics for a specific simulation.
482
+
483
+ Args:
484
+ sim_id: The simulation ID
485
+
486
+ Returns:
487
+ dict: Simulation result metrics
488
+
489
+ Example:
490
+ >>> metrics = client.simulation.get_sim_metrics(sim_id)
491
+ >>> print(metrics)
492
+ """
493
+ return self._client._request("GET", f"{RESULTS_PATH}/{sim_id}/metrics")
494
+
495
+ def get_sim_data(self, sim_id: str, filename: str = "sim_data.parquet"):
496
+ """
497
+ Download simulation data as a Polars DataFrame.
498
+
499
+ Args:
500
+ sim_id: The simulation ID
501
+ filename: File to download. Options:
502
+ - "sim_data.parquet": Full simulation output (LOB + orders)
503
+ - "mid_price_by_min.parquet": Mid-price by minute
504
+ - "l2_by_second.parquet": L2 order book (10 levels) sampled per second
505
+ - "exec_schedule.parquet": Execution schedule (if algo)
506
+ - "schedule_by_min.parquet": Algo schedule by minute (if algo)
507
+ - "exec_results.parquet": Execution cost metrics (if algo)
508
+
509
+ Returns:
510
+ polars.DataFrame: The simulation data
511
+
512
+ Example:
513
+ >>> df = client.simulation.get_sim_data(sim_id)
514
+ >>> print(df.shape)
515
+ >>> print(df.head())
516
+
517
+ >>> # Get mid-price data
518
+ >>> mid_df = client.simulation.get_sim_data(sim_id, "mid_price_by_min.parquet")
519
+ """
520
+ import polars as pl
521
+
522
+ url = f"{self._client.base_url}{RESULTS_PATH}/{sim_id}/data/{filename}"
523
+ response = self._client.session.get(url)
524
+
525
+ if not response.ok:
526
+ try:
527
+ detail = response.json().get("detail", response.text)
528
+ except ValueError:
529
+ detail = response.text
530
+ raise PulseAPIError(response.status_code, detail)
531
+
532
+ return pl.read_parquet(io.BytesIO(response.content))
533
+
534
+ def list_cached(
535
+ self,
536
+ symbol: str = None,
537
+ date: str = None,
538
+ scenario: str = None,
539
+ ):
540
+ """
541
+ List cached baseline simulations available to free tier users.
542
+
543
+ Returns aggregated simulation metadata for baseline (non-exec algo) simulations.
544
+ Use the returned sim_id to retrieve data via get_sim_data(), get_sim_params(), etc.
545
+
546
+ Free tier users can only access baseline simulations - no execution algorithms.
547
+
548
+ Args:
549
+ symbol: Filter by symbol (e.g., "700.HK", "9999.HK")
550
+ date: Filter by date (e.g., "2025-09-02")
551
+ scenario: Filter by scenario (e.g., "normal", "flash_crash")
552
+
553
+ Returns:
554
+ dict: Contains:
555
+ - simulations (list): List of cached simulation groups, each with:
556
+ - example_sim_id (str): A sim_id from this group (use with get_sim_data)
557
+ - symbol (str): Trading symbol
558
+ - date (str): Calibration date
559
+ - scenario (str): Scenario name
560
+ - n_runs (int): Number of available runs
561
+ - cal_hash (str): Calibration parameter hash
562
+ - sim_hash (str): Simulation parameter hash
563
+ - time_range (str): Trading time range
564
+ - total (int): Number of unique symbol/date/scenario combinations
565
+
566
+ Example - List all cached simulations:
567
+ >>> cached = client.simulation.list_cached()
568
+ >>> print(f"Found {cached['total']} cached simulation groups")
569
+ >>>
570
+ >>> for sim in cached["simulations"]:
571
+ ... print(f"{sim['symbol']} {sim['date']} {sim['scenario']}: {sim['n_runs']} runs")
572
+ ... print(f" Use sim_id: {sim['example_sim_id']}")
573
+
574
+ Example - Filter by symbol:
575
+ >>> cached = client.simulation.list_cached(symbol="700.HK")
576
+ >>> for sim in cached["simulations"]:
577
+ ... print(f"{sim['date']} {sim['scenario']}: {sim['n_runs']} runs")
578
+
579
+ Example - Get data from a cached simulation:
580
+ >>> cached = client.simulation.list_cached(symbol="9999.HK", scenario="flash_crash")
581
+ >>> if cached["simulations"]:
582
+ ... sim_id = cached["simulations"][0]["example_sim_id"]
583
+ ... df = client.simulation.get_sim_data(sim_id)
584
+ ... print(df.head())
585
+ """
586
+ params = {}
587
+ if symbol:
588
+ params["symbol"] = symbol
589
+ if date:
590
+ params["date"] = date
591
+ if scenario:
592
+ params["scenario"] = scenario
593
+
594
+ return self._client._request("GET", CACHED_PATH, params=params)
595
+
596
+ def get_sample_data(self, path: str = "simulation_sample.zip"):
597
+ """
598
+ Download a sample dataset for 700.HK — 5 Monte Carlo runs, no configuration needed.
599
+
600
+ The server picks the best available scenario (normal preferred) and returns
601
+ sim_data.parquet and mid_price_by_min.parquet for each run as a ZIP.
602
+
603
+ Args:
604
+ path: File path to save the ZIP to (default: "simulation_sample.zip").
605
+ Pass None to return raw bytes instead.
606
+
607
+ Returns:
608
+ bytes: ZIP file content if path is None, otherwise None (file written to disk).
609
+
610
+ Example - Save to disk:
611
+ >>> client.simulation.get_sample_data()
612
+ # writes simulation_sample.zip to current directory
613
+
614
+ Example - Load directly into DataFrames:
615
+ >>> import zipfile, io
616
+ >>> import polars as pl
617
+ >>>
618
+ >>> data = client.simulation.get_sample_data(path=None)
619
+ >>> with zipfile.ZipFile(io.BytesIO(data)) as zf:
620
+ ... for name in zf.namelist():
621
+ ... df = pl.read_parquet(io.BytesIO(zf.read(name)))
622
+ ... print(f"{name}: {df.shape}")
623
+ """
624
+ url = f"{self._client.base_url}{SAMPLE_PATH}"
625
+ response = self._client.session.get(url)
626
+
627
+ if not response.ok:
628
+ try:
629
+ detail = response.json().get("detail", response.text)
630
+ except ValueError:
631
+ detail = response.text
632
+ raise PulseAPIError(response.status_code, detail)
633
+
634
+ if path is None:
635
+ return response.content
636
+
637
+ with open(path, "wb") as f:
638
+ f.write(response.content)
639
+
640
+ def get_bulk_data(
641
+ self,
642
+ sim_ids: list,
643
+ include_sim_data: bool = True,
644
+ include_mid_price: bool = False,
645
+ include_l2_by_second: bool = False,
646
+ ):
647
+ """
648
+ Download data for multiple simulations as a ZIP file.
649
+
650
+ Args:
651
+ sim_ids: List of simulation IDs to download
652
+ include_sim_data: Include sim_data.parquet files (default: True)
653
+ include_mid_price: Include mid_price_by_min.parquet files (default: False)
654
+ include_l2_by_second: Include l2_by_second.parquet files (default: False)
655
+
656
+ Returns:
657
+ bytes: ZIP file content containing requested parquet files
658
+
659
+ Free-tier quota: bulk downloads are capped per rolling 24-hour window
660
+ (default 3). One unit is consumed per distinct simulation group (all
661
+ Monte Carlo runs of one scenario) — not per call, run, or file format,
662
+ and re-fetching a group already counted in the window is free. The API
663
+ returns HTTP 429 when a request would exceed the allowance. Check your
664
+ remaining quota with ``client.profile.downloads()``. Pro and demo
665
+ tiers are unlimited.
666
+
667
+ Example - Download sim_data for multiple sims:
668
+ >>> cached = client.simulation.list_cached(symbol="700.HK")
669
+ >>> sim_ids = [s["example_sim_id"] for s in cached["simulations"]]
670
+ >>>
671
+ >>> zip_bytes = client.simulation.get_bulk_data(sim_ids)
672
+ >>> with open("simulation_data.zip", "wb") as f:
673
+ ... f.write(zip_bytes)
674
+
675
+ Example - Download with L2 order book data:
676
+ >>> zip_bytes = client.simulation.get_bulk_data(
677
+ ... sim_ids=["sim_id_1", "sim_id_2"],
678
+ ... include_sim_data=True,
679
+ ... include_l2_by_second=True
680
+ ... )
681
+
682
+ Example - Extract and load into DataFrames:
683
+ >>> import zipfile
684
+ >>> import polars as pl
685
+ >>>
686
+ >>> zip_bytes = client.simulation.get_bulk_data(sim_ids)
687
+ >>> with zipfile.ZipFile(io.BytesIO(zip_bytes)) as zf:
688
+ ... for name in zf.namelist():
689
+ ... if name.endswith('.parquet'):
690
+ ... df = pl.read_parquet(io.BytesIO(zf.read(name)))
691
+ ... print(f"{name}: {df.shape}")
692
+ """
693
+ payload = {
694
+ "sim_ids": sim_ids,
695
+ "include_sim_data": include_sim_data,
696
+ "include_mid_price": include_mid_price,
697
+ "include_l2_by_second": include_l2_by_second,
698
+ }
699
+
700
+ url = f"{self._client.base_url}{RESULTS_PATH}/bulk"
701
+ response = self._client.session.post(url, json=payload)
702
+
703
+ if not response.ok:
704
+ try:
705
+ detail = response.json().get("detail", response.text)
706
+ except ValueError:
707
+ detail = response.text
708
+ raise PulseAPIError(response.status_code, detail)
709
+
710
+ return response.content