ivolatility-backtesting 2.147__tar.gz → 2.148__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (16) hide show
  1. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/PKG-INFO +1 -1
  2. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting/ivolatility_backtesting.py +102 -43
  3. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting.egg-info/PKG-INFO +1 -1
  4. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting.egg-info/SOURCES.txt +2 -1
  5. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/pyproject.toml +1 -1
  6. ivolatility_backtesting-2.148/tests/test_2148_loader_memory.py +223 -0
  7. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/README.md +0 -0
  8. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting/__init__.py +0 -0
  9. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting.egg-info/dependency_links.txt +0 -0
  10. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting.egg-info/requires.txt +0 -0
  11. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/ivolatility_backtesting.egg-info/top_level.txt +0 -0
  12. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/setup.cfg +0 -0
  13. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/tests/test_2142_fixes.py +0 -0
  14. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/tests/test_2144_duckdb_dedup.py +0 -0
  15. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/tests/test_2146_vix_vro.py +0 -0
  16. {ivolatility_backtesting-2.147 → ivolatility_backtesting-2.148}/tests/test_2147_cents_multiplier.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ivolatility_backtesting
3
- Version: 2.147
3
+ Version: 2.148
4
4
  Summary: A universal backtesting framework for financial strategies using the IVolatility API.
5
5
  Author-email: IVolatility <support@ivolatility.com>
6
6
  Project-URL: Homepage, https://ivolatility.com
@@ -518,9 +518,8 @@ from typing import Callable, Dict, List, Optional, Tuple, Union, Any
518
518
  # malloc_trim→250MB). malloc_trim(0) forces glibc to return free pages. Resolved
519
519
  # lazily once; a no-op on musl/macOS (returns None / raises → swallowed).
520
520
  _libc_malloc_trim = None
521
- def _release_freed_memory():
522
- """gc.collect() + glibc malloc_trim(0) — actually shrinks RSS, unlike gc alone."""
523
- gc.collect()
521
+ def _trim_heap():
522
+ """glibc malloc_trim(0) without a gc pass — cheap enough to call after every chunk."""
524
523
  global _libc_malloc_trim
525
524
  if _libc_malloc_trim is None:
526
525
  try:
@@ -534,6 +533,67 @@ def _release_freed_memory():
534
533
  except Exception:
535
534
  pass
536
535
 
536
+
537
+ def _release_freed_memory():
538
+ """gc.collect() + glibc malloc_trim(0) — actually shrinks RSS, unlike gc alone."""
539
+ gc.collect()
540
+ _trim_heap()
541
+
542
+
543
+ import threading
544
+
545
+ # Admission control by cgroup free memory; the first download always passes so a load never stalls.
546
+ def _env_int(name, default):
547
+ try:
548
+ return int(os.environ.get(name, default))
549
+ except ValueError:
550
+ return default
551
+
552
+
553
+ _API_MEM_EST_MB = _env_int('IVOL_API_MEM_EST_MB', 150)
554
+ _API_MEM_SAFETY_MB = _env_int('IVOL_API_MEM_SAFETY_MB', 300)
555
+ _api_mem_lock = threading.Lock()
556
+ _api_mem_inflight = 0
557
+
558
+
559
+ def _cgroup_free_mb():
560
+ """Container free memory in MB (limit minus usage excluding inactive file cache); None without a cgroup v2 limit."""
561
+ try:
562
+ with open('/sys/fs/cgroup/memory.max') as f:
563
+ raw = f.read().strip()
564
+ if raw == 'max':
565
+ return None
566
+ with open('/sys/fs/cgroup/memory.current') as f:
567
+ current = int(f.read())
568
+ inactive = 0
569
+ with open('/sys/fs/cgroup/memory.stat') as f:
570
+ for line in f:
571
+ if line.startswith('inactive_file '):
572
+ inactive = int(line.split()[1])
573
+ break
574
+ return (int(raw) - (current - inactive)) >> 20
575
+ except (OSError, ValueError):
576
+ return None
577
+
578
+
579
+ def _api_mem_acquire():
580
+ """Block until the container can fit one more in-flight download."""
581
+ global _api_mem_inflight
582
+ while True:
583
+ with _api_mem_lock:
584
+ free_mb = _cgroup_free_mb()
585
+ if (free_mb is None or _api_mem_inflight == 0
586
+ or free_mb - _api_mem_inflight * _API_MEM_EST_MB >= _API_MEM_EST_MB + _API_MEM_SAFETY_MB):
587
+ _api_mem_inflight += 1
588
+ return
589
+ time.sleep(0.5)
590
+
591
+
592
+ def _api_mem_release():
593
+ global _api_mem_inflight
594
+ with _api_mem_lock:
595
+ _api_mem_inflight = max(0, _api_mem_inflight - 1)
596
+
537
597
  # Trim once at import: if the lib is re-imported into a kernel that already ran
538
598
  # heavy work, hand back any glibc arenas before this session allocates.
539
599
  _release_freed_memory()
@@ -7415,8 +7475,7 @@ class APIHelper:
7415
7475
  if response.empty:
7416
7476
  return None
7417
7477
 
7418
- records = response.to_dict('records')
7419
- return {'data': records, 'status': 'success'}
7478
+ return {'data': response, 'status': 'success', '_is_df': True}
7420
7479
 
7421
7480
  if debug:
7422
7481
  print(f"[APIHelper] Unexpected type: {type(response)}")
@@ -20193,23 +20252,28 @@ def _api_call_logged(endpoint, cache_config, skip_parquet_cache=False, debuginfo
20193
20252
 
20194
20253
  # Make API call with debug level
20195
20254
  # debuginfo: 0=silent, 1=basic, 2=detailed (URLs), 3=verbose timing
20196
- response = api_call(
20197
- endpoint,
20198
- cache_config,
20199
- debug=debuginfo, # Pass integer directly
20200
- skip_parquet_cache=skip_parquet_cache,
20201
- _chunk_info=_chunk_info,
20202
- **normalized_params
20203
- )
20204
-
20255
+ _api_mem_acquire()
20256
+ try:
20257
+ response = api_call(
20258
+ endpoint,
20259
+ cache_config,
20260
+ debug=debuginfo, # Pass integer directly
20261
+ skip_parquet_cache=skip_parquet_cache,
20262
+ _chunk_info=_chunk_info,
20263
+ **normalized_params
20264
+ )
20265
+ finally:
20266
+ _api_mem_release()
20267
+
20205
20268
  # Parse response
20206
20269
  if response is None:
20207
20270
  return None
20208
-
20271
+
20209
20272
  if isinstance(response, pd.DataFrame):
20210
20273
  return response if not response.empty else None
20211
20274
  elif isinstance(response, dict) and 'data' in response:
20212
- df = pd.DataFrame(response['data'])
20275
+ data = response['data']
20276
+ df = data if isinstance(data, pd.DataFrame) else pd.DataFrame(data)
20213
20277
  return df if not df.empty else None
20214
20278
 
20215
20279
  return None
@@ -20308,16 +20372,14 @@ def _load_options_to_duckdb(config, cache_config, symbol, start_date, end_date):
20308
20372
  print(f" ⚡ PARALLEL mode: {total_requests} requests with {max_workers} workers")
20309
20373
 
20310
20374
  all_data = []
20375
+ # Legacy (pandas-indicator) mode still returns the sample if reading DuckDB back fails.
20376
+ _keep_sample = not config.get('use_duckdb_indicators', True)
20311
20377
  total_rows = 0
20312
20378
  api_debuginfo = debuginfo
20313
20379
  # Resolve snapshot-aware endpoint ONCE for both fetch and save
20314
20380
  _save_ep = _get_options_endpoints(_get_options_snapshot_mode(config))['filtered']
20315
20381
 
20316
- # MEMFIX(backpressure): воркеры качают быстрее, чем главный поток успевает
20317
- # писать чанк в DuckDB (запись ~3с), поэтому готовые, но ещё не обработанные
20318
- # DataFrame копились в памяти. Семафор держит очередь готовых результатов
20319
- # ограниченной: воркер не отдаёт результат, пока главный поток не разгрёб
20320
- # предыдущие. Это ограничивает пик числом воркеров, а не длиной периода.
20382
+ # Caps downloaded-but-not-yet-written chunks; each loop must release after writing a non-empty chunk.
20321
20383
  import threading as _mf_thr
20322
20384
  _mf_inflight = _mf_thr.Semaphore(max(2, max_workers))
20323
20385
 
@@ -20438,29 +20500,27 @@ def _load_options_to_duckdb(config, cache_config, symbol, start_date, end_date):
20438
20500
  _safe_print(f" ⏱️ [SLOW DUCKDB WRITE] {_db_elapsed:.1f}s for {len(df)} rows (chunk {completed})")
20439
20501
  rows_saved_total += rows_saved
20440
20502
  total_rows += len(df)
20441
-
20442
- if len(all_data) < 5:
20503
+
20504
+ # Sample is only a fallback for a failed DuckDB write.
20505
+ if rows_saved_total > 0 and not _keep_sample:
20506
+ all_data.clear()
20507
+ elif len(all_data) < 5:
20443
20508
  all_data.append(df)
20444
-
20509
+
20445
20510
  if df is not None and not getattr(df, 'empty', True):
20446
20511
  try:
20447
20512
  _mf_inflight.release()
20448
20513
  except Exception:
20449
20514
  pass
20450
- # MEMFIX: release the chunk once it is persisted — the Future keeps a
20451
- # reference to its result until the executor block exits, so without this
20452
- # the whole fetch accumulates in RAM regardless of DuckDB writes.
20515
+ # The Future holds its result until the executor exits; drop it once persisted.
20453
20516
  try:
20454
20517
  future._result = None
20455
20518
  except Exception:
20456
20519
  pass
20457
20520
  df = None
20521
+ _trim_heap()
20458
20522
  if completed % 25 == 0:
20459
- # MEMFIX(page-cache): страницы записанного .duckdb оседают в file-кэше
20460
- # cgroup и подтягивают memory.current к лимиту. Кэш вытесняемый, но
20461
- # держит счётчик у потолка и сокращает буфер до OOM. Сбрасываем его:
20462
- # сначала fdatasync (грязные страницы иначе не освободить), затем
20463
- # POSIX_FADV_DONTNEED на файл БД и его WAL.
20523
+ # Written .duckdb pages count toward the cgroup limit: fdatasync, then drop them from page cache.
20464
20524
  try:
20465
20525
  import os as _os4
20466
20526
  _dbp = None
@@ -20487,13 +20547,7 @@ def _load_options_to_duckdb(config, cache_config, symbol, start_date, end_date):
20487
20547
  pass
20488
20548
  except Exception:
20489
20549
  pass
20490
- import gc as _gc
20491
- _gc.collect()
20492
- try:
20493
- import ctypes as _ct
20494
- _ct.CDLL("libc.so.6").malloc_trim(0)
20495
- except Exception:
20496
- pass
20550
+ _release_freed_memory()
20497
20551
 
20498
20552
  _watchdog_completed[0] = completed
20499
20553
  if completed % 10 == 0 or completed == total_requests:
@@ -20534,11 +20588,16 @@ def _load_options_to_duckdb(config, cache_config, symbol, start_date, end_date):
20534
20588
  )
20535
20589
  rows_saved_total += rows_saved
20536
20590
  total_rows += len(df)
20537
-
20538
- # Keep small sample for return
20539
- if len(all_data) < 5:
20591
+
20592
+ # Sample is only a fallback for a failed DuckDB write.
20593
+ if rows_saved_total > 0 and not _keep_sample:
20594
+ all_data.clear()
20595
+ elif len(all_data) < 5:
20540
20596
  all_data.append(df)
20541
-
20597
+ _mf_inflight.release()
20598
+ df = None
20599
+ _trim_heap()
20600
+
20542
20601
  # Progress summary every 10 requests
20543
20602
  if (i + 1) % 10 == 0 or i == len(all_requests) - 1:
20544
20603
  print(f" 📦 Request {i + 1}/{total_requests}: {total_rows:,} rows so far")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ivolatility_backtesting
3
- Version: 2.147
3
+ Version: 2.148
4
4
  Summary: A universal backtesting framework for financial strategies using the IVolatility API.
5
5
  Author-email: IVolatility <support@ivolatility.com>
6
6
  Project-URL: Homepage, https://ivolatility.com
@@ -11,4 +11,5 @@ ivolatility_backtesting.egg-info/top_level.txt
11
11
  tests/test_2142_fixes.py
12
12
  tests/test_2144_duckdb_dedup.py
13
13
  tests/test_2146_vix_vro.py
14
- tests/test_2147_cents_multiplier.py
14
+ tests/test_2147_cents_multiplier.py
15
+ tests/test_2148_loader_memory.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "ivolatility_backtesting"
7
- version = "2.147"
7
+ version = "2.148"
8
8
  description = "A universal backtesting framework for financial strategies using the IVolatility API."
9
9
  readme = "README.md"
10
10
  authors = [
@@ -0,0 +1,223 @@
1
+ # Loader memory: DataFrame passthrough, download admission by cgroup free memory, heap trim.
2
+ # Run: python3 tests/test_2148_loader_memory.py (or pytest)
3
+ import os
4
+ import sys
5
+ import threading
6
+ import time
7
+ import traceback
8
+
9
+ import matplotlib
10
+ matplotlib.use('Agg')
11
+
12
+ _REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
13
+ sys.path.insert(0, os.path.join(_REPO, 'ivolatility_backtesting'))
14
+ import ivolatility_backtesting as lib # noqa: E402
15
+
16
+ import pandas as pd # noqa: E402
17
+
18
+
19
+ def _chain(n=3):
20
+ return pd.DataFrame({
21
+ 'date': ['2024-01-02'] * n,
22
+ 'expiration': ['2024-01-19'] * n,
23
+ 'strike': [100.0 + i for i in range(n)],
24
+ 'Call/Put': ['C'] * n,
25
+ 'bid': [1.0 + i for i in range(n)],
26
+ 'ask': [1.1 + i for i in range(n)],
27
+ })
28
+
29
+
30
+ def test_normalize_keeps_same_dataframe():
31
+ df = _chain()
32
+ out = lib.APIHelper.normalize_response(df)
33
+ assert out['status'] == 'success'
34
+ assert out['_is_df'] is True
35
+ assert out['data'] is df
36
+
37
+
38
+ def test_normalize_empty_dataframe_is_none():
39
+ assert lib.APIHelper.normalize_response(pd.DataFrame()) is None
40
+
41
+
42
+ def test_normalize_dict_passthrough_unchanged():
43
+ resp = {'data': [{'a': 1}], 'status': 'success'}
44
+ assert lib.APIHelper.normalize_response(resp) is resp
45
+
46
+
47
+ def test_api_call_logged_returns_frame_without_copy():
48
+ df = _chain()
49
+ orig = lib.api_call
50
+ try:
51
+ lib.api_call = lambda *a, **k: {'data': df, 'status': 'success', '_is_df': True}
52
+ got = lib._api_call_logged('/equities/eod/options-rawiv', {}, symbol='SPY',
53
+ from_='2024-01-02', to='2024-01-02')
54
+ assert got is df
55
+ lib.api_call = lambda *a, **k: {'data': df.to_dict('records'), 'status': 'success'}
56
+ got = lib._api_call_logged('/equities/eod/options-rawiv', {}, symbol='SPY',
57
+ from_='2024-01-02', to='2024-01-02')
58
+ pd.testing.assert_frame_equal(got, df)
59
+ lib.api_call = lambda *a, **k: {'data': pd.DataFrame(), 'status': 'success', '_is_df': True}
60
+ assert lib._api_call_logged('/equities/eod/options-rawiv', {}, symbol='SPY') is None
61
+ finally:
62
+ lib.api_call = orig
63
+ assert lib._api_mem_inflight == 0
64
+
65
+
66
+ def test_api_call_logged_releases_slot_on_error():
67
+ orig = lib.api_call
68
+
69
+ def boom(*a, **k):
70
+ raise RuntimeError('network')
71
+ try:
72
+ lib.api_call = boom
73
+ try:
74
+ lib._api_call_logged('/equities/eod/options-rawiv', {}, symbol='SPY')
75
+ except RuntimeError:
76
+ pass
77
+ finally:
78
+ lib.api_call = orig
79
+ assert lib._api_mem_inflight == 0
80
+
81
+
82
+ def test_gate_first_download_always_passes():
83
+ orig = lib._cgroup_free_mb
84
+ try:
85
+ lib._cgroup_free_mb = lambda: 0
86
+ lib._api_mem_acquire()
87
+ assert lib._api_mem_inflight == 1
88
+ lib._api_mem_release()
89
+ assert lib._api_mem_inflight == 0
90
+ finally:
91
+ lib._cgroup_free_mb = orig
92
+
93
+
94
+ def test_gate_waits_for_memory_then_admits():
95
+ free = {'mb': 400}
96
+ orig = lib._cgroup_free_mb
97
+ try:
98
+ lib._cgroup_free_mb = lambda: free['mb']
99
+ lib._api_mem_acquire()
100
+ t = threading.Thread(target=lib._api_mem_acquire, daemon=True)
101
+ t.start()
102
+ t.join(1.2)
103
+ assert t.is_alive(), 'second download must wait: 400 - 150 < 150 + 300'
104
+ assert lib._api_mem_inflight == 1
105
+ free['mb'] = 2000
106
+ t.join(2.0)
107
+ assert not t.is_alive()
108
+ assert lib._api_mem_inflight == 2
109
+ finally:
110
+ lib._cgroup_free_mb = orig
111
+ lib._api_mem_release()
112
+ lib._api_mem_release()
113
+ assert lib._api_mem_inflight == 0
114
+
115
+
116
+ def test_gate_reserves_per_inflight_download():
117
+ orig = lib._cgroup_free_mb
118
+ try:
119
+ # 900 free: 1st passes (inflight 0), 2nd 900-150=750 >= 450, 3rd 900-300=600 >= 450,
120
+ # 4th 900-450=450 >= 450, 5th 900-600=300 < 450 → waits.
121
+ lib._cgroup_free_mb = lambda: 900
122
+ for _ in range(4):
123
+ lib._api_mem_acquire()
124
+ assert lib._api_mem_inflight == 4
125
+ t = threading.Thread(target=lib._api_mem_acquire, daemon=True)
126
+ t.start()
127
+ t.join(1.2)
128
+ assert t.is_alive()
129
+ lib._api_mem_release()
130
+ t.join(2.0)
131
+ assert not t.is_alive()
132
+ finally:
133
+ lib._cgroup_free_mb = orig
134
+ for _ in range(4):
135
+ lib._api_mem_release()
136
+ assert lib._api_mem_inflight == 0
137
+
138
+
139
+ def test_gate_is_noop_without_cgroup_limit():
140
+ orig = lib._cgroup_free_mb
141
+ try:
142
+ lib._cgroup_free_mb = lambda: None
143
+ start = time.time()
144
+ for _ in range(8):
145
+ lib._api_mem_acquire()
146
+ assert time.time() - start < 0.5
147
+ finally:
148
+ lib._cgroup_free_mb = orig
149
+ for _ in range(8):
150
+ lib._api_mem_release()
151
+ assert lib._api_mem_inflight == 0
152
+
153
+
154
+ def test_release_never_goes_negative():
155
+ lib._api_mem_release()
156
+ assert lib._api_mem_inflight == 0
157
+
158
+
159
+ def test_cgroup_free_mb_type():
160
+ v = lib._cgroup_free_mb()
161
+ assert v is None or isinstance(v, int)
162
+
163
+
164
+ def _run_loader(config, calls, timeout=20):
165
+ saved = []
166
+ orig_call, orig_save = lib._api_call_logged, lib._save_to_duckdb_storage
167
+
168
+ def fake_call(endpoint, cache_config, **params):
169
+ calls.append(params['cp'])
170
+ return _chain(4)
171
+
172
+ def fake_save(df, endpoint, cache_config, debug=False):
173
+ saved.append(len(df))
174
+ return len(df)
175
+ box = {}
176
+ try:
177
+ lib._api_call_logged, lib._save_to_duckdb_storage = fake_call, fake_save
178
+ t = threading.Thread(target=lambda: box.setdefault('r', lib._load_options_to_duckdb(
179
+ config, {}, 'SPY', '2024-01-01', '2024-01-30')), daemon=True)
180
+ t.start()
181
+ t.join(timeout)
182
+ return (not t.is_alive()), saved
183
+ finally:
184
+ lib._api_call_logged, lib._save_to_duckdb_storage = orig_call, orig_save
185
+
186
+
187
+ def test_sequential_loader_does_not_stall():
188
+ # 6 chunks x C/P = 12 requests > semaphore size max(2, workers=2)
189
+ cfg = {'strategy_type': 'STRADDLE', 'chunk_days_options': 5, 'parallel_mode': False,
190
+ 'parallel_max_workers': 2, 'debuginfo': 0}
191
+ calls = []
192
+ finished, saved = _run_loader(cfg, calls)
193
+ assert finished, f'sequential load stalled after {len(calls)} requests'
194
+ assert len(calls) == 12 and saved == [4] * 12
195
+
196
+
197
+ def test_parallel_loader_saves_every_chunk():
198
+ cfg = {'strategy_type': 'STRADDLE', 'chunk_days_options': 5, 'parallel_mode': True,
199
+ 'parallel_max_workers': 3, 'debuginfo': 0}
200
+ calls = []
201
+ finished, saved = _run_loader(cfg, calls)
202
+ assert finished
203
+ assert len(calls) == 12 and saved == [4] * 12
204
+
205
+
206
+ def test_trim_heap_safe_everywhere():
207
+ lib._trim_heap()
208
+ lib._trim_heap()
209
+ lib._release_freed_memory()
210
+
211
+
212
+ if __name__ == '__main__':
213
+ failed = 0
214
+ for name, fn in sorted((n, f) for n, f in globals().items() if n.startswith('test_') and callable(f)):
215
+ try:
216
+ fn()
217
+ print(f'PASS {name}')
218
+ except Exception:
219
+ failed += 1
220
+ print(f'FAIL {name}')
221
+ traceback.print_exc()
222
+ print(f'{failed} failed')
223
+ sys.exit(1 if failed else 0)