sharedbox 0.1.0__cp313-cp313-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sharedbox/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ from ._shareddict import SharedDict
2
+
3
+ __all__ = ["SharedDict"]
@@ -0,0 +1,32 @@
1
+ from typing import Any, Generic, TypeVar
2
+
3
+ K = TypeVar("K")
4
+ V = TypeVar("V")
5
+
6
+ class SharedDict(Generic[K, V]):
7
+ def __init__(
8
+ self,
9
+ name: str,
10
+ data: dict[str, Any] | None = None,
11
+ /,
12
+ *,
13
+ size: int = 128 * 1024 * 1024,
14
+ create: bool = False,
15
+ max_keys: int = 128,
16
+ ) -> None: ...
17
+ def close(self) -> None: ...
18
+ def unlink(self) -> None: ...
19
+ def is_closed(self) -> bool: ...
20
+ def __len__(self) -> int: ...
21
+ def __contains__(self, key: K) -> bool: ...
22
+ def __getitem__(self, key: K) -> V: ...
23
+ def __setitem__(self, key: K, value: V) -> None: ...
24
+ def __delitem__(self, key: K) -> None: ...
25
+ def __iter__(self) -> Any: ...
26
+ def get(self, key: K, default: V = None) -> V: ...
27
+ def keys(self) -> list[K]: ...
28
+ def keys_atomic(self) -> list[K]: ...
29
+ def items(self) -> list[tuple[K, V]]: ...
30
+ def recommend_sizing(
31
+ self, target_entries: int | None = None
32
+ ) -> dict[str, object]: ...
@@ -0,0 +1,350 @@
1
+ # distutils: language = c++
2
+ from libcpp.string cimport string
3
+ from libcpp.vector cimport vector
4
+ from libcpp cimport bool as cbool
5
+ import pickle
6
+ import struct
7
+ from cpython.bytes cimport PyBytes_FromStringAndSize
8
+ from cpython.unicode cimport PyUnicode_AsUTF8AndSize, PyUnicode_DecodeUTF8
9
+ from .utils import SegmentSizer, LockTuner
10
+ import numpy as np
11
+
12
+ cdef inline int calculate_target_entries(int current_entries, int min_entries) nogil:
13
+ """Pure C function to calculate target entries with 10x growth."""
14
+ cdef int growth_target = current_entries * 10
15
+ return growth_target if growth_target > min_entries else min_entries
16
+
17
+ cdef inline str _decode_to_str(const string& key_string):
18
+ """Encode a C++ string object to a Python string."""
19
+ return PyUnicode_DecodeUTF8(key_string.c_str(), key_string.size(), NULL)
20
+
21
+ cdef inline string _encode_to_string(str key):
22
+ """Encode a Python string to a C++ string using UTF-8 encoding."""
23
+ cdef const char* key_ptr
24
+ cdef Py_ssize_t key_len
25
+ key_ptr = PyUnicode_AsUTF8AndSize(key, &key_len)
26
+ return string(key_ptr, key_len)
27
+
28
+
29
+ cdef extern from "shared_dict.hpp" namespace "shared_memory":
30
+ cdef cppclass SharedMemoryDict:
31
+ SharedMemoryDict(const string& name, size_t size, cbool create, size_t max_keys) except +
32
+ void set(const string& k, const string& v) except +
33
+ cbool get(const string& k, string& out) const
34
+ cbool erase(const string& k)
35
+ cbool contains(const string& k) const
36
+ size_t size() const
37
+ vector[string] keys() const
38
+ void close() except +
39
+ void unlink() except +
40
+ cbool is_closed() const
41
+
42
+ cdef class SharedDict:
43
+ cdef SharedMemoryDict* c_map
44
+ cdef str name
45
+
46
+ def __cinit__(self, str name, dict data = None, /, *, int size = 128 * 1024 * 1024, cbool create = True, int max_keys = 128) -> None:
47
+ self.name = name
48
+ cdef string nm = _encode_to_string(name)
49
+ self.c_map = new SharedMemoryDict(nm, <size_t>size, <cbool>create, <size_t>max_keys)
50
+
51
+ # Initialize with provided data if specified
52
+ if data is not None:
53
+ self._initialize_data(data)
54
+
55
+ def __dealloc__(self) -> None:
56
+ """Destructor that releases the connection but does NOT remove shared memory.
57
+
58
+ This follows the multiprocessing.SharedMemory API pattern:
59
+ - Destructor only closes the connection (like close())
60
+ - Shared memory removal must be done explicitly via unlink()
61
+ - This prevents race conditions when multiple processes use the same segment
62
+ """
63
+ if self.c_map is not NULL:
64
+ # The C++ destructor will call close() if not already closed
65
+ del self.c_map
66
+ self.c_map = NULL
67
+
68
+ cpdef void close(self):
69
+ """Close access to shared memory without removing it.
70
+
71
+ After calling close(), this SharedDict instance becomes unusable
72
+ but other processes can still access the shared memory.
73
+ Similar to multiprocessing.SharedMemory.close().
74
+ """
75
+ if self.c_map is not NULL:
76
+ self.c_map.close()
77
+
78
+ cpdef void unlink(self):
79
+ """Remove the shared memory segment entirely.
80
+
81
+ This removes the shared memory from the system, making it
82
+ inaccessible to all processes. Similar to multiprocessing.SharedMemory.unlink().
83
+ Only call this from the process that created the segment.
84
+ """
85
+ if self.c_map is not NULL:
86
+ if not self.c_map.is_closed():
87
+ raise RuntimeError("Cannot unlink a SharedDict that is still open. Call close() first.")
88
+ self.c_map.unlink()
89
+
90
+ cpdef cbool is_closed(self):
91
+ """Check if this SharedDict connection has been closed.
92
+
93
+ Returns True if close() has been called or the object has been destructed.
94
+ A closed SharedDict cannot perform any operations.
95
+ """
96
+ if self.c_map is NULL:
97
+ return True
98
+ return self.c_map.is_closed()
99
+
100
+ cdef bytes _dumps_value(self, object obj):
101
+ """Value serialization with numpy array support."""
102
+ if isinstance(obj, np.ndarray):
103
+ return self._serialize_numpy_array(obj)
104
+ else:
105
+ # Fallback to pickle for other types
106
+ header = b'\x00' # Non-numpy marker
107
+ data = pickle.dumps(obj, protocol=pickle.HIGHEST_PROTOCOL)
108
+ return header + data
109
+
110
+ cdef object _loads_value(self, bytes b):
111
+ """Value deserialization with numpy array support."""
112
+ if len(b) == 0:
113
+ raise ValueError("Empty data cannot be deserialized")
114
+
115
+ cdef unsigned char marker = b[0]
116
+ if marker == 1: # Numpy array marker
117
+ return self._deserialize_numpy_array(b[1:])
118
+ else: # Pickle marker (0) or legacy data
119
+ if marker == 0:
120
+ return pickle.loads(b[1:]) # Skip marker
121
+ else:
122
+ return pickle.loads(b) # Legacy data without marker
123
+
124
+ cdef bytes _serialize_numpy_array(self, object arr):
125
+ """Efficiently serialize numpy array without pickle."""
126
+ cdef bytes dtype_str = str(arr.dtype).encode('utf-8')
127
+
128
+ # Get raw array data - ensure it's contiguous
129
+ cdef bytes array_data
130
+ if arr.flags.c_contiguous:
131
+ array_data = arr.tobytes()
132
+ else:
133
+ # Make contiguous copy
134
+ array_data = np.ascontiguousarray(arr).tobytes()
135
+
136
+ # Pack header: marker(1) + dtype_len(4) + ndim(4) + data_len(4)
137
+ cdef bytes header = struct.pack('<BIII',
138
+ 1, # Numpy marker
139
+ len(dtype_str),
140
+ arr.ndim,
141
+ len(array_data))
142
+
143
+ # Pack shape data
144
+ cdef bytes shape_data = b''
145
+ for dim in arr.shape:
146
+ shape_data += struct.pack('<Q', dim)
147
+
148
+ # Combine all parts
149
+ return header + dtype_str + shape_data + array_data
150
+
151
+ cdef object _deserialize_numpy_array(self, bytes data):
152
+ """Efficiently deserialize numpy array."""
153
+ cdef int offset = 0
154
+
155
+ # Unpack header: dtype_len(4) + ndim(4) + data_len(4)
156
+ dtype_len, ndim, data_len = struct.unpack('<III', data[offset:offset+12])
157
+ offset += 12
158
+
159
+ # Extract dtype string
160
+ dtype_str = data[offset:offset+dtype_len].decode('utf-8')
161
+ offset += dtype_len
162
+
163
+ # Extract shape
164
+ cdef list shape = []
165
+ for i in range(ndim):
166
+ dim = struct.unpack('<Q', data[offset:offset+8])[0]
167
+ shape.append(dim)
168
+ offset += 8
169
+
170
+ # Extract array data
171
+ array_data = data[offset:offset+data_len]
172
+
173
+ # Reconstruct numpy array
174
+ arr = np.frombuffer(array_data, dtype=dtype_str).reshape(tuple(shape))
175
+
176
+ # Make a copy to ensure proper memory ownership
177
+ return np.array(arr)
178
+
179
+ cdef void _initialize_data(self, dict data):
180
+ """Initialize SharedDict with provided data.
181
+
182
+ Args:
183
+ data: Dictionary with string keys and values to populate SharedDict.
184
+ Values must be serializable (built-in types, numpy arrays, etc.)
185
+
186
+ Raises:
187
+ TypeError: If data is not a dictionary or contains invalid keys/values
188
+ ValueError: If serialization fails for any value
189
+ """
190
+ if not isinstance(data, dict):
191
+ raise TypeError("Initialization data must be a dictionary")
192
+
193
+ cdef int initialized_count = 0
194
+ cdef object key_obj
195
+ cdef str key = ""
196
+ cdef object value
197
+
198
+ try:
199
+ for key_obj, value in data.items():
200
+ if not isinstance(key_obj, str):
201
+ raise TypeError(f"All keys must be strings, got {type(key_obj)} for key: {key_obj}")
202
+
203
+ key = <str>key_obj # Now we know it's safe to cast
204
+ # Use the existing __setitem__ method for consistency
205
+ self[key] = value
206
+ initialized_count += 1
207
+
208
+ except TypeError as e:
209
+ # Re-raise TypeError with original message for key type errors
210
+ if "All keys must be strings" in str(e):
211
+ raise e
212
+ else:
213
+ raise ValueError(
214
+ f"Failed to initialize SharedDict during key iteration: {e}"
215
+ ) from e
216
+ except Exception as e:
217
+ # If initialization fails partway through, provide helpful context
218
+ if key:
219
+ raise ValueError(
220
+ f"Failed to initialize SharedDict after {initialized_count} items. "
221
+ f"Error on key '{key}': {e}"
222
+ ) from e
223
+ else:
224
+ raise ValueError(
225
+ f"Failed to initialize SharedDict during iteration: {e}"
226
+ ) from e
227
+
228
+ def __len__(self) -> int:
229
+ return <int> self.c_map.size()
230
+
231
+ def __contains__(self, str key) -> bool:
232
+ cdef string ks = _encode_to_string(key)
233
+ return bool(self.c_map.contains(ks))
234
+
235
+ def __getitem__(self, str key) -> object:
236
+ cdef string ks = _encode_to_string(key)
237
+ cdef string out
238
+ if not self.c_map.get(ks, out):
239
+ raise KeyError(key)
240
+ # Convert std::string (raw bytes) to Python bytes then unpickle
241
+ cdef bytes b = <bytes>PyBytes_FromStringAndSize(out.c_str(), out.size())
242
+ return self._loads_value(b)
243
+
244
+ def get(self, str key, object default = None) -> object:
245
+ try:
246
+ return self[key]
247
+ except KeyError:
248
+ return default
249
+
250
+ def __setitem__(self, str key, object value) -> None:
251
+ cdef string ks = _encode_to_string(key)
252
+ cdef bytes vb = self._dumps_value(value)
253
+ cdef string vs = vb
254
+ self.c_map.set(ks, vs)
255
+
256
+ def __delitem__(self, str key) -> None:
257
+ cdef string ks = _encode_to_string(key)
258
+ if not self.c_map.erase(ks):
259
+ raise KeyError(key)
260
+
261
+ def __iter__(self):
262
+ cdef vector[string] ks = self.c_map.keys() # Use atomic full-lock version instead of striped
263
+ cdef string s
264
+ for i in range(ks.size()):
265
+ s = ks[i]
266
+ yield _decode_to_str(s)
267
+
268
+ def keys(self) -> list[str]:
269
+ return list(iter(self))
270
+
271
+ def keys_atomic(self) -> list[str]:
272
+ """Get keys with full atomic snapshot (locks all stripes at once)."""
273
+ cdef vector[string] ks = self.c_map.keys() # Full lock version
274
+ cdef string s
275
+ result: list[str] = []
276
+ for i in range(ks.size()):
277
+ s = ks[i]
278
+ result.append(_decode_to_str(s))
279
+ return result
280
+
281
+ def items(self) -> list[tuple[str, object]]:
282
+ return [(k, self[k]) for k in self]
283
+
284
+ def values(self) -> list[object]:
285
+ return [self[k] for k in self]
286
+
287
+ def get_stats(self) -> dict[str, object]:
288
+ """Get runtime statistics and diagnostic information."""
289
+ # Sample some keys to estimate sizes
290
+ cdef vector[string] sample_keys = self.c_map.keys()
291
+ cdef string key_bytes, value_bytes
292
+ sample_size = min(<int>sample_keys.size(), 100)
293
+
294
+ total_key_bytes = 0
295
+ total_value_bytes = 0
296
+
297
+ if sample_size > 0:
298
+ for i in range(sample_size):
299
+ key_bytes = sample_keys[i]
300
+ total_key_bytes += key_bytes.size()
301
+ if self.c_map.get(key_bytes, value_bytes):
302
+ total_value_bytes += value_bytes.size()
303
+
304
+ avg_key_bytes = total_key_bytes / sample_size if sample_size > 0 else 0
305
+ avg_value_bytes = total_value_bytes / sample_size if sample_size > 0 else 0
306
+
307
+ return {
308
+ 'total_entries': <int>self.c_map.size(),
309
+ 'sample_size': sample_size,
310
+ 'avg_key_utf8_bytes': avg_key_bytes,
311
+ 'avg_value_pickle_bytes': avg_value_bytes,
312
+ 'estimated_data_bytes': <int>self.c_map.size() * (avg_key_bytes + avg_value_bytes),
313
+ 'segment_name': self.name,
314
+ }
315
+
316
+ def recommend_sizing(self, target_entries: int | None = None) -> dict[str, object]:
317
+ """Get sizing recommendations based on current usage."""
318
+ stats = self.get_stats()
319
+ current_entries = stats['total_entries']
320
+
321
+ if target_entries is None:
322
+ target_entries = calculate_target_entries(current_entries, 10000)
323
+
324
+ if current_entries == 0:
325
+ return {
326
+ 'current_stats': stats,
327
+ 'target_entries': target_entries,
328
+ 'sizing_recommendation': None,
329
+ 'lock_recommendation': None,
330
+ 'message': 'No data in SharedMemoryDict yet - cannot provide recommendations'
331
+ }
332
+
333
+ # Size analysis
334
+ sizing = SegmentSizer.calculate_segment_size(
335
+ target_entries,
336
+ int(stats['avg_key_utf8_bytes']),
337
+ int(stats['avg_value_pickle_bytes'])
338
+ )
339
+
340
+ # Lock analysis
341
+ lock_rec = LockTuner.recommend_lock_count(target_entries)
342
+
343
+ # Return data instead of printing
344
+
345
+ return {
346
+ 'current_stats': stats,
347
+ 'target_entries': target_entries,
348
+ 'sizing_recommendation': sizing,
349
+ 'lock_recommendation': lock_rec
350
+ }
sharedbox/_version.py ADDED
@@ -0,0 +1,34 @@
1
+ # file generated by setuptools-scm
2
+ # don't change, don't track in version control
3
+
4
+ __all__ = [
5
+ "__version__",
6
+ "__version_tuple__",
7
+ "version",
8
+ "version_tuple",
9
+ "__commit_id__",
10
+ "commit_id",
11
+ ]
12
+
13
+ TYPE_CHECKING = False
14
+ if TYPE_CHECKING:
15
+ from typing import Tuple
16
+ from typing import Union
17
+
18
+ VERSION_TUPLE = Tuple[Union[int, str], ...]
19
+ COMMIT_ID = Union[str, None]
20
+ else:
21
+ VERSION_TUPLE = object
22
+ COMMIT_ID = object
23
+
24
+ version: str
25
+ __version__: str
26
+ __version_tuple__: VERSION_TUPLE
27
+ version_tuple: VERSION_TUPLE
28
+ commit_id: COMMIT_ID
29
+ __commit_id__: COMMIT_ID
30
+
31
+ __version__ = version = '0.1.0'
32
+ __version_tuple__ = version_tuple = (0, 1, 0)
33
+
34
+ __commit_id__ = commit_id = 'gec7f23c35'
Binary file
sharedbox/utils.py ADDED
@@ -0,0 +1,293 @@
1
+ import math
2
+ import pickle
3
+ from typing import Any
4
+
5
+
6
+ class SegmentSizer:
7
+ """Helper class for calculating optimal shared memory segment sizes."""
8
+
9
+ # Conservative estimates for overhead per entry
10
+ MAP_NODE_OVERHEAD = 128 # bytes per map entry for pointers, alignment, etc.
11
+ ALLOCATOR_OVERHEAD_RATIO = 0.05 # 5% of total segment for allocator metadata
12
+ MUTEX_SIZE = 96 # bytes per interprocess_mutex (conservative estimate)
13
+
14
+ @classmethod
15
+ def estimate_pickle_size(cls, obj: Any) -> int:
16
+ """Estimate the pickle size of an object."""
17
+ try:
18
+ return len(pickle.dumps(obj, protocol=pickle.HIGHEST_PROTOCOL))
19
+ except Exception:
20
+ # Fallback for unpicklable objects
21
+ return len(str(obj).encode("utf-8"))
22
+
23
+ @classmethod
24
+ def calculate_segment_size(
25
+ cls,
26
+ num_entries: int,
27
+ avg_key_size: int,
28
+ avg_value_size: int,
29
+ num_locks: int = 2048,
30
+ safety_margin: float = 1.5,
31
+ ) -> dict[str, int]:
32
+ """
33
+ Calculate recommended segment size based on workload characteristics.
34
+
35
+ Args:
36
+ num_entries: Expected number of dictionary entries
37
+ avg_key_size: Average size of pickled keys in bytes
38
+ avg_value_size: Average size of pickled values in bytes
39
+ num_locks: Number of lock stripes (default 2048)
40
+ safety_margin: Multiplier for safety/fragmentation (default 1.5x)
41
+
42
+ Returns:
43
+ dict with size breakdown and recommendation
44
+ """
45
+ # Calculate component sizes
46
+ data_size: int = num_entries * (avg_key_size + avg_value_size)
47
+ entry_overhead: int = num_entries * cls.MAP_NODE_OVERHEAD
48
+ locks_overhead: int = num_locks * cls.MUTEX_SIZE
49
+
50
+ # Base size before safety margin
51
+ base_size: int = data_size + entry_overhead + locks_overhead
52
+
53
+ # Add allocator overhead
54
+ allocator_overhead: int = int(base_size * cls.ALLOCATOR_OVERHEAD_RATIO)
55
+ base_size += allocator_overhead
56
+
57
+ # Apply safety margin for fragmentation and growth
58
+ recommended_size: int = int(base_size * safety_margin)
59
+
60
+ # Round up to next power of 2 for better allocation
61
+ recommended_size_pow2: int = 2 ** math.ceil(
62
+ math.log2(max(recommended_size, 1024 * 1024))
63
+ )
64
+
65
+ return {
66
+ "data_bytes": data_size,
67
+ "entry_overhead_bytes": entry_overhead,
68
+ "locks_overhead_bytes": locks_overhead,
69
+ "allocator_overhead_bytes": allocator_overhead,
70
+ "base_size_bytes": base_size,
71
+ "recommended_size_bytes": recommended_size,
72
+ "recommended_size_pow2_bytes": recommended_size_pow2,
73
+ "safety_margin": safety_margin,
74
+ }
75
+
76
+ @classmethod
77
+ def size_for_workload(cls, workload_type: str) -> dict[str, Any]:
78
+ """
79
+ Get pre-calculated size recommendations for common workload patterns.
80
+
81
+ Args:
82
+ workload_type: One of 'small', 'medium', 'large', 'xlarge'
83
+
84
+ Returns:
85
+ dict with workload parameters and size recommendations
86
+ """
87
+ workloads = {
88
+ "small": {
89
+ "description": "1K entries, ~100 bytes avg per key+value",
90
+ "num_entries": 1_000,
91
+ "avg_key_size": 50,
92
+ "avg_value_size": 50,
93
+ "num_locks": 1024,
94
+ },
95
+ "medium": {
96
+ "description": "100K entries, ~500 bytes avg per key+value",
97
+ "num_entries": 100_000,
98
+ "avg_key_size": 100,
99
+ "avg_value_size": 400,
100
+ "num_locks": 2048,
101
+ },
102
+ "large": {
103
+ "description": "1M entries, ~1KB avg per key+value",
104
+ "num_entries": 1_000_000,
105
+ "avg_key_size": 200,
106
+ "avg_value_size": 800,
107
+ "num_locks": 8192,
108
+ },
109
+ "xlarge": {
110
+ "description": "10M entries, ~2KB avg per key+value",
111
+ "num_entries": 10_000_000,
112
+ "avg_key_size": 400,
113
+ "avg_value_size": 1600,
114
+ "num_locks": 16384,
115
+ },
116
+ }
117
+
118
+ if workload_type not in workloads:
119
+ raise ValueError(
120
+ f"Unknown workload type '{workload_type}'. "
121
+ f"Available: {list(workloads.keys())}"
122
+ )
123
+
124
+ params = workloads[workload_type]
125
+ sizing = cls.calculate_segment_size(
126
+ params["num_entries"],
127
+ params["avg_key_size"],
128
+ params["avg_value_size"],
129
+ params["num_locks"],
130
+ )
131
+
132
+ return {**params, **sizing}
133
+
134
+ @classmethod
135
+ def analyze_sample_data(
136
+ cls, sample_keys: list[Any], sample_values: list[Any]
137
+ ) -> dict[str, Any]:
138
+ """
139
+ Analyze sample data to estimate sizing requirements.
140
+
141
+ Args:
142
+ sample_keys: Representative sample of keys
143
+ sample_values: Representative sample of values
144
+
145
+ Returns:
146
+ dict with analysis results and size recommendations
147
+ """
148
+ if len(sample_keys) != len(sample_values):
149
+ raise ValueError("sample_keys and sample_values must have same length")
150
+
151
+ if not sample_keys:
152
+ raise ValueError("Empty samples provided")
153
+
154
+ # Calculate pickle sizes
155
+ key_sizes = [cls.estimate_pickle_size(k) for k in sample_keys]
156
+ value_sizes = [cls.estimate_pickle_size(v) for v in sample_values]
157
+
158
+ # Statistical analysis
159
+ avg_key_size = sum(key_sizes) // len(key_sizes)
160
+ avg_value_size = sum(value_sizes) // len(value_sizes)
161
+ max_key_size = max(key_sizes)
162
+ max_value_size = max(value_sizes)
163
+
164
+ # Generate recommendations for different scales
165
+ recommendations = {}
166
+ for scale, multiplier in [
167
+ ("current", 1),
168
+ ("10x", 10),
169
+ ("100x", 100),
170
+ ("1000x", 1000),
171
+ ]:
172
+ num_entries = len(sample_keys) * multiplier
173
+ sizing = cls.calculate_segment_size(
174
+ num_entries, avg_key_size, avg_value_size
175
+ )
176
+ recommendations[scale] = sizing
177
+
178
+ return {
179
+ "sample_size": len(sample_keys),
180
+ "avg_key_size_bytes": avg_key_size,
181
+ "avg_value_size_bytes": avg_value_size,
182
+ "max_key_size_bytes": max_key_size,
183
+ "max_value_size_bytes": max_value_size,
184
+ "key_sizes": key_sizes,
185
+ "value_sizes": value_sizes,
186
+ "recommendations": recommendations,
187
+ }
188
+
189
+
190
+ class LockTuner:
191
+ """Helper class for lock count tuning and performance optimization."""
192
+
193
+ @classmethod
194
+ def recommend_lock_count(
195
+ cls,
196
+ num_entries: int,
197
+ write_concurrency: int = 4,
198
+ target_entries_per_lock: int = 500,
199
+ ) -> dict[str, Any]:
200
+ """
201
+ Recommend optimal lock count based on workload characteristics.
202
+
203
+ Args:
204
+ num_entries: Expected number of dictionary entries
205
+ write_concurrency: Expected number of concurrent writers
206
+ target_entries_per_lock: Target entries per lock stripe
207
+
208
+ Returns:
209
+ dict with recommendations and rationale
210
+ """
211
+ # Calculate lock count based on entries per lock target
212
+ locks_for_entries: int = max(64, num_entries // target_entries_per_lock)
213
+
214
+ # Ensure enough locks for write concurrency (at least 4x writers)
215
+ locks_for_concurrency: int = write_concurrency * 4
216
+
217
+ # Take the maximum and round to next power of 2
218
+ min_locks: int = max(locks_for_entries, locks_for_concurrency, 64)
219
+ recommended_locks: int = 2 ** math.ceil(math.log2(min_locks))
220
+
221
+ # Cap at reasonable maximum (too many locks waste memory)
222
+ max_reasonable = 65536
223
+ if recommended_locks > max_reasonable:
224
+ recommended_locks = max_reasonable
225
+
226
+ # Calculate actual entries per lock with recommendation
227
+ actual_entries_per_lock = (
228
+ num_entries / recommended_locks if recommended_locks else 0
229
+ )
230
+
231
+ # Memory cost of locks
232
+ lock_memory_kb = (recommended_locks * SegmentSizer.MUTEX_SIZE) // 1024
233
+
234
+ return {
235
+ "recommended_lock_count": recommended_locks,
236
+ "actual_entries_per_lock": actual_entries_per_lock,
237
+ "lock_memory_kb": lock_memory_kb,
238
+ "rationale": {
239
+ "target_entries_per_lock": target_entries_per_lock,
240
+ "locks_for_entries": locks_for_entries,
241
+ "locks_for_concurrency": locks_for_concurrency,
242
+ "write_concurrency": write_concurrency,
243
+ },
244
+ }
245
+
246
+ @classmethod
247
+ def performance_presets(cls) -> dict[str, dict[str, Any]]:
248
+ """
249
+ Get pre-configured lock count recommendations for different performance profiles.
250
+
251
+ Returns:
252
+ dict mapping preset names to lock configurations
253
+ """
254
+ return {
255
+ "memory_optimized": {
256
+ "description": "Minimal memory usage, lower concurrency",
257
+ "base_locks": 256,
258
+ "max_locks": 1024,
259
+ "entries_per_lock": 2000,
260
+ "suitable_for": "Memory-constrained environments, low write concurrency",
261
+ },
262
+ "balanced": {
263
+ "description": "Good balance of memory and performance",
264
+ "base_locks": 1024,
265
+ "max_locks": 8192,
266
+ "entries_per_lock": 500,
267
+ "suitable_for": "Most general-purpose applications",
268
+ },
269
+ "performance_optimized": {
270
+ "description": "Maximum concurrency, higher memory usage",
271
+ "base_locks": 4096,
272
+ "max_locks": 32768,
273
+ "entries_per_lock": 100,
274
+ "suitable_for": "High-concurrency, write-heavy workloads",
275
+ },
276
+ "extreme_performance": {
277
+ "description": "Ultra-high concurrency for specialized use cases",
278
+ "base_locks": 16384,
279
+ "max_locks": 65536,
280
+ "entries_per_lock": 50,
281
+ "suitable_for": "Specialized high-throughput applications",
282
+ },
283
+ }
284
+
285
+
286
+ def format_size(size_bytes: int) -> str:
287
+ """Format byte size in human-readable format."""
288
+ size_float: float = size_bytes
289
+ for unit in ["B", "KB", "MB", "GB"]:
290
+ if size_float < 1024:
291
+ return f"{size_float:.1f}{unit}"
292
+ size_float /= 1024
293
+ return f"{size_float:.1f}TB"
@@ -0,0 +1,155 @@
1
+ Metadata-Version: 2.4
2
+ Name: sharedbox
3
+ Version: 0.1.0
4
+ Summary: Python containers using shared memory.
5
+ Author-email: Jacopo Abramo <jacopo.abramo@gmail.com>
6
+ License-Expression: Apache-2.0
7
+ Classifier: Development Status :: 3 - Alpha
8
+ Classifier: Intended Audience :: Developers
9
+ Classifier: Programming Language :: Python :: 3.10
10
+ Classifier: Programming Language :: Python :: 3.11
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Programming Language :: C++
14
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
15
+ Requires-Python: >=3.10
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: numpy>=2.2.6
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=8.4.2; extra == "dev"
21
+ Requires-Dist: ruff>=0.13.1; extra == "dev"
22
+ Dynamic: license-file
23
+
24
+ # `sharedbox`
25
+
26
+ > [!WARNING]
27
+ > This project is a work in progress; be patient or feel free to contribute.
28
+
29
+ Python inter-process shared containers leveraging `boost::interprocess`.
30
+
31
+ ## Installation
32
+
33
+ It is reccomended to install `sharedbox` in a virtual environment; for example using `uv`:
34
+
35
+ ```sh
36
+ uv venv --python 3.10
37
+ .venv\Scripts\activate
38
+ uv pip install sharedbox
39
+ ```
40
+
41
+ ## Quick Start
42
+
43
+ ```python
44
+ import multiprocessing as mp
45
+ from sharedbox import SharedDict
46
+
47
+ # Use in child processes
48
+ def worker(segment_name):
49
+ d = SharedDict(segment_name, create=False) # Connect to existing
50
+ d["worker_data"] = "Hello from worker!"
51
+ d.close() # Close in child process
52
+
53
+ if __name__ == "__main__":
54
+ # Create a shared dictionary
55
+ d = SharedDict("my_segment", create=True, size=10*1024*1024)
56
+ d["hello"] = "world"
57
+ d["data"] = [1, 2, 3, 4, 5]
58
+
59
+ # Start worker
60
+ p = mp.Process(target=worker, args=("my_segment",))
61
+ p.start()
62
+ p.join()
63
+
64
+ print(d["worker_data"]) # "Hello from worker!"
65
+ d.close() # Close in the main process
66
+ d.unlink() # Unlink (free resources)
67
+ ```
68
+
69
+ ## Initialization with Data
70
+
71
+ You can initialize SharedDict with existing data for convenient setup:
72
+
73
+ ```python
74
+ import numpy as np
75
+ from sharedbox import SharedDict
76
+
77
+ # Initialize with mixed data types
78
+ config_data = {
79
+ "app_name": "MyApp",
80
+ "version": "1.0",
81
+ "max_users": 1000,
82
+ "features": ["auth", "logging"],
83
+ "model_weights": np.array([0.1, 0.3, 0.6])
84
+ }
85
+
86
+ # Create SharedDict with initial data
87
+ shared_config = SharedDict("config", config_data, create=True)
88
+
89
+ # Data is immediately available
90
+ print(shared_config["app_name"]) # "MyApp"
91
+ print(shared_config["model_weights"]) # numpy array
92
+
93
+ # when a child process doesn't need the memory anymore call "close"
94
+ shared_config.close()
95
+
96
+ # the main process is in charge of cleaning up; call "unlink" to do so,
97
+ # similarly to a regular python SharedMemory object;
98
+ # make sure that the main process calls "close" before as well
99
+ shared_config.unlink()
100
+ ```
101
+
102
+ ## Limitations
103
+
104
+ - Nested dictionaries are "currently" unsupported
105
+ - Project is quite unstable, might not provide great performance boost at this time
106
+ - macOS unsupported
107
+
108
+ ## Examples
109
+
110
+ The `examples/` folder contains some code examples on how to use the package.
111
+
112
+ ## Building locally
113
+ ### Requirements
114
+
115
+ - [`git`](https://git-scm.com/downloads)
116
+ - [`uv`](https://docs.astral.sh/uv/getting-started/installation/)
117
+ - Python >= 3.10
118
+ - [`vcpkg`](https://vcpkg.io/en/)
119
+ - `cython>=3.1`
120
+ - A C++17 compatible compiler
121
+
122
+ First [install and boostrap vcpkg](https://learn.microsoft.com/en-us/vcpkg/get_started/get-started-vscode?pivots=shell-cmd) somewhere in your system.
123
+
124
+ ```sh
125
+ # for Windows it is reccomended to install it in C:\
126
+ cd C:\
127
+
128
+ git clone https://github.com/microsoft/vcpkg.git
129
+ cd vcpkg && bootstrap-vcpkg.bat
130
+ ```
131
+
132
+ Don't forget to add it to your `PATH`.
133
+
134
+ Also, add an environment variable pointing to the root of `vcpkg`:
135
+
136
+ ```sh
137
+ set "VCPKG_ROOT=C:\path\to\vcpkg"
138
+ ```
139
+
140
+ Clone this repository, then install `boost::interprocess`:
141
+
142
+ ```
143
+ vcpkg install boost-interprocess
144
+ ```
145
+
146
+ Finally, through `uv`, install the package in a virtual environment:
147
+
148
+ ```sh
149
+ uv venv --python 3.10
150
+ uv pip install -e .[dev]
151
+ ```
152
+
153
+ ## License
154
+
155
+ Licensed under [Apache 2.0](./LICENSE)
@@ -0,0 +1,12 @@
1
+ sharedbox/__init__.py,sha256=eXnGQaG6CewxDJ7sGZjKbQ8zRXYzllcwDnx8HK1lBQY,65
2
+ sharedbox/_shareddict.cp313-win_amd64.pyd,sha256=n2sH_RPpyDWr9rmzMZJHrSIj96jHhe3-DpN77AD04TU,227328
3
+ sharedbox/_shareddict.pyi,sha256=bnSxe1V2H28o4OmKYKUmWM7aOLt3PfeyN10swaixTiY,1029
4
+ sharedbox/_shareddict.pyx,sha256=hrU1MqcozLBooxTR5gLoju-h2dvBVuYOaCcKv0sNQtA,13879
5
+ sharedbox/_version.py,sha256=7VAwr_Y4I30gkfUzxKxH2NbnLcgKYS2eJ7xUHD2f5t4,746
6
+ sharedbox/utils.cp313-win_amd64.pyd,sha256=5TeGVW-K8yPibdcuIBktD_own90f7JZV5G74kzhhVPQ,81408
7
+ sharedbox/utils.py,sha256=9sTzgKYMj4dj_Ks2Q1dbTCWDKxeD-aNAkDog6CI_aNc,10893
8
+ sharedbox-0.1.0.dist-info/licenses/LICENSE,sha256=lj5YNz2g66TL33fWs0aLU5qLgALPnOvnb0xVk6whWbA,11546
9
+ sharedbox-0.1.0.dist-info/METADATA,sha256=LCTMS_3Wooxzx0aXsxl1E3Y3tucZdu0kL5ZxANhHO64,4239
10
+ sharedbox-0.1.0.dist-info/WHEEL,sha256=qV0EIPljj1XC_vuSatRWjn02nZIz3N1t8jsZz7HBr2U,101
11
+ sharedbox-0.1.0.dist-info/top_level.txt,sha256=efvcA1LLG6nAoGlbTnKFlWWtIau3MXustUMsATPQMZE,10
12
+ sharedbox-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (80.9.0)
3
+ Root-Is-Purelib: false
4
+ Tag: cp313-cp313-win_amd64
5
+
@@ -0,0 +1,201 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [2025] [Jacopo Abramo]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
@@ -0,0 +1 @@
1
+ sharedbox