PyCyBase 0.1.8.post4__cp313-cp313-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cbase/__infra__.pxd +6 -0
- cbase/__init__.pxd +6 -0
- cbase/__init__.py +28 -0
- cbase/allocator_protocol/__infra__.pxd +58 -0
- cbase/allocator_protocol/__init__.pxd +58 -0
- cbase/allocator_protocol/__init__.py +45 -0
- cbase/allocator_protocol/c_allocator_protocol.c +31818 -0
- cbase/allocator_protocol/c_allocator_protocol.cp313-win_amd64.pyd +0 -0
- cbase/allocator_protocol/c_allocator_protocol.h +352 -0
- cbase/allocator_protocol/c_allocator_protocol.pxd +64 -0
- cbase/allocator_protocol/c_allocator_protocol.pyi +144 -0
- cbase/allocator_protocol/c_allocator_protocol.pyx +209 -0
- cbase/allocator_protocol/c_heap_allocator.c +38144 -0
- cbase/allocator_protocol/c_heap_allocator.cp313-win_amd64.pyd +0 -0
- cbase/allocator_protocol/c_heap_allocator.h +444 -0
- cbase/allocator_protocol/c_heap_allocator.pxd +97 -0
- cbase/allocator_protocol/c_heap_allocator.pyi +161 -0
- cbase/allocator_protocol/c_heap_allocator.pyx +300 -0
- cbase/allocator_protocol/c_nt_shm_allocator.c +33399 -0
- cbase/allocator_protocol/c_nt_shm_allocator.cp313-win_amd64.pyd +0 -0
- cbase/allocator_protocol/c_nt_shm_allocator.h +996 -0
- cbase/allocator_protocol/c_nt_shm_allocator.pxd +102 -0
- cbase/allocator_protocol/c_nt_shm_allocator.pyx +223 -0
- cbase/allocator_protocol/c_shm_allocator.h +1168 -0
- cbase/allocator_protocol/c_shm_allocator.pxd +142 -0
- cbase/allocator_protocol/c_shm_allocator.pyi +299 -0
- cbase/allocator_protocol/c_shm_allocator.pyx +442 -0
- cbase/allocator_protocol/c_shm_comp.c +4643 -0
- cbase/allocator_protocol/c_shm_comp.cp313-win_amd64.pyd +0 -0
- cbase/allocator_protocol/c_shm_comp.h +49 -0
- cbase/allocator_protocol/c_shm_comp.pxd +38 -0
- cbase/allocator_protocol/c_shm_comp.pyx +1 -0
- cbase/backports/__infra__.pxd +17 -0
- cbase/backports/__init__.pxd +17 -0
- cbase/backports/__init__.py +9 -0
- cbase/backports/pydict.h +55 -0
- cbase/backports/pydict.pxd +5 -0
- cbase/backports/pylong.c +9946 -0
- cbase/backports/pylong.cp313-win_amd64.pyd +0 -0
- cbase/backports/pylong.h +79 -0
- cbase/backports/pylong.pxd +23 -0
- cbase/backports/pylong.pyx +61 -0
- cbase/bytemap/__infra__.pxd +42 -0
- cbase/bytemap/__init__.pxd +42 -0
- cbase/bytemap/__init__.py +21 -0
- cbase/bytemap/c_bytemap.c +52076 -0
- cbase/bytemap/c_bytemap.cp313-win_amd64.pyd +0 -0
- cbase/bytemap/c_bytemap.h +867 -0
- cbase/bytemap/c_bytemap.pxd +370 -0
- cbase/bytemap/c_bytemap.pyx +1758 -0
- cbase/bytemap/xxh3.h +55 -0
- cbase/bytemap/xxhash.c +42 -0
- cbase/bytemap/xxhash.h +7487 -0
- cbase/env.c +11109 -0
- cbase/env.cp313-win_amd64.pyd +0 -0
- cbase/env.pxd +8 -0
- cbase/env.pyi +48 -0
- cbase/env.pyx +46 -0
- cbase/includes/cbase/allocator_protocol/c_allocator_protocol.c +31818 -0
- cbase/includes/cbase/allocator_protocol/c_allocator_protocol.h +352 -0
- cbase/includes/cbase/allocator_protocol/c_heap_allocator.c +38144 -0
- cbase/includes/cbase/allocator_protocol/c_heap_allocator.h +444 -0
- cbase/includes/cbase/allocator_protocol/c_nt_shm_allocator.c +33399 -0
- cbase/includes/cbase/allocator_protocol/c_nt_shm_allocator.h +996 -0
- cbase/includes/cbase/allocator_protocol/c_shm_allocator.h +1168 -0
- cbase/includes/cbase/allocator_protocol/c_shm_comp.c +4643 -0
- cbase/includes/cbase/allocator_protocol/c_shm_comp.h +49 -0
- cbase/includes/cbase/backports/pydict.h +55 -0
- cbase/includes/cbase/backports/pylong.c +9946 -0
- cbase/includes/cbase/backports/pylong.h +79 -0
- cbase/includes/cbase/bytemap/c_bytemap.c +52076 -0
- cbase/includes/cbase/bytemap/c_bytemap.h +867 -0
- cbase/includes/cbase/bytemap/xxh3.h +55 -0
- cbase/includes/cbase/bytemap/xxhash.c +42 -0
- cbase/includes/cbase/bytemap/xxhash.h +7487 -0
- cbase/includes/cbase/env.c +11109 -0
- cbase/includes/cbase/int128.h +81 -0
- cbase/includes/cbase/intern_string/c_intern_string.c +28081 -0
- cbase/includes/cbase/intern_string/c_intern_string.h +614 -0
- cbase/includes/cbase/nt/pthread_nt_compat.h +102 -0
- cbase/includes/cbase/nt/regex_nt_compat.h +133 -0
- cbase/includes/cbase/nt/time_sched_nt_compat.h +32 -0
- cbase/int128.h +81 -0
- cbase/int128.pxd +15 -0
- cbase/intern_string/BENCHMARK.md +157 -0
- cbase/intern_string/__infra__.pxd +11 -0
- cbase/intern_string/__init__.pxd +11 -0
- cbase/intern_string/__init__.py +19 -0
- cbase/intern_string/c_intern_string.c +28081 -0
- cbase/intern_string/c_intern_string.cp313-win_amd64.pyd +0 -0
- cbase/intern_string/c_intern_string.h +614 -0
- cbase/intern_string/c_intern_string.pxd +118 -0
- cbase/intern_string/c_intern_string.pyi +376 -0
- cbase/intern_string/c_intern_string.pyx +765 -0
- cbase/nt/pthread_nt_compat.h +102 -0
- cbase/nt/regex_nt_compat.h +133 -0
- cbase/nt/time_sched_nt_compat.h +32 -0
- pycybase-0.1.8.post4.dist-info/METADATA +180 -0
- pycybase-0.1.8.post4.dist-info/RECORD +101 -0
- pycybase-0.1.8.post4.dist-info/WHEEL +5 -0
- pycybase-0.1.8.post4.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
#ifndef REGEX_NT_COMPAT_H
|
|
2
|
+
#define REGEX_NT_COMPAT_H
|
|
3
|
+
|
|
4
|
+
#ifndef _WIN32
|
|
5
|
+
#error "regex_nt_compat.h is intended for Windows builds only"
|
|
6
|
+
#endif
|
|
7
|
+
|
|
8
|
+
#include <errno.h>
|
|
9
|
+
#include <stddef.h>
|
|
10
|
+
|
|
11
|
+
#if defined(__has_include)
|
|
12
|
+
#if __has_include(<Python.h>)
|
|
13
|
+
#include <Python.h>
|
|
14
|
+
#define REGEX_NT_COMPAT_HAS_PYTHON 1
|
|
15
|
+
#else
|
|
16
|
+
#define REGEX_NT_COMPAT_HAS_PYTHON 0
|
|
17
|
+
#endif
|
|
18
|
+
#else
|
|
19
|
+
#include <Python.h>
|
|
20
|
+
#define REGEX_NT_COMPAT_HAS_PYTHON 1
|
|
21
|
+
#endif
|
|
22
|
+
|
|
23
|
+
#ifndef REG_EXTENDED
|
|
24
|
+
#define REG_EXTENDED 1
|
|
25
|
+
#endif
|
|
26
|
+
|
|
27
|
+
#ifndef REG_NOMATCH
|
|
28
|
+
#define REG_NOMATCH 1
|
|
29
|
+
#endif
|
|
30
|
+
|
|
31
|
+
#ifndef REG_BADPAT
|
|
32
|
+
#define REG_BADPAT 2
|
|
33
|
+
#endif
|
|
34
|
+
|
|
35
|
+
typedef struct regex_t {
|
|
36
|
+
PyObject* compiled;
|
|
37
|
+
} regex_t;
|
|
38
|
+
|
|
39
|
+
typedef struct regmatch_t {
|
|
40
|
+
ptrdiff_t rm_so;
|
|
41
|
+
ptrdiff_t rm_eo;
|
|
42
|
+
} regmatch_t;
|
|
43
|
+
|
|
44
|
+
#if REGEX_NT_COMPAT_HAS_PYTHON
|
|
45
|
+
|
|
46
|
+
static inline int regcomp(regex_t* regex, const char* pattern, int cflags) {
|
|
47
|
+
(void) cflags;
|
|
48
|
+
if (!regex || !pattern) {
|
|
49
|
+
return REG_BADPAT;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
regex->compiled = NULL;
|
|
53
|
+
|
|
54
|
+
PyGILState_STATE gil_state = PyGILState_Ensure();
|
|
55
|
+
PyObject* re_module = PyImport_ImportModule("re");
|
|
56
|
+
if (!re_module) {
|
|
57
|
+
PyErr_Clear();
|
|
58
|
+
PyGILState_Release(gil_state);
|
|
59
|
+
return REG_BADPAT;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
PyObject* compiled = PyObject_CallMethod(re_module, "compile", "s", pattern);
|
|
63
|
+
Py_DECREF(re_module);
|
|
64
|
+
if (!compiled) {
|
|
65
|
+
PyErr_Clear();
|
|
66
|
+
PyGILState_Release(gil_state);
|
|
67
|
+
return REG_BADPAT;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
regex->compiled = compiled;
|
|
71
|
+
PyGILState_Release(gil_state);
|
|
72
|
+
return 0;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
static inline int regexec(const regex_t* regex, const char* string, size_t nmatch, regmatch_t pmatch[], int eflags) {
|
|
76
|
+
(void) nmatch;
|
|
77
|
+
(void) pmatch;
|
|
78
|
+
(void) eflags;
|
|
79
|
+
|
|
80
|
+
if (!regex || !regex->compiled || !string) {
|
|
81
|
+
return REG_NOMATCH;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
PyGILState_STATE gil_state = PyGILState_Ensure();
|
|
85
|
+
PyObject* match = PyObject_CallMethod(regex->compiled, "search", "s", string);
|
|
86
|
+
if (!match) {
|
|
87
|
+
PyErr_Clear();
|
|
88
|
+
PyGILState_Release(gil_state);
|
|
89
|
+
return REG_NOMATCH;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
int result = (match == Py_None) ? REG_NOMATCH : 0;
|
|
93
|
+
Py_DECREF(match);
|
|
94
|
+
PyGILState_Release(gil_state);
|
|
95
|
+
return result;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
static inline void regfree(regex_t* regex) {
|
|
99
|
+
if (!regex || !regex->compiled) {
|
|
100
|
+
return;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
PyGILState_STATE gil_state = PyGILState_Ensure();
|
|
104
|
+
Py_DECREF(regex->compiled);
|
|
105
|
+
regex->compiled = NULL;
|
|
106
|
+
PyGILState_Release(gil_state);
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
#else
|
|
110
|
+
|
|
111
|
+
static inline int regcomp(regex_t* regex, const char* pattern, int cflags) {
|
|
112
|
+
(void) regex;
|
|
113
|
+
(void) pattern;
|
|
114
|
+
(void) cflags;
|
|
115
|
+
return REG_BADPAT;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
static inline int regexec(const regex_t* regex, const char* string, size_t nmatch, regmatch_t pmatch[], int eflags) {
|
|
119
|
+
(void) regex;
|
|
120
|
+
(void) string;
|
|
121
|
+
(void) nmatch;
|
|
122
|
+
(void) pmatch;
|
|
123
|
+
(void) eflags;
|
|
124
|
+
return REG_NOMATCH;
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
static inline void regfree(regex_t* regex) {
|
|
128
|
+
(void) regex;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
#endif
|
|
132
|
+
|
|
133
|
+
#endif /* REGEX_NT_COMPAT_H */
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
#ifndef TIME_SCHED_NT_COMPAT_H
|
|
2
|
+
#define TIME_SCHED_NT_COMPAT_H
|
|
3
|
+
|
|
4
|
+
#ifndef _WIN32
|
|
5
|
+
#error "time_sched_nt_compat.h is intended for Windows builds only"
|
|
6
|
+
#endif
|
|
7
|
+
|
|
8
|
+
#include <time.h>
|
|
9
|
+
|
|
10
|
+
#ifndef WIN32_LEAN_AND_MEAN
|
|
11
|
+
#define WIN32_LEAN_AND_MEAN
|
|
12
|
+
#endif
|
|
13
|
+
#include <windows.h>
|
|
14
|
+
|
|
15
|
+
#ifndef CLOCK_REALTIME
|
|
16
|
+
#define CLOCK_REALTIME 0
|
|
17
|
+
#endif
|
|
18
|
+
|
|
19
|
+
static inline int clock_gettime(int clock_id, struct timespec* ts) {
|
|
20
|
+
(void) clock_id;
|
|
21
|
+
if (!ts) {
|
|
22
|
+
return -1;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
return timespec_get(ts, TIME_UTC) == TIME_UTC ? 0 : -1;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
static inline void sched_yield(void) {
|
|
29
|
+
SwitchToThread();
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
#endif /* TIME_SCHED_NT_COMPAT_H */
|
cbase/int128.h
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
#ifndef C_CBASE_INT128_H
|
|
2
|
+
#define C_CBASE_INT128_H
|
|
3
|
+
|
|
4
|
+
#include <stdint.h>
|
|
5
|
+
#include <string.h>
|
|
6
|
+
|
|
7
|
+
#if defined(__SIZEOF_INT128__)
|
|
8
|
+
typedef __int128_t int128_t;
|
|
9
|
+
typedef __uint128_t uint128_t;
|
|
10
|
+
|
|
11
|
+
static const uint128_t UINT128_MAX = (((uint128_t) 1) << 127) * 2 - 1;
|
|
12
|
+
static const int128_t INT128_MAX = (int128_t) (UINT128_MAX >> 1);
|
|
13
|
+
static const int128_t INT128_MIN = -((int128_t) (UINT128_MAX)) - 1;
|
|
14
|
+
|
|
15
|
+
static inline int c_u128_cmp(const uint128_t v1, const uint128_t v2) {
|
|
16
|
+
if (v1 < v2) return -1;
|
|
17
|
+
if (v1 > v2) return 1;
|
|
18
|
+
return 0;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
static inline int c_i128_cmp(const int128_t v1, const int128_t v2) {
|
|
22
|
+
if (v1 < v2) return -1;
|
|
23
|
+
if (v1 > v2) return 1;
|
|
24
|
+
return 0;
|
|
25
|
+
}
|
|
26
|
+
#else
|
|
27
|
+
/**
|
|
28
|
+
* @brief MSVC has no native 128-bit integer -- emulate with a two-limb
|
|
29
|
+
* little-endian struct (memory layout identical to __uint128_t on
|
|
30
|
+
* x86-64, so 16-byte ID slots stay byte-compatible).
|
|
31
|
+
*
|
|
32
|
+
* The Cython boundary converts via int.to_bytes/from_bytes, so no 128-bit
|
|
33
|
+
* arithmetic is required -- only 16-byte round-trips and ordered comparison.
|
|
34
|
+
*/
|
|
35
|
+
typedef struct {
|
|
36
|
+
uint64_t lo;
|
|
37
|
+
int64_t hi;
|
|
38
|
+
} int128_t;
|
|
39
|
+
typedef struct {
|
|
40
|
+
uint64_t lo;
|
|
41
|
+
uint64_t hi;
|
|
42
|
+
} uint128_t;
|
|
43
|
+
|
|
44
|
+
static const uint128_t UINT128_MAX = {UINT64_MAX, UINT64_MAX};
|
|
45
|
+
static const int128_t INT128_MAX = {UINT64_MAX, INT64_MAX};
|
|
46
|
+
static const int128_t INT128_MIN = {0u, INT64_MIN};
|
|
47
|
+
|
|
48
|
+
static inline int c_u128_cmp(const uint128_t v1, const uint128_t v2) {
|
|
49
|
+
if (v1.hi != v2.hi) return v1.hi < v2.hi ? -1 : 1;
|
|
50
|
+
if (v1.lo != v2.lo) return v1.lo < v2.lo ? -1 : 1;
|
|
51
|
+
return 0;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
static inline int c_i128_cmp(const int128_t v1, const int128_t v2) {
|
|
55
|
+
if (v1.hi != v2.hi) return v1.hi < v2.hi ? -1 : 1;
|
|
56
|
+
if (v1.lo != v2.lo) return v1.lo < v2.lo ? -1 : 1;
|
|
57
|
+
return 0;
|
|
58
|
+
}
|
|
59
|
+
#endif
|
|
60
|
+
|
|
61
|
+
static inline void c_write_uint128(void* data, uint128_t value) {
|
|
62
|
+
memcpy(data, &value, sizeof(uint128_t));
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
static inline uint128_t c_read_uint128(const void* data) {
|
|
66
|
+
uint128_t value;
|
|
67
|
+
memcpy(&value, data, sizeof(uint128_t));
|
|
68
|
+
return value;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
static inline void c_write_int128(void* data, int128_t value) {
|
|
72
|
+
memcpy(data, &value, sizeof(int128_t));
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
static inline int128_t c_read_int128(const void* data) {
|
|
76
|
+
int128_t value;
|
|
77
|
+
memcpy(&value, data, sizeof(int128_t));
|
|
78
|
+
return value;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
#endif // C_CBASE_INT128_H
|
cbase/int128.pxd
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
cdef extern from "cbase/int128.h":
|
|
2
|
+
ctypedef unsigned long long uint128_t
|
|
3
|
+
ctypedef long long int128_t
|
|
4
|
+
|
|
5
|
+
const uint128_t UINT128_MAX
|
|
6
|
+
const int128_t INT128_MAX
|
|
7
|
+
const int128_t INT128_MIN
|
|
8
|
+
|
|
9
|
+
uint128_t c_read_uint128(const void* data)
|
|
10
|
+
void c_write_uint128(void* data, uint128_t value)
|
|
11
|
+
int128_t c_read_int128(const void* data)
|
|
12
|
+
void c_write_int128(void* data, int128_t value)
|
|
13
|
+
|
|
14
|
+
int c_u128_cmp(uint128_t v1, uint128_t v2)
|
|
15
|
+
int c_i128_cmp(int128_t v1, int128_t v2)
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# InternString Performance Benchmark
|
|
2
|
+
|
|
3
|
+
## Realistic Workload — Miss-Rate Controlled
|
|
4
|
+
|
|
5
|
+
**Scenario:** 1,000,000 operations with controlled miss rates from 1/1K to 1/1M
|
|
6
|
+
**Config:** 256 MiB buffer, 1M available segments, 32-byte max, 10 iterations
|
|
7
|
+
**Method:** Misses (new unique strings) evenly distributed; hits randomly pick from already-seen set
|
|
8
|
+
|
|
9
|
+
| Miss Rate | n_misses | istr (ns) | py_unicode (ns) | Speedup |
|
|
10
|
+
| ----------- | -------: | --------: | --------------: | -------: |
|
|
11
|
+
| 1/1,000 | 1,000 | 16.7 ns | 28.5 ns | **1.7×** |
|
|
12
|
+
| 1/10,000 | 100 | 14.2 ns | 26.6 ns | **1.9×** |
|
|
13
|
+
| 1/100,000 | 10 | 11.1 ns | 21.1 ns | **1.9×** |
|
|
14
|
+
| 1/1,000,000 | 1 | 7.3 ns | 16.4 ns | **2.2×** |
|
|
15
|
+
|
|
16
|
+
### Analysis
|
|
17
|
+
|
|
18
|
+
- At **1/1M** miss rate (typical for production ticker/data feeds): istr is **2.2× faster** than Python `str` creation
|
|
19
|
+
- As miss rate → 0, istr latency approaches the raw hash+probe cost (7.3 ns) — almost all operations are hash-table hits
|
|
20
|
+
- Python `str` creation is independent of miss rate — it always allocates a new object (~16–28 ns)
|
|
21
|
+
- The crossover point where istr becomes faster than Python `str` is at approximately **1/10 miss rate** (10% new strings)
|
|
22
|
+
- Below 1/100 miss rate, istr provides **1.7–2.2× speedup** PLUS string deduplication, stable C pointers, and cached FNV-1a hashes
|
|
23
|
+
|
|
24
|
+
### What istr provides beyond raw speed
|
|
25
|
+
|
|
26
|
+
| Feature | istr | Python str |
|
|
27
|
+
| -------------------- | ---------------------------------- | ------------------------------ |
|
|
28
|
+
| String deduplication | ✅ Same pointer for same key | ❌ New object every time |
|
|
29
|
+
| Stable C pointer | ✅ `const char*` usable as hash key | ❌ Must call `PyUnicode_AsUTF8` |
|
|
30
|
+
| Cached hash | ✅ FNV-1a stored in entry | ❌ Computed on first `hash()` |
|
|
31
|
+
| Memory | One copy per unique string | One copy per creation |
|
|
32
|
+
|
|
33
|
+
---
|
|
34
|
+
|
|
35
|
+
## Backend Comparison — Pure C-level
|
|
36
|
+
|
|
37
|
+
**Dataset:** 256 MiB buffer, 50,000 segments, max 64 bytes/segment, 10 iterations
|
|
38
|
+
**Total key bytes:** ~1.6 MB
|
|
39
|
+
|
|
40
|
+
### Per-operation latency (ns)
|
|
41
|
+
|
|
42
|
+
| Benchmark | Native | Bytemap | Winner |
|
|
43
|
+
| ------------------------ | -------: | -------: | --------------- |
|
|
44
|
+
| `fnv1a_hash` (raw C) | 48.8 ns | 38.3 ns | Bytemap (1.27×) |
|
|
45
|
+
| `istr_intern` (unlocked) | 291.6 ns | 532.8 ns | Native (1.83×) |
|
|
46
|
+
| `istr_intern_synced` | 192.2 ns | 390.5 ns | Native (2.03×) |
|
|
47
|
+
| `istr_lookup` (unlocked) | 57.3 ns | 28.2 ns | Bytemap (2.03×) |
|
|
48
|
+
| `istr_lookup_synced` | 54.5 ns | 26.3 ns | Bytemap (2.07×) |
|
|
49
|
+
| `istr_eq` (Python) | 215.1 ns | 197.7 ns | Bytemap (1.09×) |
|
|
50
|
+
| `py_unicode` (create) | 52.4 ns | 46.0 ns | — |
|
|
51
|
+
| `py_hash` (str) | 3.8 ns | 4.2 ns | — |
|
|
52
|
+
| `py_eq` (str) | 4.7 ns | 4.7 ns | — |
|
|
53
|
+
|
|
54
|
+
### Mutex overhead (synced − unlocked, per op)
|
|
55
|
+
|
|
56
|
+
| Operation | Native | Bytemap |
|
|
57
|
+
| --------- | ------- | ------- |
|
|
58
|
+
| Intern | *noise* | *noise* |
|
|
59
|
+
| Lookup | ~1.0 ns | ~1.9 ns |
|
|
60
|
+
|
|
61
|
+
> Mutex overhead is within measurement noise — the pthread_mutex_lock/unlock cost
|
|
62
|
+
> (~15–25 ns) is dwarfed by the hash+probe cost at 50k iterations. The negative
|
|
63
|
+
> values sometimes seen are run-ordering artifacts (heap warm-up from prior
|
|
64
|
+
> unlocked run).
|
|
65
|
+
|
|
66
|
+
### Analysis
|
|
67
|
+
|
|
68
|
+
| Operation | Result |
|
|
69
|
+
| -------------------- | ---------------------------------------------------------------------------------------- |
|
|
70
|
+
| **Intern speed** | Native 1.83× faster — simpler insert path (no tombstone logic, no XXH3 wrapper overhead) |
|
|
71
|
+
| **Lookup speed** | Bytemap 2.03× faster — XXH3 hash + `memcmp` with stored `key_length` beats FNV-1a |
|
|
72
|
+
| **Hash speed** | XXH3 beats FNV-1a by 1.27× on raw `const char*` |
|
|
73
|
+
| **Pure C vs Python** | Both now hit C fast path via `PyUnicode_AsUTF8AndSize` + `key_length` — zero strlen |
|
|
74
|
+
|
|
75
|
+
### Backend behavioral differences
|
|
76
|
+
|
|
77
|
+
| Behavior | Native | Bytemap |
|
|
78
|
+
| ------------------- | -------------------------- | ---------------------------------------------- |
|
|
79
|
+
| Iteration order | LIFO (reverse insertion) | FIFO (insertion order via doubly-linked list) |
|
|
80
|
+
| Empty string (`""`) | Valid (1-byte allocation) | Rejected (`BYTEMAP_ERR_INVALID_KEY`) |
|
|
81
|
+
| Hash function | FNV-1a | XXH3 (with per-map salt) |
|
|
82
|
+
| Capacity rounding | Power-of-2 | Internal (MIN_BYTEMAP_CAPACITY = 16) |
|
|
83
|
+
| Resize trigger | `size >= capacity/2` | `occupied * 2 >= capacity` |
|
|
84
|
+
| Thread safety | Internal `pthread_mutex_t` | Embedded `pthread_mutex_t` in istr_map wrapper |
|
|
85
|
+
|
|
86
|
+
### Optimizations applied
|
|
87
|
+
|
|
88
|
+
- `c_istr` / `c_istr_map_lookup` accept `size_t key_length` — `strlen` skipped when known
|
|
89
|
+
- `strcmp` replaced with `key_length` comparison + `memcmp` shortcut
|
|
90
|
+
- `PyUnicode_AsUTF8AndSize` passes pre-computed UTF-8 length from Python layer
|
|
91
|
+
- C-level benchmark routines bypass Python `str` entirely — raw `const char*` + `size_t`
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## NT Cross-Platform — Native Backend
|
|
96
|
+
|
|
97
|
+
**Quick benchmark:** 1 MiB buffer, 5,000 segments, max 64 bytes, 5 iterations
|
|
98
|
+
**Unit tests:** 220/220 pass, 0 fail, 5 skipped (fork-related)
|
|
99
|
+
|
|
100
|
+
### Per-operation latency (ns)
|
|
101
|
+
|
|
102
|
+
| Benchmark | Linux (gcc -O2) | Windows NT (MSVC /O2) | Δ |
|
|
103
|
+
| ------------------------ | --------------: | --------------------: | ---- |
|
|
104
|
+
| `fnv1a_hash` | 13.5 ns | 15.5 ns | +15% |
|
|
105
|
+
| `istr_intern` (unlocked) | 116.6 ns | 120.8 ns | +4% |
|
|
106
|
+
| `istr_intern` (synced) | 102.6 ns | 107.0 ns | +4% |
|
|
107
|
+
| `istr_lookup` (unlocked) | 24.6 ns | 30.2 ns | +23% |
|
|
108
|
+
| `istr_lookup` (synced) | 25.6 ns | 31.6 ns | +23% |
|
|
109
|
+
| `istr_eq` (Python) | 177.5 ns | 208.0 ns | +17% |
|
|
110
|
+
| `py_unicode` (create) | 27.9 ns | 34.1 ns | +22% |
|
|
111
|
+
| `py_hash` | 4.6 ns | 6.3 ns | +37% |
|
|
112
|
+
| `py_eq` | 4.6 ns | 6.8 ns | +48% |
|
|
113
|
+
|
|
114
|
+
### NT compatibility
|
|
115
|
+
|
|
116
|
+
| Component | Status |
|
|
117
|
+
| -------------------------------------------------------------- | --------------------------------------- |
|
|
118
|
+
| `pthread_nt_compat.h` (`pthread_mutex_t` → `CRITICAL_SECTION`) | ✅ |
|
|
119
|
+
| MSVC build (`/std:c17 /experimental:c11atomics`) | ✅ |
|
|
120
|
+
| Cython 3.x on Windows | ✅ |
|
|
121
|
+
| All 220 unit tests | ✅ |
|
|
122
|
+
| Thread safety (concurrent tests) | ✅ |
|
|
123
|
+
| Performance delta vs Linux | ~15–23% (expected: MSVC vs GCC codegen) |
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
## Running Benchmarks
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
# Linux — native backend (default)
|
|
131
|
+
python tests/intern_string/test_perf.py # full (1 GiB, 100k segs, 10 iters)
|
|
132
|
+
python tests/intern_string/test_perf.py --quick # quick (1 MiB, 5k segs, 5 iters)
|
|
133
|
+
|
|
134
|
+
# Linux — bytemap backend (requires rebuild)
|
|
135
|
+
# Add -DISTR_USE_BYTEMAP_BACKEND=1 to extra_compile_args in setup.py, rebuild, then:
|
|
136
|
+
python tests/intern_string/test_perf.py --quick
|
|
137
|
+
|
|
138
|
+
# Windows NT
|
|
139
|
+
python nt_build.py # full sync→build→test cycle
|
|
140
|
+
# Then run test_perf.py on the VM
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## Test Toolkit API
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from cbase.intern_string.c_intern_string import IstrTestToolkit
|
|
147
|
+
|
|
148
|
+
tk = IstrTestToolkit(
|
|
149
|
+
buf_size=2**30, # 1 GiB character buffer
|
|
150
|
+
n_seg=100_000, # number of random segments
|
|
151
|
+
max_seg_len=64, # max segment length in bytes
|
|
152
|
+
n_iters=10, # benchmark repetitions
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
results = tk.run_test()
|
|
156
|
+
# Returns dict with per-op nanosecond latencies + raw timings
|
|
157
|
+
```
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from .c_intern_string import (
|
|
2
|
+
InternString,
|
|
3
|
+
InternStringPool,
|
|
4
|
+
IstrTestToolkit,
|
|
5
|
+
POOL,
|
|
6
|
+
INTRA_POOL,
|
|
7
|
+
C_POOL,
|
|
8
|
+
C_INTRA_POOL,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
'InternString',
|
|
13
|
+
'InternStringPool',
|
|
14
|
+
'IstrTestToolkit',
|
|
15
|
+
'POOL',
|
|
16
|
+
'INTRA_POOL',
|
|
17
|
+
'C_POOL',
|
|
18
|
+
'C_INTRA_POOL',
|
|
19
|
+
]
|