numexpr 2.10.2__cp313-cp313-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
numexpr/module.cpp ADDED
@@ -0,0 +1,537 @@
1
+ // Numexpr - Fast numerical array expression evaluator for NumPy.
2
+ //
3
+ // License: MIT
4
+ // Author: See AUTHORS.txt
5
+ //
6
+ // See LICENSE.txt for details about copyright and rights to use.
7
+ //
8
+ // module.cpp contains the CPython-specific module exposure.
9
+
10
+ #define DO_NUMPY_IMPORT_ARRAY
11
+
12
+ #include "module.hpp"
13
+ #include <structmember.h>
14
+ #include <vector>
15
+
16
+ #include <signal.h>
17
+
18
+ #include "interpreter.hpp"
19
+ #include "numexpr_object.hpp"
20
+
21
+ using namespace std;
22
+
23
+ // Global state. The file interpreter.hpp also has some global state
24
+ // in its 'th_params' variable
25
+ global_state gs;
26
+ long global_max_threads=DEFAULT_MAX_THREADS;
27
+
28
+ /* Do the worker job for a certain thread */
29
+ void *th_worker(void *tidptr)
30
+ {
31
+ int tid = *(int *)tidptr;
32
+ /* Parameters for threads */
33
+ npy_intp start;
34
+ npy_intp vlen;
35
+ npy_intp block_size;
36
+ NpyIter *iter;
37
+ vm_params params;
38
+ int *pc_error;
39
+ int ret;
40
+ int n_inputs;
41
+ int n_constants;
42
+ int n_temps;
43
+ size_t memsize;
44
+ char **mem;
45
+ npy_intp *memsteps;
46
+ npy_intp istart, iend;
47
+ char **errmsg;
48
+ // For output buffering if needed
49
+ vector<char> out_buffer;
50
+
51
+ while (1) {
52
+
53
+ /* Sentinels have to be initialised yet */
54
+ gs.init_sentinels_done = 0;
55
+
56
+ /* Meeting point for all threads (wait for initialization) */
57
+ pthread_mutex_lock(&gs.count_threads_mutex);
58
+ if (gs.count_threads < gs.nthreads) {
59
+ gs.count_threads++;
60
+ /* Beware of spurious wakeups. See issue pydata/numexpr#306. */
61
+ do {
62
+ pthread_cond_wait(&gs.count_threads_cv,
63
+ &gs.count_threads_mutex);
64
+ } while (!gs.barrier_passed);
65
+ }
66
+ else {
67
+ gs.barrier_passed = 1;
68
+ pthread_cond_broadcast(&gs.count_threads_cv);
69
+ }
70
+ pthread_mutex_unlock(&gs.count_threads_mutex);
71
+
72
+ /* Check if thread has been asked to return */
73
+ if (gs.end_threads) {
74
+ return(0);
75
+ }
76
+
77
+ /* Get parameters for this thread before entering the main loop */
78
+ start = th_params.start;
79
+ vlen = th_params.vlen;
80
+ block_size = th_params.block_size;
81
+ params = th_params.params;
82
+ pc_error = th_params.pc_error;
83
+
84
+ // If output buffering is needed, allocate it
85
+ if (th_params.need_output_buffering) {
86
+ out_buffer.resize(params.memsizes[0] * BLOCK_SIZE1);
87
+ params.out_buffer = &out_buffer[0];
88
+ } else {
89
+ params.out_buffer = NULL;
90
+ }
91
+
92
+ /* Populate private data for each thread */
93
+ n_inputs = params.n_inputs;
94
+ n_constants = params.n_constants;
95
+ n_temps = params.n_temps;
96
+ memsize = (1+n_inputs+n_constants+n_temps) * sizeof(char *);
97
+ /* XXX malloc seems thread safe for POSIX, but for Win? */
98
+ mem = (char **)malloc(memsize);
99
+ memcpy(mem, params.mem, memsize);
100
+
101
+ errmsg = th_params.errmsg;
102
+
103
+ params.mem = mem;
104
+
105
+ /* Loop over blocks */
106
+ pthread_mutex_lock(&gs.count_mutex);
107
+ if (!gs.init_sentinels_done) {
108
+ /* Set sentinels and other global variables */
109
+ gs.gindex = start;
110
+ istart = gs.gindex;
111
+ iend = istart + block_size;
112
+ if (iend > vlen) {
113
+ iend = vlen;
114
+ }
115
+ gs.init_sentinels_done = 1; /* sentinels have been initialised */
116
+ gs.giveup = 0; /* no giveup initially */
117
+ } else {
118
+ gs.gindex += block_size;
119
+ istart = gs.gindex;
120
+ iend = istart + block_size;
121
+ if (iend > vlen) {
122
+ iend = vlen;
123
+ }
124
+ }
125
+ /* Grab one of the iterators */
126
+ iter = th_params.iter[tid];
127
+ if (iter == NULL) {
128
+ th_params.ret_code = -1;
129
+ gs.giveup = 1;
130
+ }
131
+ memsteps = th_params.memsteps[tid];
132
+ /* Get temporary space for each thread */
133
+ ret = get_temps_space(params, mem, BLOCK_SIZE1);
134
+ if (ret < 0) {
135
+ /* Propagate error to main thread */
136
+ th_params.ret_code = ret;
137
+ gs.giveup = 1;
138
+ }
139
+ pthread_mutex_unlock(&gs.count_mutex);
140
+
141
+ while (istart < vlen && !gs.giveup) {
142
+ /* Reset the iterator to the range for this task */
143
+ ret = NpyIter_ResetToIterIndexRange(iter, istart, iend,
144
+ errmsg);
145
+ /* Execute the task */
146
+ if (ret >= 0) {
147
+ ret = vm_engine_iter_task(iter, memsteps, params, pc_error, errmsg);
148
+ }
149
+
150
+ if (ret < 0) {
151
+ pthread_mutex_lock(&gs.count_mutex);
152
+ gs.giveup = 1;
153
+ /* Propagate error to main thread */
154
+ th_params.ret_code = ret;
155
+ pthread_mutex_unlock(&gs.count_mutex);
156
+ break;
157
+ }
158
+
159
+ pthread_mutex_lock(&gs.count_mutex);
160
+ gs.gindex += block_size;
161
+ istart = gs.gindex;
162
+ iend = istart + block_size;
163
+ if (iend > vlen) {
164
+ iend = vlen;
165
+ }
166
+ pthread_mutex_unlock(&gs.count_mutex);
167
+ }
168
+
169
+ /* Meeting point for all threads (wait for finalization) */
170
+ pthread_mutex_lock(&gs.count_threads_mutex);
171
+ if (gs.count_threads > 0) {
172
+ gs.count_threads--;
173
+ do {
174
+ pthread_cond_wait(&gs.count_threads_cv,
175
+ &gs.count_threads_mutex);
176
+ } while (gs.barrier_passed);
177
+ }
178
+ else {
179
+ gs.barrier_passed = 0;
180
+ pthread_cond_broadcast(&gs.count_threads_cv);
181
+ }
182
+ pthread_mutex_unlock(&gs.count_threads_mutex);
183
+
184
+ /* Release resources */
185
+ free_temps_space(params, mem);
186
+ free(mem);
187
+
188
+ } /* closes while(1) */
189
+
190
+ /* This should never be reached, but anyway */
191
+ return(0);
192
+ }
193
+
194
+ /* Initialize threads */
195
+ int init_threads(void)
196
+ {
197
+ int tid, rc;
198
+
199
+ if ( !(gs.nthreads > 1 && (!gs.init_threads_done || gs.pid != getpid())) ) {
200
+ /* Thread pool must always be initialized once and once only. */
201
+ return(0);
202
+ }
203
+
204
+ /* Initialize mutex and condition variable objects */
205
+ pthread_mutex_init(&gs.count_mutex, NULL);
206
+ pthread_mutex_init(&gs.parallel_mutex, NULL);
207
+
208
+ /* Barrier initialization */
209
+ pthread_mutex_init(&gs.count_threads_mutex, NULL);
210
+ pthread_cond_init(&gs.count_threads_cv, NULL);
211
+ gs.count_threads = 0; /* Reset threads counter */
212
+ gs.barrier_passed = 0;
213
+
214
+ /*
215
+ * Our worker threads should not deal with signals from the rest of the
216
+ * application - mask everything temporarily in this thread, so our workers
217
+ * can inherit that mask
218
+ */
219
+ sigset_t sigset_block_all, sigset_restore;
220
+ rc = sigfillset(&sigset_block_all);
221
+ if (rc != 0) {
222
+ fprintf(stderr, "ERROR; failed to block signals: sigfillset: %s",
223
+ strerror(rc));
224
+ exit(-1);
225
+ }
226
+ rc = pthread_sigmask( SIG_BLOCK, &sigset_block_all, &sigset_restore);
227
+ if (rc != 0) {
228
+ fprintf(stderr, "ERROR; failed to block signals: pthread_sigmask: %s",
229
+ strerror(rc));
230
+ exit(-1);
231
+ }
232
+
233
+ /* Now create the threads */
234
+ for (tid = 0; tid < gs.nthreads; tid++) {
235
+ gs.tids[tid] = tid;
236
+ rc = pthread_create(&gs.threads[tid], NULL, th_worker,
237
+ (void *)&gs.tids[tid]);
238
+ if (rc) {
239
+ fprintf(stderr,
240
+ "ERROR; return code from pthread_create() is %d\n", rc);
241
+ fprintf(stderr, "\tError detail: %s\n", strerror(rc));
242
+ exit(-1);
243
+ }
244
+ }
245
+
246
+ /*
247
+ * Restore the signal mask so the main thread can process signals as
248
+ * expected
249
+ */
250
+ rc = pthread_sigmask( SIG_SETMASK, &sigset_restore, NULL);
251
+ if (rc != 0) {
252
+ fprintf(stderr,
253
+ "ERROR: failed to restore signal mask: pthread_sigmask: %s",
254
+ strerror(rc));
255
+ exit(-1);
256
+ }
257
+
258
+ gs.init_threads_done = 1; /* Initialization done! */
259
+ gs.pid = (int)getpid(); /* save the PID for this process */
260
+
261
+ return(0);
262
+ }
263
+
264
+ /* Set the number of threads in numexpr's VM */
265
+ int numexpr_set_nthreads(int nthreads_new)
266
+ {
267
+ int nthreads_old = gs.nthreads;
268
+ int t, rc;
269
+ void *status;
270
+
271
+ // if (nthreads_new > MAX_THREADS) {
272
+ // fprintf(stderr,
273
+ // "Error. nthreads cannot be larger than MAX_THREADS (%d)",
274
+ // MAX_THREADS);
275
+ // return -1;
276
+ // }
277
+ if (nthreads_new > global_max_threads) {
278
+ fprintf(stderr,
279
+ "Error. nthreads cannot be larger than environment variable \"NUMEXPR_MAX_THREADS\" (%ld)",
280
+ global_max_threads);
281
+ return -1;
282
+ }
283
+ else if (nthreads_new <= 0) {
284
+ fprintf(stderr, "Error. nthreads must be a positive integer");
285
+ return -1;
286
+ }
287
+
288
+ /* Only join threads if they are not initialized or if our PID is
289
+ different from that in pid var (probably means that we are a
290
+ subprocess, and thus threads are non-existent). */
291
+ if (gs.nthreads > 1 && gs.init_threads_done && gs.pid == getpid()) {
292
+ /* Tell all existing threads to finish */
293
+ gs.end_threads = 1;
294
+ pthread_mutex_lock(&gs.count_threads_mutex);
295
+ if (gs.count_threads < gs.nthreads) {
296
+ gs.count_threads++;
297
+ do {
298
+ pthread_cond_wait(&gs.count_threads_cv,
299
+ &gs.count_threads_mutex);
300
+ } while (!gs.barrier_passed);
301
+ }
302
+ else {
303
+ gs.barrier_passed = 1;
304
+ pthread_cond_broadcast(&gs.count_threads_cv);
305
+ }
306
+ pthread_mutex_unlock(&gs.count_threads_mutex);
307
+
308
+ /* Join exiting threads */
309
+ for (t=0; t<gs.nthreads; t++) {
310
+ rc = pthread_join(gs.threads[t], &status);
311
+ if (rc) {
312
+ fprintf(stderr,
313
+ "ERROR; return code from pthread_join() is %d\n",
314
+ rc);
315
+ fprintf(stderr, "\tError detail: %s\n", strerror(rc));
316
+ exit(-1);
317
+ }
318
+ }
319
+ gs.init_threads_done = 0;
320
+ gs.end_threads = 0;
321
+ }
322
+
323
+ /* Launch a new pool of threads (if necessary) */
324
+ gs.nthreads = nthreads_new;
325
+ init_threads();
326
+
327
+ return nthreads_old;
328
+ }
329
+
330
+
331
+ #ifdef USE_VML
332
+
333
+ static PyObject *
334
+ _get_vml_version(PyObject *self, PyObject *args)
335
+ {
336
+ int len=198;
337
+ char buf[198];
338
+ mkl_get_version_string(buf, len);
339
+ return Py_BuildValue("s", buf);
340
+ }
341
+
342
+ static PyObject *
343
+ _set_vml_accuracy_mode(PyObject *self, PyObject *args)
344
+ {
345
+ int mode_in, mode_old;
346
+ if (!PyArg_ParseTuple(args, "i", &mode_in))
347
+ return NULL;
348
+ mode_old = vmlGetMode() & VML_ACCURACY_MASK;
349
+ vmlSetMode((mode_in & VML_ACCURACY_MASK) | VML_ERRMODE_IGNORE );
350
+ return Py_BuildValue("i", mode_old);
351
+ }
352
+
353
+ static PyObject *
354
+ _set_vml_num_threads(PyObject *self, PyObject *args)
355
+ {
356
+ int max_num_threads;
357
+ if (!PyArg_ParseTuple(args, "i", &max_num_threads))
358
+ return NULL;
359
+ mkl_domain_set_num_threads(max_num_threads, MKL_DOMAIN_VML);
360
+ Py_RETURN_NONE;
361
+ }
362
+
363
+ static PyObject *
364
+ _get_vml_num_threads(PyObject *self, PyObject *args)
365
+ {
366
+ int max_num_threads = mkl_domain_get_max_threads (MKL_DOMAIN_VML);
367
+ return Py_BuildValue("i", max_num_threads);
368
+ }
369
+
370
+ #endif
371
+
372
+ static PyObject*
373
+ Py_set_num_threads(PyObject *self, PyObject *args)
374
+ {
375
+ int num_threads, nthreads_old;
376
+ if (!PyArg_ParseTuple(args, "i", &num_threads))
377
+ return NULL;
378
+ nthreads_old = numexpr_set_nthreads(num_threads);
379
+ return Py_BuildValue("i", nthreads_old);
380
+ }
381
+
382
+ static PyObject*
383
+ Py_get_num_threads(PyObject *self, PyObject *args)
384
+ {
385
+ int n_thread;
386
+ n_thread = gs.nthreads;
387
+ return Py_BuildValue("i", n_thread);
388
+ }
389
+
390
+ static PyMethodDef module_methods[] = {
391
+ #ifdef USE_VML
392
+ {"_get_vml_version", _get_vml_version, METH_VARARGS,
393
+ "Get the VML/MKL library version."},
394
+ {"_set_vml_accuracy_mode", _set_vml_accuracy_mode, METH_VARARGS,
395
+ "Set accuracy mode for VML functions."},
396
+ {"_set_vml_num_threads", _set_vml_num_threads, METH_VARARGS,
397
+ "Suggests a maximum number of threads to be used in VML operations."},
398
+ {"_get_vml_num_threads", _get_vml_num_threads, METH_VARARGS,
399
+ "Gets the maximum number of threads to be used in VML operations."},
400
+ #endif
401
+ {"_set_num_threads", Py_set_num_threads, METH_VARARGS,
402
+ "Suggests a maximum number of threads to be used in operations."},
403
+ {"_get_num_threads", Py_get_num_threads, METH_VARARGS,
404
+ "Gets the maximum number of threads currently in use for operations."},
405
+ {NULL}
406
+ };
407
+
408
+ static int
409
+ add_symbol(PyObject *d, const char *sname, int name, const char* routine_name)
410
+ {
411
+ PyObject *o, *s;
412
+ int r;
413
+
414
+ if (!sname) {
415
+ return 0;
416
+ }
417
+
418
+ o = PyLong_FromLong(name);
419
+ s = PyBytes_FromString(sname);
420
+ if (!o || !s) {
421
+ PyErr_SetString(PyExc_RuntimeError, routine_name);
422
+ r = -1;
423
+ }
424
+ else {
425
+ r = PyDict_SetItem(d, s, o);
426
+ }
427
+ Py_XDECREF(o);
428
+ Py_XDECREF(s);
429
+ return r;
430
+ }
431
+
432
+ #ifdef __cplusplus
433
+ extern "C" {
434
+ #endif
435
+
436
+ /* XXX: handle the "global_state" state via moduledef */
437
+ static struct PyModuleDef moduledef = {
438
+ PyModuleDef_HEAD_INIT,
439
+ "interpreter",
440
+ NULL,
441
+ -1, /* sizeof(struct global_state), */
442
+ module_methods,
443
+ NULL,
444
+ NULL, /* module_traverse, */
445
+ NULL, /* module_clear, */
446
+ NULL
447
+ };
448
+
449
+ #define INITERROR return NULL
450
+
451
+ PyObject *
452
+ PyInit_interpreter(void) {
453
+ PyObject *m, *d;
454
+
455
+
456
+ char *max_thread_str = getenv("NUMEXPR_MAX_THREADS");
457
+ char *end;
458
+ if (max_thread_str != NULL) {
459
+ global_max_threads = strtol(max_thread_str, &end, 10);
460
+ }
461
+
462
+
463
+
464
+ th_params.memsteps = (npy_intp**)calloc(sizeof(npy_intp*), global_max_threads);
465
+ th_params.iter = (NpyIter**)calloc(sizeof(NpyIter*), global_max_threads);
466
+ th_params.reduce_iter = (NpyIter**)calloc(sizeof(NpyIter*), global_max_threads);
467
+ gs.threads = (pthread_t*)calloc(sizeof(pthread_t), global_max_threads);
468
+ gs.tids = (int*)calloc(sizeof(int), global_max_threads);
469
+ // TODO: for Py3, deallocate: https://docs.python.org/3/c-api/module.html#c.PyModuleDef.m_free
470
+ // For Python 2.7, people have to exit the process to reclaim the memory.
471
+
472
+ if (PyType_Ready(&NumExprType) < 0)
473
+ INITERROR;
474
+
475
+ m = PyModule_Create(&moduledef);
476
+
477
+ if (m == NULL)
478
+ INITERROR;
479
+
480
+ Py_INCREF(&NumExprType);
481
+ PyModule_AddObject(m, "NumExpr", (PyObject *)&NumExprType);
482
+
483
+ import_array();
484
+
485
+ d = PyDict_New();
486
+ if (!d) INITERROR;
487
+
488
+ #define OPCODE(n, name, sname, ...) \
489
+ if (add_symbol(d, sname, name, "add_op") < 0) { INITERROR; }
490
+ #include "opcodes.hpp"
491
+ #undef OPCODE
492
+
493
+ if (PyModule_AddObject(m, "opcodes", d) < 0) INITERROR;
494
+
495
+ d = PyDict_New();
496
+ if (!d) INITERROR;
497
+
498
+ #define add_func(name, sname) \
499
+ if (add_symbol(d, sname, name, "add_func") < 0) { INITERROR; }
500
+ #define FUNC_FF(name, sname, ...) add_func(name, sname);
501
+ #define FUNC_FFF(name, sname, ...) add_func(name, sname);
502
+ #define FUNC_DD(name, sname, ...) add_func(name, sname);
503
+ #define FUNC_DDD(name, sname, ...) add_func(name, sname);
504
+ #define FUNC_CC(name, sname, ...) add_func(name, sname);
505
+ #define FUNC_CCC(name, sname, ...) add_func(name, sname);
506
+ #include "functions.hpp"
507
+ #undef FUNC_CCC
508
+ #undef FUNC_CC
509
+ #undef FUNC_DDD
510
+ #undef FUNC_DD
511
+ #undef FUNC_DD
512
+ #undef FUNC_FFF
513
+ #undef FUNC_FF
514
+ #undef add_func
515
+
516
+ if (PyModule_AddObject(m, "funccodes", d) < 0) INITERROR;
517
+
518
+ if (PyModule_AddObject(m, "allaxes", PyLong_FromLong(255)) < 0) INITERROR;
519
+ if (PyModule_AddObject(m, "maxdims", PyLong_FromLong(NPY_MAXDIMS)) < 0) INITERROR;
520
+
521
+ if(PyModule_AddIntConstant(m, "MAX_THREADS", global_max_threads) < 0) INITERROR;
522
+
523
+ // Let's export the block sizes to Python side for benchmarking comparisons
524
+ if(PyModule_AddIntConstant(m, "__BLOCK_SIZE1__", BLOCK_SIZE1) < 0) INITERROR;
525
+ // Export if we are using VML or not
526
+ #ifdef USE_VML
527
+ if(PyModule_AddObject(m, "use_vml", Py_True) < 0) INITERROR;
528
+ #else
529
+ if(PyModule_AddObject(m, "use_vml", Py_False) < 0) INITERROR;
530
+ #endif
531
+
532
+ return m;
533
+ }
534
+
535
+ #ifdef __cplusplus
536
+ } // extern "C"
537
+ #endif
numexpr/module.hpp ADDED
@@ -0,0 +1,60 @@
1
+ #ifndef NUMEXPR_MODULE_HPP
2
+ #define NUMEXPR_MODULE_HPP
3
+
4
+ // Deal with the clunky numpy import mechanism
5
+ // by inverting the logic of the NO_IMPORT_ARRAY symbol.
6
+ #define PY_ARRAY_UNIQUE_SYMBOL numexpr_ARRAY_API
7
+ #ifndef DO_NUMPY_IMPORT_ARRAY
8
+ # define NO_IMPORT_ARRAY
9
+ #endif
10
+
11
+ #define NPY_NO_DEPRECATED_API NPY_API_VERSION
12
+
13
+ #include <Python.h>
14
+ #include <numpy/ndarrayobject.h>
15
+ #include <numpy/arrayscalars.h>
16
+
17
+ #include "numexpr_config.hpp"
18
+
19
+ struct global_state {
20
+ /* Global variables for threads */
21
+ int nthreads; /* number of desired threads in pool */
22
+ int init_threads_done; /* pool of threads initialized? */
23
+ int end_threads; /* should exisiting threads end? */
24
+ // pthread_t threads[MAX_THREADS]; /* opaque structure for threads */
25
+ // int tids[MAX_THREADS]; /* ID per each thread */
26
+ /* NOTE: threads and tids are arrays, they MUST be allocated to length
27
+ `global_max_threads` before module load. */
28
+ pthread_t *threads; /* opaque structure for threads */
29
+ int *tids; /* ID per each thread */
30
+ npy_intp gindex; /* global index for all threads */
31
+ int init_sentinels_done; /* sentinels initialized? */
32
+ int giveup; /* should parallel code giveup? */
33
+ int force_serial; /* force serial code instead of parallel? */
34
+ int pid; /* the PID for this process */
35
+
36
+ /* Synchronization variables for threadpool state */
37
+ pthread_mutex_t count_mutex;
38
+ int count_threads;
39
+ int barrier_passed; /* indicates if the thread pool's thread barrier
40
+ is unlocked and ready for the VM to process.*/
41
+ pthread_mutex_t count_threads_mutex;
42
+ pthread_cond_t count_threads_cv;
43
+
44
+ /* Mutual exclusion for access to global thread params (th_params) */
45
+ pthread_mutex_t parallel_mutex;
46
+
47
+ global_state() {
48
+ nthreads = 1;
49
+ init_threads_done = 0;
50
+ barrier_passed = 0;
51
+ end_threads = 0;
52
+ pid = 0;
53
+ }
54
+ };
55
+
56
+ extern global_state gs;
57
+
58
+ int numexpr_set_nthreads(int nthreads_new);
59
+
60
+ #endif // NUMEXPR_MODULE_HPP