polymorph-ai 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ """polymorph_ai: pick sequential / threading / multiprocessing automatically.
2
+
3
+ from polymorph_ai import adaptive_exec
4
+
5
+ @adaptive_exec
6
+ def process(item):
7
+ ...
8
+
9
+ results = process.map(items)
10
+ print(process.last_run)
11
+ """
12
+
13
+ from importlib.metadata import PackageNotFoundError, version
14
+
15
+ from .decorator import Decision, adaptive_exec, run, shutdown, warm_up
16
+ from .features import FEATURE_NAMES, extract_features
17
+
18
+ try:
19
+ __version__ = version("polymorph-ai")
20
+ except PackageNotFoundError: # running from a source checkout that is not installed
21
+ __version__ = "unknown"
22
+
23
+ __all__ = ["adaptive_exec", "run", "warm_up", "shutdown", "Decision",
24
+ "FEATURE_NAMES", "extract_features"]
@@ -0,0 +1,480 @@
1
+ """@adaptive_exec: run a per-item function over a list in the best mode.
2
+
3
+ from polymorph_ai import adaptive_exec
4
+
5
+ @adaptive_exec
6
+ def process(item):
7
+ ...
8
+
9
+ process(one_item) # a normal call, nothing changes
10
+ results = process.map(items) # polymorph_ai picks sequential / threading / multiprocessing
11
+ print(process.last_run) # what it picked and why
12
+
13
+ How one .map() call works:
14
+ 1. safety checks: empty / tiny list, one worker, nested call, can items and
15
+ the function be sent to other processes at all?
16
+ 2. probe: run the first items for ~50 ms and measure them (features.py, the
17
+ exact code used to collect the training data). Their results are kept.
18
+ 3. the MLP (model.py) turns the 13 features into a mode.
19
+ 4. the remaining items run in that mode on a pool that stays alive for the
20
+ next call (the model was trained on warm-pool timings).
21
+ """
22
+
23
+ import atexit
24
+ import functools
25
+ import importlib
26
+ import logging
27
+ import math
28
+ import multiprocessing
29
+ import os
30
+ import pickle
31
+ import sys
32
+ import threading
33
+ import time
34
+ from dataclasses import dataclass, field
35
+ from multiprocessing.pool import ThreadPool
36
+
37
+ try:
38
+ import cloudpickle # sends functions written in Jupyter/REPL to worker processes by value
39
+ except ImportError: # pragma: no cover
40
+ cloudpickle = None
41
+
42
+ from .features import PROBE_MIN_ITEMS, ast_features, extract_features, thread_cpu_time, usable_cpu_count
43
+ from .model import ModeModel
44
+
45
+ log = logging.getLogger("polymorph_ai")
46
+
47
+ MODES = ("sequential", "threading", "multiprocessing")
48
+ # Below this estimated sequential time (whole job, or what is left after the probe), skip the model:
49
+ # no pool can win back its own overhead (the training data starts at 10 ms).
50
+ SMALL_JOB_S = 0.01
51
+ # A cached decision is reused this many times, then the job is probed again.
52
+ CACHE_USES = 50
53
+ # Median time to start a process pool in our dataset (pool_startup_s):
54
+ # Windows (spawn) 0.25 s, Linux (fork/forkserver) 0.07 s. Used only when the pool is not running yet.
55
+ COLD_START_S = {"spawn": 0.25, "fork": 0.07, "forkserver": 0.07}
56
+
57
+
58
+ def _start_method():
59
+ # Read the method without fixing it, so the user can still call set_start_method() later.
60
+ # get_all_start_methods() lists the platform default first.
61
+ return multiprocessing.get_start_method(allow_none=True) or multiprocessing.get_all_start_methods()[0]
62
+
63
+
64
+ # ------------------------------------------------------------------- result
65
+
66
+ @dataclass
67
+ class Decision:
68
+ """What one .map() call did. `print(decision)` gives a one-line summary."""
69
+ function: str
70
+ mode: str
71
+ reason: str
72
+ n_items: int
73
+ n_workers: int
74
+ probabilities: dict = None # model output, None if the model was not asked
75
+ features: dict = None # the 13 features, None if not measured
76
+ probed_items: int = 0 # items already computed during the probe
77
+ overhead_s: float = 0.0 # time spent deciding that did no useful work
78
+ total_s: float = 0.0
79
+ notes: list = field(default_factory=list)
80
+
81
+ def __str__(self):
82
+ p = f" p={self.probabilities[self.mode]:.2f}" if self.probabilities else ""
83
+ return (f"[polymorph_ai] {self.function}: {self.n_items:,} items -> {self.mode}{p} "
84
+ f"({self.reason}) | overhead {self.overhead_s * 1e3:.1f} ms | total {self.total_s:.3f} s")
85
+
86
+
87
+ # -------------------------------------------------------------------- pools
88
+
89
+ _pools = {}
90
+ _pools_lock = threading.Lock()
91
+ _worker = threading.local() # .active is True inside our pool workers
92
+ _decisions = {} # cache key -> [mode, uses left]
93
+ _model = None
94
+ _model_lock = threading.Lock()
95
+
96
+
97
+ def _mark_worker():
98
+ _worker.active = True
99
+
100
+
101
+ class _ThreadPool(ThreadPool):
102
+ """ThreadPool that does not lock the program's default start method
103
+ (the stock one calls multiprocessing.get_context(), which does)."""
104
+
105
+ def __init__(self, processes, initializer):
106
+ multiprocessing.pool.Pool.__init__(self, processes, initializer,
107
+ context=multiprocessing.get_context(_start_method()))
108
+
109
+
110
+ def _get_pool(mode, n_workers):
111
+ with _pools_lock:
112
+ pool = _pools.get((mode, n_workers))
113
+ if pool is None:
114
+ if mode == "threading":
115
+ pool = _ThreadPool(n_workers, _mark_worker)
116
+ else: # explicit context: same reason as _ThreadPool
117
+ pool = multiprocessing.get_context(_start_method()).Pool(n_workers, initializer=_mark_worker)
118
+ _pools[(mode, n_workers)] = pool
119
+ _register_shutdown()
120
+ return pool
121
+
122
+
123
+ _shutdown_registered = False
124
+
125
+
126
+ def _register_shutdown():
127
+ # atexit runs the last registered function first. multiprocessing registers its own
128
+ # exit hook when the first pool starts; if that hook ran before ours, it would kill
129
+ # the pool's workers, the pool would start new ones, and those would be left running
130
+ # after the program (or Jupyter kernel) exits. Registering after the first pool
131
+ # exists makes our shutdown() run first.
132
+ global _shutdown_registered
133
+ if not _shutdown_registered:
134
+ atexit.register(shutdown)
135
+ _shutdown_registered = True
136
+
137
+
138
+ def _pool_is_warm(mode, n_workers):
139
+ return (mode, n_workers) in _pools
140
+
141
+
142
+ def warm_up(n_workers=None, modes=("threading", "multiprocessing")):
143
+ """Start the pools (and the Windows CPU-clock calibration) now, so the first
144
+ .map() call does not pay these one-off costs."""
145
+ thread_cpu_time()
146
+ _get_model()
147
+ n_workers = n_workers or usable_cpu_count()
148
+ for mode in modes:
149
+ _get_pool(mode, n_workers).map(int, range(n_workers), chunksize=1)
150
+
151
+
152
+ def shutdown():
153
+ """Stop all pools (called automatically when the program exits)."""
154
+ with _pools_lock:
155
+ for pool in _pools.values():
156
+ # pool.terminate() can hang while the interpreter is shutting down (seen in
157
+ # Jupyter kernels on Windows), which left the worker processes running after
158
+ # the kernel was gone. Give it a few seconds, then kill the workers directly.
159
+ workers = list(getattr(pool, "_pool", []))
160
+ stopper = threading.Thread(target=pool.terminate, daemon=True)
161
+ stopper.start()
162
+ stopper.join(timeout=3)
163
+ for p in workers + list(getattr(pool, "_pool", [])):
164
+ if hasattr(p, "kill") and p.is_alive():
165
+ p.kill()
166
+ _pools.clear()
167
+
168
+
169
+ def _get_model():
170
+ global _model
171
+ with _model_lock:
172
+ if _model is None:
173
+ _model = ModeModel()
174
+ return _model
175
+
176
+
177
+ # ------------------------------------------------- can we use processes?
178
+
179
+ def _find(module, qualname):
180
+ """Unpickle helper: look a decorated function up by name in a worker process."""
181
+ obj = importlib.import_module(module)
182
+ for part in qualname.split("."):
183
+ obj = getattr(obj, part)
184
+ return obj
185
+
186
+
187
+ def _defined_without_file(target):
188
+ """True for a function typed into Jupyter or a REPL: worker processes started
189
+ with spawn (Windows, macOS) cannot import it by name."""
190
+ module = sys.modules.get(getattr(target, "__module__", None))
191
+ return (module is not None and module.__name__ == "__main__" and not hasattr(module, "__file__")
192
+ and _start_method() != "fork")
193
+
194
+
195
+ def _process_problem(target, sample_item):
196
+ """None if target(item) can run in a worker process, else the reason why not."""
197
+ if multiprocessing.parent_process() is not None or getattr(_worker, "active", False):
198
+ return "already inside a worker"
199
+ qualname = getattr(target, "__qualname__", "")
200
+ if "<locals>" in qualname or "<lambda>" in qualname:
201
+ return "function is not defined at module top level"
202
+ module = sys.modules.get(getattr(target, "__module__", None))
203
+ if module is None:
204
+ return "function's module cannot be found"
205
+ if _defined_without_file(target):
206
+ if cloudpickle is None or not isinstance(target, AdaptiveExec):
207
+ return "function defined in Jupyter/REPL cannot be sent to worker processes"
208
+ try: # it travels by value: check that this really works
209
+ pickle.loads(pickle.dumps(target))
210
+ except Exception as e:
211
+ return f"function defined in Jupyter/REPL cannot be pickled ({type(e).__name__})"
212
+ else:
213
+ try:
214
+ found = _find(module.__name__, qualname)
215
+ except AttributeError:
216
+ found = None
217
+ if found is not target:
218
+ return "function cannot be found by its name"
219
+ try:
220
+ pickle.dumps(sample_item)
221
+ except Exception:
222
+ return "items cannot be pickled"
223
+ return None
224
+
225
+
226
+ def _pickled_size(obj):
227
+ """Bytes after pickling, or None if obj cannot be pickled."""
228
+ try:
229
+ return len(pickle.dumps(obj))
230
+ except Exception:
231
+ return None
232
+
233
+
234
+ def _rule_on_source(func):
235
+ """Fallback when the probe cannot run: threads if the code waits or calls GIL-free C code."""
236
+ static = ast_features(func)
237
+ return "threading" if static["has_io_call"] or static["has_gil_release_call"] else "sequential"
238
+
239
+
240
+ class _Recorder:
241
+ """Wraps the user function during the probe.
242
+
243
+ - keeps the outputs, so if probing fails half-way (an output that cannot be
244
+ pickled) the items already computed are not lost or run twice;
245
+ - keeps exceptions raised in the probe's helper threads, which Python would
246
+ otherwise only print and swallow.
247
+ It hashes/compares equal to the wrapped function, so the cached AST
248
+ features of that function are reused.
249
+ """
250
+
251
+ def __init__(self, func):
252
+ self.func = func
253
+ self.__wrapped__ = func # inspect.getsource() follows this
254
+ self.caller = threading.get_ident()
255
+ self.outputs = []
256
+ self.errors = []
257
+
258
+ def __hash__(self):
259
+ return hash(self.func)
260
+
261
+ def __eq__(self, other):
262
+ return other is self.func or (isinstance(other, _Recorder) and other.func is self.func)
263
+
264
+ def __call__(self, item):
265
+ in_caller = threading.get_ident() == self.caller
266
+ try:
267
+ out = self.func(item)
268
+ except BaseException as e:
269
+ self.errors.append(e)
270
+ if in_caller:
271
+ raise
272
+ return None
273
+ if in_caller:
274
+ self.outputs.append(out)
275
+ return out
276
+
277
+
278
+ # ----------------------------------------------------------------- the core
279
+
280
+ def _run_mode(mode, func, target, items, n_workers):
281
+ if not items:
282
+ return []
283
+ if mode == "sequential":
284
+ return [func(item) for item in items]
285
+ pool = _get_pool(mode, n_workers)
286
+ # default chunksize, same as in the data collection
287
+ return pool.map(target if mode == "multiprocessing" else func, items)
288
+
289
+
290
+ def _best_allowed(probabilities, allowed):
291
+ return max(allowed, key=lambda m: probabilities[m])
292
+
293
+
294
+ def run(func, items, n_workers=None, mode=None, cache=True, *, _target=None, _name=None):
295
+ """Compute [func(item) for item in items] in the best mode.
296
+
297
+ Returns (results, Decision). `mode` forces one mode (also settable for the
298
+ whole program with the environment variable POLYMORPH_MODE). With `cache`,
299
+ a later call that looks the same (similar item count, first-item time and
300
+ item size) reuses the decision instead of probing again.
301
+ """
302
+ t_start = time.perf_counter()
303
+ target = _target or func # what multiprocessing pickles
304
+ if not isinstance(items, (list, tuple)):
305
+ items = list(items)
306
+ n_workers = max(1, int(n_workers or usable_cpu_count()))
307
+ d = Decision(function=_name or getattr(func, "__qualname__", repr(func)),
308
+ mode="sequential", reason="", n_items=len(items), n_workers=n_workers)
309
+
310
+ def finish(results):
311
+ d.total_s = time.perf_counter() - t_start
312
+ log.debug("%s", d)
313
+ return results, d
314
+
315
+ forced = mode or os.environ.get("POLYMORPH_MODE")
316
+ if forced:
317
+ if forced not in MODES:
318
+ raise ValueError(f"mode must be one of {MODES}, got {forced!r}")
319
+ if forced == "multiprocessing" and items:
320
+ problem = _process_problem(target, items[0])
321
+ if problem:
322
+ raise ValueError(f"cannot use multiprocessing: {problem}")
323
+ d.mode, d.reason = forced, "forced"
324
+ return finish(list(_run_mode(forced, func, target, list(items), n_workers)))
325
+
326
+ # ---- 1. cases where measuring is pointless or impossible
327
+ if getattr(_worker, "active", False):
328
+ d.reason = "nested call inside a polymorph_ai worker"
329
+ return finish([func(item) for item in items])
330
+ if len(items) <= PROBE_MIN_ITEMS or n_workers == 1:
331
+ d.reason = "too few items" if n_workers > 1 else "only 1 worker"
332
+ return finish([func(item) for item in items])
333
+
334
+ allowed = list(MODES)
335
+ problem = _process_problem(target, items[0])
336
+ if problem:
337
+ allowed.remove("multiprocessing")
338
+ d.notes.append("no multiprocessing: " + problem)
339
+
340
+ item_bytes = _pickled_size(items[0])
341
+ if item_bytes is None:
342
+ # The probe measures pickling, so it cannot run. Fall back to the code itself.
343
+ d.mode, d.reason = _rule_on_source(func), "items cannot be pickled -> rule on the source code"
344
+ return finish(_run_mode(d.mode, func, target, list(items), n_workers))
345
+
346
+ # ---- 2. time the first item (useful work): is the whole job tiny?
347
+ t0 = time.perf_counter()
348
+ first = [func(items[0])]
349
+ first_s = time.perf_counter() - t0
350
+ if first_s * len(items) < SMALL_JOB_S:
351
+ d.reason, d.probed_items = "tiny job (from the first item's time)", 1
352
+ return finish(first + [func(item) for item in items[1:]])
353
+
354
+ # ---- 3. same kind of call as before? reuse that decision, skip the probe
355
+ key = (func, n_workers, round(math.log2(len(items))), round(math.log2(item_bytes + 1) / 2),
356
+ round(math.log2(max(first_s, 1e-7)))) if cache else None
357
+ cached = None
358
+ if key: # one timing can land just across a bucket edge, so accept the neighbouring buckets too
359
+ cached = next((_decisions[k] for k in (key, key[:-1] + (key[-1] - 1,), key[:-1] + (key[-1] + 1,))
360
+ if k in _decisions), None)
361
+ if cached and cached[1] > 0 and (cached[0] != "multiprocessing" or "multiprocessing" in allowed):
362
+ cached[1] -= 1
363
+ d.mode, d.reason, d.probed_items = cached[0], "same as an earlier similar call (cached)", 1
364
+ d.overhead_s = time.perf_counter() - t0 - first_s
365
+ return finish(first + _run_mode(d.mode, func, target, list(items[1:]), n_workers))
366
+
367
+ # ---- 4. probe the next items (they are really computed and kept)
368
+ ast_features(func) # cache under func itself, not the recorder
369
+ rec = _Recorder(func)
370
+ try:
371
+ features, done, _, extra = extract_features(rec, items[1:], n_workers)
372
+ except Exception:
373
+ if rec.errors: # the user's function raised: behave like a plain loop
374
+ raise
375
+ # an output could not be pickled: keep what was computed, finish without processes
376
+ done = first + rec.outputs
377
+ d.mode, d.reason = _rule_on_source(func), "results cannot be pickled -> rule on the source code"
378
+ d.probed_items = len(done)
379
+ return finish(done + _run_mode(d.mode, func, target, list(items[len(done):]), n_workers))
380
+ if rec.errors: # raised inside a thread-probe thread
381
+ raise rec.errors[0]
382
+
383
+ features["n_items"] = len(items) # the model is about the whole job
384
+ done = first + done
385
+ d.features, d.probed_items = features, len(done)
386
+ rest = list(items[len(done):])
387
+ t_decide = time.perf_counter()
388
+
389
+ # ---- 5. decide
390
+ if not rest:
391
+ d.reason = "probe already finished every item"
392
+ elif features["time_per_item"] * len(rest) < SMALL_JOB_S:
393
+ d.reason = "remaining work too small for any pool"
394
+ if key:
395
+ _decisions[key] = ["sequential", CACHE_USES]
396
+ else:
397
+ best, d.probabilities = _get_model().predict(features)
398
+ d.mode, d.reason = _best_allowed(d.probabilities, allowed), "model"
399
+ if d.mode != best:
400
+ d.reason = f"model ({best} not possible)"
401
+ if key: # cached before the start-up check: a repeated job is worth starting the pool for
402
+ _decisions[key] = [d.mode, CACHE_USES]
403
+ if d.mode == "multiprocessing" and not _pool_is_warm("multiprocessing", n_workers):
404
+ # The model learned on warm pools. If even a perfect speed-up would
405
+ # save less than starting the pool costs, do not start it for this call.
406
+ ideal_saving = features["time_per_item"] * len(rest) * (1 - 1 / min(n_workers, features["cpu_count"]))
407
+ if ideal_saving < COLD_START_S.get(_start_method(), 0.25):
408
+ d.mode = _best_allowed(d.probabilities, ["sequential", "threading"])
409
+ d.reason = "model said multiprocessing, but pool start-up would cost more than it saves"
410
+ d.overhead_s = extra + (time.perf_counter() - t_decide)
411
+
412
+ # ---- 6. run the rest
413
+ return finish(done + _run_mode(d.mode, func, target, rest, n_workers))
414
+
415
+
416
+ # ---------------------------------------------------------------- decorator
417
+
418
+ class AdaptiveExec:
419
+ """The object @adaptive_exec puts in place of your function."""
420
+
421
+ def __init__(self, func, n_workers=None, mode=None, cache=True, verbose=False):
422
+ functools.update_wrapper(self, func)
423
+ self.func = func
424
+ self.n_workers = n_workers
425
+ self.mode = mode
426
+ self.cache = cache
427
+ self.verbose = verbose
428
+ self.last_run = None
429
+
430
+ def __call__(self, item):
431
+ return self.func(item)
432
+
433
+ def map(self, items, n_workers=None, mode=None, cache=None):
434
+ """[self(item) for item in items], run in the mode polymorph_ai picks."""
435
+ results, self.last_run = run(self.func, items, n_workers or self.n_workers, mode or self.mode,
436
+ self.cache if cache is None else cache,
437
+ _target=self, _name=self.__qualname__)
438
+ if self.verbose:
439
+ print(self.last_run, file=sys.stderr)
440
+ return results
441
+
442
+ def __reduce__(self):
443
+ if _defined_without_file(self) and cloudpickle is not None:
444
+ # Written in Jupyter/REPL: there is no file to import it from, so send
445
+ # the code itself. Inside that cloudpickle call, a reference back to
446
+ # this wrapper (e.g. a recursive function calling itself by name) is
447
+ # rebuilt from the function, which cloudpickle has already memoised.
448
+ if getattr(_by_value, "active", False):
449
+ return _rewrap, (self.func, self.n_workers, self.mode, self.cache, self.verbose)
450
+ _by_value.active = True
451
+ try:
452
+ data = cloudpickle.dumps(self)
453
+ finally:
454
+ _by_value.active = False
455
+ return cloudpickle.loads, (data,)
456
+ # Worker processes import the module and look the function up by name,
457
+ # which finds this object again (pickling self.func by name would fail:
458
+ # its name now points at this wrapper).
459
+ return _find, (self.__module__, self.__qualname__)
460
+
461
+ def __repr__(self):
462
+ return f"<adaptive_exec {self.__module__}.{self.__qualname__}>"
463
+
464
+
465
+ _by_value = threading.local()
466
+
467
+
468
+ def _rewrap(func, n_workers, mode, cache, verbose):
469
+ return AdaptiveExec(func, n_workers, mode, cache, verbose)
470
+
471
+
472
+ def adaptive_exec(func=None, *, n_workers=None, mode=None, cache=True, verbose=False):
473
+ """Decorator. Use as @adaptive_exec or @adaptive_exec(n_workers=4, verbose=True).
474
+
475
+ n_workers: pool size (default: usable CPUs) mode: force one mode
476
+ cache: reuse decisions for similar calls verbose: print every decision
477
+ """
478
+ if func is None:
479
+ return lambda f: AdaptiveExec(f, n_workers, mode, cache, verbose)
480
+ return AdaptiveExec(func, n_workers, mode, cache, verbose)
@@ -0,0 +1,289 @@
1
+ """Feature extraction shared by the data collector and the decorator.
2
+
3
+ The same code must be used in both places, otherwise the model is trained on
4
+ features that the decorator measures differently at run time.
5
+
6
+ 13 features in 4 groups (see CLAUDE.md):
7
+ call-site : n_items, input_bytes, cpu_count, n_workers
8
+ ast : max_loop_depth, arith_op_count, has_io_call, has_gil_release_call
9
+ probe : time_per_item, cpu_ratio, time_cv, pickle_time_ratio
10
+ thread probe : thread_probe_speedup
11
+
12
+ A workload is "apply func to every item in a list", like map(func, items).
13
+ """
14
+
15
+ import ast
16
+ import functools
17
+ import inspect
18
+ import os
19
+ import pickle
20
+ import statistics
21
+ import textwrap
22
+ import threading
23
+ import time
24
+
25
+ FEATURE_NAMES = [
26
+ # call-site
27
+ "n_items", "input_bytes", "cpu_count", "n_workers",
28
+ # ast
29
+ "max_loop_depth", "arith_op_count", "has_io_call", "has_gil_release_call",
30
+ # probe
31
+ "time_per_item", "cpu_ratio", "time_cv", "pickle_time_ratio",
32
+ # thread probe
33
+ "thread_probe_speedup",
34
+ ]
35
+
36
+ # The probe runs real items one by one until it has used this much time
37
+ # (at least PROBE_MIN_ITEMS, at most a quarter of all items or PROBE_MAX_ITEMS). The outputs are
38
+ # kept, so the decorator does not waste this work.
39
+ PROBE_TARGET_S = 0.05
40
+ PROBE_MIN_ITEMS = 3
41
+ PROBE_MAX_ITEMS = 2000 # tiny items: more adds bookkeeping, not information
42
+ PICKLE_SAMPLE = 20 # items used to measure the pickling cost
43
+
44
+
45
+ # ---------------------------------------------------------------- cpu clock
46
+ # CPU time used by the current thread, in seconds.
47
+ # Linux: time.thread_time() is precise. Windows: thread_time()/process_time()
48
+ # only tick every 15.6 ms, useless for a probe of a few ms. Instead we read
49
+ # the thread's CPU cycle counter (QueryThreadCycleTime) and convert cycles to
50
+ # seconds with a rate measured once by a short busy loop.
51
+
52
+ if os.name == "nt":
53
+ import ctypes
54
+ from ctypes import wintypes
55
+
56
+ _kernel32 = ctypes.WinDLL("kernel32")
57
+ _kernel32.GetCurrentThread.restype = wintypes.HANDLE
58
+ _kernel32.QueryThreadCycleTime.argtypes = [wintypes.HANDLE,
59
+ ctypes.POINTER(ctypes.c_ulonglong)]
60
+
61
+ def _thread_cycles():
62
+ cycles = ctypes.c_ulonglong()
63
+ _kernel32.QueryThreadCycleTime(_kernel32.GetCurrentThread(), ctypes.byref(cycles))
64
+ return cycles.value
65
+
66
+ @functools.lru_cache(maxsize=1)
67
+ def _cycles_per_second():
68
+ # Busy loop: the thread uses 100% CPU, so cycles / wall = rate.
69
+ c0, t0 = _thread_cycles(), time.perf_counter()
70
+ while time.perf_counter() - t0 < 0.02:
71
+ pass
72
+ return (_thread_cycles() - c0) / (time.perf_counter() - t0)
73
+
74
+ def thread_cpu_time():
75
+ rate = _cycles_per_second() # calibrate first, outside the reading
76
+ return _thread_cycles() / rate
77
+ else:
78
+ thread_cpu_time = time.thread_time
79
+
80
+
81
+ # ---------------------------------------------------------------- call-site
82
+
83
+ def usable_cpu_count():
84
+ """Number of CPUs this process may actually use (not just installed)."""
85
+ if hasattr(os, "process_cpu_count"): # Python 3.13+
86
+ n = os.process_cpu_count()
87
+ elif hasattr(os, "sched_getaffinity"): # Linux
88
+ n = len(os.sched_getaffinity(0))
89
+ else: # Windows, older Python
90
+ n = os.cpu_count()
91
+ return n or 1
92
+
93
+
94
+ # ---------------------------------------------------------------------- ast
95
+
96
+ # Calls that mean "waits for I/O". Matched against the dotted call name,
97
+ # e.g. `time.sleep(...)` -> "time.sleep".
98
+ IO_CALLS = {"open", "print", "input", "time.sleep", "os.read", "os.write",
99
+ "os.fsync", "asyncio.sleep"}
100
+ IO_PREFIXES = ("socket.", "requests.", "urllib.", "http.", "subprocess.",
101
+ "shutil.", "ftplib.", "smtplib.", "sqlite3.")
102
+ # Method names that are almost always file/socket I/O: f.read(), s.recv() ...
103
+ IO_METHODS = {"read", "write", "readline", "readlines", "recv", "send",
104
+ "sendall", "flush"}
105
+
106
+ # C functions that do heavy work and release the GIL, so threads can run
107
+ # them in parallel. Only functions we have checked; plain `math` is not here.
108
+ GIL_RELEASE_CALLS = {"np.dot", "numpy.dot", "np.matmul", "numpy.matmul",
109
+ "np.sort", "numpy.sort"}
110
+ GIL_RELEASE_PREFIXES = ("zlib.", "bz2.", "lzma.", "hashlib.",
111
+ "np.linalg.", "numpy.linalg.", "np.fft.", "numpy.fft.")
112
+
113
+ LOOP_NODES = (ast.For, ast.AsyncFor, ast.While)
114
+ COMPREHENSIONS = (ast.ListComp, ast.SetComp, ast.DictComp, ast.GeneratorExp)
115
+
116
+
117
+ def _dotted_name(node):
118
+ """`zlib.compress` -> "zlib.compress", `f.read` -> "?.read" if f is complex."""
119
+ parts = []
120
+ while isinstance(node, ast.Attribute):
121
+ parts.append(node.attr)
122
+ node = node.value
123
+ parts.append(node.id if isinstance(node, ast.Name) else "?")
124
+ return ".".join(reversed(parts))
125
+
126
+
127
+ def _loop_depth(node, depth=0):
128
+ if isinstance(node, LOOP_NODES):
129
+ depth += 1
130
+ elif isinstance(node, COMPREHENSIONS):
131
+ depth += len(node.generators)
132
+ child_depths = [_loop_depth(child, depth) for child in ast.iter_child_nodes(node)]
133
+ return max([depth] + child_depths)
134
+
135
+
136
+ @functools.lru_cache(maxsize=256)
137
+ def ast_features(func):
138
+ """Static features read from the function's source code (cached per function).
139
+
140
+ Only the function's own body is inspected, not the functions it calls.
141
+ """
142
+ try:
143
+ source = textwrap.dedent(inspect.getsource(func))
144
+ tree = ast.parse(source)
145
+ except (OSError, TypeError, SyntaxError):
146
+ # No source available (builtin, lambda in REPL, ...): neutral values.
147
+ return {"max_loop_depth": 0, "arith_op_count": 0,
148
+ "has_io_call": 0, "has_gil_release_call": 0}
149
+
150
+ arith = 0
151
+ has_io = False
152
+ has_gil_release = False
153
+ for node in ast.walk(tree):
154
+ if isinstance(node, (ast.BinOp, ast.AugAssign)):
155
+ arith += 1
156
+ if isinstance(node.op, ast.MatMult): # a @ b on numpy arrays
157
+ has_gil_release = True
158
+ elif isinstance(node, ast.Call):
159
+ name = _dotted_name(node.func)
160
+ method = name.rsplit(".", 1)[-1]
161
+ if (name in IO_CALLS or name.startswith(IO_PREFIXES)
162
+ or ("." in name and method in IO_METHODS)):
163
+ has_io = True
164
+ if name in GIL_RELEASE_CALLS or name.startswith(GIL_RELEASE_PREFIXES):
165
+ has_gil_release = True
166
+
167
+ return {"max_loop_depth": _loop_depth(tree), "arith_op_count": arith,
168
+ "has_io_call": int(has_io), "has_gil_release_call": int(has_gil_release)}
169
+
170
+
171
+ # -------------------------------------------------------------------- probe
172
+
173
+ def probe(func, items):
174
+ """Run the first few real items sequentially and measure them.
175
+
176
+ Returns (features, results) where results are the outputs of the probed
177
+ items, so the decorator can reuse them instead of computing them twice.
178
+ """
179
+ max_items = min(len(items), PROBE_MAX_ITEMS, max(PROBE_MIN_ITEMS, len(items) // 4))
180
+ item_times, results, pickle_times, input_sizes = [], [], [], []
181
+
182
+ cpu_start = thread_cpu_time()
183
+ wall_start = time.perf_counter()
184
+ for item in items[:max_items]:
185
+ t0 = time.perf_counter()
186
+ out = func(item)
187
+ item_times.append(time.perf_counter() - t0)
188
+ results.append(out)
189
+ if (len(item_times) >= PROBE_MIN_ITEMS
190
+ and time.perf_counter() - wall_start >= PROBE_TARGET_S):
191
+ break
192
+ wall = time.perf_counter() - wall_start
193
+ cpu = thread_cpu_time() - cpu_start
194
+
195
+ # What multiprocessing would pay per item: send input, get output back.
196
+ # Measured outside the timed loop above so it does not inflate cpu_ratio.
197
+ # A spread-out sample is enough for the mean; tiny items can mean
198
+ # tens of thousands of probed items.
199
+ step = max(1, len(results) // PICKLE_SAMPLE)
200
+ for item, out in list(zip(items, results))[::step][:PICKLE_SAMPLE]:
201
+ t0 = time.perf_counter()
202
+ data = pickle.dumps(item)
203
+ pickle.loads(data)
204
+ pickle.loads(pickle.dumps(out))
205
+ pickle_times.append(time.perf_counter() - t0)
206
+ input_sizes.append(len(data))
207
+
208
+ time_per_item = statistics.mean(item_times)
209
+ features = {
210
+ "input_bytes": statistics.mean(input_sizes),
211
+ "time_per_item": time_per_item,
212
+ "cpu_ratio": min(cpu / wall, 1.0) if wall > 0 else 0.0,
213
+ "time_cv": (statistics.pstdev(item_times) / time_per_item
214
+ if time_per_item > 0 else 0.0),
215
+ "pickle_time_ratio": (statistics.mean(pickle_times) / time_per_item
216
+ if time_per_item > 0 else 0.0),
217
+ }
218
+ return features, results, item_times
219
+
220
+
221
+ def thread_probe(func, items, time_per_item):
222
+ """Run the NEXT items (not the probed ones again) on 2 threads.
223
+
224
+ Compares their wall time with what they would take sequentially
225
+ (len(items) * time_per_item from the probe):
226
+ ~2.0 means threads really run in parallel (I/O or GIL-releasing code),
227
+ ~1.0 or less means the GIL serialises them.
228
+ The outputs are kept, so this work is not wasted. Items are never run
229
+ twice, so functions with side effects (writing files, ...) stay correct.
230
+
231
+ Returns (speedup, results). With fewer than 2 items left: (1.0, []).
232
+ """
233
+ if len(items) < 2:
234
+ return 1.0, []
235
+ results = [None] * len(items)
236
+
237
+ def work(indexes):
238
+ for i in indexes:
239
+ results[i] = func(items[i])
240
+
241
+ halves = [range(0, len(items), 2), range(1, len(items), 2)]
242
+ threads = [threading.Thread(target=work, args=(h,)) for h in halves]
243
+ t0 = time.perf_counter()
244
+ for t in threads:
245
+ t.start()
246
+ for t in threads:
247
+ t.join()
248
+ wall = time.perf_counter() - t0
249
+ speedup = len(items) * time_per_item / wall if wall > 0 else 1.0
250
+ return speedup, results
251
+
252
+
253
+ # ---------------------------------------------------------------------- all
254
+
255
+ def extract_features(func, items, n_workers):
256
+ """All 13 features for running func over items with n_workers workers.
257
+
258
+ Returns (features, probe_results, overhead_s, extra_s).
259
+ probe_results are the outputs of the first len(probe_results) items.
260
+ overhead_s is the total time spent here; extra_s is the part that did
261
+ no useful work (pickle test, AST, bookkeeping), i.e. overhead_s minus
262
+ the time spent computing real items.
263
+ """
264
+ t0 = time.perf_counter()
265
+ probe_feats, results, item_times = probe(func, items)
266
+ t_probe = time.perf_counter()
267
+ # Skip the first probed item when estimating sequential speed: it may
268
+ # include one-off warm-up.
269
+ steady = item_times[1:] if len(item_times) >= 3 else item_times
270
+ next_items = items[len(results):len(results) + len(results)]
271
+ speedup, thread_results = thread_probe(func, next_items, statistics.mean(steady))
272
+ t_thread = time.perf_counter()
273
+ results = results + thread_results
274
+
275
+ features = {
276
+ "n_items": len(items),
277
+ "input_bytes": probe_feats["input_bytes"],
278
+ "cpu_count": usable_cpu_count(),
279
+ "n_workers": n_workers,
280
+ **ast_features(func),
281
+ "time_per_item": probe_feats["time_per_item"],
282
+ "cpu_ratio": probe_feats["cpu_ratio"],
283
+ "time_cv": probe_feats["time_cv"],
284
+ "pickle_time_ratio": probe_feats["pickle_time_ratio"],
285
+ "thread_probe_speedup": speedup,
286
+ }
287
+ overhead = time.perf_counter() - t0
288
+ useful = sum(item_times) + (t_thread - t_probe if thread_results else 0.0)
289
+ return features, results, overhead, max(overhead - useful, 0.0)
polymorph_ai/model.npz ADDED
Binary file
polymorph_ai/model.py ADDED
@@ -0,0 +1,48 @@
1
+ """The trained mode-selection MLP, run with plain numpy.
2
+
3
+ The weights come from notebooks/04_export_model.ipynb (Keras, trained on the
4
+ collected dataset). Re-implementing the forward pass here keeps the library's
5
+ only dependency at numpy: no Keras/JAX import (seconds) at run time, and one
6
+ decision takes microseconds.
7
+
8
+ Steps, identical to training:
9
+ log10 of the heavy-tailed features -> standardise -> Dense+ReLU ... -> Dense -> softmax
10
+ """
11
+
12
+ import json
13
+ import os
14
+
15
+ import numpy as np
16
+
17
+ DEFAULT_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "model.npz")
18
+
19
+
20
+ class ModeModel:
21
+ def __init__(self, path=DEFAULT_PATH):
22
+ data = np.load(path)
23
+ self.feature_names = [str(f) for f in data["feature_names"]]
24
+ self.modes = [str(m) for m in data["modes"]]
25
+ self.meta = json.loads(str(data["meta"]))
26
+ log_features = {str(f) for f in data["log_features"]}
27
+ self._is_log = np.array([f in log_features for f in self.feature_names])
28
+ self._mean = data["scaler_mean"]
29
+ self._scale = data["scaler_scale"]
30
+ n = int(data["n_layers"])
31
+ self._layers = [(data[f"w{2 * i}"], data[f"w{2 * i + 1}"]) for i in range(n)]
32
+
33
+ def predict_proba(self, features):
34
+ """features: dict with the 13 feature names -> probabilities in self.modes order."""
35
+ x = np.array([float(features[f]) for f in self.feature_names])
36
+ x[self._is_log] = np.log10(np.maximum(x[self._is_log], 1e-9))
37
+ h = (x - self._mean) / self._scale
38
+ for i, (w, b) in enumerate(self._layers):
39
+ h = h @ w + b
40
+ if i < len(self._layers) - 1:
41
+ h = np.maximum(h, 0.0) # ReLU (dropout is off at inference)
42
+ e = np.exp(h - h.max())
43
+ return e / e.sum() # softmax
44
+
45
+ def predict(self, features):
46
+ """Most likely mode and the probability of every mode (dict)."""
47
+ p = self.predict_proba(features)
48
+ return self.modes[int(p.argmax())], dict(zip(self.modes, p.tolist()))
@@ -0,0 +1,128 @@
1
+ Metadata-Version: 2.4
2
+ Name: polymorph-ai
3
+ Version: 1.1.0
4
+ Summary: A decorator that uses a trained neural network to run your function sequentially, with threads, or with processes - whichever is fastest.
5
+ Author: Worachat Songmuangnu
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/Worachat-Songmuangnu/polymorph-ai
8
+ Project-URL: Source, https://github.com/Worachat-Songmuangnu/polymorph-ai
9
+ Project-URL: Issues, https://github.com/Worachat-Songmuangnu/polymorph-ai/issues
10
+ Keywords: parallel,multiprocessing,threading,decorator,machine-learning,performance
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Operating System :: Microsoft :: Windows
14
+ Classifier: Operating System :: POSIX :: Linux
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Software Development :: Libraries
17
+ Classifier: Topic :: System :: Distributed Computing
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: numpy
22
+ Requires-Dist: cloudpickle
23
+ Dynamic: license-file
24
+
25
+ # polymorph-ai
26
+
27
+ `@adaptive_exec` runs a function over a list of items **sequentially, with threads, or with processes**, and a small trained neural network picks whichever mode should be fastest for that job on that machine.
28
+
29
+ ```python
30
+ from polymorph_ai import adaptive_exec
31
+
32
+ @adaptive_exec
33
+ def process(item):
34
+ ...
35
+
36
+ if __name__ == "__main__": # required on Windows/macOS for processes
37
+ results = process.map(items) # same as [process(x) for x in items], in order
38
+ print(process.last_run)
39
+ # [polymorph_ai] process: 5,000 items -> multiprocessing p=0.97 (model) | overhead 0.7 ms | total 1.84 s
40
+ ```
41
+
42
+ `process(x)` still works as a normal function call. Only `.map()` adds the automatic part.
43
+
44
+ ## Install
45
+
46
+ ```bash
47
+ pip install polymorph-ai
48
+ ```
49
+
50
+ Dependencies: numpy and cloudpickle. Tested on Linux with Python 3.10, 3.12 and 3.14, and on Windows with Python 3.14.
51
+
52
+ Works in Jupyter too: a function written in a notebook cell is sent to the worker processes by value (with cloudpickle), so it can still use multiprocessing on Windows and macOS.
53
+
54
+ ## How it decides
55
+
56
+ Every `.map(items)` call goes through these steps:
57
+
58
+ 1. **Safety checks.** Empty or very short lists, one worker, or a call nested inside another polymorph_ai worker all just run as a plain loop. If the function or the items cannot be sent to another process (a lambda, a function defined inside another function, unpicklable items), multiprocessing is ruled out.
59
+ 2. **First item.** The first item runs and is timed. If `time × number of items` is under 10 ms, the job is too small for any pool to pay off, so the rest runs as a plain loop.
60
+ 3. **Cache.** A call that looks like an earlier one (similar number of items, first-item time, and item size) reuses that decision and skips the probe.
61
+ 4. **Probe.** The next items run for about 50 ms, first one by one and then on 2 threads, to measure the 13 features: call site, static code analysis (AST), runtime probe, and thread probe. The probe uses the same code (`polymorph_ai/features.py`) that collected the training data. **Probed items are real work: their results are kept and no item ever runs twice.**
62
+ 5. **Model.** An MLP (13 → 225 ReLU → 3 softmax, 3,828 weights) turns the features into probabilities. It is trained in Keras and runs here in numpy, at about 30 µs per decision.
63
+ 6. **Run.** The remaining items run in the chosen mode on a pool that stays alive for the next call.
64
+
65
+ ## Options
66
+
67
+ ```python
68
+ @adaptive_exec(n_workers=4, verbose=True, cache=True) # verbose prints every decision
69
+ def process(item): ...
70
+
71
+ process.map(items, mode="threading") # force one mode for this call
72
+ # POLYMORPH_MODE=sequential python app.py # force one mode for the whole program (A/B testing, debugging)
73
+
74
+ from polymorph_ai import run, warm_up
75
+ results, decision = run(some_function, items) # for functions you cannot decorate
76
+ warm_up() # start the pools early (e.g. at server start-up)
77
+ ```
78
+
79
+ `process.last_run` (a `Decision`) records the chosen mode, why it was chosen, the model's probabilities, the 13 features, how many items the probe computed, and the overhead.
80
+
81
+ ## Results
82
+
83
+ **Model** (13,603 jobs collected on 16 machine settings, see [`dataset/`](https://github.com/Worachat-Songmuangnu/polymorph-ai/tree/main/dataset), GroupKFold by workload template, so every test job comes from code the model never trained on). Time lost compared with always picking the fastest mode:
84
+
85
+ - Always multiprocessing: +22.1%
86
+ - if-else rules: +13.0%
87
+ - Random Forest: +5.8%
88
+ - **MLP: +4.3%**
89
+
90
+ **Library** ([`examples/demo.py`](https://github.com/Worachat-Songmuangnu/polymorph-ai/blob/main/examples/demo.py): 8 everyday jobs that are not in the training data, 8 CPUs). Total time compared with perfect picks:
91
+
92
+ | | Windows | Linux (WSL) |
93
+ |---|---|---|
94
+ | always sequential | +349% | +426% |
95
+ | always threading | +117% | +138% |
96
+ | always multiprocessing | +19% | +30% |
97
+ | polymorph_ai, first call | +52% | +66% |
98
+ | polymorph_ai, repeated call | **+13%** | **+24%** |
99
+
100
+ **When does it pay off?** The first call on a job pays about 0.2 s for the probe (`python examples/demo.py --overhead`).
101
+
102
+ - Compared with a plain loop, polymorph_ai is already faster on jobs from about 0.3–0.6 s.
103
+ - Compared with a perfect choice of mode, the first call is within 15% once the job takes 5 s or more.
104
+ - Repeated calls skip the probe, so they cost almost nothing.
105
+
106
+ ## Limitations
107
+
108
+ - The decorated function takes **one item** and is applied to every item (`map`). polymorph_ai cannot parallelise code that does not have this shape.
109
+ - On Windows and macOS, the script that uses processes needs the `if __name__ == "__main__":` guard (a Python rule for processes, not specific to polymorph_ai).
110
+ - The first item and the probe items run in the calling thread, so a job with a few very long items loses one item's worth of parallel time.
111
+ - The model was trained on synthetic workloads from 13 families. Code that behaves unlike all of them can still be mispredicted. In the demo, sorting 20k-number lists stays sequential when processes would be twice as fast.
112
+ - asyncio is not supported: a decorator cannot turn ordinary code into `async def` code.
113
+ - In Jupyter on Windows, call `shutdown()` before you close or restart the kernel. If the kernel is stopped straight after the cell that started the process pool, the worker processes can be left running.
114
+
115
+ ## Tests
116
+
117
+ ```bash
118
+ python -m pytest tests -v # 28 tests, pass on Windows and Linux
119
+ ```
120
+
121
+ ## Project layout
122
+
123
+ ```
124
+ polymorph_ai/ the library: decorator.py, features.py, model.py + model.npz (the trained network)
125
+ tests/ pytest suite
126
+ examples/ demo.py (benchmark vs fixed modes) and its results/
127
+ dataset/ the training dataset (13,603 jobs) and the real-code results (92 jobs)
128
+ ```
@@ -0,0 +1,10 @@
1
+ polymorph_ai/__init__.py,sha256=1aFA5RwshexIWECw1ihJu1b7A9FWZKvZCV06iF5q6lM,738
2
+ polymorph_ai/decorator.py,sha256=zB5MnztiSfOOWXYmwXBbrAgWSlKD2g_ziR2gSngr_ig,19980
3
+ polymorph_ai/features.py,sha256=JgSjN6FW6elrlYmwXNmvj_VavwxT6aty7MeBIFHOpFw,11818
4
+ polymorph_ai/model.npz,sha256=PYXE0nZhIMtUiSw5JjXBhk-rPuWl2I_qpclF5ITfJRg,21120
5
+ polymorph_ai/model.py,sha256=GrOTV-TpepgZlaDamEDiLpWEiUi0-iT0XdZTg1OEOKw,2089
6
+ polymorph_ai-1.1.0.dist-info/licenses/LICENSE,sha256=H0aLa4UfHVK6fzZHZ_aEMtgywY18sh6wa71_q0H4thg,1081
7
+ polymorph_ai-1.1.0.dist-info/METADATA,sha256=I1DTMUvZb-tR2eV-FvGCkJzuafExt_14JGIRDyP8fHg,6998
8
+ polymorph_ai-1.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
9
+ polymorph_ai-1.1.0.dist-info/top_level.txt,sha256=5F0pgkilc2sBAikhdTIpRrxPIxDMS2D1Ae9qrLO6K5A,13
10
+ polymorph_ai-1.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 the polymorph-ai authors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ polymorph_ai