polymorph-ai 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- polymorph_ai/__init__.py +24 -0
- polymorph_ai/decorator.py +480 -0
- polymorph_ai/features.py +289 -0
- polymorph_ai/model.npz +0 -0
- polymorph_ai/model.py +48 -0
- polymorph_ai-1.1.0.dist-info/METADATA +128 -0
- polymorph_ai-1.1.0.dist-info/RECORD +10 -0
- polymorph_ai-1.1.0.dist-info/WHEEL +5 -0
- polymorph_ai-1.1.0.dist-info/licenses/LICENSE +21 -0
- polymorph_ai-1.1.0.dist-info/top_level.txt +1 -0
polymorph_ai/__init__.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""polymorph_ai: pick sequential / threading / multiprocessing automatically.
|
|
2
|
+
|
|
3
|
+
from polymorph_ai import adaptive_exec
|
|
4
|
+
|
|
5
|
+
@adaptive_exec
|
|
6
|
+
def process(item):
|
|
7
|
+
...
|
|
8
|
+
|
|
9
|
+
results = process.map(items)
|
|
10
|
+
print(process.last_run)
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
14
|
+
|
|
15
|
+
from .decorator import Decision, adaptive_exec, run, shutdown, warm_up
|
|
16
|
+
from .features import FEATURE_NAMES, extract_features
|
|
17
|
+
|
|
18
|
+
try:
|
|
19
|
+
__version__ = version("polymorph-ai")
|
|
20
|
+
except PackageNotFoundError: # running from a source checkout that is not installed
|
|
21
|
+
__version__ = "unknown"
|
|
22
|
+
|
|
23
|
+
__all__ = ["adaptive_exec", "run", "warm_up", "shutdown", "Decision",
|
|
24
|
+
"FEATURE_NAMES", "extract_features"]
|
|
@@ -0,0 +1,480 @@
|
|
|
1
|
+
"""@adaptive_exec: run a per-item function over a list in the best mode.
|
|
2
|
+
|
|
3
|
+
from polymorph_ai import adaptive_exec
|
|
4
|
+
|
|
5
|
+
@adaptive_exec
|
|
6
|
+
def process(item):
|
|
7
|
+
...
|
|
8
|
+
|
|
9
|
+
process(one_item) # a normal call, nothing changes
|
|
10
|
+
results = process.map(items) # polymorph_ai picks sequential / threading / multiprocessing
|
|
11
|
+
print(process.last_run) # what it picked and why
|
|
12
|
+
|
|
13
|
+
How one .map() call works:
|
|
14
|
+
1. safety checks: empty / tiny list, one worker, nested call, can items and
|
|
15
|
+
the function be sent to other processes at all?
|
|
16
|
+
2. probe: run the first items for ~50 ms and measure them (features.py, the
|
|
17
|
+
exact code used to collect the training data). Their results are kept.
|
|
18
|
+
3. the MLP (model.py) turns the 13 features into a mode.
|
|
19
|
+
4. the remaining items run in that mode on a pool that stays alive for the
|
|
20
|
+
next call (the model was trained on warm-pool timings).
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
import atexit
|
|
24
|
+
import functools
|
|
25
|
+
import importlib
|
|
26
|
+
import logging
|
|
27
|
+
import math
|
|
28
|
+
import multiprocessing
|
|
29
|
+
import os
|
|
30
|
+
import pickle
|
|
31
|
+
import sys
|
|
32
|
+
import threading
|
|
33
|
+
import time
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from multiprocessing.pool import ThreadPool
|
|
36
|
+
|
|
37
|
+
try:
|
|
38
|
+
import cloudpickle # sends functions written in Jupyter/REPL to worker processes by value
|
|
39
|
+
except ImportError: # pragma: no cover
|
|
40
|
+
cloudpickle = None
|
|
41
|
+
|
|
42
|
+
from .features import PROBE_MIN_ITEMS, ast_features, extract_features, thread_cpu_time, usable_cpu_count
|
|
43
|
+
from .model import ModeModel
|
|
44
|
+
|
|
45
|
+
log = logging.getLogger("polymorph_ai")
|
|
46
|
+
|
|
47
|
+
MODES = ("sequential", "threading", "multiprocessing")
|
|
48
|
+
# Below this estimated sequential time (whole job, or what is left after the probe), skip the model:
|
|
49
|
+
# no pool can win back its own overhead (the training data starts at 10 ms).
|
|
50
|
+
SMALL_JOB_S = 0.01
|
|
51
|
+
# A cached decision is reused this many times, then the job is probed again.
|
|
52
|
+
CACHE_USES = 50
|
|
53
|
+
# Median time to start a process pool in our dataset (pool_startup_s):
|
|
54
|
+
# Windows (spawn) 0.25 s, Linux (fork/forkserver) 0.07 s. Used only when the pool is not running yet.
|
|
55
|
+
COLD_START_S = {"spawn": 0.25, "fork": 0.07, "forkserver": 0.07}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _start_method():
|
|
59
|
+
# Read the method without fixing it, so the user can still call set_start_method() later.
|
|
60
|
+
# get_all_start_methods() lists the platform default first.
|
|
61
|
+
return multiprocessing.get_start_method(allow_none=True) or multiprocessing.get_all_start_methods()[0]
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ------------------------------------------------------------------- result
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class Decision:
|
|
68
|
+
"""What one .map() call did. `print(decision)` gives a one-line summary."""
|
|
69
|
+
function: str
|
|
70
|
+
mode: str
|
|
71
|
+
reason: str
|
|
72
|
+
n_items: int
|
|
73
|
+
n_workers: int
|
|
74
|
+
probabilities: dict = None # model output, None if the model was not asked
|
|
75
|
+
features: dict = None # the 13 features, None if not measured
|
|
76
|
+
probed_items: int = 0 # items already computed during the probe
|
|
77
|
+
overhead_s: float = 0.0 # time spent deciding that did no useful work
|
|
78
|
+
total_s: float = 0.0
|
|
79
|
+
notes: list = field(default_factory=list)
|
|
80
|
+
|
|
81
|
+
def __str__(self):
|
|
82
|
+
p = f" p={self.probabilities[self.mode]:.2f}" if self.probabilities else ""
|
|
83
|
+
return (f"[polymorph_ai] {self.function}: {self.n_items:,} items -> {self.mode}{p} "
|
|
84
|
+
f"({self.reason}) | overhead {self.overhead_s * 1e3:.1f} ms | total {self.total_s:.3f} s")
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# -------------------------------------------------------------------- pools
|
|
88
|
+
|
|
89
|
+
_pools = {}
|
|
90
|
+
_pools_lock = threading.Lock()
|
|
91
|
+
_worker = threading.local() # .active is True inside our pool workers
|
|
92
|
+
_decisions = {} # cache key -> [mode, uses left]
|
|
93
|
+
_model = None
|
|
94
|
+
_model_lock = threading.Lock()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _mark_worker():
|
|
98
|
+
_worker.active = True
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
class _ThreadPool(ThreadPool):
|
|
102
|
+
"""ThreadPool that does not lock the program's default start method
|
|
103
|
+
(the stock one calls multiprocessing.get_context(), which does)."""
|
|
104
|
+
|
|
105
|
+
def __init__(self, processes, initializer):
|
|
106
|
+
multiprocessing.pool.Pool.__init__(self, processes, initializer,
|
|
107
|
+
context=multiprocessing.get_context(_start_method()))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _get_pool(mode, n_workers):
|
|
111
|
+
with _pools_lock:
|
|
112
|
+
pool = _pools.get((mode, n_workers))
|
|
113
|
+
if pool is None:
|
|
114
|
+
if mode == "threading":
|
|
115
|
+
pool = _ThreadPool(n_workers, _mark_worker)
|
|
116
|
+
else: # explicit context: same reason as _ThreadPool
|
|
117
|
+
pool = multiprocessing.get_context(_start_method()).Pool(n_workers, initializer=_mark_worker)
|
|
118
|
+
_pools[(mode, n_workers)] = pool
|
|
119
|
+
_register_shutdown()
|
|
120
|
+
return pool
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
_shutdown_registered = False
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _register_shutdown():
|
|
127
|
+
# atexit runs the last registered function first. multiprocessing registers its own
|
|
128
|
+
# exit hook when the first pool starts; if that hook ran before ours, it would kill
|
|
129
|
+
# the pool's workers, the pool would start new ones, and those would be left running
|
|
130
|
+
# after the program (or Jupyter kernel) exits. Registering after the first pool
|
|
131
|
+
# exists makes our shutdown() run first.
|
|
132
|
+
global _shutdown_registered
|
|
133
|
+
if not _shutdown_registered:
|
|
134
|
+
atexit.register(shutdown)
|
|
135
|
+
_shutdown_registered = True
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _pool_is_warm(mode, n_workers):
|
|
139
|
+
return (mode, n_workers) in _pools
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def warm_up(n_workers=None, modes=("threading", "multiprocessing")):
|
|
143
|
+
"""Start the pools (and the Windows CPU-clock calibration) now, so the first
|
|
144
|
+
.map() call does not pay these one-off costs."""
|
|
145
|
+
thread_cpu_time()
|
|
146
|
+
_get_model()
|
|
147
|
+
n_workers = n_workers or usable_cpu_count()
|
|
148
|
+
for mode in modes:
|
|
149
|
+
_get_pool(mode, n_workers).map(int, range(n_workers), chunksize=1)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def shutdown():
|
|
153
|
+
"""Stop all pools (called automatically when the program exits)."""
|
|
154
|
+
with _pools_lock:
|
|
155
|
+
for pool in _pools.values():
|
|
156
|
+
# pool.terminate() can hang while the interpreter is shutting down (seen in
|
|
157
|
+
# Jupyter kernels on Windows), which left the worker processes running after
|
|
158
|
+
# the kernel was gone. Give it a few seconds, then kill the workers directly.
|
|
159
|
+
workers = list(getattr(pool, "_pool", []))
|
|
160
|
+
stopper = threading.Thread(target=pool.terminate, daemon=True)
|
|
161
|
+
stopper.start()
|
|
162
|
+
stopper.join(timeout=3)
|
|
163
|
+
for p in workers + list(getattr(pool, "_pool", [])):
|
|
164
|
+
if hasattr(p, "kill") and p.is_alive():
|
|
165
|
+
p.kill()
|
|
166
|
+
_pools.clear()
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _get_model():
|
|
170
|
+
global _model
|
|
171
|
+
with _model_lock:
|
|
172
|
+
if _model is None:
|
|
173
|
+
_model = ModeModel()
|
|
174
|
+
return _model
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
# ------------------------------------------------- can we use processes?
|
|
178
|
+
|
|
179
|
+
def _find(module, qualname):
|
|
180
|
+
"""Unpickle helper: look a decorated function up by name in a worker process."""
|
|
181
|
+
obj = importlib.import_module(module)
|
|
182
|
+
for part in qualname.split("."):
|
|
183
|
+
obj = getattr(obj, part)
|
|
184
|
+
return obj
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _defined_without_file(target):
|
|
188
|
+
"""True for a function typed into Jupyter or a REPL: worker processes started
|
|
189
|
+
with spawn (Windows, macOS) cannot import it by name."""
|
|
190
|
+
module = sys.modules.get(getattr(target, "__module__", None))
|
|
191
|
+
return (module is not None and module.__name__ == "__main__" and not hasattr(module, "__file__")
|
|
192
|
+
and _start_method() != "fork")
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _process_problem(target, sample_item):
|
|
196
|
+
"""None if target(item) can run in a worker process, else the reason why not."""
|
|
197
|
+
if multiprocessing.parent_process() is not None or getattr(_worker, "active", False):
|
|
198
|
+
return "already inside a worker"
|
|
199
|
+
qualname = getattr(target, "__qualname__", "")
|
|
200
|
+
if "<locals>" in qualname or "<lambda>" in qualname:
|
|
201
|
+
return "function is not defined at module top level"
|
|
202
|
+
module = sys.modules.get(getattr(target, "__module__", None))
|
|
203
|
+
if module is None:
|
|
204
|
+
return "function's module cannot be found"
|
|
205
|
+
if _defined_without_file(target):
|
|
206
|
+
if cloudpickle is None or not isinstance(target, AdaptiveExec):
|
|
207
|
+
return "function defined in Jupyter/REPL cannot be sent to worker processes"
|
|
208
|
+
try: # it travels by value: check that this really works
|
|
209
|
+
pickle.loads(pickle.dumps(target))
|
|
210
|
+
except Exception as e:
|
|
211
|
+
return f"function defined in Jupyter/REPL cannot be pickled ({type(e).__name__})"
|
|
212
|
+
else:
|
|
213
|
+
try:
|
|
214
|
+
found = _find(module.__name__, qualname)
|
|
215
|
+
except AttributeError:
|
|
216
|
+
found = None
|
|
217
|
+
if found is not target:
|
|
218
|
+
return "function cannot be found by its name"
|
|
219
|
+
try:
|
|
220
|
+
pickle.dumps(sample_item)
|
|
221
|
+
except Exception:
|
|
222
|
+
return "items cannot be pickled"
|
|
223
|
+
return None
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _pickled_size(obj):
|
|
227
|
+
"""Bytes after pickling, or None if obj cannot be pickled."""
|
|
228
|
+
try:
|
|
229
|
+
return len(pickle.dumps(obj))
|
|
230
|
+
except Exception:
|
|
231
|
+
return None
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _rule_on_source(func):
|
|
235
|
+
"""Fallback when the probe cannot run: threads if the code waits or calls GIL-free C code."""
|
|
236
|
+
static = ast_features(func)
|
|
237
|
+
return "threading" if static["has_io_call"] or static["has_gil_release_call"] else "sequential"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
class _Recorder:
|
|
241
|
+
"""Wraps the user function during the probe.
|
|
242
|
+
|
|
243
|
+
- keeps the outputs, so if probing fails half-way (an output that cannot be
|
|
244
|
+
pickled) the items already computed are not lost or run twice;
|
|
245
|
+
- keeps exceptions raised in the probe's helper threads, which Python would
|
|
246
|
+
otherwise only print and swallow.
|
|
247
|
+
It hashes/compares equal to the wrapped function, so the cached AST
|
|
248
|
+
features of that function are reused.
|
|
249
|
+
"""
|
|
250
|
+
|
|
251
|
+
def __init__(self, func):
|
|
252
|
+
self.func = func
|
|
253
|
+
self.__wrapped__ = func # inspect.getsource() follows this
|
|
254
|
+
self.caller = threading.get_ident()
|
|
255
|
+
self.outputs = []
|
|
256
|
+
self.errors = []
|
|
257
|
+
|
|
258
|
+
def __hash__(self):
|
|
259
|
+
return hash(self.func)
|
|
260
|
+
|
|
261
|
+
def __eq__(self, other):
|
|
262
|
+
return other is self.func or (isinstance(other, _Recorder) and other.func is self.func)
|
|
263
|
+
|
|
264
|
+
def __call__(self, item):
|
|
265
|
+
in_caller = threading.get_ident() == self.caller
|
|
266
|
+
try:
|
|
267
|
+
out = self.func(item)
|
|
268
|
+
except BaseException as e:
|
|
269
|
+
self.errors.append(e)
|
|
270
|
+
if in_caller:
|
|
271
|
+
raise
|
|
272
|
+
return None
|
|
273
|
+
if in_caller:
|
|
274
|
+
self.outputs.append(out)
|
|
275
|
+
return out
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
# ----------------------------------------------------------------- the core
|
|
279
|
+
|
|
280
|
+
def _run_mode(mode, func, target, items, n_workers):
|
|
281
|
+
if not items:
|
|
282
|
+
return []
|
|
283
|
+
if mode == "sequential":
|
|
284
|
+
return [func(item) for item in items]
|
|
285
|
+
pool = _get_pool(mode, n_workers)
|
|
286
|
+
# default chunksize, same as in the data collection
|
|
287
|
+
return pool.map(target if mode == "multiprocessing" else func, items)
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _best_allowed(probabilities, allowed):
|
|
291
|
+
return max(allowed, key=lambda m: probabilities[m])
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def run(func, items, n_workers=None, mode=None, cache=True, *, _target=None, _name=None):
|
|
295
|
+
"""Compute [func(item) for item in items] in the best mode.
|
|
296
|
+
|
|
297
|
+
Returns (results, Decision). `mode` forces one mode (also settable for the
|
|
298
|
+
whole program with the environment variable POLYMORPH_MODE). With `cache`,
|
|
299
|
+
a later call that looks the same (similar item count, first-item time and
|
|
300
|
+
item size) reuses the decision instead of probing again.
|
|
301
|
+
"""
|
|
302
|
+
t_start = time.perf_counter()
|
|
303
|
+
target = _target or func # what multiprocessing pickles
|
|
304
|
+
if not isinstance(items, (list, tuple)):
|
|
305
|
+
items = list(items)
|
|
306
|
+
n_workers = max(1, int(n_workers or usable_cpu_count()))
|
|
307
|
+
d = Decision(function=_name or getattr(func, "__qualname__", repr(func)),
|
|
308
|
+
mode="sequential", reason="", n_items=len(items), n_workers=n_workers)
|
|
309
|
+
|
|
310
|
+
def finish(results):
|
|
311
|
+
d.total_s = time.perf_counter() - t_start
|
|
312
|
+
log.debug("%s", d)
|
|
313
|
+
return results, d
|
|
314
|
+
|
|
315
|
+
forced = mode or os.environ.get("POLYMORPH_MODE")
|
|
316
|
+
if forced:
|
|
317
|
+
if forced not in MODES:
|
|
318
|
+
raise ValueError(f"mode must be one of {MODES}, got {forced!r}")
|
|
319
|
+
if forced == "multiprocessing" and items:
|
|
320
|
+
problem = _process_problem(target, items[0])
|
|
321
|
+
if problem:
|
|
322
|
+
raise ValueError(f"cannot use multiprocessing: {problem}")
|
|
323
|
+
d.mode, d.reason = forced, "forced"
|
|
324
|
+
return finish(list(_run_mode(forced, func, target, list(items), n_workers)))
|
|
325
|
+
|
|
326
|
+
# ---- 1. cases where measuring is pointless or impossible
|
|
327
|
+
if getattr(_worker, "active", False):
|
|
328
|
+
d.reason = "nested call inside a polymorph_ai worker"
|
|
329
|
+
return finish([func(item) for item in items])
|
|
330
|
+
if len(items) <= PROBE_MIN_ITEMS or n_workers == 1:
|
|
331
|
+
d.reason = "too few items" if n_workers > 1 else "only 1 worker"
|
|
332
|
+
return finish([func(item) for item in items])
|
|
333
|
+
|
|
334
|
+
allowed = list(MODES)
|
|
335
|
+
problem = _process_problem(target, items[0])
|
|
336
|
+
if problem:
|
|
337
|
+
allowed.remove("multiprocessing")
|
|
338
|
+
d.notes.append("no multiprocessing: " + problem)
|
|
339
|
+
|
|
340
|
+
item_bytes = _pickled_size(items[0])
|
|
341
|
+
if item_bytes is None:
|
|
342
|
+
# The probe measures pickling, so it cannot run. Fall back to the code itself.
|
|
343
|
+
d.mode, d.reason = _rule_on_source(func), "items cannot be pickled -> rule on the source code"
|
|
344
|
+
return finish(_run_mode(d.mode, func, target, list(items), n_workers))
|
|
345
|
+
|
|
346
|
+
# ---- 2. time the first item (useful work): is the whole job tiny?
|
|
347
|
+
t0 = time.perf_counter()
|
|
348
|
+
first = [func(items[0])]
|
|
349
|
+
first_s = time.perf_counter() - t0
|
|
350
|
+
if first_s * len(items) < SMALL_JOB_S:
|
|
351
|
+
d.reason, d.probed_items = "tiny job (from the first item's time)", 1
|
|
352
|
+
return finish(first + [func(item) for item in items[1:]])
|
|
353
|
+
|
|
354
|
+
# ---- 3. same kind of call as before? reuse that decision, skip the probe
|
|
355
|
+
key = (func, n_workers, round(math.log2(len(items))), round(math.log2(item_bytes + 1) / 2),
|
|
356
|
+
round(math.log2(max(first_s, 1e-7)))) if cache else None
|
|
357
|
+
cached = None
|
|
358
|
+
if key: # one timing can land just across a bucket edge, so accept the neighbouring buckets too
|
|
359
|
+
cached = next((_decisions[k] for k in (key, key[:-1] + (key[-1] - 1,), key[:-1] + (key[-1] + 1,))
|
|
360
|
+
if k in _decisions), None)
|
|
361
|
+
if cached and cached[1] > 0 and (cached[0] != "multiprocessing" or "multiprocessing" in allowed):
|
|
362
|
+
cached[1] -= 1
|
|
363
|
+
d.mode, d.reason, d.probed_items = cached[0], "same as an earlier similar call (cached)", 1
|
|
364
|
+
d.overhead_s = time.perf_counter() - t0 - first_s
|
|
365
|
+
return finish(first + _run_mode(d.mode, func, target, list(items[1:]), n_workers))
|
|
366
|
+
|
|
367
|
+
# ---- 4. probe the next items (they are really computed and kept)
|
|
368
|
+
ast_features(func) # cache under func itself, not the recorder
|
|
369
|
+
rec = _Recorder(func)
|
|
370
|
+
try:
|
|
371
|
+
features, done, _, extra = extract_features(rec, items[1:], n_workers)
|
|
372
|
+
except Exception:
|
|
373
|
+
if rec.errors: # the user's function raised: behave like a plain loop
|
|
374
|
+
raise
|
|
375
|
+
# an output could not be pickled: keep what was computed, finish without processes
|
|
376
|
+
done = first + rec.outputs
|
|
377
|
+
d.mode, d.reason = _rule_on_source(func), "results cannot be pickled -> rule on the source code"
|
|
378
|
+
d.probed_items = len(done)
|
|
379
|
+
return finish(done + _run_mode(d.mode, func, target, list(items[len(done):]), n_workers))
|
|
380
|
+
if rec.errors: # raised inside a thread-probe thread
|
|
381
|
+
raise rec.errors[0]
|
|
382
|
+
|
|
383
|
+
features["n_items"] = len(items) # the model is about the whole job
|
|
384
|
+
done = first + done
|
|
385
|
+
d.features, d.probed_items = features, len(done)
|
|
386
|
+
rest = list(items[len(done):])
|
|
387
|
+
t_decide = time.perf_counter()
|
|
388
|
+
|
|
389
|
+
# ---- 5. decide
|
|
390
|
+
if not rest:
|
|
391
|
+
d.reason = "probe already finished every item"
|
|
392
|
+
elif features["time_per_item"] * len(rest) < SMALL_JOB_S:
|
|
393
|
+
d.reason = "remaining work too small for any pool"
|
|
394
|
+
if key:
|
|
395
|
+
_decisions[key] = ["sequential", CACHE_USES]
|
|
396
|
+
else:
|
|
397
|
+
best, d.probabilities = _get_model().predict(features)
|
|
398
|
+
d.mode, d.reason = _best_allowed(d.probabilities, allowed), "model"
|
|
399
|
+
if d.mode != best:
|
|
400
|
+
d.reason = f"model ({best} not possible)"
|
|
401
|
+
if key: # cached before the start-up check: a repeated job is worth starting the pool for
|
|
402
|
+
_decisions[key] = [d.mode, CACHE_USES]
|
|
403
|
+
if d.mode == "multiprocessing" and not _pool_is_warm("multiprocessing", n_workers):
|
|
404
|
+
# The model learned on warm pools. If even a perfect speed-up would
|
|
405
|
+
# save less than starting the pool costs, do not start it for this call.
|
|
406
|
+
ideal_saving = features["time_per_item"] * len(rest) * (1 - 1 / min(n_workers, features["cpu_count"]))
|
|
407
|
+
if ideal_saving < COLD_START_S.get(_start_method(), 0.25):
|
|
408
|
+
d.mode = _best_allowed(d.probabilities, ["sequential", "threading"])
|
|
409
|
+
d.reason = "model said multiprocessing, but pool start-up would cost more than it saves"
|
|
410
|
+
d.overhead_s = extra + (time.perf_counter() - t_decide)
|
|
411
|
+
|
|
412
|
+
# ---- 6. run the rest
|
|
413
|
+
return finish(done + _run_mode(d.mode, func, target, rest, n_workers))
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
# ---------------------------------------------------------------- decorator
|
|
417
|
+
|
|
418
|
+
class AdaptiveExec:
|
|
419
|
+
"""The object @adaptive_exec puts in place of your function."""
|
|
420
|
+
|
|
421
|
+
def __init__(self, func, n_workers=None, mode=None, cache=True, verbose=False):
|
|
422
|
+
functools.update_wrapper(self, func)
|
|
423
|
+
self.func = func
|
|
424
|
+
self.n_workers = n_workers
|
|
425
|
+
self.mode = mode
|
|
426
|
+
self.cache = cache
|
|
427
|
+
self.verbose = verbose
|
|
428
|
+
self.last_run = None
|
|
429
|
+
|
|
430
|
+
def __call__(self, item):
|
|
431
|
+
return self.func(item)
|
|
432
|
+
|
|
433
|
+
def map(self, items, n_workers=None, mode=None, cache=None):
|
|
434
|
+
"""[self(item) for item in items], run in the mode polymorph_ai picks."""
|
|
435
|
+
results, self.last_run = run(self.func, items, n_workers or self.n_workers, mode or self.mode,
|
|
436
|
+
self.cache if cache is None else cache,
|
|
437
|
+
_target=self, _name=self.__qualname__)
|
|
438
|
+
if self.verbose:
|
|
439
|
+
print(self.last_run, file=sys.stderr)
|
|
440
|
+
return results
|
|
441
|
+
|
|
442
|
+
def __reduce__(self):
|
|
443
|
+
if _defined_without_file(self) and cloudpickle is not None:
|
|
444
|
+
# Written in Jupyter/REPL: there is no file to import it from, so send
|
|
445
|
+
# the code itself. Inside that cloudpickle call, a reference back to
|
|
446
|
+
# this wrapper (e.g. a recursive function calling itself by name) is
|
|
447
|
+
# rebuilt from the function, which cloudpickle has already memoised.
|
|
448
|
+
if getattr(_by_value, "active", False):
|
|
449
|
+
return _rewrap, (self.func, self.n_workers, self.mode, self.cache, self.verbose)
|
|
450
|
+
_by_value.active = True
|
|
451
|
+
try:
|
|
452
|
+
data = cloudpickle.dumps(self)
|
|
453
|
+
finally:
|
|
454
|
+
_by_value.active = False
|
|
455
|
+
return cloudpickle.loads, (data,)
|
|
456
|
+
# Worker processes import the module and look the function up by name,
|
|
457
|
+
# which finds this object again (pickling self.func by name would fail:
|
|
458
|
+
# its name now points at this wrapper).
|
|
459
|
+
return _find, (self.__module__, self.__qualname__)
|
|
460
|
+
|
|
461
|
+
def __repr__(self):
|
|
462
|
+
return f"<adaptive_exec {self.__module__}.{self.__qualname__}>"
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
_by_value = threading.local()
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
def _rewrap(func, n_workers, mode, cache, verbose):
|
|
469
|
+
return AdaptiveExec(func, n_workers, mode, cache, verbose)
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
def adaptive_exec(func=None, *, n_workers=None, mode=None, cache=True, verbose=False):
|
|
473
|
+
"""Decorator. Use as @adaptive_exec or @adaptive_exec(n_workers=4, verbose=True).
|
|
474
|
+
|
|
475
|
+
n_workers: pool size (default: usable CPUs) mode: force one mode
|
|
476
|
+
cache: reuse decisions for similar calls verbose: print every decision
|
|
477
|
+
"""
|
|
478
|
+
if func is None:
|
|
479
|
+
return lambda f: AdaptiveExec(f, n_workers, mode, cache, verbose)
|
|
480
|
+
return AdaptiveExec(func, n_workers, mode, cache, verbose)
|
polymorph_ai/features.py
ADDED
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
"""Feature extraction shared by the data collector and the decorator.
|
|
2
|
+
|
|
3
|
+
The same code must be used in both places, otherwise the model is trained on
|
|
4
|
+
features that the decorator measures differently at run time.
|
|
5
|
+
|
|
6
|
+
13 features in 4 groups (see CLAUDE.md):
|
|
7
|
+
call-site : n_items, input_bytes, cpu_count, n_workers
|
|
8
|
+
ast : max_loop_depth, arith_op_count, has_io_call, has_gil_release_call
|
|
9
|
+
probe : time_per_item, cpu_ratio, time_cv, pickle_time_ratio
|
|
10
|
+
thread probe : thread_probe_speedup
|
|
11
|
+
|
|
12
|
+
A workload is "apply func to every item in a list", like map(func, items).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import ast
|
|
16
|
+
import functools
|
|
17
|
+
import inspect
|
|
18
|
+
import os
|
|
19
|
+
import pickle
|
|
20
|
+
import statistics
|
|
21
|
+
import textwrap
|
|
22
|
+
import threading
|
|
23
|
+
import time
|
|
24
|
+
|
|
25
|
+
FEATURE_NAMES = [
|
|
26
|
+
# call-site
|
|
27
|
+
"n_items", "input_bytes", "cpu_count", "n_workers",
|
|
28
|
+
# ast
|
|
29
|
+
"max_loop_depth", "arith_op_count", "has_io_call", "has_gil_release_call",
|
|
30
|
+
# probe
|
|
31
|
+
"time_per_item", "cpu_ratio", "time_cv", "pickle_time_ratio",
|
|
32
|
+
# thread probe
|
|
33
|
+
"thread_probe_speedup",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
# The probe runs real items one by one until it has used this much time
|
|
37
|
+
# (at least PROBE_MIN_ITEMS, at most a quarter of all items or PROBE_MAX_ITEMS). The outputs are
|
|
38
|
+
# kept, so the decorator does not waste this work.
|
|
39
|
+
PROBE_TARGET_S = 0.05
|
|
40
|
+
PROBE_MIN_ITEMS = 3
|
|
41
|
+
PROBE_MAX_ITEMS = 2000 # tiny items: more adds bookkeeping, not information
|
|
42
|
+
PICKLE_SAMPLE = 20 # items used to measure the pickling cost
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
# ---------------------------------------------------------------- cpu clock
|
|
46
|
+
# CPU time used by the current thread, in seconds.
|
|
47
|
+
# Linux: time.thread_time() is precise. Windows: thread_time()/process_time()
|
|
48
|
+
# only tick every 15.6 ms, useless for a probe of a few ms. Instead we read
|
|
49
|
+
# the thread's CPU cycle counter (QueryThreadCycleTime) and convert cycles to
|
|
50
|
+
# seconds with a rate measured once by a short busy loop.
|
|
51
|
+
|
|
52
|
+
if os.name == "nt":
|
|
53
|
+
import ctypes
|
|
54
|
+
from ctypes import wintypes
|
|
55
|
+
|
|
56
|
+
_kernel32 = ctypes.WinDLL("kernel32")
|
|
57
|
+
_kernel32.GetCurrentThread.restype = wintypes.HANDLE
|
|
58
|
+
_kernel32.QueryThreadCycleTime.argtypes = [wintypes.HANDLE,
|
|
59
|
+
ctypes.POINTER(ctypes.c_ulonglong)]
|
|
60
|
+
|
|
61
|
+
def _thread_cycles():
|
|
62
|
+
cycles = ctypes.c_ulonglong()
|
|
63
|
+
_kernel32.QueryThreadCycleTime(_kernel32.GetCurrentThread(), ctypes.byref(cycles))
|
|
64
|
+
return cycles.value
|
|
65
|
+
|
|
66
|
+
@functools.lru_cache(maxsize=1)
|
|
67
|
+
def _cycles_per_second():
|
|
68
|
+
# Busy loop: the thread uses 100% CPU, so cycles / wall = rate.
|
|
69
|
+
c0, t0 = _thread_cycles(), time.perf_counter()
|
|
70
|
+
while time.perf_counter() - t0 < 0.02:
|
|
71
|
+
pass
|
|
72
|
+
return (_thread_cycles() - c0) / (time.perf_counter() - t0)
|
|
73
|
+
|
|
74
|
+
def thread_cpu_time():
|
|
75
|
+
rate = _cycles_per_second() # calibrate first, outside the reading
|
|
76
|
+
return _thread_cycles() / rate
|
|
77
|
+
else:
|
|
78
|
+
thread_cpu_time = time.thread_time
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ---------------------------------------------------------------- call-site
|
|
82
|
+
|
|
83
|
+
def usable_cpu_count():
|
|
84
|
+
"""Number of CPUs this process may actually use (not just installed)."""
|
|
85
|
+
if hasattr(os, "process_cpu_count"): # Python 3.13+
|
|
86
|
+
n = os.process_cpu_count()
|
|
87
|
+
elif hasattr(os, "sched_getaffinity"): # Linux
|
|
88
|
+
n = len(os.sched_getaffinity(0))
|
|
89
|
+
else: # Windows, older Python
|
|
90
|
+
n = os.cpu_count()
|
|
91
|
+
return n or 1
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ---------------------------------------------------------------------- ast
|
|
95
|
+
|
|
96
|
+
# Calls that mean "waits for I/O". Matched against the dotted call name,
|
|
97
|
+
# e.g. `time.sleep(...)` -> "time.sleep".
|
|
98
|
+
IO_CALLS = {"open", "print", "input", "time.sleep", "os.read", "os.write",
|
|
99
|
+
"os.fsync", "asyncio.sleep"}
|
|
100
|
+
IO_PREFIXES = ("socket.", "requests.", "urllib.", "http.", "subprocess.",
|
|
101
|
+
"shutil.", "ftplib.", "smtplib.", "sqlite3.")
|
|
102
|
+
# Method names that are almost always file/socket I/O: f.read(), s.recv() ...
|
|
103
|
+
IO_METHODS = {"read", "write", "readline", "readlines", "recv", "send",
|
|
104
|
+
"sendall", "flush"}
|
|
105
|
+
|
|
106
|
+
# C functions that do heavy work and release the GIL, so threads can run
|
|
107
|
+
# them in parallel. Only functions we have checked; plain `math` is not here.
|
|
108
|
+
GIL_RELEASE_CALLS = {"np.dot", "numpy.dot", "np.matmul", "numpy.matmul",
|
|
109
|
+
"np.sort", "numpy.sort"}
|
|
110
|
+
GIL_RELEASE_PREFIXES = ("zlib.", "bz2.", "lzma.", "hashlib.",
|
|
111
|
+
"np.linalg.", "numpy.linalg.", "np.fft.", "numpy.fft.")
|
|
112
|
+
|
|
113
|
+
LOOP_NODES = (ast.For, ast.AsyncFor, ast.While)
|
|
114
|
+
COMPREHENSIONS = (ast.ListComp, ast.SetComp, ast.DictComp, ast.GeneratorExp)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _dotted_name(node):
|
|
118
|
+
"""`zlib.compress` -> "zlib.compress", `f.read` -> "?.read" if f is complex."""
|
|
119
|
+
parts = []
|
|
120
|
+
while isinstance(node, ast.Attribute):
|
|
121
|
+
parts.append(node.attr)
|
|
122
|
+
node = node.value
|
|
123
|
+
parts.append(node.id if isinstance(node, ast.Name) else "?")
|
|
124
|
+
return ".".join(reversed(parts))
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _loop_depth(node, depth=0):
|
|
128
|
+
if isinstance(node, LOOP_NODES):
|
|
129
|
+
depth += 1
|
|
130
|
+
elif isinstance(node, COMPREHENSIONS):
|
|
131
|
+
depth += len(node.generators)
|
|
132
|
+
child_depths = [_loop_depth(child, depth) for child in ast.iter_child_nodes(node)]
|
|
133
|
+
return max([depth] + child_depths)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
@functools.lru_cache(maxsize=256)
|
|
137
|
+
def ast_features(func):
|
|
138
|
+
"""Static features read from the function's source code (cached per function).
|
|
139
|
+
|
|
140
|
+
Only the function's own body is inspected, not the functions it calls.
|
|
141
|
+
"""
|
|
142
|
+
try:
|
|
143
|
+
source = textwrap.dedent(inspect.getsource(func))
|
|
144
|
+
tree = ast.parse(source)
|
|
145
|
+
except (OSError, TypeError, SyntaxError):
|
|
146
|
+
# No source available (builtin, lambda in REPL, ...): neutral values.
|
|
147
|
+
return {"max_loop_depth": 0, "arith_op_count": 0,
|
|
148
|
+
"has_io_call": 0, "has_gil_release_call": 0}
|
|
149
|
+
|
|
150
|
+
arith = 0
|
|
151
|
+
has_io = False
|
|
152
|
+
has_gil_release = False
|
|
153
|
+
for node in ast.walk(tree):
|
|
154
|
+
if isinstance(node, (ast.BinOp, ast.AugAssign)):
|
|
155
|
+
arith += 1
|
|
156
|
+
if isinstance(node.op, ast.MatMult): # a @ b on numpy arrays
|
|
157
|
+
has_gil_release = True
|
|
158
|
+
elif isinstance(node, ast.Call):
|
|
159
|
+
name = _dotted_name(node.func)
|
|
160
|
+
method = name.rsplit(".", 1)[-1]
|
|
161
|
+
if (name in IO_CALLS or name.startswith(IO_PREFIXES)
|
|
162
|
+
or ("." in name and method in IO_METHODS)):
|
|
163
|
+
has_io = True
|
|
164
|
+
if name in GIL_RELEASE_CALLS or name.startswith(GIL_RELEASE_PREFIXES):
|
|
165
|
+
has_gil_release = True
|
|
166
|
+
|
|
167
|
+
return {"max_loop_depth": _loop_depth(tree), "arith_op_count": arith,
|
|
168
|
+
"has_io_call": int(has_io), "has_gil_release_call": int(has_gil_release)}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
# -------------------------------------------------------------------- probe
|
|
172
|
+
|
|
173
|
+
def probe(func, items):
|
|
174
|
+
"""Run the first few real items sequentially and measure them.
|
|
175
|
+
|
|
176
|
+
Returns (features, results) where results are the outputs of the probed
|
|
177
|
+
items, so the decorator can reuse them instead of computing them twice.
|
|
178
|
+
"""
|
|
179
|
+
max_items = min(len(items), PROBE_MAX_ITEMS, max(PROBE_MIN_ITEMS, len(items) // 4))
|
|
180
|
+
item_times, results, pickle_times, input_sizes = [], [], [], []
|
|
181
|
+
|
|
182
|
+
cpu_start = thread_cpu_time()
|
|
183
|
+
wall_start = time.perf_counter()
|
|
184
|
+
for item in items[:max_items]:
|
|
185
|
+
t0 = time.perf_counter()
|
|
186
|
+
out = func(item)
|
|
187
|
+
item_times.append(time.perf_counter() - t0)
|
|
188
|
+
results.append(out)
|
|
189
|
+
if (len(item_times) >= PROBE_MIN_ITEMS
|
|
190
|
+
and time.perf_counter() - wall_start >= PROBE_TARGET_S):
|
|
191
|
+
break
|
|
192
|
+
wall = time.perf_counter() - wall_start
|
|
193
|
+
cpu = thread_cpu_time() - cpu_start
|
|
194
|
+
|
|
195
|
+
# What multiprocessing would pay per item: send input, get output back.
|
|
196
|
+
# Measured outside the timed loop above so it does not inflate cpu_ratio.
|
|
197
|
+
# A spread-out sample is enough for the mean; tiny items can mean
|
|
198
|
+
# tens of thousands of probed items.
|
|
199
|
+
step = max(1, len(results) // PICKLE_SAMPLE)
|
|
200
|
+
for item, out in list(zip(items, results))[::step][:PICKLE_SAMPLE]:
|
|
201
|
+
t0 = time.perf_counter()
|
|
202
|
+
data = pickle.dumps(item)
|
|
203
|
+
pickle.loads(data)
|
|
204
|
+
pickle.loads(pickle.dumps(out))
|
|
205
|
+
pickle_times.append(time.perf_counter() - t0)
|
|
206
|
+
input_sizes.append(len(data))
|
|
207
|
+
|
|
208
|
+
time_per_item = statistics.mean(item_times)
|
|
209
|
+
features = {
|
|
210
|
+
"input_bytes": statistics.mean(input_sizes),
|
|
211
|
+
"time_per_item": time_per_item,
|
|
212
|
+
"cpu_ratio": min(cpu / wall, 1.0) if wall > 0 else 0.0,
|
|
213
|
+
"time_cv": (statistics.pstdev(item_times) / time_per_item
|
|
214
|
+
if time_per_item > 0 else 0.0),
|
|
215
|
+
"pickle_time_ratio": (statistics.mean(pickle_times) / time_per_item
|
|
216
|
+
if time_per_item > 0 else 0.0),
|
|
217
|
+
}
|
|
218
|
+
return features, results, item_times
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def thread_probe(func, items, time_per_item):
|
|
222
|
+
"""Run the NEXT items (not the probed ones again) on 2 threads.
|
|
223
|
+
|
|
224
|
+
Compares their wall time with what they would take sequentially
|
|
225
|
+
(len(items) * time_per_item from the probe):
|
|
226
|
+
~2.0 means threads really run in parallel (I/O or GIL-releasing code),
|
|
227
|
+
~1.0 or less means the GIL serialises them.
|
|
228
|
+
The outputs are kept, so this work is not wasted. Items are never run
|
|
229
|
+
twice, so functions with side effects (writing files, ...) stay correct.
|
|
230
|
+
|
|
231
|
+
Returns (speedup, results). With fewer than 2 items left: (1.0, []).
|
|
232
|
+
"""
|
|
233
|
+
if len(items) < 2:
|
|
234
|
+
return 1.0, []
|
|
235
|
+
results = [None] * len(items)
|
|
236
|
+
|
|
237
|
+
def work(indexes):
|
|
238
|
+
for i in indexes:
|
|
239
|
+
results[i] = func(items[i])
|
|
240
|
+
|
|
241
|
+
halves = [range(0, len(items), 2), range(1, len(items), 2)]
|
|
242
|
+
threads = [threading.Thread(target=work, args=(h,)) for h in halves]
|
|
243
|
+
t0 = time.perf_counter()
|
|
244
|
+
for t in threads:
|
|
245
|
+
t.start()
|
|
246
|
+
for t in threads:
|
|
247
|
+
t.join()
|
|
248
|
+
wall = time.perf_counter() - t0
|
|
249
|
+
speedup = len(items) * time_per_item / wall if wall > 0 else 1.0
|
|
250
|
+
return speedup, results
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ---------------------------------------------------------------------- all
|
|
254
|
+
|
|
255
|
+
def extract_features(func, items, n_workers):
|
|
256
|
+
"""All 13 features for running func over items with n_workers workers.
|
|
257
|
+
|
|
258
|
+
Returns (features, probe_results, overhead_s, extra_s).
|
|
259
|
+
probe_results are the outputs of the first len(probe_results) items.
|
|
260
|
+
overhead_s is the total time spent here; extra_s is the part that did
|
|
261
|
+
no useful work (pickle test, AST, bookkeeping), i.e. overhead_s minus
|
|
262
|
+
the time spent computing real items.
|
|
263
|
+
"""
|
|
264
|
+
t0 = time.perf_counter()
|
|
265
|
+
probe_feats, results, item_times = probe(func, items)
|
|
266
|
+
t_probe = time.perf_counter()
|
|
267
|
+
# Skip the first probed item when estimating sequential speed: it may
|
|
268
|
+
# include one-off warm-up.
|
|
269
|
+
steady = item_times[1:] if len(item_times) >= 3 else item_times
|
|
270
|
+
next_items = items[len(results):len(results) + len(results)]
|
|
271
|
+
speedup, thread_results = thread_probe(func, next_items, statistics.mean(steady))
|
|
272
|
+
t_thread = time.perf_counter()
|
|
273
|
+
results = results + thread_results
|
|
274
|
+
|
|
275
|
+
features = {
|
|
276
|
+
"n_items": len(items),
|
|
277
|
+
"input_bytes": probe_feats["input_bytes"],
|
|
278
|
+
"cpu_count": usable_cpu_count(),
|
|
279
|
+
"n_workers": n_workers,
|
|
280
|
+
**ast_features(func),
|
|
281
|
+
"time_per_item": probe_feats["time_per_item"],
|
|
282
|
+
"cpu_ratio": probe_feats["cpu_ratio"],
|
|
283
|
+
"time_cv": probe_feats["time_cv"],
|
|
284
|
+
"pickle_time_ratio": probe_feats["pickle_time_ratio"],
|
|
285
|
+
"thread_probe_speedup": speedup,
|
|
286
|
+
}
|
|
287
|
+
overhead = time.perf_counter() - t0
|
|
288
|
+
useful = sum(item_times) + (t_thread - t_probe if thread_results else 0.0)
|
|
289
|
+
return features, results, overhead, max(overhead - useful, 0.0)
|
polymorph_ai/model.npz
ADDED
|
Binary file
|
polymorph_ai/model.py
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
"""The trained mode-selection MLP, run with plain numpy.
|
|
2
|
+
|
|
3
|
+
The weights come from notebooks/04_export_model.ipynb (Keras, trained on the
|
|
4
|
+
collected dataset). Re-implementing the forward pass here keeps the library's
|
|
5
|
+
only dependency at numpy: no Keras/JAX import (seconds) at run time, and one
|
|
6
|
+
decision takes microseconds.
|
|
7
|
+
|
|
8
|
+
Steps, identical to training:
|
|
9
|
+
log10 of the heavy-tailed features -> standardise -> Dense+ReLU ... -> Dense -> softmax
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
|
|
15
|
+
import numpy as np
|
|
16
|
+
|
|
17
|
+
DEFAULT_PATH = os.path.join(os.path.dirname(os.path.abspath(__file__)), "model.npz")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class ModeModel:
|
|
21
|
+
def __init__(self, path=DEFAULT_PATH):
|
|
22
|
+
data = np.load(path)
|
|
23
|
+
self.feature_names = [str(f) for f in data["feature_names"]]
|
|
24
|
+
self.modes = [str(m) for m in data["modes"]]
|
|
25
|
+
self.meta = json.loads(str(data["meta"]))
|
|
26
|
+
log_features = {str(f) for f in data["log_features"]}
|
|
27
|
+
self._is_log = np.array([f in log_features for f in self.feature_names])
|
|
28
|
+
self._mean = data["scaler_mean"]
|
|
29
|
+
self._scale = data["scaler_scale"]
|
|
30
|
+
n = int(data["n_layers"])
|
|
31
|
+
self._layers = [(data[f"w{2 * i}"], data[f"w{2 * i + 1}"]) for i in range(n)]
|
|
32
|
+
|
|
33
|
+
def predict_proba(self, features):
|
|
34
|
+
"""features: dict with the 13 feature names -> probabilities in self.modes order."""
|
|
35
|
+
x = np.array([float(features[f]) for f in self.feature_names])
|
|
36
|
+
x[self._is_log] = np.log10(np.maximum(x[self._is_log], 1e-9))
|
|
37
|
+
h = (x - self._mean) / self._scale
|
|
38
|
+
for i, (w, b) in enumerate(self._layers):
|
|
39
|
+
h = h @ w + b
|
|
40
|
+
if i < len(self._layers) - 1:
|
|
41
|
+
h = np.maximum(h, 0.0) # ReLU (dropout is off at inference)
|
|
42
|
+
e = np.exp(h - h.max())
|
|
43
|
+
return e / e.sum() # softmax
|
|
44
|
+
|
|
45
|
+
def predict(self, features):
|
|
46
|
+
"""Most likely mode and the probability of every mode (dict)."""
|
|
47
|
+
p = self.predict_proba(features)
|
|
48
|
+
return self.modes[int(p.argmax())], dict(zip(self.modes, p.tolist()))
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: polymorph-ai
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: A decorator that uses a trained neural network to run your function sequentially, with threads, or with processes - whichever is fastest.
|
|
5
|
+
Author: Worachat Songmuangnu
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Worachat-Songmuangnu/polymorph-ai
|
|
8
|
+
Project-URL: Source, https://github.com/Worachat-Songmuangnu/polymorph-ai
|
|
9
|
+
Project-URL: Issues, https://github.com/Worachat-Songmuangnu/polymorph-ai/issues
|
|
10
|
+
Keywords: parallel,multiprocessing,threading,decorator,machine-learning,performance
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Operating System :: Microsoft :: Windows
|
|
14
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
17
|
+
Classifier: Topic :: System :: Distributed Computing
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: numpy
|
|
22
|
+
Requires-Dist: cloudpickle
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# polymorph-ai
|
|
26
|
+
|
|
27
|
+
`@adaptive_exec` runs a function over a list of items **sequentially, with threads, or with processes**, and a small trained neural network picks whichever mode should be fastest for that job on that machine.
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
from polymorph_ai import adaptive_exec
|
|
31
|
+
|
|
32
|
+
@adaptive_exec
|
|
33
|
+
def process(item):
|
|
34
|
+
...
|
|
35
|
+
|
|
36
|
+
if __name__ == "__main__": # required on Windows/macOS for processes
|
|
37
|
+
results = process.map(items) # same as [process(x) for x in items], in order
|
|
38
|
+
print(process.last_run)
|
|
39
|
+
# [polymorph_ai] process: 5,000 items -> multiprocessing p=0.97 (model) | overhead 0.7 ms | total 1.84 s
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
`process(x)` still works as a normal function call. Only `.map()` adds the automatic part.
|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install polymorph-ai
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Dependencies: numpy and cloudpickle. Tested on Linux with Python 3.10, 3.12 and 3.14, and on Windows with Python 3.14.
|
|
51
|
+
|
|
52
|
+
Works in Jupyter too: a function written in a notebook cell is sent to the worker processes by value (with cloudpickle), so it can still use multiprocessing on Windows and macOS.
|
|
53
|
+
|
|
54
|
+
## How it decides
|
|
55
|
+
|
|
56
|
+
Every `.map(items)` call goes through these steps:
|
|
57
|
+
|
|
58
|
+
1. **Safety checks.** Empty or very short lists, one worker, or a call nested inside another polymorph_ai worker all just run as a plain loop. If the function or the items cannot be sent to another process (a lambda, a function defined inside another function, unpicklable items), multiprocessing is ruled out.
|
|
59
|
+
2. **First item.** The first item runs and is timed. If `time × number of items` is under 10 ms, the job is too small for any pool to pay off, so the rest runs as a plain loop.
|
|
60
|
+
3. **Cache.** A call that looks like an earlier one (similar number of items, first-item time, and item size) reuses that decision and skips the probe.
|
|
61
|
+
4. **Probe.** The next items run for about 50 ms, first one by one and then on 2 threads, to measure the 13 features: call site, static code analysis (AST), runtime probe, and thread probe. The probe uses the same code (`polymorph_ai/features.py`) that collected the training data. **Probed items are real work: their results are kept and no item ever runs twice.**
|
|
62
|
+
5. **Model.** An MLP (13 → 225 ReLU → 3 softmax, 3,828 weights) turns the features into probabilities. It is trained in Keras and runs here in numpy, at about 30 µs per decision.
|
|
63
|
+
6. **Run.** The remaining items run in the chosen mode on a pool that stays alive for the next call.
|
|
64
|
+
|
|
65
|
+
## Options
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
@adaptive_exec(n_workers=4, verbose=True, cache=True) # verbose prints every decision
|
|
69
|
+
def process(item): ...
|
|
70
|
+
|
|
71
|
+
process.map(items, mode="threading") # force one mode for this call
|
|
72
|
+
# POLYMORPH_MODE=sequential python app.py # force one mode for the whole program (A/B testing, debugging)
|
|
73
|
+
|
|
74
|
+
from polymorph_ai import run, warm_up
|
|
75
|
+
results, decision = run(some_function, items) # for functions you cannot decorate
|
|
76
|
+
warm_up() # start the pools early (e.g. at server start-up)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
`process.last_run` (a `Decision`) records the chosen mode, why it was chosen, the model's probabilities, the 13 features, how many items the probe computed, and the overhead.
|
|
80
|
+
|
|
81
|
+
## Results
|
|
82
|
+
|
|
83
|
+
**Model** (13,603 jobs collected on 16 machine settings, see [`dataset/`](https://github.com/Worachat-Songmuangnu/polymorph-ai/tree/main/dataset), GroupKFold by workload template, so every test job comes from code the model never trained on). Time lost compared with always picking the fastest mode:
|
|
84
|
+
|
|
85
|
+
- Always multiprocessing: +22.1%
|
|
86
|
+
- if-else rules: +13.0%
|
|
87
|
+
- Random Forest: +5.8%
|
|
88
|
+
- **MLP: +4.3%**
|
|
89
|
+
|
|
90
|
+
**Library** ([`examples/demo.py`](https://github.com/Worachat-Songmuangnu/polymorph-ai/blob/main/examples/demo.py): 8 everyday jobs that are not in the training data, 8 CPUs). Total time compared with perfect picks:
|
|
91
|
+
|
|
92
|
+
| | Windows | Linux (WSL) |
|
|
93
|
+
|---|---|---|
|
|
94
|
+
| always sequential | +349% | +426% |
|
|
95
|
+
| always threading | +117% | +138% |
|
|
96
|
+
| always multiprocessing | +19% | +30% |
|
|
97
|
+
| polymorph_ai, first call | +52% | +66% |
|
|
98
|
+
| polymorph_ai, repeated call | **+13%** | **+24%** |
|
|
99
|
+
|
|
100
|
+
**When does it pay off?** The first call on a job pays about 0.2 s for the probe (`python examples/demo.py --overhead`).
|
|
101
|
+
|
|
102
|
+
- Compared with a plain loop, polymorph_ai is already faster on jobs from about 0.3–0.6 s.
|
|
103
|
+
- Compared with a perfect choice of mode, the first call is within 15% once the job takes 5 s or more.
|
|
104
|
+
- Repeated calls skip the probe, so they cost almost nothing.
|
|
105
|
+
|
|
106
|
+
## Limitations
|
|
107
|
+
|
|
108
|
+
- The decorated function takes **one item** and is applied to every item (`map`). polymorph_ai cannot parallelise code that does not have this shape.
|
|
109
|
+
- On Windows and macOS, the script that uses processes needs the `if __name__ == "__main__":` guard (a Python rule for processes, not specific to polymorph_ai).
|
|
110
|
+
- The first item and the probe items run in the calling thread, so a job with a few very long items loses one item's worth of parallel time.
|
|
111
|
+
- The model was trained on synthetic workloads from 13 families. Code that behaves unlike all of them can still be mispredicted. In the demo, sorting 20k-number lists stays sequential when processes would be twice as fast.
|
|
112
|
+
- asyncio is not supported: a decorator cannot turn ordinary code into `async def` code.
|
|
113
|
+
- In Jupyter on Windows, call `shutdown()` before you close or restart the kernel. If the kernel is stopped straight after the cell that started the process pool, the worker processes can be left running.
|
|
114
|
+
|
|
115
|
+
## Tests
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
python -m pytest tests -v # 28 tests, pass on Windows and Linux
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
## Project layout
|
|
122
|
+
|
|
123
|
+
```
|
|
124
|
+
polymorph_ai/ the library: decorator.py, features.py, model.py + model.npz (the trained network)
|
|
125
|
+
tests/ pytest suite
|
|
126
|
+
examples/ demo.py (benchmark vs fixed modes) and its results/
|
|
127
|
+
dataset/ the training dataset (13,603 jobs) and the real-code results (92 jobs)
|
|
128
|
+
```
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
polymorph_ai/__init__.py,sha256=1aFA5RwshexIWECw1ihJu1b7A9FWZKvZCV06iF5q6lM,738
|
|
2
|
+
polymorph_ai/decorator.py,sha256=zB5MnztiSfOOWXYmwXBbrAgWSlKD2g_ziR2gSngr_ig,19980
|
|
3
|
+
polymorph_ai/features.py,sha256=JgSjN6FW6elrlYmwXNmvj_VavwxT6aty7MeBIFHOpFw,11818
|
|
4
|
+
polymorph_ai/model.npz,sha256=PYXE0nZhIMtUiSw5JjXBhk-rPuWl2I_qpclF5ITfJRg,21120
|
|
5
|
+
polymorph_ai/model.py,sha256=GrOTV-TpepgZlaDamEDiLpWEiUi0-iT0XdZTg1OEOKw,2089
|
|
6
|
+
polymorph_ai-1.1.0.dist-info/licenses/LICENSE,sha256=H0aLa4UfHVK6fzZHZ_aEMtgywY18sh6wa71_q0H4thg,1081
|
|
7
|
+
polymorph_ai-1.1.0.dist-info/METADATA,sha256=I1DTMUvZb-tR2eV-FvGCkJzuafExt_14JGIRDyP8fHg,6998
|
|
8
|
+
polymorph_ai-1.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
9
|
+
polymorph_ai-1.1.0.dist-info/top_level.txt,sha256=5F0pgkilc2sBAikhdTIpRrxPIxDMS2D1Ae9qrLO6K5A,13
|
|
10
|
+
polymorph_ai-1.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 the polymorph-ai authors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
polymorph_ai
|