openzoo 0.49.8 → 0.49.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,809 @@
1
+ """Route a task to the CHEAPEST model that will probably finish it (MR-3).
2
+
3
+ THE OBJECTIVE, stated once and obeyed everywhere below
4
+ ------------------------------------------------------
5
+ Not "what topic is this" and not "what is the best model". The decision is:
6
+
7
+ pick argmin cost(m, task) subject to P_success(m | task) >= bar(task)
8
+ m in feasible(task)
9
+
10
+ ... and if NOTHING clears the bar, say so and return the highest-P option, labelled as such,
11
+ rather than silently pretending the cheapest one was fine.
12
+
13
+ Everything else in this file exists to supply one of those four terms:
14
+ feasible() -- HARD constraints from the request: image input, tool calling, strict JSON, context
15
+ length. These are metadata facts, never guesses; a model without an image input
16
+ modality cannot read a screenshot at any price.
17
+ P_success() -- a PRIOR, honestly labelled. See the next section.
18
+ cost() -- $/task = price_in * estimated prompt tokens + price_out * estimated completion
19
+ tokens. "Relative cost" is this number divided by the cheapest feasible candidate's,
20
+ so the ranking is scale-free.
21
+ bar() -- an authored policy knob per (class, difficulty). Not measured. Stated, not hidden.
22
+
23
+ WHERE P_success COMES FROM, AND WHAT IT IS NOT
24
+ ----------------------------------------------
25
+ There is no free API that reports "model m completes task class c with probability p". What exists is
26
+ OpenRouter's per-category leaderboard: for 12 categories, the top-20 models BY USAGE. That is REVEALED
27
+ PREFERENCE -- thousands of developers repeatedly choosing a model for that kind of work and not
28
+ switching away -- which correlates with "it finishes the job", but it is NOT a benchmark. Popularity
29
+ tracks price, marketing, and defaults too.
30
+
31
+ So the prior is built as: leaderboard rank in the categories this task class maps to, shrunk toward a
32
+ metadata floor for the ~87% of the catalogue that appears on no leaderboard at all. And then the
33
+ important part: `Outcomes` turns the prior into a MEASUREMENT over time. Every real completion can be
34
+ recorded with record_outcome(class, model, ok); P_success becomes a Beta posterior whose prior weight
35
+ is the leaderboard number and whose data is your own results. Day one it is a proxy and says so
36
+ (`evidence: "prior"`); after a few hundred real calls it is yours (`evidence: "measured(n=...)"`).
37
+
38
+ KEPT NEGATIVE (a trap this design walks around): price is tempting as a capability proxy -- expensive
39
+ models are usually stronger. It is deliberately NOT used that way, because cost is the thing being
40
+ minimised; letting price raise P_success too would make the objective partly cancel itself and quietly
41
+ re-rank toward expensive models for no measured reason. Price appears in cost(), nowhere else.
42
+
43
+ THE CLASSIFIER (from scratch, no pretrained anything)
44
+ -----------------------------------------------------
45
+ Text -> hypervector by RANDOM INDEXING (Kanerva): each token deterministically hashes to k sparse +-1
46
+ positions in a d-dim space; a request is the bundle of its token atoms over three separately-normalised
47
+ channels (words, word bigrams, char 4-grams, so a misspelling degrades a vector instead of erasing a
48
+ word), weighted by a hashed IDF table.
49
+
50
+ Two gradient-free heads produce the SAME artifact shape, a K x d matrix:
51
+ ridge (shipped) -- closed-form one-vs-rest least squares in the dual, W = X^T (XX^T + lam I)^-1 Y.
52
+ One 535x535 solve. No epochs, no learning rate, no shuffling, no seed to get
53
+ lucky with.
54
+ adapthd (kept) -- perceptron prototypes, the reference the ridge head is checked against.
55
+
56
+ MEASURED (tools/modelroute/train_router.py --ablate, 10 classes, 400 authored requests + anchors,
57
+ held-out split written in a deliberately different register): ridge+IDF+anchors 0.642 held-out /
58
+ 0.772 5-fold, against a bag-of-words nearest-centroid baseline at 0.333 and majority at 0.100. Every
59
+ one of the three choices was measured separately and each earns its place; the whole learned artifact
60
+ is 40 KB int8 + a 16 KB IDF table. No token table, no embedding model, no torch. Deterministic under
61
+ any PYTHONHASHSEED because every hash is sha256, never Python's salted hash().
62
+
63
+ But class accuracy is NOT the headline -- see tools/modelroute/routing_regret.py, which measures the
64
+ classifier in dollars and failures instead: 5.0% of held-out requests route below the bar their true
65
+ class would have demanded, at a median 1.12x the oracle route's cost.
66
+
67
+ The classifier's OUTPUT IS NOT THE ANSWER -- it sets the bar and picks which leaderboards to trust. The
68
+ 414-model catalogue is a lookup table, not a softmax, so a model added tomorrow is routable today.
69
+ """
70
+ import hashlib
71
+ import json
72
+ import os
73
+ import re
74
+
75
+ import numpy as np
76
+
77
+ from holographic.misc.holographic_determinism import argmax_tiebreak
78
+
79
+ # ---------------------------------------------------------------------------------------------------
80
+ # artifact locations (all small, all optional-at-import)
81
+ _HERE = os.path.dirname(os.path.abspath(__file__))
82
+ _DATA = os.path.abspath(os.path.join(_HERE, "..", "..", "lecore_data", "modelroute"))
83
+ CATALOG_PATH = os.path.join(_DATA, "catalog.npz")
84
+ ROUTER_PATH = os.path.join(_DATA, "router.npz")
85
+ OUTCOMES_PATH = os.path.join(_DATA, "outcomes.json")
86
+
87
+ MOD_TEXT, MOD_IMAGE, MOD_AUDIO, MOD_FILE = 1, 2, 4, 8
88
+
89
+ # ---------------------------------------------------------------------------------------------------
90
+ # 1. THE ENCODER -- text to hypervector, no model, no vocabulary file
91
+ # ---------------------------------------------------------------------------------------------------
92
+ _WORD = re.compile(r"[a-z0-9]+")
93
+
94
+
95
+ def _atom_positions(token, dim, k, seed):
96
+ """Deterministic sparse atom for a token: k (index, sign) pairs from one sha256 digest.
97
+
98
+ Random indexing, not a lookup table -- so the encoder has NO vocabulary and an unseen word is
99
+ encoded exactly like a seen one (into a near-orthogonal direction) instead of being dropped.
100
+ sha256, never hash(): the engine's determinism rule."""
101
+ h = hashlib.sha256(f"{seed}|{token}".encode()).digest()
102
+ # 8 bytes per (index, sign) pair; a 32-byte digest carries 4 pairs, so re-hash as needed for k>4.
103
+ idx, sgn, buf, salt = [], [], h, 0
104
+ while len(idx) < k:
105
+ for off in range(0, 32, 8):
106
+ if len(idx) >= k:
107
+ break
108
+ chunk = int.from_bytes(buf[off:off + 8], "big")
109
+ idx.append(chunk % dim)
110
+ sgn.append(1.0 if (chunk >> 63) & 1 else -1.0)
111
+ salt += 1
112
+ buf = hashlib.sha256(f"{seed}|{token}|{salt}".encode()).digest()
113
+ return np.array(idx, dtype=np.int64), np.array(sgn, dtype=np.float64)
114
+
115
+
116
+ class TextEncoder:
117
+ """Request text -> unit hypervector. Words + word bigrams + char 4-grams, bundled.
118
+
119
+ Char 4-grams are what make the held-out register survivable: 'fucntion' shares most of its 4-grams
120
+ with 'function', so a typo degrades the vector instead of erasing the word. Bigrams carry the small
121
+ amount of order that matters for routing ('unit test', 'json only', 'step by step')."""
122
+
123
+ # (words, word-bigrams, char-n-grams). Words carry the topic, bigrams the little order that
124
+ # matters ('unit test', 'json only'), char-grams the robustness to typos and unseen words.
125
+ CHANNEL_W = (1.0, 0.6, 0.45)
126
+
127
+ IDF_BUCKETS = 8192 # hashed IDF table size; a table, never a vocabulary
128
+
129
+ def __init__(self, dim=4096, k=8, seed=17, char_n=4, use_bigrams=True, cache=True, idf=None):
130
+ self.dim, self.k, self.seed, self.char_n = int(dim), int(k), int(seed), int(char_n)
131
+ self.use_bigrams = bool(use_bigrams)
132
+ self._cache = {} if cache else None
133
+ self.idf = None if idf is None else np.asarray(idf, dtype=np.float64)
134
+
135
+ def _idf_bucket(self, tok):
136
+ return int.from_bytes(hashlib.sha256(f"idf|{self.seed}|{tok}".encode()).digest()[:8],
137
+ "big") % self.IDF_BUCKETS
138
+
139
+ def fit_idf(self, texts):
140
+ """Learn HASHED inverse document frequency from the training texts.
141
+
142
+ WHY: without it, 'this', 'the', 'i' contribute as much direction as 'docker' or 'translate',
143
+ and since every class's requests are mostly function words the prototypes end up mostly
144
+ parallel. Standard IDF needs a vocabulary file; hashing the token into a fixed 8192-bucket
145
+ table keeps the encoder vocabulary-free (an unseen word still gets a weight -- the average of
146
+ whatever collided into its bucket) at a cost of 32 KB. Collisions are noise, not error: a rare
147
+ word colliding with a common one is merely under-weighted."""
148
+ df = np.zeros(self.IDF_BUCKETS)
149
+ for t in texts:
150
+ for b in {self._idf_bucket(tok) for tok in self.tokens(t)}:
151
+ df[b] += 1.0
152
+ n = max(1, len(texts))
153
+ self.idf = np.log((n + 1.0) / (df + 1.0)) + 1.0
154
+ return self
155
+
156
+ def _weight(self, tok):
157
+ return 1.0 if self.idf is None else float(self.idf[self._idf_bucket(tok)])
158
+
159
+ def channels(self, text):
160
+ """The three token channels of a request, kept SEPARATE on purpose.
161
+
162
+ MEASURED, the hard way: bundling all three into one bag collapses the classifier -- a 60-char
163
+ request has ~10 words but ~57 char-4-grams, so the char channel outvotes the words 6:1 and
164
+ every class prototype drifts toward generic English. Normalising each channel to unit length
165
+ first and then mixing by CHANNEL_W makes the contribution independent of how many tokens the
166
+ channel happens to emit. (Held-out accuracy before the split: 0.18. After: see the report.)"""
167
+ words = _WORD.findall(text.lower())
168
+ bigrams = [f"{a}_{b}" for a, b in zip(words, words[1:])] if self.use_bigrams else []
169
+ flat = " ".join(words)
170
+ n = self.char_n
171
+ chars = [f"#{flat[i:i + n]}" for i in range(max(0, len(flat) - n + 1))]
172
+ return words, bigrams, chars
173
+
174
+ def tokens(self, text):
175
+ """Flat token multiset (the union of the channels) -- kept for inspection and tests."""
176
+ w, b, c = self.channels(text)
177
+ return w + b + c
178
+
179
+ def _atom(self, tok):
180
+ if self._cache is None:
181
+ return _atom_positions(tok, self.dim, self.k, self.seed)
182
+ got = self._cache.get(tok)
183
+ if got is None:
184
+ got = _atom_positions(tok, self.dim, self.k, self.seed)
185
+ self._cache[tok] = got
186
+ return got
187
+
188
+ def _bundle(self, toks):
189
+ v = np.zeros(self.dim, dtype=np.float64)
190
+ for tok in toks:
191
+ idx, sgn = self._atom(tok)
192
+ np.add.at(v, idx, sgn * self._weight(tok))
193
+ n = np.linalg.norm(v)
194
+ return v if n < 1e-12 else v / n
195
+
196
+ def encode(self, text):
197
+ """Per-channel unit bundle, then a weighted mix, then unit-normalise. Empty/unparseable text
198
+ -> zero vector (the classifier then abstains rather than picking a class from nothing)."""
199
+ v = np.zeros(self.dim, dtype=np.float64)
200
+ for w, toks in zip(self.CHANNEL_W, self.channels(text)):
201
+ if toks:
202
+ v += w * self._bundle(toks)
203
+ n = np.linalg.norm(v)
204
+ return v if n < 1e-12 else v / n
205
+
206
+ def encode_many(self, texts):
207
+ return np.stack([self.encode(t) for t in texts]) if texts else np.zeros((0, self.dim))
208
+
209
+ def config(self):
210
+ return dict(dim=self.dim, k=self.k, seed=self.seed, char_n=self.char_n,
211
+ use_bigrams=self.use_bigrams)
212
+
213
+
214
+ # ---------------------------------------------------------------------------------------------------
215
+ # 2. THE CLASSIFIER -- gradient-free HDC prototypes with AdaptHD retraining
216
+ # ---------------------------------------------------------------------------------------------------
217
+ class TaskClassifier:
218
+ """Nearest-prototype classifier over hypervectors, trained by perceptron-style nudges.
219
+
220
+ fit(): bundle each class's examples into a prototype (one-shot centroid), then for `epochs` passes
221
+ over the data, every MISCLASSIFIED example is added to its true class prototype and subtracted from
222
+ the one that wrongly won. No gradients, no optimiser, no learning rate schedule beyond `lr`.
223
+
224
+ predict(): cosine against every prototype. Returns (label, confidence, margin) where margin is the
225
+ gap to the runner-up -- the abstention signal. A low margin means the request straddles two classes
226
+ (they often genuinely do), and the router widens its bar instead of committing."""
227
+
228
+ def __init__(self, classes, encoder):
229
+ self.classes = list(classes)
230
+ self.enc = encoder
231
+ self.P = np.zeros((len(self.classes), encoder.dim))
232
+
233
+ # -- training ------------------------------------------------------------------------------
234
+ def fit(self, texts, labels, method="ridge", **kw):
235
+ """Train the K x dim matrix. Two heads, both gradient-free, both the same artifact shape:
236
+
237
+ 'ridge' -- closed-form one-vs-rest least squares in the DUAL form
238
+ W = X^T (X X^T + lam I)^-1 Y. With n=400 examples that is a 400x400 solve, so
239
+ 'training' is one linear system, not an optimisation. Deterministic by
240
+ construction: no epochs, no shuffling, no learning rate.
241
+ 'adapthd' -- the perceptron prototypes (fit_adapthd below).
242
+
243
+ The ridge head is the fast path and adapthd is its REFERENCE (the F22 rule): two independent
244
+ learners over the same features, so a feature bug shows up as both of them failing, not as one
245
+ of them silently compensating."""
246
+ return (self.fit_ridge(texts, labels, **kw) if method == "ridge"
247
+ else self.fit_adapthd(texts, labels, **kw))
248
+
249
+ def fit_ridge(self, texts, labels, lam=1.0):
250
+ X = self.enc.encode_many(texts)
251
+ y = np.array([self.classes.index(l) for l in labels])
252
+ Y = -np.ones((len(y), len(self.classes))) # one-vs-rest targets in {-1, +1}
253
+ Y[np.arange(len(y)), y] = 1.0
254
+ G = X @ X.T
255
+ alpha = np.linalg.solve(G + lam * np.eye(len(X)), Y) # dual coefficients, n x K
256
+ self.P = self._unit((X.T @ alpha).T) # K x dim, unit rows for cosine scoring
257
+ self.train_acc_ = float(np.mean(
258
+ [int(argmax_tiebreak(self.P @ X[i])) == y[i] for i in range(len(X))]))
259
+ return self
260
+
261
+ def fit_adapthd(self, texts, labels, epochs=30, lr=0.35, seed=0):
262
+ """THREE things here are load-bearing, each of them learned by getting it wrong first:
263
+
264
+ 1. UPDATE THE RAW ACCUMULATOR, score with a normalised view. Re-normalising the prototype
265
+ matrix inside the update loop throws away the accumulated evidence every step, so late
266
+ examples overwrite early ones instead of adding to them.
267
+ 2. SHUFFLE (seeded). With the corpus in class order and a full-size update, the last class
268
+ processed wins everything: the first run of this collapsed all 120 held-out requests onto
269
+ `advice`, the final class in the file. Seeded permutation = shuffled AND reproducible.
270
+ 3. KEEP THE BEST EPOCH. The perceptron does not converge monotonically on data that is not
271
+ linearly separable; the last epoch is not the best one. Snapshot on improvement."""
272
+ X = self.enc.encode_many(texts)
273
+ y = np.array([self.classes.index(l) for l in labels])
274
+ K = len(self.classes)
275
+ A = np.zeros((K, self.enc.dim)) # raw accumulator, never re-normalised
276
+ for c in range(K): # stage 1: one-shot bundle (the centroid)
277
+ rows = X[y == c]
278
+ if len(rows):
279
+ A[c] = rows.sum(0) / max(1, len(rows))
280
+ rng = np.random.default_rng(seed)
281
+ best, best_acc = self._unit(A).copy(), -1.0
282
+ for _ in range(epochs):
283
+ order = rng.permutation(len(X))
284
+ errs = 0
285
+ for i in order:
286
+ P = self._unit(A)
287
+ pred = int(argmax_tiebreak(P @ X[i]))
288
+ if pred != y[i]:
289
+ A[y[i]] += lr * X[i] # toward the truth
290
+ A[pred] -= lr * X[i] # away from the impostor
291
+ errs += 1
292
+ acc = 1.0 - errs / max(1, len(X))
293
+ if acc > best_acc:
294
+ best_acc, best = acc, self._unit(A).copy()
295
+ if errs == 0: # separated: further passes are no-ops
296
+ break
297
+ self.P = best
298
+ self.train_acc_ = best_acc
299
+ return self
300
+
301
+ @staticmethod
302
+ def _unit(M):
303
+ return M / (np.linalg.norm(M, axis=1, keepdims=True) + 1e-12)
304
+
305
+ # -- inference -----------------------------------------------------------------------------
306
+ def scores(self, text):
307
+ v = self.enc.encode(text)
308
+ return self.P @ v
309
+
310
+ def predict(self, text):
311
+ s = self.scores(text)
312
+ i = int(argmax_tiebreak(s))
313
+ srt = np.sort(s)[::-1]
314
+ margin = float(srt[0] - srt[1]) if len(srt) > 1 else float(srt[0])
315
+ return self.classes[i], float(s[i]), margin
316
+
317
+ def predict_batch(self, texts):
318
+ return [self.predict(t)[0] for t in texts]
319
+
320
+ # -- persistence ---------------------------------------------------------------------------
321
+ def save(self, path):
322
+ """int8 prototypes + per-row scale. Cosine ranking survives per-row quantisation because a
323
+ positive per-row scale cannot reorder that row's own dot products against a shared query."""
324
+ scale = np.abs(self.P).max(axis=1, keepdims=True) + 1e-12
325
+ q = np.round(self.P / scale * 127.0).astype(np.int8)
326
+ os.makedirs(os.path.dirname(path) or ".", exist_ok=True)
327
+ payload = dict(classes=np.array(self.classes), q=q, scale=scale.astype(np.float32),
328
+ config=np.array([json.dumps(self.enc.config())]))
329
+ if self.enc.idf is not None:
330
+ payload["idf"] = self.enc.idf.astype(np.float16) # 16 KB; weights, not a vocabulary
331
+ np.savez_compressed(path, **payload)
332
+ return path
333
+
334
+ @classmethod
335
+ def load(cls, path=ROUTER_PATH):
336
+ z = np.load(path, allow_pickle=False)
337
+ cfg = json.loads(str(z["config"][0]))
338
+ idf = z["idf"].astype(np.float64) if "idf" in z.files else None
339
+ obj = cls([str(c) for c in z["classes"]], TextEncoder(idf=idf, **cfg))
340
+ obj.P = obj._unit(z["q"].astype(np.float64) * z["scale"].astype(np.float64) / 127.0)
341
+ return obj
342
+
343
+
344
+ # ---------------------------------------------------------------------------------------------------
345
+ # 3. HARD CONSTRAINTS -- facts about the request, not opinions about it
346
+ # ---------------------------------------------------------------------------------------------------
347
+ _IMG_CUE = re.compile(r"\b(image|images|screenshot|screenshots|photo|photos|pic|picture|pictures|"
348
+ r"scan|scanned|attached|attachment|diagram|chart|graph|x[- ]?ray|mockup|"
349
+ r"handwrit\w*|snap|frame|logo|infographic)\b")
350
+ _TOOL_CUE = re.compile(r"\b(tool call\w*|tool_choice|function call\w*|call the (tool|api|endpoint|"
351
+ r"function)|use the \w+ tool|invoke the|agent(ic)? loop|orchestrat\w*|"
352
+ r"tools? (and|then|as needed)|agent that)\b")
353
+ # TIGHT on purpose. The first version matched a bare 'json', so "write a function that dumps json" --
354
+ # a CODE request whose output format is irrelevant to the model -- silently applied a hard filter and
355
+ # removed 49 of 414 models from consideration. A hard constraint must be triggered by a statement
356
+ # ABOUT THE MODEL'S OUTPUT, not by the topic. When in doubt the caller passes needs_json=True.
357
+ _JSON_CUE = re.compile(r"\b(json (only|schema|output|mode)|(valid|strict|only) json|"
358
+ r"structured output\w*|json ?schema|response_format|pydantic|"
359
+ r"machine[ -]readable|no prose|conform\w* to this schema)\b")
360
+
361
+
362
+ def extract_constraints(text, has_image=False, needs_tools=None, needs_json=None,
363
+ input_tokens=None, output_tokens=None, task_class=None):
364
+ """Turn a request into the HARD facts a candidate must satisfy.
365
+
366
+ Cues are read from the text only as a FALLBACK; a caller that actually knows (it is holding an
367
+ image, it is running a tool loop) should pass the flag explicitly, and the explicit value always
368
+ wins. Token counts default to a chars/4 estimate of the prompt plus a per-class output estimate --
369
+ both are estimates and the returned dict says so, because cost() multiplies by them."""
370
+ low = text.lower()
371
+ est_in = int(input_tokens if input_tokens is not None else max(16, len(text) / 4))
372
+ est_out = int(output_tokens if output_tokens is not None else _OUT_TOKENS.get(task_class, 400))
373
+ return dict(
374
+ needs_image=bool(has_image or _IMG_CUE.search(low)),
375
+ needs_tools=bool(_TOOL_CUE.search(low)) if needs_tools is None else bool(needs_tools),
376
+ needs_json=bool(_JSON_CUE.search(low)) if needs_json is None else bool(needs_json),
377
+ est_in=est_in, est_out=est_out,
378
+ min_context=int((est_in + est_out) * 1.25) + 512, # 25% headroom, then a floor
379
+ estimated=(input_tokens is None, output_tokens is None),
380
+ )
381
+
382
+
383
+ # per-class expected completion length (tokens). Authored, not measured -- but cost is linear in it,
384
+ # so it is stated here in one place instead of hiding inside the scorer.
385
+ _OUT_TOKENS = {"code": 900, "reasoning": 1200, "longctx": 1000, "vision": 350, "bulk": 60,
386
+ "creative": 900, "translate": 500, "agentic": 250, "chat": 250, "advice": 700}
387
+
388
+ # task class -> the OpenRouter leaderboards whose revealed preference is evidence for this class,
389
+ # with weights. 'vision' and 'bulk' have NO natural leaderboard: vision is decided by a hard modality
390
+ # filter, bulk by price among the feasible. Their empty maps are the honest statement of that.
391
+ _CLASS_CATEGORIES = {
392
+ "code": {"programming": 1.0, "technology": 0.5},
393
+ "reasoning": {"science": 1.0, "academia": 0.7, "trivia": 0.2},
394
+ "longctx": {"academia": 0.8, "legal": 0.5, "technology": 0.4},
395
+ # PROXY, and labelled as one: OpenRouter has no vision leaderboard. With an empty map every
396
+ # vision-capable model sat at the P_UNKNOWN floor, which is below the vision bar, so EVERY image
397
+ # request answered "nothing clears the bar" -- honest but useless. 'technology' is the nearest
398
+ # revealed-preference signal (general working use), and it is weighted below 1.0 to say that it is
399
+ # borrowed evidence. The hard modality filter still does the real work here.
400
+ "vision": {"technology": 0.6, "trivia": 0.2},
401
+ # DELIBERATELY EMPTY: for bulk work the leaderboards are the wrong evidence -- popularity tracks
402
+ # capability, and this class explicitly does not want capability, it wants the cheapest thing that
403
+ # can hold a label in its head. The bulk bar sits BELOW P_UNKNOWN so the objective degenerates to
404
+ # "cheapest feasible", which is the correct answer for the class rather than a missing one.
405
+ "bulk": {},
406
+ "creative": {"roleplay": 1.0, "marketing": 0.7, "marketing/seo": 0.4},
407
+ "translate": {"translation": 1.0},
408
+ "agentic": {"programming": 0.8, "technology": 0.8},
409
+ "chat": {"trivia": 1.0, "technology": 0.3},
410
+ "advice": {"legal": 0.8, "health": 0.8, "finance": 0.8},
411
+ }
412
+
413
+ # the bar: minimum P_success a candidate must clear, by class and difficulty. AUTHORED POLICY, the one
414
+ # number a user should tune. High where a wrong answer is expensive (advice, reasoning), low where a
415
+ # retry is cheap (bulk, chat).
416
+ _BAR = {
417
+ "code": (0.58, 0.72), "reasoning": (0.65, 0.82), "longctx": (0.60, 0.74),
418
+ "vision": (0.55, 0.70), "bulk": (0.40, 0.52), "creative": (0.50, 0.66),
419
+ "translate": (0.52, 0.68), "agentic": (0.60, 0.76), "chat": (0.40, 0.52),
420
+ "advice": (0.68, 0.80),
421
+ }
422
+
423
+ # temperature of the top-2 blend, in cosine units. At gap=0 the two classes weigh 0.5/0.5; at gap
424
+ # equal to BLEND_TEMP the runner-up weighs 0.27; by gap=3*BLEND_TEMP it is 0.05 and the hedge is
425
+ # effectively off. Measured margins on the held-out split sit around 0.01-0.15, so 0.05 puts the
426
+ # transition where the classifier actually gets torn.
427
+ BLEND_TEMP = 0.05
428
+
429
+ # ---------------------------------------------------------------------------------------------------
430
+ # BIND-FIRST: the correction that openzoo forces
431
+ #
432
+ # The first version treated "this is a huge document" as a MODEL-SELECTION problem: filter to models
433
+ # with a big context window, pay for the big window. That is how you use openrouter.ai directly, and
434
+ # it is NOT how this stack works. openzoo binds a corpus into a leCore context and retrieves against
435
+ # it -- which is why the gateway advertises context_length=128,000,000 for a model whose real window
436
+ # (max_model_len) is 1,048,576. The 128M is the bind path talking, not the model.
437
+ #
438
+ # So the right decision for a large input is not "find a bigger model", it is "BIND IT, then route the
439
+ # question -- which is now small -- on its own merits". That makes context a preprocessing step
440
+ # instead of a routing constraint, and it stops a 300-page document from forcing an expensive
441
+ # long-window model when a cheap one will answer the retrieved slice perfectly well.
442
+ #
443
+ # KEPT NEGATIVE: this is right for openzoo and WRONG for a caller hitting a provider directly with no
444
+ # bind path. Hence `bindable` -- default True here because this repo's stack has one, and callers
445
+ # without one pass bindable=False and get the old context-window filter.
446
+ BIND_ABOVE_TOKENS = 60_000 # above this, binding beats buying a bigger window
447
+ BIND_SLICE_TOKENS = 8_000 # what a retrieval against a bound context actually puts on the wire
448
+
449
+ _HARD_CUE = re.compile(r"\b(prove|proof|derive|rigorous|production|carefully|step by step|"
450
+ r"complex|subtle|edge case\w*|optimis\w*|optimiz\w*|architect\w*|"
451
+ r"security|correctness|exactly|strictly|must)\b")
452
+
453
+
454
+ def difficulty(text, margin=1.0, est_in=0):
455
+ """easy|hard for the request. A HEURISTIC, not a learned head -- there are no difficulty labels in
456
+ the corpus, so inventing a trained one would be dressing a guess as a model. Three observable
457
+ signals: explicit rigour cues, a long prompt, and a low classifier margin (the request straddles
458
+ classes, so a weaker model is likelier to pick the wrong reading)."""
459
+ hits = bool(_HARD_CUE.search(text.lower())) + (est_in > 2000) + (margin < 0.05)
460
+ return "hard" if hits >= 1 else "easy"
461
+
462
+
463
+ # ---------------------------------------------------------------------------------------------------
464
+ # 4. OUTCOMES -- how the prior becomes a measurement
465
+ # ---------------------------------------------------------------------------------------------------
466
+ class Outcomes:
467
+ """Per-(class, model) success counts, so P_success stops being a proxy.
468
+
469
+ Beta posterior mean with the leaderboard prior as pseudo-counts: p = (a0*p0 + s) / (a0 + n), a0 =
470
+ PRIOR_STRENGTH. With no data this returns the prior exactly; after n real results the prior's pull
471
+ decays like a0/(a0+n). This is the only path in the file by which a number becomes earned."""
472
+
473
+ PRIOR_STRENGTH = 6.0
474
+
475
+ def __init__(self, path=OUTCOMES_PATH):
476
+ self.path = path
477
+ self.tab = {}
478
+ if path and os.path.exists(path):
479
+ with open(path) as f:
480
+ self.tab = json.load(f)
481
+
482
+ @staticmethod
483
+ def _key(task_class, model_id):
484
+ return f"{task_class}|{model_id}"
485
+
486
+ def record(self, task_class, model_id, ok):
487
+ k = self._key(task_class, model_id)
488
+ s, n = self.tab.get(k, [0, 0])
489
+ self.tab[k] = [s + (1 if ok else 0), n + 1]
490
+ if self.path:
491
+ os.makedirs(os.path.dirname(self.path) or ".", exist_ok=True)
492
+ with open(self.path, "w") as f:
493
+ json.dump(self.tab, f, indent=0, sort_keys=True)
494
+ return self.tab[k]
495
+
496
+ def posterior(self, task_class, model_id, prior):
497
+ s, n = self.tab.get(self._key(task_class, model_id), [0, 0])
498
+ a0 = self.PRIOR_STRENGTH
499
+ p = (a0 * prior + s) / (a0 + n)
500
+ return float(p), int(n)
501
+
502
+
503
+ # ---------------------------------------------------------------------------------------------------
504
+ # 5. THE CATALOGUE -- 414 models as columns, and the decision over them
505
+ # ---------------------------------------------------------------------------------------------------
506
+ class Catalog:
507
+ """The OpenRouter snapshot as numpy columns, plus feasibility, cost, and P_success."""
508
+
509
+ # prior mapped from leaderboard rank 1..20; rank 1 -> P_TOP, rank 20 -> P_BOT.
510
+ P_TOP, P_BOT = 0.88, 0.62
511
+ # a model on NO relevant leaderboard: we do not know. This floor is deliberately below every bar
512
+ # except bulk/chat, so an unmeasured model can win cheap high-volume work (where being wrong is
513
+ # cheap) but cannot silently win a proof or a legal question.
514
+ P_UNKNOWN = 0.45
515
+ P_UNKNOWN_TOOLS = 0.05 # + for declaring tools/structured output when the task needs them
516
+ P_UNKNOWN_REASON = 0.06 # + for declaring reasoning support on a reasoning-class task
517
+
518
+ def __init__(self, path=CATALOG_PATH):
519
+ z = np.load(path, allow_pickle=False)
520
+ self.ids = [str(s) for s in z["ids"]]
521
+ self.categories = [str(s) for s in z["categories"]]
522
+ self.ctx = z["ctx"]
523
+ self.price_in = z["price_in"].astype(np.float64) # USD per 1M prompt tokens
524
+ self.price_out = z["price_out"].astype(np.float64)
525
+ self.modality = z["modality"]
526
+ self.tools = z["tools"].astype(bool)
527
+ self.jsonmode = z["jsonmode"].astype(bool)
528
+ self.reasoning = z["reasoning"].astype(bool)
529
+ self.ranks = z["ranks"] # M x C, 0 = unranked
530
+ self.stamp = str(z["stamp"][0]) if "stamp" in z.files else "?"
531
+ self._cat_ix = {c: i for i, c in enumerate(self.categories)}
532
+
533
+ def __len__(self):
534
+ return len(self.ids)
535
+
536
+ # -- the four terms of the objective -------------------------------------------------------
537
+ def feasible(self, cons, allow_free=True):
538
+ """Boolean mask: models that CAN do the job at all. Pure metadata, no scoring.
539
+
540
+ allow_free=False drops $0 endpoints. THEY WIN EVERYTHING OTHERWISE -- a free model that clears
541
+ the bar has zero cost, so the objective picks it every time, correctly and uselessly if you
542
+ cannot live with free-tier rate limits and data policies. That is a deployment fact the router
543
+ cannot read from the catalogue, so it is a caller decision, defaulting to the honest reading of
544
+ the objective (free is cheapest)."""
545
+ ok = np.ones(len(self.ids), dtype=bool)
546
+ if not allow_free:
547
+ ok &= (self.price_in > 0) | (self.price_out > 0)
548
+ if cons["needs_image"]:
549
+ ok &= (self.modality & MOD_IMAGE) > 0
550
+ if cons["needs_tools"]:
551
+ ok &= self.tools
552
+ if cons["needs_json"]:
553
+ ok &= self.jsonmode
554
+ ok &= self.ctx >= cons["min_context"]
555
+ # unpriced or NEGATIVELY priced -> uncostable. isfinite alone is NOT enough: a -3 sailed
556
+ # through it once and would have won every route by being "cheaper" than free.
557
+ ok &= np.isfinite(self.price_in) & np.isfinite(self.price_out)
558
+ ok &= (self.price_in >= 0) & (self.price_out >= 0)
559
+ return ok
560
+
561
+ def cost(self, cons):
562
+ """USD per task for every model, from the token estimates. Vectorised, no filtering."""
563
+ return self.price_in * cons["est_in"] / 1e6 + self.price_out * cons["est_out"] / 1e6
564
+
565
+ def prior(self, task_class, cons):
566
+ """P_success prior for every model. Leaderboard-derived where available, floored where not."""
567
+ weights = _CLASS_CATEGORIES.get(task_class, {})
568
+ p = np.full(len(self.ids), self.P_UNKNOWN)
569
+ if weights:
570
+ num = np.zeros(len(self.ids))
571
+ den = np.zeros(len(self.ids))
572
+ span = self.P_TOP - self.P_BOT
573
+ for cat, w in weights.items():
574
+ j = self._cat_ix.get(cat)
575
+ if j is None:
576
+ continue
577
+ r = self.ranks[:, j].astype(np.float64)
578
+ seen = r > 0
579
+ pc = self.P_TOP - (r - 1.0) / 19.0 * span
580
+ num[seen] += w * pc[seen]
581
+ den[seen] += w
582
+ hit = den > 0
583
+ p[hit] = num[hit] / den[hit]
584
+ # metadata nudges apply ONLY to the unknowns: a declared capability is weak evidence, and it
585
+ # must never outrank an actual leaderboard placement.
586
+ unknown = p == self.P_UNKNOWN
587
+ if cons["needs_tools"] or cons["needs_json"]:
588
+ p[unknown & self.tools] += self.P_UNKNOWN_TOOLS
589
+ if task_class in ("reasoning", "code"):
590
+ p[unknown & self.reasoning] += self.P_UNKNOWN_REASON
591
+ return np.clip(p, 0.0, 1.0)
592
+
593
+ def bar(self, task_class, diff):
594
+ lo, hi = _BAR.get(task_class, (0.55, 0.70))
595
+ return hi if diff == "hard" else lo
596
+
597
+
598
+ def route(text, catalog=None, classifier=None, outcomes=None, k=5, bar_shift=0.0,
599
+ allow_free=True, blend=True, context=None, bindable=True, **cons_kw):
600
+ """The whole decision. Returns a dict: the chosen model, the shortlist, and every input to the
601
+ choice so a human can disagree with a specific number instead of with a vibe.
602
+
603
+ bar_shift raises (>0, be pickier) or lowers (<0, be cheaper) the success bar. It is the one knob
604
+ that changes the answer without changing the code."""
605
+ catalog = catalog if catalog is not None else Catalog()
606
+ classifier = classifier if classifier is not None else TaskClassifier.load()
607
+ outcomes = outcomes if outcomes is not None else Outcomes()
608
+
609
+ # CLASSIFY on text+context, but SIZE the job from `text`. A live agent turn is usually a fragment
610
+ # ('fix it', 'did this work?') that carries no routable task alone, so the caller passes the
611
+ # conversation so far as `context`; it informs WHAT the task is without inflating the token
612
+ # estimate for THIS call, which the caller should supply via input_tokens when it knows.
613
+ classify_on = f"{context}\n{text}" if context else text
614
+ s = np.atleast_1d(np.asarray(classifier.scores(classify_on), dtype=np.float64))
615
+ order2 = np.argsort(-s)[:2]
616
+ if len(order2) == 1: # a single-class classifier: nothing to hedge
617
+ order2 = np.array([order2[0], order2[0]])
618
+ cls = classifier.classes[int(argmax_tiebreak(s))]
619
+ conf, margin = float(s[order2[0]]), float(s[order2[0]] - s[order2[1]])
620
+ cons = extract_constraints(text, task_class=cls, **cons_kw)
621
+
622
+ # BIND-FIRST. Decide this BEFORE feasibility, because it changes what the models are being asked
623
+ # to hold. A bound corpus never reaches the model whole; a retrieved slice does.
624
+ raw_in = cons["est_in"]
625
+ bind_first = bool(bindable and raw_in > BIND_ABOVE_TOKENS)
626
+ if bind_first:
627
+ cons = dict(cons, est_in=BIND_SLICE_TOKENS,
628
+ min_context=int((BIND_SLICE_TOKENS + cons["est_out"]) * 1.25) + 512)
629
+ cons["bind_first"] = bind_first
630
+ cons["raw_in"] = raw_in
631
+
632
+ diff = difficulty(classify_on, margin=margin, est_in=raw_in)
633
+
634
+ # BLEND THE TOP TWO CLASSES by their score share. When the classifier is confident the runner-up
635
+ # weight is negligible and this is identical to committing; when it is torn -- which on the
636
+ # held-out split is exactly where it is wrong -- the prior and the bar become the mixture instead
637
+ # of a coin flip. Uncertainty should cost a little money, not a failed task.
638
+ c2 = classifier.classes[int(order2[1])] if len(order2) > 1 else cls
639
+ # Softmax over the top-2 with a COSINE-SCALE temperature. Normalising the raw scores instead
640
+ # (w = s / sum(s)) was the first version and it is wrong: cosines 0.30 vs 0.29 are a confident
641
+ # call, but that formula reports 0.51/0.49 and hedges every single request. The gap is the signal,
642
+ # so the weight has to be a function of the gap, not of the magnitudes.
643
+ gap = float(s[order2[0]] - s[order2[1]])
644
+ w2 = float(np.exp(-max(0.0, gap) / BLEND_TEMP))
645
+ w = np.array([1.0, w2]) / (1.0 + w2)
646
+ if not blend:
647
+ w = np.array([1.0, 0.0])
648
+
649
+ feas = catalog.feasible(cons, allow_free=allow_free)
650
+ cost = catalog.cost(cons)
651
+ prior = w[0] * catalog.prior(cls, cons) + w[1] * catalog.prior(c2, cons)
652
+ bar = float(np.clip(w[0] * catalog.bar(cls, diff) + w[1] * catalog.bar(c2, diff) + bar_shift,
653
+ 0.0, 1.0))
654
+ post = np.array([outcomes.posterior(cls, mid, prior[i])[0] for i, mid in enumerate(catalog.ids)])
655
+ nobs = np.array([outcomes.posterior(cls, mid, prior[i])[1] for i, mid in enumerate(catalog.ids)])
656
+
657
+ idx = np.flatnonzero(feas)
658
+ if len(idx) == 0:
659
+ # EVERY return from route() carries the same keys. The first version omitted bind_first
660
+ # here, so a caller that always reads r["bind_first"] crashed on exactly the path where it
661
+ # most needed an answer. A test caught it; the contract is now uniform.
662
+ return dict(model=None, task_class=cls, runner_up=c2, bind_first=cons["bind_first"],
663
+ difficulty=diff, bar=bar, constraints=cons, shortlist=[],
664
+ reason="no model in the catalogue satisfies the hard constraints",
665
+ cleared_bar=False, feasible_models=0, catalog_stamp=catalog.stamp)
666
+
667
+ clears = idx[post[idx] >= bar]
668
+ cleared = len(clears) > 0
669
+ if cleared:
670
+ # THE OBJECTIVE: cheapest that clears. Ties in cost break by higher P, then by id -- so the
671
+ # answer never depends on catalogue order.
672
+ pool = clears
673
+ order = sorted(pool, key=lambda i: (round(float(cost[i]), 12), -post[i], catalog.ids[i]))
674
+ else:
675
+ # Nothing clears: fall back to the strongest available and SAY SO. Cheapness is not a
676
+ # consolation prize when the task probably will not get done.
677
+ pool = idx
678
+ order = sorted(pool, key=lambda i: (-post[i], round(float(cost[i]), 12), catalog.ids[i]))
679
+
680
+ # Relative cost against the cheapest FEASIBLE model, floored at $1-per-million-tasks. Without the
681
+ # floor a free ($0) baseline makes every ratio 1e8-ish nonsense; with it, "37x" against a free
682
+ # model means "37 dollars per million tasks more than free", which is a real thing to know.
683
+ COST_FLOOR = 1e-6
684
+ cheapest_feasible = max(float(np.min(cost[idx])), COST_FLOOR)
685
+ short = []
686
+ for i in order[:k]:
687
+ short.append(dict(
688
+ model=catalog.ids[i],
689
+ p_success=round(float(post[i]), 3),
690
+ evidence=(f"measured(n={int(nobs[i])})" if nobs[i] else "prior"),
691
+ usd_per_task=round(float(cost[i]), 6),
692
+ relative_cost=round(float(cost[i]) / cheapest_feasible, 2) if cheapest_feasible else None,
693
+ context=int(catalog.ctx[i]),
694
+ ))
695
+ top = order[0]
696
+ return dict(
697
+ model=catalog.ids[top], task_class=cls, runner_up=c2, bind_first=cons["bind_first"],
698
+ blend=(round(float(w[0]), 3), round(float(w[1]), 3)),
699
+ class_confidence=round(conf, 3),
700
+ class_margin=round(margin, 3), difficulty=diff, bar=round(bar, 3),
701
+ p_success=round(float(post[top]), 3), usd_per_task=round(float(cost[top]), 6),
702
+ cleared_bar=cleared, feasible_models=int(len(idx)), shortlist=short,
703
+ constraints=cons, catalog_stamp=catalog.stamp,
704
+ reason=(("bind %d tokens to leCore first, then " % cons["raw_in"] if cons["bind_first"]
705
+ else "")
706
+ + "cheapest of %d feasible models clearing P>=%.2f for a %s %s task"
707
+ % (int((post[idx] >= bar).sum()), bar, diff, cls) if cleared else
708
+ "NO feasible model clears P>=%.2f for a %s %s task; returning highest-P instead"
709
+ % (bar, diff, cls)),
710
+ )
711
+
712
+
713
+ # ---------------------------------------------------------------------------------------------------
714
+ def _selftest():
715
+ """Assert the CONTRACTS, not the vibes: determinism of the encoder, that the objective actually
716
+ minimises cost subject to the bar, that hard constraints are hard, and that recorded outcomes move
717
+ P away from the prior. Uses a small synthetic catalogue so it needs no network and no artifact."""
718
+ import tempfile
719
+
720
+ # 1) encoder: deterministic, vocabulary-free, typo-tolerant
721
+ enc = TextEncoder(dim=1024, seed=3)
722
+ a, b = enc.encode("write a python function"), enc.encode("write a python function")
723
+ assert np.array_equal(a, b), "encoder must be deterministic"
724
+ assert enc.encode("zzzz qqqq wwww").shape == (1024,), "unseen words must still encode"
725
+ typo = float(enc.encode("write a python fucntion") @ a)
726
+ unrel = float(enc.encode("translate this into japanese") @ a)
727
+ assert typo > unrel + 0.2, (typo, unrel) # char n-grams carry the misspelling
728
+
729
+ # 2) classifier: learns a separable toy task and abstains sensibly on a straddle
730
+ toy_x = ["write a python function", "debug this rust program", "fix the failing test",
731
+ "translate this into french", "how do you say hello in italian",
732
+ "localise these strings for german"]
733
+ toy_y = ["code", "code", "code", "translate", "translate", "translate"]
734
+ # BOTH heads must separate the toy task -- the fast path and its reference (F22). If only one
735
+ # does, the features are fine and the head is broken; if neither does, the features are broken.
736
+ for method, kw in (("ridge", {}), ("adapthd", dict(epochs=20))):
737
+ c = TaskClassifier(["code", "translate"], enc).fit(toy_x, toy_y, method=method, **kw)
738
+ assert c.predict("refactor this function")[0] == "code", method
739
+ assert c.predict("translate this paragraph into spanish")[0] == "translate", method
740
+ clf = TaskClassifier(["code", "translate"], enc).fit(toy_x, toy_y)
741
+
742
+ # 3) round-trip through int8 must not change the decision
743
+ fd, p = tempfile.mkstemp(suffix=".npz"); os.close(fd)
744
+ clf.save(p)
745
+ clf2 = TaskClassifier.load(p)
746
+ assert clf2.predict("refactor this function")[0] == "code", "int8 round-trip changed the label"
747
+ os.remove(p)
748
+
749
+ # 4) the objective, on a synthetic catalogue: cheap+capable must beat expensive+capable, and a
750
+ # model that fails a HARD constraint must be unreachable at any price or probability.
751
+ fd, cp = tempfile.mkstemp(suffix=".npz"); os.close(fd)
752
+ ids = ["cheap/good", "pricey/good", "cheap/blind", "cheap/weak"]
753
+ np.savez(cp, ids=np.array(ids), ctx=np.array([200000] * 4), categories=np.array(["programming"]),
754
+ price_in=np.array([0.5, 10.0, 0.4, 0.3], dtype=np.float32),
755
+ price_out=np.array([1.0, 30.0, 0.8, 0.6], dtype=np.float32),
756
+ modality=np.array([MOD_TEXT | MOD_IMAGE, MOD_TEXT | MOD_IMAGE, MOD_TEXT, MOD_TEXT],
757
+ dtype=np.uint8),
758
+ tools=np.array([True, True, True, False]), jsonmode=np.array([True] * 4),
759
+ reasoning=np.array([True] * 4),
760
+ ranks=np.array([[2], [1], [3], [0]], dtype=np.uint8),
761
+ stamp=np.array(["selftest"]))
762
+ cat = Catalog(cp)
763
+ cons = extract_constraints("write a python function", task_class="code")
764
+ prior = cat.prior("code", cons)
765
+ assert prior[3] == Catalog.P_UNKNOWN + Catalog.P_UNKNOWN_REASON, prior # unranked -> floor + nudge
766
+ assert prior[1] > prior[0] > prior[2], prior # rank order preserved
767
+
768
+ clf3 = TaskClassifier(["code"], enc).fit(["write a python function"], ["code"])
769
+ out = Outcomes(path=None)
770
+ r = route("write a python function", catalog=cat, classifier=clf3, outcomes=out)
771
+ assert r["cleared_bar"] is True
772
+ # the contract is the OBJECTIVE itself, computed here independently: among the models clearing the
773
+ # bar, the chosen one is the cheapest -- NOT the highest-P one. pricey/good has the best prior and
774
+ # must lose anyway; that losing is the entire point of the router.
775
+ clearing = [s for s in r["shortlist"] if s["p_success"] >= r["bar"]]
776
+ assert r["model"] == min(clearing, key=lambda s: s["usd_per_task"])["model"], r
777
+ assert r["model"] != "pricey/good", r
778
+ assert r["p_success"] >= r["bar"], r
779
+
780
+ rv = route("what is in this screenshot", catalog=cat, classifier=clf3, outcomes=out,
781
+ has_image=True)
782
+ assert all(s["model"] in ("cheap/good", "pricey/good") for s in rv["shortlist"]), rv
783
+ assert rv["feasible_models"] == 2, rv # the two text-only models are GONE, not demoted
784
+
785
+ # 5) a bar nobody clears must be reported, not papered over
786
+ rhard = route("prove this rigorously", catalog=cat, classifier=clf3, outcomes=out, bar_shift=0.5)
787
+ assert rhard["cleared_bar"] is False and "NO feasible model" in rhard["reason"], rhard
788
+ assert rhard["model"] == "pricey/good", rhard # highest P, explicitly labelled as a fallback
789
+
790
+ # 6) outcomes move the number: 12 failures on the cheap model must dethrone it
791
+ out2 = Outcomes(path=None)
792
+ for _ in range(12):
793
+ out2.record("code", "cheap/good", False)
794
+ r2 = route("write a python function", catalog=cat, classifier=clf3, outcomes=out2)
795
+ assert r2["model"] != "cheap/good", r2
796
+ # the demoted model fell OUT of the clearing set entirely, so it is absent from the shortlist --
797
+ # check the mechanism at its source rather than through the survivors.
798
+ p_before = float(cat.prior("code", cons)[ids.index("cheap/good")])
799
+ p_after, n_after = out2.posterior("code", "cheap/good", p_before)
800
+ assert n_after == 12 and p_after < p_before - 0.3, (p_before, p_after, n_after)
801
+ assert "cheap/good" not in [s["model"] for s in r2["shortlist"]], r2
802
+ os.remove(cp)
803
+ print("holographic_modelroute selftest: OK (deterministic encoder; int8 round-trip preserves the "
804
+ "label; cheapest-clearing-the-bar wins; hard constraints remove candidates; an uncleared bar "
805
+ "is reported; recorded outcomes override the prior)")
806
+
807
+
808
+ if __name__ == "__main__":
809
+ _selftest()