vesma-cortex 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,447 @@
1
+ """Candidate N — small MLP head over frozen vesma-embed-v1 vectors.
2
+
3
+ ADR 0001 V1: head ≤ ~0.5M parameters over [a, b, |a−b|, a⊙b] (4×384)
4
+ plus the scalar feature block; corruption pretrain (cortex.pretrain),
5
+ NO own encoder, NO external pretrain weights, cosine distillation
6
+ forbidden. Prod wiring is GATED on the canon ADR 0004 addendum (vectors
7
+ in CanonState); the eval runner (A5) reads vectors itself and is not
8
+ blocked. torch — train extra only (CPU; XPU walkthrough per charter §6).
9
+
10
+ A3b implementation notes (frozen here):
11
+
12
+ - torch is imported INSIDE every method: ``import cortex.candidates`` must
13
+ stay green in a torch-less environment (default ``uv run pytest`` —
14
+ tests gate themselves with ``pytest.importorskip("torch")``).
15
+ - Determinism: ``torch.manual_seed(config.seed)`` before module creation,
16
+ single-threaded CPU fit, deterministic algorithms enforced for the fit
17
+ window (previous global state restored afterwards). Same config + data
18
+ → identical weights (pinned by tests).
19
+ - Vector block: [a, b, |a−b|, a⊙b] flattened (4×384=1536) concatenated
20
+ with the scalar block (FEATURE_NAMES) — corruption pairs reuse the BASE
21
+ record's store vector on both sides (self-pair convention,
22
+ cortex.pretrain.corruption), so the head must read divergence from the
23
+ scalars: exactly the W4c lesson as training signal.
24
+ - Pretrain is SUPERVISED-on-construction (weak positive=1, hard
25
+ negative=0, labels by construction — allowed for PRETRAIN, ADR 0001
26
+ V2/P3); ``train()`` then fine-tunes from the pretrained state so owner
27
+ labels override the initialization (V2: "pretrain-инициализация обязана
28
+ перебиваться владельческими метками").
29
+ - ``pretrain()`` gained the ``labels`` keyword (A3b signature fix): the
30
+ corruption labels-by-construction ARE the signal; silently assuming a
31
+ label would be dishonest.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import json
37
+ from collections.abc import Mapping, Sequence
38
+ from dataclasses import asdict, dataclass
39
+ from pathlib import Path
40
+ from typing import Final
41
+
42
+ import numpy as np
43
+
44
+ from cortex.features.pair import FEATURE_NAMES, FeatureVector
45
+
46
+ __all__ = ["GRID_N", "MAX_HEAD_PARAMS", "N_EPOCHS", "NGridConfig", "NHeadModel"]
47
+
48
+ #: Parameter budget (ADR 0001 V1: "голова ≤ ~0.5M параметров").
49
+ MAX_HEAD_PARAMS: Final[int] = 500_000
50
+
51
+ #: Vector block layout fed to the head: [a, b, |a−b|, a⊙b] → 4 × 384.
52
+ VECTOR_BLOCK_ROWS: Final[int] = 4
53
+ VECTOR_DIM: Final[int] = 384
54
+
55
+ #: Full-batch Adam epochs — fixed (not a grid axis) so the frozen grid
56
+ #: stays ≤ 8 configurations and fits stay deterministic.
57
+ N_EPOCHS: Final[int] = 100
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class NGridConfig:
62
+ """One frozen grid point. Hidden layers 1–2, widths 32–128."""
63
+
64
+ name: str
65
+ hidden_dims: tuple[int, ...]
66
+ dropout: float
67
+ learning_rate: float
68
+ pretrain: bool # ablation axis: with/without corruption pretrain
69
+ seed: int
70
+
71
+
72
+ #: Frozen grid, ≤ 8 configurations (ADR 0001 V1). Three architectures over
73
+ #: the plain supervised path plus two corruption-pretrain points (the
74
+ #: ADR-mandated с/без-претрейна ablation).
75
+ GRID_N: Final[tuple[NGridConfig, ...]] = (
76
+ NGridConfig(
77
+ "n-h64",
78
+ hidden_dims=(64,),
79
+ dropout=0.1,
80
+ learning_rate=1e-3,
81
+ pretrain=False,
82
+ seed=1,
83
+ ),
84
+ NGridConfig(
85
+ "n-h128",
86
+ hidden_dims=(128,),
87
+ dropout=0.1,
88
+ learning_rate=1e-3,
89
+ pretrain=False,
90
+ seed=1,
91
+ ),
92
+ NGridConfig(
93
+ "n-h64-32",
94
+ hidden_dims=(64, 32),
95
+ dropout=0.1,
96
+ learning_rate=1e-3,
97
+ pretrain=False,
98
+ seed=1,
99
+ ),
100
+ NGridConfig(
101
+ "n-h64-pt",
102
+ hidden_dims=(64,),
103
+ dropout=0.1,
104
+ learning_rate=1e-3,
105
+ pretrain=True,
106
+ seed=1,
107
+ ),
108
+ NGridConfig(
109
+ "n-h128-64-pt",
110
+ hidden_dims=(128, 64),
111
+ dropout=0.1,
112
+ learning_rate=1e-3,
113
+ pretrain=True,
114
+ seed=1,
115
+ ),
116
+ )
117
+
118
+
119
+ def vector_block(vec_a: Sequence[float], vec_b: Sequence[float]) -> np.ndarray:
120
+ """Assemble the [a, b, |a−b|, a⊙b] block — shape (4, dim), float32.
121
+
122
+ Store vectors must be non-empty twins; unit-normalization is the store
123
+ contract (vesma-embed-v1) and is NOT re-imposed here (never re-measure
124
+ — the block mirrors what the engine would pass).
125
+ """
126
+ a = np.asarray(vec_a, dtype=np.float32)
127
+ b = np.asarray(vec_b, dtype=np.float32)
128
+ if a.ndim != 1 or b.ndim != 1 or a.size == 0 or a.shape != b.shape:
129
+ raise ValueError(
130
+ f"vector pair must be non-empty 1-D twins, got {a.shape} vs {b.shape}"
131
+ )
132
+ return np.stack([a, b, np.abs(a - b), a * b])
133
+
134
+
135
+ class NHeadModel:
136
+ """MLP head over frozen store vectors + scalar features.
137
+
138
+ Shares the three-method candidate surface with DBoostModel. Vectors
139
+ enter as the (4, 384) block; scalars as FEATURE_NAMES block — the
140
+ exported graph takes BOTH inputs (inference-v1.md §4, variant N).
141
+ """
142
+
143
+ feature_names: tuple[str, ...] = FEATURE_NAMES
144
+
145
+ def __init__(self) -> None:
146
+ self._module: object | None = None # torch.nn.Module (lazy import)
147
+ self._config: NGridConfig | None = None
148
+ self._pretrained: bool = False
149
+
150
+ # ── torch plumbing (lazy import boundary) ─────────────────────────────────
151
+
152
+ def _build_module(self, config: NGridConfig) -> object:
153
+ import torch
154
+ from torch import nn
155
+
156
+ class Head(nn.Module):
157
+ def __init__(self, scalar_dim: int) -> None:
158
+ super().__init__()
159
+ dims = [
160
+ scalar_dim + VECTOR_BLOCK_ROWS * VECTOR_DIM,
161
+ *config.hidden_dims,
162
+ ]
163
+ layers: list[nn.Module] = []
164
+ for in_dim, out_dim in zip(dims, dims[1:]):
165
+ layers.append(nn.Linear(in_dim, out_dim))
166
+ layers.append(nn.ReLU())
167
+ if config.dropout > 0:
168
+ layers.append(nn.Dropout(config.dropout))
169
+ layers.append(nn.Linear(dims[-1], 1))
170
+ self.net = nn.Sequential(*layers)
171
+
172
+ def forward(self, scalars, vectors):
173
+ flat = torch.flatten(vectors, start_dim=-2)
174
+ x = torch.cat([scalars, flat], dim=-1)
175
+ return torch.sigmoid(self.net(x))
176
+
177
+ torch.manual_seed(config.seed)
178
+ module = Head(len(FEATURE_NAMES))
179
+ n_params = sum(p.numel() for p in module.parameters())
180
+ if n_params > MAX_HEAD_PARAMS:
181
+ raise ValueError(
182
+ f"head has {n_params} parameters — exceeds the "
183
+ f"{MAX_HEAD_PARAMS} budget (ADR 0001 V1)"
184
+ )
185
+ return module
186
+
187
+ def _fit_loop(
188
+ self,
189
+ scalars: np.ndarray,
190
+ blocks: np.ndarray,
191
+ labels: np.ndarray,
192
+ config: NGridConfig,
193
+ ) -> None:
194
+ """Full-batch Adam BCE fit — deterministic under config.seed."""
195
+ import torch
196
+
197
+ y = torch.as_tensor(labels, dtype=torch.float32).unsqueeze(-1)
198
+ x_scalars = torch.as_tensor(scalars, dtype=torch.float32)
199
+ x_vectors = torch.as_tensor(blocks, dtype=torch.float32)
200
+
201
+ previous_threads = torch.get_num_threads()
202
+ previous_deterministic = torch.are_deterministic_algorithms_enabled()
203
+ torch.set_num_threads(1)
204
+ torch.use_deterministic_algorithms(True)
205
+ try:
206
+ if (
207
+ self._module is not None
208
+ and self._config is not None
209
+ and self._config.hidden_dims != config.hidden_dims
210
+ ):
211
+ raise ValueError(
212
+ "architecture changed between pretrain and train — "
213
+ f"{self._config.hidden_dims} vs {config.hidden_dims}; "
214
+ "pretrained weights cannot carry over (rebuild explicitly)"
215
+ )
216
+ if self._module is None:
217
+ self._module = self._build_module(config)
218
+ module = self._module
219
+ module.train()
220
+ optimizer = torch.optim.Adam(module.parameters(), lr=config.learning_rate)
221
+ loss_fn = torch.nn.BCELoss()
222
+ for _ in range(N_EPOCHS):
223
+ optimizer.zero_grad()
224
+ probs = module(x_scalars, x_vectors)
225
+ loss = loss_fn(probs, y)
226
+ loss.backward()
227
+ optimizer.step()
228
+ finally:
229
+ torch.set_num_threads(previous_threads)
230
+ torch.use_deterministic_algorithms(previous_deterministic)
231
+ self._config = config
232
+
233
+ def _validate_input(
234
+ self,
235
+ vectors: Sequence[FeatureVector],
236
+ vector_blocks: Sequence[np.ndarray],
237
+ labels: Sequence[int] | None = None,
238
+ ) -> tuple[np.ndarray, np.ndarray, np.ndarray | None]:
239
+ if not vectors:
240
+ raise ValueError("training/inference requires at least one pair")
241
+ names = {v.names for v in vectors}
242
+ if len(names) != 1 or names != {FEATURE_NAMES}:
243
+ raise ValueError(
244
+ f"N head consumes the CORE scalar block ({len(FEATURE_NAMES)} features), "
245
+ f"got {len(next(iter(names))) if len(names) == 1 else 'mixed'}"
246
+ )
247
+ blocks = np.asarray(vector_blocks, dtype=np.float32)
248
+ if blocks.shape != (len(vectors), VECTOR_BLOCK_ROWS, VECTOR_DIM):
249
+ raise ValueError(
250
+ "vector_blocks must have shape "
251
+ f"({len(vectors)}, {VECTOR_BLOCK_ROWS}, {VECTOR_DIM}), got {blocks.shape}"
252
+ )
253
+ scalars = np.asarray([v.values for v in vectors], dtype=np.float32)
254
+ if labels is None:
255
+ return scalars, blocks, None
256
+ y = np.asarray(labels, dtype=np.int64)
257
+ if y.shape != (len(vectors),):
258
+ raise ValueError(f"labels length {len(y)} != vectors length {len(vectors)}")
259
+ if not np.isin(y, (0, 1)).all():
260
+ raise ValueError("labels must be binary {0, 1}")
261
+ if len(np.unique(y)) < 2:
262
+ raise ValueError("training requires both classes present")
263
+ return scalars, blocks, y
264
+
265
+ # ── candidate surface ─────────────────────────────────────────────────────
266
+
267
+ def train(
268
+ self,
269
+ vectors: Sequence[FeatureVector],
270
+ labels: Sequence[int],
271
+ config: NGridConfig,
272
+ *,
273
+ vector_blocks: Sequence[np.ndarray],
274
+ ) -> None:
275
+ """Supervised fit; MUST override any pretrain initialization when
276
+ pretrain weights are present (ADR 0001 V1: owner labels win).
277
+
278
+ Override semantics: supervised training runs ON TOP of the
279
+ corruption-initialized weights (two-stage scheme, addendum P3) —
280
+ the final stage is fitted on owner labels only, so they dominate
281
+ the deployed behavior.
282
+ """
283
+ scalars, blocks, y = self._validate_input(vectors, vector_blocks, labels)
284
+ self._fit_loop(scalars, blocks, y, config)
285
+
286
+ def pretrain(
287
+ self,
288
+ pairs: Sequence[tuple[FeatureVector, np.ndarray]],
289
+ config: NGridConfig,
290
+ *,
291
+ labels: Sequence[int],
292
+ ) -> None:
293
+ """Corruption-based pretrain (cortex.pretrain.corruption pairs).
294
+ Distillation of the cosine is FORBIDDEN (ADR 0001 V1).
295
+
296
+ ``pairs`` carry (scalar features, vector block); ``labels`` are the
297
+ labels-BY-CONSTRUCTION (weak positive=1, hard negative=0) — the
298
+ only supervision allowed at this stage (ADR 0001 V2/P3).
299
+ """
300
+ if not pairs:
301
+ raise ValueError("pretrain requires at least one corruption pair")
302
+ vectors = [pair[0] for pair in pairs]
303
+ blocks = [pair[1] for pair in pairs]
304
+ scalars, block_array, y = self._validate_input(vectors, blocks, labels)
305
+ self._fit_loop(scalars, block_array, y, config)
306
+ self._pretrained = True
307
+
308
+ def _require_module(self) -> object:
309
+ if self._module is None:
310
+ raise RuntimeError(
311
+ "NHeadModel is not fitted — call train()/pretrain() first"
312
+ )
313
+ return self._module
314
+
315
+ def predict_proba(
316
+ self, vectors: Sequence[FeatureVector], *, vector_blocks: Sequence[np.ndarray]
317
+ ) -> np.ndarray:
318
+ """P(duplicate) per pair, shape (n,), values in [0, 1]."""
319
+ import torch
320
+
321
+ module = self._require_module()
322
+ scalars, blocks, _ = self._validate_input(vectors, vector_blocks)
323
+ module.eval()
324
+ with torch.no_grad():
325
+ probs = module(
326
+ torch.as_tensor(scalars, dtype=torch.float32),
327
+ torch.as_tensor(blocks, dtype=torch.float32),
328
+ )
329
+ return probs.squeeze(-1).numpy().astype(np.float64)
330
+
331
+ def export_onnx(
332
+ self, path: Path, *, metadata_props: Mapping[str, str] | None = None
333
+ ) -> Path:
334
+ """torch.onnx export: inputs scalars float32[K] + vectors
335
+ float32[4, 384] → probability[1]; raises RuntimeError when the
336
+ param budget (MAX_HEAD_PARAMS) or the artifact size gate is
337
+ exceeded."""
338
+ import torch
339
+
340
+ module = self._require_module()
341
+ module.eval()
342
+ scalars_example = torch.zeros(len(FEATURE_NAMES), dtype=torch.float32)
343
+ vectors_example = torch.zeros(
344
+ VECTOR_BLOCK_ROWS, VECTOR_DIM, dtype=torch.float32
345
+ )
346
+ out_path = Path(path)
347
+ out_path.parent.mkdir(parents=True, exist_ok=True)
348
+ # dynamo=False — the legacy exporter owns opset-15 static-shape
349
+ # export (the dynamo path targets opset 18+; torch >= 2.9 defaults
350
+ # to it, so pin the legacy path explicitly).
351
+ torch.onnx.export(
352
+ module,
353
+ (scalars_example, vectors_example),
354
+ str(out_path),
355
+ opset_version=15,
356
+ input_names=["scalars", "vectors"],
357
+ output_names=["probability"],
358
+ dynamo=False,
359
+ )
360
+ self._smoke_onnx(out_path)
361
+ if metadata_props:
362
+ import onnx
363
+
364
+ from cortex.artifacts import set_onnx_metadata
365
+
366
+ model = onnx.load(str(out_path))
367
+ set_onnx_metadata(model, metadata_props)
368
+ out_path.write_bytes(model.SerializeToString())
369
+ from cortex.artifacts import assert_artifact_size
370
+
371
+ assert_artifact_size(out_path)
372
+ return out_path
373
+
374
+ def _smoke_onnx(self, path: Path) -> None:
375
+ """Eager ORT validation: zeros-in → probability[1] ∈ [0, 1],
376
+ consistent with the library forward pass."""
377
+ import onnxruntime as ort
378
+
379
+ session = ort.InferenceSession(str(path), providers=["CPUExecutionProvider"])
380
+ feeds = {
381
+ "scalars": np.zeros(len(FEATURE_NAMES), dtype=np.float32),
382
+ "vectors": np.zeros((VECTOR_BLOCK_ROWS, VECTOR_DIM), dtype=np.float32),
383
+ }
384
+ (prob,) = session.run(None, feeds)
385
+ if prob.shape != (1,):
386
+ raise RuntimeError(f"exported graph output shape {prob.shape} != (1,)")
387
+ if not (0.0 <= float(prob[0]) <= 1.0):
388
+ raise RuntimeError(f"exported probability {float(prob[0])} outside [0, 1]")
389
+ reference = float(
390
+ self.predict_proba(
391
+ [FeatureVector(FEATURE_NAMES, (0.0,) * len(FEATURE_NAMES))],
392
+ vector_blocks=[
393
+ np.zeros((VECTOR_BLOCK_ROWS, VECTOR_DIM), dtype=np.float32)
394
+ ],
395
+ )[0]
396
+ )
397
+ if abs(float(prob[0]) - reference) > 1e-5:
398
+ raise RuntimeError(
399
+ f"exported graph diverges from library prediction: {float(prob[0])} vs {reference}"
400
+ )
401
+
402
+ # ── persistence (dev pipeline: npz weights, no pickle) ────────────────────
403
+
404
+ def save(self, directory: Path) -> Path:
405
+ """Persist fitted state as numpy weights (npz) + meta json — no
406
+ pickle anywhere in the dev hand-off (defense in depth on top of
407
+ the artifact-side pickle ban)."""
408
+
409
+ module = self._require_module()
410
+ out_dir = Path(directory)
411
+ out_dir.mkdir(parents=True, exist_ok=True)
412
+ arrays = {
413
+ name: param.detach().numpy() for name, param in module.state_dict().items()
414
+ }
415
+ np.savez(out_dir / "weights.npz", **arrays)
416
+ meta = {
417
+ "candidate": "n-head",
418
+ "config": asdict(self._config) if self._config else None,
419
+ "feature_names": list(FEATURE_NAMES),
420
+ "pretrained": self._pretrained,
421
+ }
422
+ (out_dir / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
423
+ return out_dir
424
+
425
+ @classmethod
426
+ def load(cls, directory: Path) -> NHeadModel:
427
+ import torch
428
+
429
+ in_dir = Path(directory)
430
+ meta = json.loads((in_dir / "meta.json").read_text(encoding="utf-8"))
431
+ if meta.get("candidate") != "n-head":
432
+ raise ValueError(f"{in_dir} is not an n-head model directory")
433
+ config = NGridConfig(**meta["config"]) if meta.get("config") else None
434
+ if config is None:
435
+ raise ValueError(
436
+ f"{in_dir} carries no grid config — cannot rebuild the head"
437
+ )
438
+ model = cls()
439
+ model._module = model._build_module(config)
440
+ state = {
441
+ name: torch.as_tensor(array)
442
+ for name, array in np.load(in_dir / "weights.npz").items()
443
+ }
444
+ model._module.load_state_dict(state)
445
+ model._config = config
446
+ model._pretrained = bool(meta.get("pretrained", False))
447
+ return model
cortex/cli/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """CLI package — argparse stubs (A3a); bodies land in A3b+/A2+."""
2
+
3
+ from cortex.cli.main import main
4
+
5
+ __all__ = ["main"]