vesma-cortex 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cortex/__init__.py ADDED
@@ -0,0 +1,26 @@
1
+ """cortex — development library for the vesma-cortex decision model.
2
+
3
+ Dual-epoch delivery (ADR 0001, owner addendum P2):
4
+
5
+ - THIS repo is the train epoch: corpus generation, pair features, D/N
6
+ candidate training, internal CV selection, ONNX export, single-shot
7
+ evaluation (preregistration v2, frozen).
8
+ - The runtime epoch is a single self-contained ONNX artifact
9
+ (``vesma-cortex-v1``) bundled into the engine by the NanoProvider
10
+ pattern — the engine never imports this package.
11
+
12
+ Contracts: docs/specs/inference-v1.md (artifact), docs/specs/data-contract.md
13
+ (data). Slice A3a ships contracts and skeletons only; algorithm bodies land
14
+ in A3b+ (every stub raises NotImplementedError by design — silent partial
15
+ implementations would violate the honest-skeleton rule).
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ __version__ = "0.1.0"
21
+
22
+ #: Re-exported so callers can pin the artifact identity from one place
23
+ #: (single-source rule: defined in cortex.artifacts, A3a).
24
+ from cortex.artifacts import ARTIFACT_NAME
25
+
26
+ __all__ = ["ARTIFACT_NAME", "__version__"]
@@ -0,0 +1,160 @@
1
+ """Artifact contract: vesma-cortex-v1 identity, metadata, fingerprint.
2
+
3
+ Single source of truth for the artifact NAME (ADR 0001 П1: the constant
4
+ lives in exactly ONE place — see tests/test_skeleton.py guard). The full
5
+ inference contract is docs/specs/inference-v1.md; this module is its code
6
+ anchor: metadata_props assembly, the size gate, and the weights fingerprint
7
+ (sha256 over the .onnx bytes — the engine-side twin is
8
+ ``mnema_weights_sha256`` in ``src/vesmaro/embeddings/__init__.py``).
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import hashlib
14
+ from collections.abc import Mapping, Sequence
15
+ from pathlib import Path
16
+ from typing import Final
17
+
18
+ __all__ = [
19
+ "ARTIFACT_NAME",
20
+ "CANDIDATE_D",
21
+ "CANDIDATE_N",
22
+ "MAX_ARTIFACT_BYTES",
23
+ "METADATA_KEYS",
24
+ "METADATA_VERSION",
25
+ "MODEL_NAME",
26
+ "assert_artifact_size",
27
+ "build_metadata_props",
28
+ "features_digest",
29
+ "set_onnx_metadata",
30
+ "sha256_file",
31
+ ]
32
+
33
+ #: Artifact identity, ONE constant for the whole repo (ADR 0001 П1).
34
+ #: The engine bundles it under src/vesmaro/models/<ARTIFACT_NAME>/.
35
+ ARTIFACT_NAME: Final[str] = "vesma-cortex-v1"
36
+
37
+ #: Model name inside metadata_props (name ≠ artifact dir name: the dir
38
+ #: carries the major, the metadata carries the family name — engine
39
+ #: precedent: vesma-embed-v1 bundle, manifest "name" field).
40
+ MODEL_NAME: Final[str] = "vesma-cortex"
41
+
42
+ #: Metadata schema major. Weight refresh within v1 bumps to "1.<n>" with a
43
+ #: new sha256 = recalibration event (inference-v1.md §8).
44
+ METADATA_VERSION: Final[str] = "1"
45
+
46
+ #: Hard size gate (ADR 0001 V3: single ONNX ≤ 5 MB, self-contained).
47
+ MAX_ARTIFACT_BYTES: Final[int] = 5 * 1024 * 1024
48
+
49
+ #: metadata_props keys (inference-v1.md §4). Frozen set — W5d validates
50
+ #: against exactly these.
51
+ METADATA_KEYS: Final[tuple[str, ...]] = (
52
+ "name",
53
+ "version",
54
+ "embedder_pin",
55
+ "corpus_fingerprint",
56
+ "trained_at",
57
+ "candidate",
58
+ "features",
59
+ "feature_set_sha256",
60
+ )
61
+
62
+ #: Which ladder candidate the artifact carries (ADR 0001 V1).
63
+ CANDIDATE_D: Final[str] = "d-boost"
64
+ CANDIDATE_N: Final[str] = "n-head"
65
+
66
+ #: sha256 over the "\n"-joined feature names — the compact feature-contract
67
+ #: assert (inference-v1.md §4, W5d step 8).
68
+ FEATURES_JOIN: Final[str] = "\n"
69
+
70
+
71
+ def sha256_file(path: Path) -> str:
72
+ """sha256 over file bytes — the weights fingerprint discipline.
73
+
74
+ Same scheme as the engine's ``mnema_weights_sha256`` (streamed,
75
+ 1 MiB chunks): the number that changes exactly when the weights
76
+ change, i.e. the recalibration key (ADR-0021 discipline).
77
+ """
78
+ digest = hashlib.sha256()
79
+ with Path(path).open("rb") as handle:
80
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
81
+ digest.update(chunk)
82
+ return digest.hexdigest()
83
+
84
+
85
+ def features_digest(feature_names: Sequence[str]) -> str:
86
+ """sha256 hex of the "\n"-joined ordered feature names."""
87
+ joined = FEATURES_JOIN.join(feature_names)
88
+ return hashlib.sha256(joined.encode("utf-8")).hexdigest()
89
+
90
+
91
+ def build_metadata_props(
92
+ *,
93
+ version: str = METADATA_VERSION,
94
+ embedder_pin: str,
95
+ corpus_fingerprint: str,
96
+ trained_at: str,
97
+ candidate: str,
98
+ feature_names: tuple[str, ...],
99
+ ) -> dict[str, str]:
100
+ """Assemble the ONNX metadata_props for a vesma-cortex artifact.
101
+
102
+ Contract (inference-v1.md §4): ``embedder_pin`` MUST be the live
103
+ engine fingerprint string (``nano:sha256:<hex>``), ``corpus_fingerprint``
104
+ the BLAKE2b-256 of the training-pair manifest, ``feature_names`` the
105
+ frozen ordered list (cortex.features.pair.FEATURE_NAMES) joined by
106
+ newlines plus its sha256. Raises ValueError on unknown candidate.
107
+ """
108
+ if candidate not in (CANDIDATE_D, CANDIDATE_N):
109
+ raise ValueError(
110
+ f"unknown ladder candidate {candidate!r} — expected {CANDIDATE_D!r} or {CANDIDATE_N!r}"
111
+ )
112
+ if not embedder_pin:
113
+ raise ValueError(
114
+ "embedder_pin is required (live engine fingerprint, nano:sha256:<hex>)"
115
+ )
116
+ if not corpus_fingerprint:
117
+ raise ValueError(
118
+ "corpus_fingerprint is required (BLAKE2b-256 of the pair manifest)"
119
+ )
120
+ if not feature_names:
121
+ raise ValueError("feature_names must be a non-empty ordered sequence")
122
+ return {
123
+ "name": MODEL_NAME,
124
+ "version": version,
125
+ "embedder_pin": embedder_pin,
126
+ "corpus_fingerprint": corpus_fingerprint,
127
+ "trained_at": trained_at,
128
+ "candidate": candidate,
129
+ "features": FEATURES_JOIN.join(feature_names),
130
+ "feature_set_sha256": features_digest(feature_names),
131
+ }
132
+
133
+
134
+ def assert_artifact_size(onnx_path: Path) -> None:
135
+ """Enforce the ≤5 MB gate at export time (fail-loud, pre-bundle)."""
136
+ size = Path(onnx_path).stat().st_size
137
+ if size > MAX_ARTIFACT_BYTES:
138
+ raise RuntimeError(
139
+ f"artifact {onnx_path} is {size} bytes — exceeds the "
140
+ f"{MAX_ARTIFACT_BYTES}-byte gate (ADR 0001 V3, inference-v1.md §4)"
141
+ )
142
+
143
+
144
+ def set_onnx_metadata(model_proto, props: Mapping[str, str]) -> None:
145
+ """Write ``props`` into the ONNX model metadata_props (in place).
146
+
147
+ Lazy ``onnx`` import: the artifact module stays import-light for
148
+ callers that never export (onnx arrives with the skl2onnx/onnxmltools
149
+ train chain, never as a runtime dep of the engine).
150
+ """
151
+ import onnx # local: export-path only
152
+
153
+ existing = {entry.key: entry for entry in model_proto.metadata_props}
154
+ for key, value in props.items():
155
+ if key in existing:
156
+ existing[key].value = value
157
+ else:
158
+ model_proto.metadata_props.append(
159
+ onnx.StringStringEntryProto(key=key, value=value)
160
+ )
@@ -0,0 +1,13 @@
1
+ """Candidates package — the D/N ladder (ADR 0001 V1)."""
2
+
3
+ from cortex.candidates.d_boost import GRID_D, DBoostModel, DGridConfig
4
+ from cortex.candidates.n_head import GRID_N, NGridConfig, NHeadModel
5
+
6
+ __all__ = [
7
+ "GRID_D",
8
+ "GRID_N",
9
+ "DBoostModel",
10
+ "DGridConfig",
11
+ "NGridConfig",
12
+ "NHeadModel",
13
+ ]
@@ -0,0 +1,429 @@
1
+ """Candidate D — gradient boosting over the frozen pair features.
2
+
3
+ ADR 0001 V1: D is an EQUAL ADOPT-candidate and the internal bar, not a
4
+ control afterthought. LightGBM here (train env only — never a runtime dep
5
+ of the engine, addendum P2). Grids ≤ 8 configurations, frozen in code at
6
+ A3 (this file); selection protocol lives in cortex.select.cv.
7
+
8
+ A3b implementation notes (frozen here):
9
+
10
+ - Determinism: every LightGBM fit runs single-threaded with
11
+ ``deterministic=True`` + ``force_col_wise=True`` under the config seed —
12
+ two fits on identical data produce identical margins (pinned by tests).
13
+ - Calibration: Platt (sigmoid over the raw margin, p = σ(A·z+B)) fitted on
14
+ TRAIN ONLY via out-of-fold margins (internal StratifiedKFold, seed =
15
+ config seed). Falls back to identity (A=1, B=0) with a warning when the
16
+ train split cannot support CV (a class with < 2 members). Isotonic was
17
+ rejected for ~140 labeled pairs (overfits the tails; Platt is the
18
+ low-variance choice, matching the ladder philosophy).
19
+ - ONNX export (skl2onnx chain — the LightGBM converter ships in
20
+ ``onnxmltools`` since skl2onnx 1.20): the converted
21
+ TreeEnsembleClassifier emits σ(z) via post_transform; this module
22
+ rewrites post_transform to NONE and appends the Platt transform
23
+ (Mul/Add/Sigmoid) as explicit graph nodes, so the exported probability
24
+ is EXACTLY the calibrated library ``predict_proba`` (cross-checked at
25
+ export time through an onnxruntime smoke inference).
26
+ - Graph I/O contract (inference-v1.md §4, variant D): input
27
+ ``features: float32[K]`` (1-D, single pair) → output
28
+ ``probability: float32[1]``; opset 15 + ai.onnx.ml 1, IR 8.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import json
34
+ import warnings
35
+ from collections.abc import Mapping, Sequence
36
+ from dataclasses import asdict, dataclass
37
+ from pathlib import Path
38
+ from typing import Final
39
+
40
+ import numpy as np
41
+
42
+ from cortex.features.pair import FEATURE_NAMES, FIELD_COSINE_FEATURES, FeatureVector
43
+
44
+ __all__ = ["D_N_ESTIMATORS", "GRID_D", "DBoostModel", "DGridConfig"]
45
+
46
+ #: Tree count of every grid point — deliberately NOT a grid axis (the grid
47
+ #: cap is 8 configurations; depth/lr/min-data carry the capacity trade).
48
+ D_N_ESTIMATORS: Final[int] = 200
49
+
50
+ #: Internal CV folds for the Platt calibration (train-only signal).
51
+ PLATT_CV_FOLDS: Final[int] = 5
52
+
53
+
54
+ @dataclass(frozen=True)
55
+ class DGridConfig:
56
+ """One frozen grid point (hyperparameters + feature variant)."""
57
+
58
+ name: str
59
+ num_leaves: int
60
+ learning_rate: float
61
+ min_data_in_leaf: int
62
+ field_cosines: bool # ablation axis: with/without field cosines
63
+ seed: int
64
+
65
+
66
+ #: Frozen grid, ≤ 8 configurations (ADR 0001 V1: "гриды ≤ 8 конфигураций,
67
+ #: замораживаются в коде A3"). Two depths × two learning rates over the
68
+ #: core features, plus one field-cosines ablation point (the ADR-mandated
69
+ #: "D-без-полевых is the cheapest runtime" axis).
70
+ GRID_D: Final[tuple[DGridConfig, ...]] = (
71
+ DGridConfig(
72
+ "d-l7-lr005",
73
+ num_leaves=7,
74
+ learning_rate=0.05,
75
+ min_data_in_leaf=5,
76
+ field_cosines=False,
77
+ seed=1,
78
+ ),
79
+ DGridConfig(
80
+ "d-l15-lr005",
81
+ num_leaves=15,
82
+ learning_rate=0.05,
83
+ min_data_in_leaf=5,
84
+ field_cosines=False,
85
+ seed=1,
86
+ ),
87
+ DGridConfig(
88
+ "d-l7-lr010",
89
+ num_leaves=7,
90
+ learning_rate=0.10,
91
+ min_data_in_leaf=10,
92
+ field_cosines=False,
93
+ seed=1,
94
+ ),
95
+ DGridConfig(
96
+ "d-l15-lr010",
97
+ num_leaves=15,
98
+ learning_rate=0.10,
99
+ min_data_in_leaf=10,
100
+ field_cosines=False,
101
+ seed=1,
102
+ ),
103
+ DGridConfig(
104
+ "d-l15-lr005-fc",
105
+ num_leaves=15,
106
+ learning_rate=0.05,
107
+ min_data_in_leaf=5,
108
+ field_cosines=True,
109
+ seed=1,
110
+ ),
111
+ )
112
+
113
+ _CORE_NAMES: Final[tuple[str, ...]] = FEATURE_NAMES
114
+ _EXTENDED_NAMES: Final[tuple[str, ...]] = FEATURE_NAMES + FIELD_COSINE_FEATURES
115
+
116
+
117
+ def _sigmoid(z: np.ndarray) -> np.ndarray:
118
+ return 1.0 / (1.0 + np.exp(-z))
119
+
120
+
121
+ class DBoostModel:
122
+ """LightGBM classifier over FEATURE_NAMES — train/predict/export API.
123
+
124
+ The three-method surface is the candidate contract shared with
125
+ NHeadModel: cv_select drives both through it, export-artifact writes
126
+ the winner as a single ONNX (≤ 5 MB, metadata_props per
127
+ cortex.artifacts).
128
+ """
129
+
130
+ feature_names: tuple[str, ...] = FEATURE_NAMES
131
+
132
+ def __init__(self) -> None:
133
+ self._booster: object | None = None # lightgbm.Booster
134
+ self._platt: tuple[float, float] = (1.0, 0.0)
135
+ self._config: DGridConfig | None = None
136
+
137
+ # ── fitting ───────────────────────────────────────────────────────────────
138
+
139
+ @staticmethod
140
+ def _lgbm_kwargs(config: DGridConfig) -> dict[str, object]:
141
+ return {
142
+ "objective": "binary",
143
+ "n_estimators": D_N_ESTIMATORS,
144
+ "num_leaves": config.num_leaves,
145
+ "learning_rate": config.learning_rate,
146
+ "min_child_samples": config.min_data_in_leaf,
147
+ "random_state": config.seed,
148
+ "deterministic": True,
149
+ "force_col_wise": True,
150
+ "n_jobs": 1,
151
+ "verbosity": -1,
152
+ }
153
+
154
+ def train(
155
+ self,
156
+ vectors: Sequence[FeatureVector],
157
+ labels: Sequence[int],
158
+ config: DGridConfig,
159
+ *,
160
+ calibrate: bool = True,
161
+ ) -> None:
162
+ """Fit on labeled train pairs (labels ∈ {0, 1}); deterministic
163
+ under config.seed. Train-env only.
164
+
165
+ ``calibrate=True`` (default) fits the Platt sigmoid on TRAIN-ONLY
166
+ out-of-fold margins; the selection protocol passes
167
+ ``calibrate=False`` (nested CV would spend 6× the budget on a
168
+ monotone transform — the ranking metric is computed on the raw
169
+ candidate, final training always calibrates).
170
+ """
171
+ import lightgbm as lgb # local: train-env import (addendum P2)
172
+
173
+ matrix, y = self._validate_train_input(vectors, labels, config)
174
+ expected = _EXTENDED_NAMES if config.field_cosines else _CORE_NAMES
175
+ clf = lgb.LGBMClassifier(**self._lgbm_kwargs(config))
176
+ clf.fit(matrix, y)
177
+ self._booster = clf.booster_
178
+ self._config = config
179
+ self.feature_names = expected
180
+ self._platt = (
181
+ self._fit_platt_params(matrix, y, config) if calibrate else (1.0, 0.0)
182
+ )
183
+
184
+ def _validate_train_input(
185
+ self,
186
+ vectors: Sequence[FeatureVector],
187
+ labels: Sequence[int],
188
+ config: DGridConfig,
189
+ ) -> tuple[np.ndarray, np.ndarray]:
190
+ if not vectors:
191
+ raise ValueError("training requires at least one labeled pair")
192
+ observed = {v.names for v in vectors}
193
+ if len(observed) != 1:
194
+ raise ValueError(
195
+ "all training vectors must share one feature-name contract"
196
+ )
197
+ expected = _EXTENDED_NAMES if config.field_cosines else _CORE_NAMES
198
+ names = observed.pop()
199
+ if names != expected:
200
+ kind = "with" if config.field_cosines else "without"
201
+ raise ValueError(
202
+ f"config {config.name} expects field cosines {kind} "
203
+ f"({len(expected)} features), vectors carry {len(names)}"
204
+ )
205
+ y = np.asarray(labels, dtype=np.int64)
206
+ if y.shape != (len(vectors),):
207
+ raise ValueError(f"labels length {len(y)} != vectors length {len(vectors)}")
208
+ if not np.isin(y, (0, 1)).all():
209
+ raise ValueError("labels must be binary {0, 1}")
210
+ if len(np.unique(y)) < 2:
211
+ raise ValueError("training requires both classes present")
212
+ matrix = np.asarray([v.values for v in vectors], dtype=np.float64)
213
+ return matrix, y
214
+
215
+ def _fit_platt_params(
216
+ self, matrix: np.ndarray, y: np.ndarray, config: DGridConfig
217
+ ) -> tuple[float, float]:
218
+ """Platt sigmoid (A, B) over TRAIN-ONLY out-of-fold raw margins."""
219
+ import lightgbm as lgb
220
+ from sklearn.linear_model import LogisticRegression
221
+ from sklearn.model_selection import StratifiedKFold
222
+
223
+ counts = np.bincount(y)
224
+ if counts.size < 2 or int(counts.min()) < 2:
225
+ warnings.warn(
226
+ "a class has < 2 members — Platt calibration skipped (identity)",
227
+ stacklevel=2,
228
+ )
229
+ return (1.0, 0.0)
230
+ n_splits = int(min(PLATT_CV_FOLDS, counts.min()))
231
+ skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=config.seed)
232
+ margins = np.zeros(len(y), dtype=np.float64)
233
+ for train_idx, valid_idx in skf.split(matrix, y):
234
+ fold = lgb.LGBMClassifier(**self._lgbm_kwargs(config))
235
+ fold.fit(matrix[train_idx], y[train_idx])
236
+ margins[valid_idx] = fold.predict(matrix[valid_idx], raw_score=True)
237
+ lr = LogisticRegression(C=float("inf"), solver="lbfgs") # unpenalized Platt fit
238
+ lr.fit(margins.reshape(-1, 1), y)
239
+ return (float(lr.coef_[0, 0]), float(lr.intercept_[0]))
240
+
241
+ # ── inference ─────────────────────────────────────────────────────────────
242
+
243
+ def _require_fitted(self) -> object:
244
+ if self._booster is None:
245
+ raise RuntimeError("DBoostModel is not fitted — call train() first")
246
+ return self._booster
247
+
248
+ def predict_proba(self, vectors: Sequence[FeatureVector]) -> np.ndarray:
249
+ """P(duplicate) per pair, float64 array shape (n,), values in [0, 1]."""
250
+ booster = self._require_fitted()
251
+ if not vectors:
252
+ return np.zeros(0, dtype=np.float64)
253
+ first = vectors[0].names
254
+ if first != self.feature_names:
255
+ raise ValueError(
256
+ f"vectors carry {len(first)} features, model expects {len(self.feature_names)}"
257
+ )
258
+ matrix = np.asarray([v.values for v in vectors], dtype=np.float64)
259
+ margins = booster.predict(matrix, raw_score=True)
260
+ platt_a, platt_b = self._platt
261
+ return _sigmoid(platt_a * margins + platt_b)
262
+
263
+ # ── export ────────────────────────────────────────────────────────────────
264
+
265
+ def export_onnx(
266
+ self, path: Path, *, metadata_props: Mapping[str, str] | None = None
267
+ ) -> Path:
268
+ """Export the fitted model as a single ONNX (skl2onnx chain), opset
269
+ 15, input float32[K] → probability[1].
270
+
271
+ The Platt calibration is folded into the graph as explicit nodes
272
+ (see module docstring). Raises RuntimeError if unfitted or the file
273
+ would exceed cortex.artifacts.MAX_ARTIFACT_BYTES.
274
+ """
275
+ booster = self._require_fitted()
276
+ from onnxmltools.convert import convert_lightgbm
277
+ from onnxmltools.convert.common.data_types import FloatTensorType
278
+
279
+ model = self._build_onnx_model(booster, convert_lightgbm, FloatTensorType)
280
+ if metadata_props:
281
+ from cortex.artifacts import set_onnx_metadata
282
+
283
+ set_onnx_metadata(model, metadata_props)
284
+ out_path = Path(path)
285
+ out_path.parent.mkdir(parents=True, exist_ok=True)
286
+ out_path.write_bytes(model.SerializeToString())
287
+ from cortex.artifacts import assert_artifact_size
288
+
289
+ assert_artifact_size(out_path)
290
+ return out_path
291
+
292
+ def _build_onnx_model(self, booster, convert_lightgbm, float_tensor_type):
293
+ import numpy as _np
294
+ import onnx
295
+ from onnx import TensorProto, helper
296
+
297
+ k = len(self.feature_names)
298
+ converted = convert_lightgbm(
299
+ booster,
300
+ initial_types=[("features_in", float_tensor_type([None, k]))],
301
+ target_opset=15,
302
+ )
303
+ tree = next(
304
+ (n for n in converted.graph.node if n.op_type == "TreeEnsembleClassifier"),
305
+ None,
306
+ )
307
+ if tree is None:
308
+ raise RuntimeError("conversion produced no TreeEnsembleClassifier node")
309
+
310
+ platt_a, platt_b = self._platt
311
+
312
+ def tensor(name: str, values, dtype) -> onnx.TensorProto:
313
+ array = _np.asarray(values)
314
+ return helper.make_tensor(
315
+ name, dtype, dims=array.shape, vals=array.flatten().tolist()
316
+ )
317
+
318
+ nodes = [
319
+ helper.make_node("Reshape", ["features", "shape_2d"], ["features_2d"]),
320
+ tree,
321
+ helper.make_node(
322
+ "Slice",
323
+ ["raw_pair", "slice_start", "slice_end", "slice_axis"],
324
+ ["raw_margin"],
325
+ ),
326
+ helper.make_node("Mul", ["raw_margin", "platt_a"], ["scaled_margin"]),
327
+ helper.make_node("Add", ["scaled_margin", "platt_b"], ["calibrated_logit"]),
328
+ helper.make_node("Sigmoid", ["calibrated_logit"], ["probability_2d"]),
329
+ helper.make_node(
330
+ "Reshape", ["probability_2d", "shape_out"], ["probability"]
331
+ ),
332
+ ]
333
+ # Rewire the tree node into the wrapped graph (raw margins out).
334
+ # TreeEnsembleClassifier REQUIRES two outputs (label + scores) — the
335
+ # label twin stays unwired on purpose.
336
+ del tree.input[:]
337
+ tree.input.extend(["features_2d"])
338
+ del tree.output[:]
339
+ tree.output.extend(["label_unused", "raw_pair"])
340
+ for attr in tree.attribute:
341
+ if attr.name == "post_transform":
342
+ attr.s = b"NONE" # Platt lives in explicit graph nodes
343
+
344
+ initializers = [
345
+ tensor("shape_2d", [1, k], TensorProto.INT64),
346
+ tensor("slice_start", [1], TensorProto.INT64),
347
+ tensor("slice_end", [2], TensorProto.INT64),
348
+ tensor("slice_axis", [1], TensorProto.INT64),
349
+ tensor("platt_a", [platt_a], TensorProto.FLOAT),
350
+ tensor("platt_b", [platt_b], TensorProto.FLOAT),
351
+ tensor("shape_out", [1], TensorProto.INT64),
352
+ ]
353
+ graph = helper.make_graph(
354
+ nodes,
355
+ name="vesma-cortex-d-boost",
356
+ inputs=[helper.make_tensor_value_info("features", TensorProto.FLOAT, [k])],
357
+ outputs=[
358
+ helper.make_tensor_value_info("probability", TensorProto.FLOAT, [1])
359
+ ],
360
+ initializer=initializers,
361
+ )
362
+ model = helper.make_model(
363
+ graph,
364
+ opset_imports=[
365
+ helper.make_opsetid("", 15),
366
+ helper.make_opsetid("ai.onnx.ml", 1),
367
+ ],
368
+ )
369
+ model.ir_version = 8
370
+ onnx.checker.check_model(model)
371
+ self._smoke_onnx(model)
372
+ return model
373
+
374
+ def _smoke_onnx(self, model) -> None:
375
+ """Eager ORT validation: the graph must reproduce the calibrated
376
+ library probability on a synthetic zero feature row."""
377
+ import onnxruntime as ort
378
+
379
+ k = len(self.feature_names)
380
+ session = ort.InferenceSession(
381
+ model.SerializeToString(), providers=["CPUExecutionProvider"]
382
+ )
383
+ zeros = np.zeros(k, dtype=np.float32)
384
+ graph_prob = float(session.run(None, {"features": zeros})[0][0])
385
+ reference = float(
386
+ self.predict_proba(
387
+ [FeatureVector(self.feature_names, tuple(zeros.tolist()))]
388
+ )[0]
389
+ )
390
+ if not np.isclose(graph_prob, reference, atol=1e-5):
391
+ raise RuntimeError(
392
+ f"exported graph diverges from library prediction: {graph_prob} vs {reference}"
393
+ )
394
+
395
+ # ── persistence (dev pipeline: no pickle — LightGBM text format) ──────────
396
+
397
+ def save(self, directory: Path) -> Path:
398
+ """Persist fitted state (booster text + meta json). Dev-only format —
399
+ NEVER the artifact (sklearn/lightgbm pickle artifacts are forbidden,
400
+ ADR 0001 V3; this text dump is the train-epoch hand-off between the
401
+ train and export CLI steps)."""
402
+ booster = self._require_fitted()
403
+ out_dir = Path(directory)
404
+ out_dir.mkdir(parents=True, exist_ok=True)
405
+ booster_path = out_dir / "booster.txt"
406
+ booster.save_model(str(booster_path))
407
+ meta = {
408
+ "candidate": "d-boost",
409
+ "config": asdict(self._config) if self._config else None,
410
+ "feature_names": list(self.feature_names),
411
+ "platt": list(self._platt),
412
+ }
413
+ (out_dir / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
414
+ return out_dir
415
+
416
+ @classmethod
417
+ def load(cls, directory: Path) -> DBoostModel:
418
+ import lightgbm as lgb
419
+
420
+ in_dir = Path(directory)
421
+ meta = json.loads((in_dir / "meta.json").read_text(encoding="utf-8"))
422
+ if meta.get("candidate") != "d-boost":
423
+ raise ValueError(f"{in_dir} is not a d-boost model directory")
424
+ model = cls()
425
+ model._booster = lgb.Booster(model_file=str(in_dir / "booster.txt"))
426
+ model._platt = (float(meta["platt"][0]), float(meta["platt"][1]))
427
+ model.feature_names = tuple(meta["feature_names"])
428
+ model._config = DGridConfig(**meta["config"]) if meta.get("config") else None
429
+ return model