vesma-cortex 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cortex/__init__.py +26 -0
- cortex/artifacts/__init__.py +160 -0
- cortex/candidates/__init__.py +13 -0
- cortex/candidates/d_boost.py +429 -0
- cortex/candidates/n_head.py +447 -0
- cortex/cli/__init__.py +5 -0
- cortex/cli/main.py +940 -0
- cortex/data/__init__.py +21 -0
- cortex/data/corner_qa.py +619 -0
- cortex/data/field_cosines.py +169 -0
- cortex/data/fingerprints.py +81 -0
- cortex/data/holdout.py +137 -0
- cortex/data/store_export.py +1366 -0
- cortex/eval/__init__.py +21 -0
- cortex/eval/runner.py +323 -0
- cortex/eval/sanity.py +696 -0
- cortex/evalsets/__init__.py +76 -0
- cortex/evalsets/generate.py +740 -0
- cortex/evalsets/runner.py +405 -0
- cortex/evalsets/taxonomy.py +205 -0
- cortex/evalsets/topics.py +256 -0
- cortex/features/__init__.py +34 -0
- cortex/features/graph.py +307 -0
- cortex/features/pair.py +299 -0
- cortex/models/vesma-cortex-v1/manifest.json +28 -0
- cortex/models/vesma-cortex-v1/model.onnx +0 -0
- cortex/pretrain/__init__.py +17 -0
- cortex/pretrain/corruption.py +356 -0
- cortex/select/__init__.py +19 -0
- cortex/select/cv.py +179 -0
- cortex/synth/__init__.py +96 -0
- cortex/synth/generate.py +1375 -0
- vesma_cortex-0.2.0.dist-info/METADATA +126 -0
- vesma_cortex-0.2.0.dist-info/RECORD +37 -0
- vesma_cortex-0.2.0.dist-info/WHEEL +4 -0
- vesma_cortex-0.2.0.dist-info/entry_points.txt +2 -0
- vesma_cortex-0.2.0.dist-info/licenses/LICENSE +202 -0
cortex/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""cortex — development library for the vesma-cortex decision model.
|
|
2
|
+
|
|
3
|
+
Dual-epoch delivery (ADR 0001, owner addendum P2):
|
|
4
|
+
|
|
5
|
+
- THIS repo is the train epoch: corpus generation, pair features, D/N
|
|
6
|
+
candidate training, internal CV selection, ONNX export, single-shot
|
|
7
|
+
evaluation (preregistration v2, frozen).
|
|
8
|
+
- The runtime epoch is a single self-contained ONNX artifact
|
|
9
|
+
(``vesma-cortex-v1``) bundled into the engine by the NanoProvider
|
|
10
|
+
pattern — the engine never imports this package.
|
|
11
|
+
|
|
12
|
+
Contracts: docs/specs/inference-v1.md (artifact), docs/specs/data-contract.md
|
|
13
|
+
(data). Slice A3a ships contracts and skeletons only; algorithm bodies land
|
|
14
|
+
in A3b+ (every stub raises NotImplementedError by design — silent partial
|
|
15
|
+
implementations would violate the honest-skeleton rule).
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
__version__ = "0.1.0"
|
|
21
|
+
|
|
22
|
+
#: Re-exported so callers can pin the artifact identity from one place
|
|
23
|
+
#: (single-source rule: defined in cortex.artifacts, A3a).
|
|
24
|
+
from cortex.artifacts import ARTIFACT_NAME
|
|
25
|
+
|
|
26
|
+
__all__ = ["ARTIFACT_NAME", "__version__"]
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Artifact contract: vesma-cortex-v1 identity, metadata, fingerprint.
|
|
2
|
+
|
|
3
|
+
Single source of truth for the artifact NAME (ADR 0001 П1: the constant
|
|
4
|
+
lives in exactly ONE place — see tests/test_skeleton.py guard). The full
|
|
5
|
+
inference contract is docs/specs/inference-v1.md; this module is its code
|
|
6
|
+
anchor: metadata_props assembly, the size gate, and the weights fingerprint
|
|
7
|
+
(sha256 over the .onnx bytes — the engine-side twin is
|
|
8
|
+
``mnema_weights_sha256`` in ``src/vesmaro/embeddings/__init__.py``).
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
from collections.abc import Mapping, Sequence
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Final
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"ARTIFACT_NAME",
|
|
20
|
+
"CANDIDATE_D",
|
|
21
|
+
"CANDIDATE_N",
|
|
22
|
+
"MAX_ARTIFACT_BYTES",
|
|
23
|
+
"METADATA_KEYS",
|
|
24
|
+
"METADATA_VERSION",
|
|
25
|
+
"MODEL_NAME",
|
|
26
|
+
"assert_artifact_size",
|
|
27
|
+
"build_metadata_props",
|
|
28
|
+
"features_digest",
|
|
29
|
+
"set_onnx_metadata",
|
|
30
|
+
"sha256_file",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
#: Artifact identity, ONE constant for the whole repo (ADR 0001 П1).
|
|
34
|
+
#: The engine bundles it under src/vesmaro/models/<ARTIFACT_NAME>/.
|
|
35
|
+
ARTIFACT_NAME: Final[str] = "vesma-cortex-v1"
|
|
36
|
+
|
|
37
|
+
#: Model name inside metadata_props (name ≠ artifact dir name: the dir
|
|
38
|
+
#: carries the major, the metadata carries the family name — engine
|
|
39
|
+
#: precedent: vesma-embed-v1 bundle, manifest "name" field).
|
|
40
|
+
MODEL_NAME: Final[str] = "vesma-cortex"
|
|
41
|
+
|
|
42
|
+
#: Metadata schema major. Weight refresh within v1 bumps to "1.<n>" with a
|
|
43
|
+
#: new sha256 = recalibration event (inference-v1.md §8).
|
|
44
|
+
METADATA_VERSION: Final[str] = "1"
|
|
45
|
+
|
|
46
|
+
#: Hard size gate (ADR 0001 V3: single ONNX ≤ 5 MB, self-contained).
|
|
47
|
+
MAX_ARTIFACT_BYTES: Final[int] = 5 * 1024 * 1024
|
|
48
|
+
|
|
49
|
+
#: metadata_props keys (inference-v1.md §4). Frozen set — W5d validates
|
|
50
|
+
#: against exactly these.
|
|
51
|
+
METADATA_KEYS: Final[tuple[str, ...]] = (
|
|
52
|
+
"name",
|
|
53
|
+
"version",
|
|
54
|
+
"embedder_pin",
|
|
55
|
+
"corpus_fingerprint",
|
|
56
|
+
"trained_at",
|
|
57
|
+
"candidate",
|
|
58
|
+
"features",
|
|
59
|
+
"feature_set_sha256",
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
#: Which ladder candidate the artifact carries (ADR 0001 V1).
|
|
63
|
+
CANDIDATE_D: Final[str] = "d-boost"
|
|
64
|
+
CANDIDATE_N: Final[str] = "n-head"
|
|
65
|
+
|
|
66
|
+
#: sha256 over the "\n"-joined feature names — the compact feature-contract
|
|
67
|
+
#: assert (inference-v1.md §4, W5d step 8).
|
|
68
|
+
FEATURES_JOIN: Final[str] = "\n"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def sha256_file(path: Path) -> str:
|
|
72
|
+
"""sha256 over file bytes — the weights fingerprint discipline.
|
|
73
|
+
|
|
74
|
+
Same scheme as the engine's ``mnema_weights_sha256`` (streamed,
|
|
75
|
+
1 MiB chunks): the number that changes exactly when the weights
|
|
76
|
+
change, i.e. the recalibration key (ADR-0021 discipline).
|
|
77
|
+
"""
|
|
78
|
+
digest = hashlib.sha256()
|
|
79
|
+
with Path(path).open("rb") as handle:
|
|
80
|
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
|
81
|
+
digest.update(chunk)
|
|
82
|
+
return digest.hexdigest()
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def features_digest(feature_names: Sequence[str]) -> str:
|
|
86
|
+
"""sha256 hex of the "\n"-joined ordered feature names."""
|
|
87
|
+
joined = FEATURES_JOIN.join(feature_names)
|
|
88
|
+
return hashlib.sha256(joined.encode("utf-8")).hexdigest()
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def build_metadata_props(
|
|
92
|
+
*,
|
|
93
|
+
version: str = METADATA_VERSION,
|
|
94
|
+
embedder_pin: str,
|
|
95
|
+
corpus_fingerprint: str,
|
|
96
|
+
trained_at: str,
|
|
97
|
+
candidate: str,
|
|
98
|
+
feature_names: tuple[str, ...],
|
|
99
|
+
) -> dict[str, str]:
|
|
100
|
+
"""Assemble the ONNX metadata_props for a vesma-cortex artifact.
|
|
101
|
+
|
|
102
|
+
Contract (inference-v1.md §4): ``embedder_pin`` MUST be the live
|
|
103
|
+
engine fingerprint string (``nano:sha256:<hex>``), ``corpus_fingerprint``
|
|
104
|
+
the BLAKE2b-256 of the training-pair manifest, ``feature_names`` the
|
|
105
|
+
frozen ordered list (cortex.features.pair.FEATURE_NAMES) joined by
|
|
106
|
+
newlines plus its sha256. Raises ValueError on unknown candidate.
|
|
107
|
+
"""
|
|
108
|
+
if candidate not in (CANDIDATE_D, CANDIDATE_N):
|
|
109
|
+
raise ValueError(
|
|
110
|
+
f"unknown ladder candidate {candidate!r} — expected {CANDIDATE_D!r} or {CANDIDATE_N!r}"
|
|
111
|
+
)
|
|
112
|
+
if not embedder_pin:
|
|
113
|
+
raise ValueError(
|
|
114
|
+
"embedder_pin is required (live engine fingerprint, nano:sha256:<hex>)"
|
|
115
|
+
)
|
|
116
|
+
if not corpus_fingerprint:
|
|
117
|
+
raise ValueError(
|
|
118
|
+
"corpus_fingerprint is required (BLAKE2b-256 of the pair manifest)"
|
|
119
|
+
)
|
|
120
|
+
if not feature_names:
|
|
121
|
+
raise ValueError("feature_names must be a non-empty ordered sequence")
|
|
122
|
+
return {
|
|
123
|
+
"name": MODEL_NAME,
|
|
124
|
+
"version": version,
|
|
125
|
+
"embedder_pin": embedder_pin,
|
|
126
|
+
"corpus_fingerprint": corpus_fingerprint,
|
|
127
|
+
"trained_at": trained_at,
|
|
128
|
+
"candidate": candidate,
|
|
129
|
+
"features": FEATURES_JOIN.join(feature_names),
|
|
130
|
+
"feature_set_sha256": features_digest(feature_names),
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def assert_artifact_size(onnx_path: Path) -> None:
|
|
135
|
+
"""Enforce the ≤5 MB gate at export time (fail-loud, pre-bundle)."""
|
|
136
|
+
size = Path(onnx_path).stat().st_size
|
|
137
|
+
if size > MAX_ARTIFACT_BYTES:
|
|
138
|
+
raise RuntimeError(
|
|
139
|
+
f"artifact {onnx_path} is {size} bytes — exceeds the "
|
|
140
|
+
f"{MAX_ARTIFACT_BYTES}-byte gate (ADR 0001 V3, inference-v1.md §4)"
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def set_onnx_metadata(model_proto, props: Mapping[str, str]) -> None:
|
|
145
|
+
"""Write ``props`` into the ONNX model metadata_props (in place).
|
|
146
|
+
|
|
147
|
+
Lazy ``onnx`` import: the artifact module stays import-light for
|
|
148
|
+
callers that never export (onnx arrives with the skl2onnx/onnxmltools
|
|
149
|
+
train chain, never as a runtime dep of the engine).
|
|
150
|
+
"""
|
|
151
|
+
import onnx # local: export-path only
|
|
152
|
+
|
|
153
|
+
existing = {entry.key: entry for entry in model_proto.metadata_props}
|
|
154
|
+
for key, value in props.items():
|
|
155
|
+
if key in existing:
|
|
156
|
+
existing[key].value = value
|
|
157
|
+
else:
|
|
158
|
+
model_proto.metadata_props.append(
|
|
159
|
+
onnx.StringStringEntryProto(key=key, value=value)
|
|
160
|
+
)
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Candidates package — the D/N ladder (ADR 0001 V1)."""
|
|
2
|
+
|
|
3
|
+
from cortex.candidates.d_boost import GRID_D, DBoostModel, DGridConfig
|
|
4
|
+
from cortex.candidates.n_head import GRID_N, NGridConfig, NHeadModel
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"GRID_D",
|
|
8
|
+
"GRID_N",
|
|
9
|
+
"DBoostModel",
|
|
10
|
+
"DGridConfig",
|
|
11
|
+
"NGridConfig",
|
|
12
|
+
"NHeadModel",
|
|
13
|
+
]
|
|
@@ -0,0 +1,429 @@
|
|
|
1
|
+
"""Candidate D — gradient boosting over the frozen pair features.
|
|
2
|
+
|
|
3
|
+
ADR 0001 V1: D is an EQUAL ADOPT-candidate and the internal bar, not a
|
|
4
|
+
control afterthought. LightGBM here (train env only — never a runtime dep
|
|
5
|
+
of the engine, addendum P2). Grids ≤ 8 configurations, frozen in code at
|
|
6
|
+
A3 (this file); selection protocol lives in cortex.select.cv.
|
|
7
|
+
|
|
8
|
+
A3b implementation notes (frozen here):
|
|
9
|
+
|
|
10
|
+
- Determinism: every LightGBM fit runs single-threaded with
|
|
11
|
+
``deterministic=True`` + ``force_col_wise=True`` under the config seed —
|
|
12
|
+
two fits on identical data produce identical margins (pinned by tests).
|
|
13
|
+
- Calibration: Platt (sigmoid over the raw margin, p = σ(A·z+B)) fitted on
|
|
14
|
+
TRAIN ONLY via out-of-fold margins (internal StratifiedKFold, seed =
|
|
15
|
+
config seed). Falls back to identity (A=1, B=0) with a warning when the
|
|
16
|
+
train split cannot support CV (a class with < 2 members). Isotonic was
|
|
17
|
+
rejected for ~140 labeled pairs (overfits the tails; Platt is the
|
|
18
|
+
low-variance choice, matching the ladder philosophy).
|
|
19
|
+
- ONNX export (skl2onnx chain — the LightGBM converter ships in
|
|
20
|
+
``onnxmltools`` since skl2onnx 1.20): the converted
|
|
21
|
+
TreeEnsembleClassifier emits σ(z) via post_transform; this module
|
|
22
|
+
rewrites post_transform to NONE and appends the Platt transform
|
|
23
|
+
(Mul/Add/Sigmoid) as explicit graph nodes, so the exported probability
|
|
24
|
+
is EXACTLY the calibrated library ``predict_proba`` (cross-checked at
|
|
25
|
+
export time through an onnxruntime smoke inference).
|
|
26
|
+
- Graph I/O contract (inference-v1.md §4, variant D): input
|
|
27
|
+
``features: float32[K]`` (1-D, single pair) → output
|
|
28
|
+
``probability: float32[1]``; opset 15 + ai.onnx.ml 1, IR 8.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
import warnings
|
|
35
|
+
from collections.abc import Mapping, Sequence
|
|
36
|
+
from dataclasses import asdict, dataclass
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Final
|
|
39
|
+
|
|
40
|
+
import numpy as np
|
|
41
|
+
|
|
42
|
+
from cortex.features.pair import FEATURE_NAMES, FIELD_COSINE_FEATURES, FeatureVector
|
|
43
|
+
|
|
44
|
+
__all__ = ["D_N_ESTIMATORS", "GRID_D", "DBoostModel", "DGridConfig"]
|
|
45
|
+
|
|
46
|
+
#: Tree count of every grid point — deliberately NOT a grid axis (the grid
|
|
47
|
+
#: cap is 8 configurations; depth/lr/min-data carry the capacity trade).
|
|
48
|
+
D_N_ESTIMATORS: Final[int] = 200
|
|
49
|
+
|
|
50
|
+
#: Internal CV folds for the Platt calibration (train-only signal).
|
|
51
|
+
PLATT_CV_FOLDS: Final[int] = 5
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass(frozen=True)
|
|
55
|
+
class DGridConfig:
|
|
56
|
+
"""One frozen grid point (hyperparameters + feature variant)."""
|
|
57
|
+
|
|
58
|
+
name: str
|
|
59
|
+
num_leaves: int
|
|
60
|
+
learning_rate: float
|
|
61
|
+
min_data_in_leaf: int
|
|
62
|
+
field_cosines: bool # ablation axis: with/without field cosines
|
|
63
|
+
seed: int
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
#: Frozen grid, ≤ 8 configurations (ADR 0001 V1: "гриды ≤ 8 конфигураций,
|
|
67
|
+
#: замораживаются в коде A3"). Two depths × two learning rates over the
|
|
68
|
+
#: core features, plus one field-cosines ablation point (the ADR-mandated
|
|
69
|
+
#: "D-без-полевых is the cheapest runtime" axis).
|
|
70
|
+
GRID_D: Final[tuple[DGridConfig, ...]] = (
|
|
71
|
+
DGridConfig(
|
|
72
|
+
"d-l7-lr005",
|
|
73
|
+
num_leaves=7,
|
|
74
|
+
learning_rate=0.05,
|
|
75
|
+
min_data_in_leaf=5,
|
|
76
|
+
field_cosines=False,
|
|
77
|
+
seed=1,
|
|
78
|
+
),
|
|
79
|
+
DGridConfig(
|
|
80
|
+
"d-l15-lr005",
|
|
81
|
+
num_leaves=15,
|
|
82
|
+
learning_rate=0.05,
|
|
83
|
+
min_data_in_leaf=5,
|
|
84
|
+
field_cosines=False,
|
|
85
|
+
seed=1,
|
|
86
|
+
),
|
|
87
|
+
DGridConfig(
|
|
88
|
+
"d-l7-lr010",
|
|
89
|
+
num_leaves=7,
|
|
90
|
+
learning_rate=0.10,
|
|
91
|
+
min_data_in_leaf=10,
|
|
92
|
+
field_cosines=False,
|
|
93
|
+
seed=1,
|
|
94
|
+
),
|
|
95
|
+
DGridConfig(
|
|
96
|
+
"d-l15-lr010",
|
|
97
|
+
num_leaves=15,
|
|
98
|
+
learning_rate=0.10,
|
|
99
|
+
min_data_in_leaf=10,
|
|
100
|
+
field_cosines=False,
|
|
101
|
+
seed=1,
|
|
102
|
+
),
|
|
103
|
+
DGridConfig(
|
|
104
|
+
"d-l15-lr005-fc",
|
|
105
|
+
num_leaves=15,
|
|
106
|
+
learning_rate=0.05,
|
|
107
|
+
min_data_in_leaf=5,
|
|
108
|
+
field_cosines=True,
|
|
109
|
+
seed=1,
|
|
110
|
+
),
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
_CORE_NAMES: Final[tuple[str, ...]] = FEATURE_NAMES
|
|
114
|
+
_EXTENDED_NAMES: Final[tuple[str, ...]] = FEATURE_NAMES + FIELD_COSINE_FEATURES
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _sigmoid(z: np.ndarray) -> np.ndarray:
|
|
118
|
+
return 1.0 / (1.0 + np.exp(-z))
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class DBoostModel:
|
|
122
|
+
"""LightGBM classifier over FEATURE_NAMES — train/predict/export API.
|
|
123
|
+
|
|
124
|
+
The three-method surface is the candidate contract shared with
|
|
125
|
+
NHeadModel: cv_select drives both through it, export-artifact writes
|
|
126
|
+
the winner as a single ONNX (≤ 5 MB, metadata_props per
|
|
127
|
+
cortex.artifacts).
|
|
128
|
+
"""
|
|
129
|
+
|
|
130
|
+
feature_names: tuple[str, ...] = FEATURE_NAMES
|
|
131
|
+
|
|
132
|
+
def __init__(self) -> None:
|
|
133
|
+
self._booster: object | None = None # lightgbm.Booster
|
|
134
|
+
self._platt: tuple[float, float] = (1.0, 0.0)
|
|
135
|
+
self._config: DGridConfig | None = None
|
|
136
|
+
|
|
137
|
+
# ── fitting ───────────────────────────────────────────────────────────────
|
|
138
|
+
|
|
139
|
+
@staticmethod
|
|
140
|
+
def _lgbm_kwargs(config: DGridConfig) -> dict[str, object]:
|
|
141
|
+
return {
|
|
142
|
+
"objective": "binary",
|
|
143
|
+
"n_estimators": D_N_ESTIMATORS,
|
|
144
|
+
"num_leaves": config.num_leaves,
|
|
145
|
+
"learning_rate": config.learning_rate,
|
|
146
|
+
"min_child_samples": config.min_data_in_leaf,
|
|
147
|
+
"random_state": config.seed,
|
|
148
|
+
"deterministic": True,
|
|
149
|
+
"force_col_wise": True,
|
|
150
|
+
"n_jobs": 1,
|
|
151
|
+
"verbosity": -1,
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
def train(
|
|
155
|
+
self,
|
|
156
|
+
vectors: Sequence[FeatureVector],
|
|
157
|
+
labels: Sequence[int],
|
|
158
|
+
config: DGridConfig,
|
|
159
|
+
*,
|
|
160
|
+
calibrate: bool = True,
|
|
161
|
+
) -> None:
|
|
162
|
+
"""Fit on labeled train pairs (labels ∈ {0, 1}); deterministic
|
|
163
|
+
under config.seed. Train-env only.
|
|
164
|
+
|
|
165
|
+
``calibrate=True`` (default) fits the Platt sigmoid on TRAIN-ONLY
|
|
166
|
+
out-of-fold margins; the selection protocol passes
|
|
167
|
+
``calibrate=False`` (nested CV would spend 6× the budget on a
|
|
168
|
+
monotone transform — the ranking metric is computed on the raw
|
|
169
|
+
candidate, final training always calibrates).
|
|
170
|
+
"""
|
|
171
|
+
import lightgbm as lgb # local: train-env import (addendum P2)
|
|
172
|
+
|
|
173
|
+
matrix, y = self._validate_train_input(vectors, labels, config)
|
|
174
|
+
expected = _EXTENDED_NAMES if config.field_cosines else _CORE_NAMES
|
|
175
|
+
clf = lgb.LGBMClassifier(**self._lgbm_kwargs(config))
|
|
176
|
+
clf.fit(matrix, y)
|
|
177
|
+
self._booster = clf.booster_
|
|
178
|
+
self._config = config
|
|
179
|
+
self.feature_names = expected
|
|
180
|
+
self._platt = (
|
|
181
|
+
self._fit_platt_params(matrix, y, config) if calibrate else (1.0, 0.0)
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
def _validate_train_input(
|
|
185
|
+
self,
|
|
186
|
+
vectors: Sequence[FeatureVector],
|
|
187
|
+
labels: Sequence[int],
|
|
188
|
+
config: DGridConfig,
|
|
189
|
+
) -> tuple[np.ndarray, np.ndarray]:
|
|
190
|
+
if not vectors:
|
|
191
|
+
raise ValueError("training requires at least one labeled pair")
|
|
192
|
+
observed = {v.names for v in vectors}
|
|
193
|
+
if len(observed) != 1:
|
|
194
|
+
raise ValueError(
|
|
195
|
+
"all training vectors must share one feature-name contract"
|
|
196
|
+
)
|
|
197
|
+
expected = _EXTENDED_NAMES if config.field_cosines else _CORE_NAMES
|
|
198
|
+
names = observed.pop()
|
|
199
|
+
if names != expected:
|
|
200
|
+
kind = "with" if config.field_cosines else "without"
|
|
201
|
+
raise ValueError(
|
|
202
|
+
f"config {config.name} expects field cosines {kind} "
|
|
203
|
+
f"({len(expected)} features), vectors carry {len(names)}"
|
|
204
|
+
)
|
|
205
|
+
y = np.asarray(labels, dtype=np.int64)
|
|
206
|
+
if y.shape != (len(vectors),):
|
|
207
|
+
raise ValueError(f"labels length {len(y)} != vectors length {len(vectors)}")
|
|
208
|
+
if not np.isin(y, (0, 1)).all():
|
|
209
|
+
raise ValueError("labels must be binary {0, 1}")
|
|
210
|
+
if len(np.unique(y)) < 2:
|
|
211
|
+
raise ValueError("training requires both classes present")
|
|
212
|
+
matrix = np.asarray([v.values for v in vectors], dtype=np.float64)
|
|
213
|
+
return matrix, y
|
|
214
|
+
|
|
215
|
+
def _fit_platt_params(
|
|
216
|
+
self, matrix: np.ndarray, y: np.ndarray, config: DGridConfig
|
|
217
|
+
) -> tuple[float, float]:
|
|
218
|
+
"""Platt sigmoid (A, B) over TRAIN-ONLY out-of-fold raw margins."""
|
|
219
|
+
import lightgbm as lgb
|
|
220
|
+
from sklearn.linear_model import LogisticRegression
|
|
221
|
+
from sklearn.model_selection import StratifiedKFold
|
|
222
|
+
|
|
223
|
+
counts = np.bincount(y)
|
|
224
|
+
if counts.size < 2 or int(counts.min()) < 2:
|
|
225
|
+
warnings.warn(
|
|
226
|
+
"a class has < 2 members — Platt calibration skipped (identity)",
|
|
227
|
+
stacklevel=2,
|
|
228
|
+
)
|
|
229
|
+
return (1.0, 0.0)
|
|
230
|
+
n_splits = int(min(PLATT_CV_FOLDS, counts.min()))
|
|
231
|
+
skf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=config.seed)
|
|
232
|
+
margins = np.zeros(len(y), dtype=np.float64)
|
|
233
|
+
for train_idx, valid_idx in skf.split(matrix, y):
|
|
234
|
+
fold = lgb.LGBMClassifier(**self._lgbm_kwargs(config))
|
|
235
|
+
fold.fit(matrix[train_idx], y[train_idx])
|
|
236
|
+
margins[valid_idx] = fold.predict(matrix[valid_idx], raw_score=True)
|
|
237
|
+
lr = LogisticRegression(C=float("inf"), solver="lbfgs") # unpenalized Platt fit
|
|
238
|
+
lr.fit(margins.reshape(-1, 1), y)
|
|
239
|
+
return (float(lr.coef_[0, 0]), float(lr.intercept_[0]))
|
|
240
|
+
|
|
241
|
+
# ── inference ─────────────────────────────────────────────────────────────
|
|
242
|
+
|
|
243
|
+
def _require_fitted(self) -> object:
|
|
244
|
+
if self._booster is None:
|
|
245
|
+
raise RuntimeError("DBoostModel is not fitted — call train() first")
|
|
246
|
+
return self._booster
|
|
247
|
+
|
|
248
|
+
def predict_proba(self, vectors: Sequence[FeatureVector]) -> np.ndarray:
|
|
249
|
+
"""P(duplicate) per pair, float64 array shape (n,), values in [0, 1]."""
|
|
250
|
+
booster = self._require_fitted()
|
|
251
|
+
if not vectors:
|
|
252
|
+
return np.zeros(0, dtype=np.float64)
|
|
253
|
+
first = vectors[0].names
|
|
254
|
+
if first != self.feature_names:
|
|
255
|
+
raise ValueError(
|
|
256
|
+
f"vectors carry {len(first)} features, model expects {len(self.feature_names)}"
|
|
257
|
+
)
|
|
258
|
+
matrix = np.asarray([v.values for v in vectors], dtype=np.float64)
|
|
259
|
+
margins = booster.predict(matrix, raw_score=True)
|
|
260
|
+
platt_a, platt_b = self._platt
|
|
261
|
+
return _sigmoid(platt_a * margins + platt_b)
|
|
262
|
+
|
|
263
|
+
# ── export ────────────────────────────────────────────────────────────────
|
|
264
|
+
|
|
265
|
+
def export_onnx(
|
|
266
|
+
self, path: Path, *, metadata_props: Mapping[str, str] | None = None
|
|
267
|
+
) -> Path:
|
|
268
|
+
"""Export the fitted model as a single ONNX (skl2onnx chain), opset
|
|
269
|
+
15, input float32[K] → probability[1].
|
|
270
|
+
|
|
271
|
+
The Platt calibration is folded into the graph as explicit nodes
|
|
272
|
+
(see module docstring). Raises RuntimeError if unfitted or the file
|
|
273
|
+
would exceed cortex.artifacts.MAX_ARTIFACT_BYTES.
|
|
274
|
+
"""
|
|
275
|
+
booster = self._require_fitted()
|
|
276
|
+
from onnxmltools.convert import convert_lightgbm
|
|
277
|
+
from onnxmltools.convert.common.data_types import FloatTensorType
|
|
278
|
+
|
|
279
|
+
model = self._build_onnx_model(booster, convert_lightgbm, FloatTensorType)
|
|
280
|
+
if metadata_props:
|
|
281
|
+
from cortex.artifacts import set_onnx_metadata
|
|
282
|
+
|
|
283
|
+
set_onnx_metadata(model, metadata_props)
|
|
284
|
+
out_path = Path(path)
|
|
285
|
+
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
286
|
+
out_path.write_bytes(model.SerializeToString())
|
|
287
|
+
from cortex.artifacts import assert_artifact_size
|
|
288
|
+
|
|
289
|
+
assert_artifact_size(out_path)
|
|
290
|
+
return out_path
|
|
291
|
+
|
|
292
|
+
def _build_onnx_model(self, booster, convert_lightgbm, float_tensor_type):
|
|
293
|
+
import numpy as _np
|
|
294
|
+
import onnx
|
|
295
|
+
from onnx import TensorProto, helper
|
|
296
|
+
|
|
297
|
+
k = len(self.feature_names)
|
|
298
|
+
converted = convert_lightgbm(
|
|
299
|
+
booster,
|
|
300
|
+
initial_types=[("features_in", float_tensor_type([None, k]))],
|
|
301
|
+
target_opset=15,
|
|
302
|
+
)
|
|
303
|
+
tree = next(
|
|
304
|
+
(n for n in converted.graph.node if n.op_type == "TreeEnsembleClassifier"),
|
|
305
|
+
None,
|
|
306
|
+
)
|
|
307
|
+
if tree is None:
|
|
308
|
+
raise RuntimeError("conversion produced no TreeEnsembleClassifier node")
|
|
309
|
+
|
|
310
|
+
platt_a, platt_b = self._platt
|
|
311
|
+
|
|
312
|
+
def tensor(name: str, values, dtype) -> onnx.TensorProto:
|
|
313
|
+
array = _np.asarray(values)
|
|
314
|
+
return helper.make_tensor(
|
|
315
|
+
name, dtype, dims=array.shape, vals=array.flatten().tolist()
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
nodes = [
|
|
319
|
+
helper.make_node("Reshape", ["features", "shape_2d"], ["features_2d"]),
|
|
320
|
+
tree,
|
|
321
|
+
helper.make_node(
|
|
322
|
+
"Slice",
|
|
323
|
+
["raw_pair", "slice_start", "slice_end", "slice_axis"],
|
|
324
|
+
["raw_margin"],
|
|
325
|
+
),
|
|
326
|
+
helper.make_node("Mul", ["raw_margin", "platt_a"], ["scaled_margin"]),
|
|
327
|
+
helper.make_node("Add", ["scaled_margin", "platt_b"], ["calibrated_logit"]),
|
|
328
|
+
helper.make_node("Sigmoid", ["calibrated_logit"], ["probability_2d"]),
|
|
329
|
+
helper.make_node(
|
|
330
|
+
"Reshape", ["probability_2d", "shape_out"], ["probability"]
|
|
331
|
+
),
|
|
332
|
+
]
|
|
333
|
+
# Rewire the tree node into the wrapped graph (raw margins out).
|
|
334
|
+
# TreeEnsembleClassifier REQUIRES two outputs (label + scores) — the
|
|
335
|
+
# label twin stays unwired on purpose.
|
|
336
|
+
del tree.input[:]
|
|
337
|
+
tree.input.extend(["features_2d"])
|
|
338
|
+
del tree.output[:]
|
|
339
|
+
tree.output.extend(["label_unused", "raw_pair"])
|
|
340
|
+
for attr in tree.attribute:
|
|
341
|
+
if attr.name == "post_transform":
|
|
342
|
+
attr.s = b"NONE" # Platt lives in explicit graph nodes
|
|
343
|
+
|
|
344
|
+
initializers = [
|
|
345
|
+
tensor("shape_2d", [1, k], TensorProto.INT64),
|
|
346
|
+
tensor("slice_start", [1], TensorProto.INT64),
|
|
347
|
+
tensor("slice_end", [2], TensorProto.INT64),
|
|
348
|
+
tensor("slice_axis", [1], TensorProto.INT64),
|
|
349
|
+
tensor("platt_a", [platt_a], TensorProto.FLOAT),
|
|
350
|
+
tensor("platt_b", [platt_b], TensorProto.FLOAT),
|
|
351
|
+
tensor("shape_out", [1], TensorProto.INT64),
|
|
352
|
+
]
|
|
353
|
+
graph = helper.make_graph(
|
|
354
|
+
nodes,
|
|
355
|
+
name="vesma-cortex-d-boost",
|
|
356
|
+
inputs=[helper.make_tensor_value_info("features", TensorProto.FLOAT, [k])],
|
|
357
|
+
outputs=[
|
|
358
|
+
helper.make_tensor_value_info("probability", TensorProto.FLOAT, [1])
|
|
359
|
+
],
|
|
360
|
+
initializer=initializers,
|
|
361
|
+
)
|
|
362
|
+
model = helper.make_model(
|
|
363
|
+
graph,
|
|
364
|
+
opset_imports=[
|
|
365
|
+
helper.make_opsetid("", 15),
|
|
366
|
+
helper.make_opsetid("ai.onnx.ml", 1),
|
|
367
|
+
],
|
|
368
|
+
)
|
|
369
|
+
model.ir_version = 8
|
|
370
|
+
onnx.checker.check_model(model)
|
|
371
|
+
self._smoke_onnx(model)
|
|
372
|
+
return model
|
|
373
|
+
|
|
374
|
+
def _smoke_onnx(self, model) -> None:
|
|
375
|
+
"""Eager ORT validation: the graph must reproduce the calibrated
|
|
376
|
+
library probability on a synthetic zero feature row."""
|
|
377
|
+
import onnxruntime as ort
|
|
378
|
+
|
|
379
|
+
k = len(self.feature_names)
|
|
380
|
+
session = ort.InferenceSession(
|
|
381
|
+
model.SerializeToString(), providers=["CPUExecutionProvider"]
|
|
382
|
+
)
|
|
383
|
+
zeros = np.zeros(k, dtype=np.float32)
|
|
384
|
+
graph_prob = float(session.run(None, {"features": zeros})[0][0])
|
|
385
|
+
reference = float(
|
|
386
|
+
self.predict_proba(
|
|
387
|
+
[FeatureVector(self.feature_names, tuple(zeros.tolist()))]
|
|
388
|
+
)[0]
|
|
389
|
+
)
|
|
390
|
+
if not np.isclose(graph_prob, reference, atol=1e-5):
|
|
391
|
+
raise RuntimeError(
|
|
392
|
+
f"exported graph diverges from library prediction: {graph_prob} vs {reference}"
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
# ── persistence (dev pipeline: no pickle — LightGBM text format) ──────────
|
|
396
|
+
|
|
397
|
+
def save(self, directory: Path) -> Path:
|
|
398
|
+
"""Persist fitted state (booster text + meta json). Dev-only format —
|
|
399
|
+
NEVER the artifact (sklearn/lightgbm pickle artifacts are forbidden,
|
|
400
|
+
ADR 0001 V3; this text dump is the train-epoch hand-off between the
|
|
401
|
+
train and export CLI steps)."""
|
|
402
|
+
booster = self._require_fitted()
|
|
403
|
+
out_dir = Path(directory)
|
|
404
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
405
|
+
booster_path = out_dir / "booster.txt"
|
|
406
|
+
booster.save_model(str(booster_path))
|
|
407
|
+
meta = {
|
|
408
|
+
"candidate": "d-boost",
|
|
409
|
+
"config": asdict(self._config) if self._config else None,
|
|
410
|
+
"feature_names": list(self.feature_names),
|
|
411
|
+
"platt": list(self._platt),
|
|
412
|
+
}
|
|
413
|
+
(out_dir / "meta.json").write_text(json.dumps(meta, indent=2), encoding="utf-8")
|
|
414
|
+
return out_dir
|
|
415
|
+
|
|
416
|
+
@classmethod
|
|
417
|
+
def load(cls, directory: Path) -> DBoostModel:
|
|
418
|
+
import lightgbm as lgb
|
|
419
|
+
|
|
420
|
+
in_dir = Path(directory)
|
|
421
|
+
meta = json.loads((in_dir / "meta.json").read_text(encoding="utf-8"))
|
|
422
|
+
if meta.get("candidate") != "d-boost":
|
|
423
|
+
raise ValueError(f"{in_dir} is not a d-boost model directory")
|
|
424
|
+
model = cls()
|
|
425
|
+
model._booster = lgb.Booster(model_file=str(in_dir / "booster.txt"))
|
|
426
|
+
model._platt = (float(meta["platt"][0]), float(meta["platt"][1]))
|
|
427
|
+
model.feature_names = tuple(meta["feature_names"])
|
|
428
|
+
model._config = DGridConfig(**meta["config"]) if meta.get("config") else None
|
|
429
|
+
return model
|