klix-engine 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- klix/__init__.py +15 -0
- klix/backbone.py +64 -0
- klix/engine.py +83 -0
- klix/heads.py +244 -0
- klix_engine-0.1.0.dist-info/METADATA +146 -0
- klix_engine-0.1.0.dist-info/RECORD +8 -0
- klix_engine-0.1.0.dist-info/WHEEL +4 -0
- klix_engine-0.1.0.dist-info/licenses/LICENSE +21 -0
klix/__init__.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Klix: Entkoppelte Entscheidungs-Köpfe (Choice, Score, Flag) auf einem geteilten semantischen Backbone."""
|
|
2
|
+
|
|
3
|
+
from klix.engine import DecisionEngine, DecisionResult
|
|
4
|
+
from klix.heads import BaseHead, Choice, Flag, Score
|
|
5
|
+
|
|
6
|
+
__version__ = "0.1.0"
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"DecisionEngine",
|
|
10
|
+
"DecisionResult",
|
|
11
|
+
"BaseHead",
|
|
12
|
+
"Choice",
|
|
13
|
+
"Score",
|
|
14
|
+
"Flag",
|
|
15
|
+
]
|
klix/backbone.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Geteilte Feature-Extraktion: ein Text geht genau einmal dense + sparse transformiert rein."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
import numpy as np
|
|
7
|
+
from fastembed import TextEmbedding
|
|
8
|
+
from sklearn.feature_extraction.text import TfidfVectorizer
|
|
9
|
+
|
|
10
|
+
# Kompakte deutsche Stoppwortliste (bewusst klein, damit Fachbegriffe ihre
|
|
11
|
+
# Information behalten; erweiterbar via build_vocabulary(..., stop_words=...)).
|
|
12
|
+
_DEFAULT_GERMAN_STOPWORDS = [
|
|
13
|
+
"die", "der", "das", "ein", "eine", "einer", "eines", "einem", "einen",
|
|
14
|
+
"im", "in", "ist", "und", "für", "von", "mit", "an", "auf", "nach", "zu",
|
|
15
|
+
"nicht", "mehr", "wird", "wie", "was", "hier", "dort",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class EncodedInput:
|
|
21
|
+
"""Ergebnis der einmaligen Feature-Extraktion für einen Eingabetext."""
|
|
22
|
+
|
|
23
|
+
text: str
|
|
24
|
+
dense_vec: np.ndarray # normalisierter Dense-Vektor (MiniLM, 384 dim)
|
|
25
|
+
sparse_vec: Any # TF-IDF Sparse-Matrix (1 x V) oder None vor compile()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class HybridBackbone:
|
|
29
|
+
"""Kapselt Dense- (FastEmbed) und Sparse- (TF-IDF) Repräsentation eines Texts.
|
|
30
|
+
|
|
31
|
+
Wird von der Engine genau einmal instanziiert. Der Text wird pro `decide()`-Aufruf
|
|
32
|
+
genau einmal encoded; alle Köpfe arbeiten anschließend auf `EncodedInput`.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
def __init__(self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"):
|
|
36
|
+
self.embed_model = TextEmbedding(model_name=model_name)
|
|
37
|
+
self.tfidf_vec: TfidfVectorizer | None = None
|
|
38
|
+
self.is_indexed = False
|
|
39
|
+
|
|
40
|
+
def build_vocabulary(
|
|
41
|
+
self,
|
|
42
|
+
all_texts: list[str],
|
|
43
|
+
stop_words: list[str] | None = None,
|
|
44
|
+
) -> None:
|
|
45
|
+
"""Baut den Sparse-Index über alle in den Köpfen hinterlegten Referenztexte auf."""
|
|
46
|
+
if stop_words is None:
|
|
47
|
+
stop_words = _DEFAULT_GERMAN_STOPWORDS
|
|
48
|
+
self.tfidf_vec = TfidfVectorizer(
|
|
49
|
+
analyzer="word",
|
|
50
|
+
token_pattern=r"(?u)\b[\w-]+\b",
|
|
51
|
+
lowercase=True,
|
|
52
|
+
stop_words=stop_words,
|
|
53
|
+
)
|
|
54
|
+
self.tfidf_vec.fit(all_texts)
|
|
55
|
+
self.is_indexed = True
|
|
56
|
+
|
|
57
|
+
def encode(self, text: str) -> EncodedInput:
|
|
58
|
+
"""Erzeugt beide Vektoren in einem Rutsch."""
|
|
59
|
+
vec = np.array(list(self.embed_model.embed([text]))[0])
|
|
60
|
+
norm = float(np.linalg.norm(vec))
|
|
61
|
+
dense_norm = vec / (norm if norm > 0 else 1.0)
|
|
62
|
+
|
|
63
|
+
sparse = self.tfidf_vec.transform([text]) if self.is_indexed else None
|
|
64
|
+
return EncodedInput(text=text, dense_vec=dense_norm, sparse_vec=sparse)
|
klix/engine.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Die Orchestrierungs-Engine: bindet Backbone und Köpfe zusammen."""
|
|
2
|
+
|
|
3
|
+
import time
|
|
4
|
+
|
|
5
|
+
from klix.backbone import HybridBackbone
|
|
6
|
+
from klix.heads import BaseHead
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class DecisionResult:
|
|
10
|
+
"""Ergebnis eines `decide()`-Aufrufs; Kopf-Werte als Attribute abrufbar."""
|
|
11
|
+
|
|
12
|
+
def __init__(self, text: str, latency_ms: float, head_data: dict):
|
|
13
|
+
self.text = text
|
|
14
|
+
self.latency_ms = latency_ms
|
|
15
|
+
self.data = head_data
|
|
16
|
+
|
|
17
|
+
def __getattr__(self, name: str):
|
|
18
|
+
if name in self.data:
|
|
19
|
+
return self.data[name]["value"]
|
|
20
|
+
raise AttributeError(f"Kopf '{name}' existiert nicht im Ergebnis.")
|
|
21
|
+
|
|
22
|
+
def details(self, name: str) -> dict:
|
|
23
|
+
"""Vollständiges Ergebnis-Dict eines einzelnen Kopfs."""
|
|
24
|
+
return self.data.get(name, {})
|
|
25
|
+
|
|
26
|
+
def __repr__(self):
|
|
27
|
+
items = [f"{key}={value['value']}" for key, value in self.data.items()]
|
|
28
|
+
return f"<DecisionResult ({self.latency_ms:.1f}ms): {', '.join(items)}>"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class DecisionEngine:
|
|
32
|
+
"""Shared-Backbone-Engine mit beliebig vielen entkoppelten Köpfen.
|
|
33
|
+
|
|
34
|
+
Der Text wird pro `decide()` genau einmal encoded; alle Köpfe evaluieren
|
|
35
|
+
danach parallel auf denselben Vektoren.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
def __init__(self, model_name: str = "sentence-transformers/paraphrase-multilingual-MiniLM-L12-v2"):
|
|
39
|
+
self.backbone = HybridBackbone(model_name=model_name)
|
|
40
|
+
self.heads: list[BaseHead] = []
|
|
41
|
+
self._compiled = False
|
|
42
|
+
|
|
43
|
+
def add_head(self, head: BaseHead) -> "DecisionEngine":
|
|
44
|
+
"""Registriert einen Kopf (fluent, verkettbar)."""
|
|
45
|
+
self.heads.append(head)
|
|
46
|
+
self._compiled = False
|
|
47
|
+
return self
|
|
48
|
+
|
|
49
|
+
def compile(self) -> None:
|
|
50
|
+
"""Sammelt alle Texte aller Köpfe ein und initialisiert die Indizes."""
|
|
51
|
+
all_texts = []
|
|
52
|
+
for head in self.heads:
|
|
53
|
+
all_texts.extend(head.get_reference_texts())
|
|
54
|
+
|
|
55
|
+
if not all_texts:
|
|
56
|
+
raise ValueError("Keine Referenztexte vorhanden: erst Köpfe via add_head() registrieren.")
|
|
57
|
+
|
|
58
|
+
self.backbone.build_vocabulary(all_texts)
|
|
59
|
+
|
|
60
|
+
for head in self.heads:
|
|
61
|
+
head.fit(self.backbone)
|
|
62
|
+
|
|
63
|
+
self._compiled = True
|
|
64
|
+
|
|
65
|
+
def decide(self, text: str) -> DecisionResult:
|
|
66
|
+
"""Encoded den Text einmal und evaluiert alle Köpfe auf den Vektoren."""
|
|
67
|
+
if not self.heads:
|
|
68
|
+
raise ValueError("Keine Köpfe registriert: erst add_head() aufrufen.")
|
|
69
|
+
if not self._compiled:
|
|
70
|
+
self.compile()
|
|
71
|
+
|
|
72
|
+
start = time.perf_counter()
|
|
73
|
+
|
|
74
|
+
# 1. Einmalige Vektorisierung (~10-12 ms).
|
|
75
|
+
encoded = self.backbone.encode(text)
|
|
76
|
+
|
|
77
|
+
# 2. Evaluation aller Köpfe (< 0.2 ms insgesamt).
|
|
78
|
+
results: dict[str, dict] = {}
|
|
79
|
+
for head in self.heads:
|
|
80
|
+
results[head.name] = head.evaluate(encoded)
|
|
81
|
+
|
|
82
|
+
elapsed_ms = (time.perf_counter() - start) * 1000
|
|
83
|
+
return DecisionResult(text=text, latency_ms=elapsed_ms, head_data=results)
|
klix/heads.py
ADDED
|
@@ -0,0 +1,244 @@
|
|
|
1
|
+
"""Die Entscheidungsköpfe: Choice (Routing), Score (Achse), Flag (Boolesch).
|
|
2
|
+
|
|
3
|
+
Alle Köpfe arbeiten ausschließlich auf den vorberechneten Vektoren aus
|
|
4
|
+
`EncodedInput` und sind dadurch voneinander entkoppelt.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from abc import ABC, abstractmethod
|
|
8
|
+
|
|
9
|
+
import numpy as np
|
|
10
|
+
|
|
11
|
+
from klix.backbone import EncodedInput, HybridBackbone
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class BaseHead(ABC):
|
|
15
|
+
"""Basisklasse für alle Köpfe.
|
|
16
|
+
|
|
17
|
+
Ein Kopf wird dreiphasig verwendet:
|
|
18
|
+
1. `get_reference_texts()` — sammelt Referenztexte für den TF-IDF-Index.
|
|
19
|
+
2. `fit(backbone)` — Vorberechnung aller Referenzvektoren (einmalig).
|
|
20
|
+
3. `evaluate(encoded)` — Bewertung pro Abfrage, < 0.1 ms Ziel.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
def __init__(self, name: str):
|
|
24
|
+
self.name = name
|
|
25
|
+
|
|
26
|
+
@abstractmethod
|
|
27
|
+
def get_reference_texts(self) -> list[str]:
|
|
28
|
+
"""Gibt alle Texte zurück, die der Backbone für den TF-IDF-Index kennen muss."""
|
|
29
|
+
|
|
30
|
+
@abstractmethod
|
|
31
|
+
def fit(self, backbone: HybridBackbone) -> None:
|
|
32
|
+
"""Vorberechnung von Referenzvektoren."""
|
|
33
|
+
|
|
34
|
+
@abstractmethod
|
|
35
|
+
def evaluate(self, encoded: EncodedInput) -> dict:
|
|
36
|
+
"""Berechnet das Ergebnis basierend auf den vorberechneten Vektoren."""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _normalize_rows(vectors: np.ndarray) -> np.ndarray:
|
|
40
|
+
"""Zeilenweise auf Einheitslänge normieren (0-Vektoren bleiben 0)."""
|
|
41
|
+
norms = np.linalg.norm(vectors, axis=1, keepdims=True)
|
|
42
|
+
return vectors / np.where(norms == 0, 1.0, norms)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Choice(BaseHead):
|
|
46
|
+
"""Klassifikation / Routing über Max-Similarity + Keyword-Boost.
|
|
47
|
+
|
|
48
|
+
Options-Label mit den meisten ähnlichen Beispielsätzen gewinnt. Die Sparse-
|
|
49
|
+
Ähnlichkeit (exakte Worttreffer, z. B. Asset-IDs wie `plc-34`) wird mit
|
|
50
|
+
`keyword_boost` auf die Dense-Ähnlichkeit addiert.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
def __init__(self, name: str, options: dict[str, list[str]], keyword_boost: float = 0.5):
|
|
54
|
+
super().__init__(name)
|
|
55
|
+
self.options = options
|
|
56
|
+
self.keyword_boost = keyword_boost
|
|
57
|
+
self.flat_texts: list[str] = []
|
|
58
|
+
self.label_map: list[str] = []
|
|
59
|
+
self.dense_matrix: np.ndarray | None = None
|
|
60
|
+
self.sparse_matrix = None
|
|
61
|
+
|
|
62
|
+
def get_reference_texts(self) -> list[str]:
|
|
63
|
+
return [text for examples in self.options.values() for text in examples]
|
|
64
|
+
|
|
65
|
+
def fit(self, backbone: HybridBackbone) -> None:
|
|
66
|
+
self.flat_texts = []
|
|
67
|
+
self.label_map = []
|
|
68
|
+
for label, examples in self.options.items():
|
|
69
|
+
for example in examples:
|
|
70
|
+
self.flat_texts.append(example)
|
|
71
|
+
self.label_map.append(label)
|
|
72
|
+
|
|
73
|
+
vecs = np.array(list(backbone.embed_model.embed(self.flat_texts)))
|
|
74
|
+
self.dense_matrix = _normalize_rows(vecs)
|
|
75
|
+
self.sparse_matrix = backbone.tfidf_vec.transform(self.flat_texts)
|
|
76
|
+
|
|
77
|
+
def evaluate(self, encoded: EncodedInput) -> dict:
|
|
78
|
+
dense_sims = self.dense_matrix @ encoded.dense_vec
|
|
79
|
+
|
|
80
|
+
# Direkter Sparse-Dot statt sklearn cosine_similarity: TfidfVectorizer
|
|
81
|
+
# normiert beide Vektoren L2 (default norm="l2"), der Dot nicht-negativer
|
|
82
|
+
# Einheitsvektoren IST die Cosine-Aehnlichkeit — aber ~5x schneller
|
|
83
|
+
# (kein sklearn-Call-Overhead pro Abfrage).
|
|
84
|
+
sparse_sims = np.asarray((encoded.sparse_vec @ self.sparse_matrix.T).todense())[0]
|
|
85
|
+
|
|
86
|
+
hybrid_sims = dense_sims + self.keyword_boost * sparse_sims
|
|
87
|
+
|
|
88
|
+
# Pro Label die beste Beispiel-Ähnlichkeit behalten.
|
|
89
|
+
category_scores: dict[str, float] = {}
|
|
90
|
+
for idx, sim in enumerate(hybrid_sims):
|
|
91
|
+
label = self.label_map[idx]
|
|
92
|
+
if label not in category_scores or sim > category_scores[label]:
|
|
93
|
+
category_scores[label] = float(sim)
|
|
94
|
+
|
|
95
|
+
best_label = max(category_scores, key=category_scores.get)
|
|
96
|
+
best_score = category_scores[best_label]
|
|
97
|
+
|
|
98
|
+
# Margin zum Runner-Up als Konfidenzkalibrierung.
|
|
99
|
+
sorted_scores = sorted(category_scores.values(), reverse=True)
|
|
100
|
+
runner_up = sorted_scores[1] if len(sorted_scores) > 1 else 0.0
|
|
101
|
+
confidence = float(np.clip((best_score - runner_up) / (best_score + 1e-5) * 1.5, 0.0, 1.0))
|
|
102
|
+
|
|
103
|
+
return {
|
|
104
|
+
"value": best_label,
|
|
105
|
+
"score": best_score,
|
|
106
|
+
"confidence": confidence,
|
|
107
|
+
"scores": category_scores,
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class Score(BaseHead):
|
|
112
|
+
"""Kontinuierliche Projektion auf eine semantische Achse.
|
|
113
|
+
|
|
114
|
+
Der Text wird gegen Low- und High-Anker similarity-gemessen; die Differenz
|
|
115
|
+
geht durch eine Sigmoid-Scherfunktion und wird auf [min_val, max_val] gemappt.
|
|
116
|
+
"""
|
|
117
|
+
|
|
118
|
+
def __init__(
|
|
119
|
+
self,
|
|
120
|
+
name: str,
|
|
121
|
+
low_anchors: list[str],
|
|
122
|
+
high_anchors: list[str],
|
|
123
|
+
min_val: float = 0.0,
|
|
124
|
+
max_val: float = 3.0,
|
|
125
|
+
sharpness: float = 8.0,
|
|
126
|
+
):
|
|
127
|
+
super().__init__(name)
|
|
128
|
+
self.low_anchors = low_anchors
|
|
129
|
+
self.high_anchors = high_anchors
|
|
130
|
+
self.min_val = min_val
|
|
131
|
+
self.max_val = max_val
|
|
132
|
+
self.sharpness = sharpness
|
|
133
|
+
self.low_matrix: np.ndarray | None = None
|
|
134
|
+
self.high_matrix: np.ndarray | None = None
|
|
135
|
+
|
|
136
|
+
def get_reference_texts(self) -> list[str]:
|
|
137
|
+
return self.low_anchors + self.high_anchors
|
|
138
|
+
|
|
139
|
+
def fit(self, backbone: HybridBackbone) -> None:
|
|
140
|
+
low_v = np.array(list(backbone.embed_model.embed(self.low_anchors)))
|
|
141
|
+
self.low_matrix = _normalize_rows(low_v)
|
|
142
|
+
|
|
143
|
+
high_v = np.array(list(backbone.embed_model.embed(self.high_anchors)))
|
|
144
|
+
self.high_matrix = _normalize_rows(high_v)
|
|
145
|
+
|
|
146
|
+
def evaluate(self, encoded: EncodedInput) -> dict:
|
|
147
|
+
# Max-Similarity zu beiden Polen.
|
|
148
|
+
s_low = float(np.max(self.low_matrix @ encoded.dense_vec))
|
|
149
|
+
s_high = float(np.max(self.high_matrix @ encoded.dense_vec))
|
|
150
|
+
|
|
151
|
+
# Sigmoid-basierte Skalierung der Differenz.
|
|
152
|
+
diff = s_high - s_low
|
|
153
|
+
ratio = 1.0 / (1.0 + np.exp(-diff * self.sharpness))
|
|
154
|
+
calculated_score = self.min_val + ratio * (self.max_val - self.min_val)
|
|
155
|
+
|
|
156
|
+
return {
|
|
157
|
+
"value": round(float(calculated_score), 2),
|
|
158
|
+
"raw_diff": diff,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class Flag(BaseHead):
|
|
163
|
+
"""Boolesche Entscheidung mit kalibrierter Wahrscheinlichkeit.
|
|
164
|
+
|
|
165
|
+
Softmax über die Similarities zu True- und False-Ankern; Temperatur steuert
|
|
166
|
+
die Schärfe der Entscheidung.
|
|
167
|
+
|
|
168
|
+
Problem ohne drittes Pol: Bei Out-of-Domain-Texten sind beide Similarities
|
|
169
|
+
niedrig und nahe beieinander -> Wahrscheinlichkeit ~0.5 und Rauschen kippt
|
|
170
|
+
die Entscheidung. Mit `neutral_anchors` wird ein 3-Klassen-Softmax genutzt;
|
|
171
|
+
der Kopf liefert dann `value=None` (statt True/False), wenn "neutral"
|
|
172
|
+
gewinnt. Standardmäßig deaktiviert (klassisches 2-Klassen-Verhalten).
|
|
173
|
+
"""
|
|
174
|
+
|
|
175
|
+
def __init__(
|
|
176
|
+
self,
|
|
177
|
+
name: str,
|
|
178
|
+
true_anchors: list[str],
|
|
179
|
+
false_anchors: list[str],
|
|
180
|
+
threshold: float = 0.5,
|
|
181
|
+
temp: float = 0.12,
|
|
182
|
+
neutral_anchors: list[str] | None = None,
|
|
183
|
+
):
|
|
184
|
+
super().__init__(name)
|
|
185
|
+
self.true_anchors = true_anchors
|
|
186
|
+
self.false_anchors = false_anchors
|
|
187
|
+
self.threshold = threshold
|
|
188
|
+
self.temp = temp
|
|
189
|
+
self.neutral_anchors = neutral_anchors or []
|
|
190
|
+
self.true_matrix: np.ndarray | None = None
|
|
191
|
+
self.false_matrix: np.ndarray | None = None
|
|
192
|
+
self.neutral_matrix: np.ndarray | None = None
|
|
193
|
+
|
|
194
|
+
def get_reference_texts(self) -> list[str]:
|
|
195
|
+
return self.true_anchors + self.false_anchors + self.neutral_anchors
|
|
196
|
+
|
|
197
|
+
def fit(self, backbone: HybridBackbone) -> None:
|
|
198
|
+
t_v = np.array(list(backbone.embed_model.embed(self.true_anchors)))
|
|
199
|
+
self.true_matrix = _normalize_rows(t_v)
|
|
200
|
+
|
|
201
|
+
f_v = np.array(list(backbone.embed_model.embed(self.false_anchors)))
|
|
202
|
+
self.false_matrix = _normalize_rows(f_v)
|
|
203
|
+
|
|
204
|
+
if self.neutral_anchors:
|
|
205
|
+
n_v = np.array(list(backbone.embed_model.embed(self.neutral_anchors)))
|
|
206
|
+
self.neutral_matrix = _normalize_rows(n_v)
|
|
207
|
+
else:
|
|
208
|
+
self.neutral_matrix = None
|
|
209
|
+
|
|
210
|
+
def evaluate(self, encoded: EncodedInput) -> dict:
|
|
211
|
+
s_true = float(np.max(self.true_matrix @ encoded.dense_vec))
|
|
212
|
+
s_false = float(np.max(self.false_matrix @ encoded.dense_vec))
|
|
213
|
+
|
|
214
|
+
# Softmax über zwei (oder drei) Klassen mit Temperatur-Skalierung.
|
|
215
|
+
logits = np.array([s_true, s_false], dtype=float)
|
|
216
|
+
if self.neutral_matrix is not None:
|
|
217
|
+
s_neutral = float(np.max(self.neutral_matrix @ encoded.dense_vec))
|
|
218
|
+
logits = np.array([s_true, s_false, s_neutral], dtype=float)
|
|
219
|
+
|
|
220
|
+
scaled = logits / self.temp
|
|
221
|
+
scaled -= scaled.max() # numerisch stabiler Softmax
|
|
222
|
+
exp = np.exp(scaled)
|
|
223
|
+
probs = exp / exp.sum()
|
|
224
|
+
|
|
225
|
+
if self.neutral_matrix is not None:
|
|
226
|
+
prob_true, prob_false, prob_neutral = (float(p) for p in probs)
|
|
227
|
+
if prob_neutral > max(prob_true, prob_false):
|
|
228
|
+
return {
|
|
229
|
+
"value": None,
|
|
230
|
+
"probability": prob_true,
|
|
231
|
+
"probabilities": {"true": prob_true, "false": prob_false, "neutral": prob_neutral},
|
|
232
|
+
}
|
|
233
|
+
value = bool(prob_true >= self.threshold)
|
|
234
|
+
return {
|
|
235
|
+
"value": value,
|
|
236
|
+
"probability": prob_true,
|
|
237
|
+
"probabilities": {"true": prob_true, "false": prob_false, "neutral": prob_neutral},
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
prob = float(probs[0])
|
|
241
|
+
return {
|
|
242
|
+
"value": bool(prob >= self.threshold),
|
|
243
|
+
"probability": prob,
|
|
244
|
+
}
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: klix-engine
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Entkoppelte Entscheidungs-Köpfe (Choice, Score, Flag) auf einem geteilten semantischen Backbone - lokal, deterministisch, ohne Modelltraining.
|
|
5
|
+
Author: Max-Christoph Hänel
|
|
6
|
+
License: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Keywords: classification,decision-engine,embeddings,nlp,routing,semantic-search
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Natural Language :: German
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: fastembed<1,>=0.3
|
|
21
|
+
Requires-Dist: numpy>=1.24
|
|
22
|
+
Requires-Dist: scikit-learn>=1.3
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# Klix
|
|
26
|
+
|
|
27
|
+
Entkoppelte Entscheidungs-Köpfe auf einem geteilten semantischen Backbone.
|
|
28
|
+
Ein Text geht **genau einmal** durch das Embedding-Modell (Dense + Sparse), beliebig viele
|
|
29
|
+
Köpfe (`Choice`, `Score`, `Flag`) arbeiten anschließend auf den vorberechneten Vektoren –
|
|
30
|
+
jeder in seinem eigenen mathematischen Raum. Kein Modelltraining, keine Slot-Limits,
|
|
31
|
+
vollständig offline und CPU-only.
|
|
32
|
+
|
|
33
|
+
## Architektur
|
|
34
|
+
|
|
35
|
+
```text
|
|
36
|
+
Text ──► HybridBackbone (FastEmbed-Dense + TF-IDF-Sparse, einmalig ~10 ms)
|
|
37
|
+
│
|
|
38
|
+
├──► Choice (Routing/Klassifikation: Max-Similarity + Keyword-Boost)
|
|
39
|
+
├──► Score (kontinuierliche Achse: Low/High-Anker + Sigmoid)
|
|
40
|
+
├──► Flag (Boolesch: 2/3-Klassen-Softmax mit Temperatur)
|
|
41
|
+
└──► eigene Köpfe (von BaseHead erben)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
- **Shared Backbone:** Der Text wird einmal embedding + einmal TF-IDF transformiert.
|
|
45
|
+
3 Köpfe oder 50 Köpfe – die Extraktionskosten bleiben gleich.
|
|
46
|
+
- **Entkoppelte Köpfe:** Neue Optionen in einem `Choice` beeinflussen weder `Score`- noch
|
|
47
|
+
`Flag`-Ergebnisse. Jeder Kopf kapselt seine eigene Logik.
|
|
48
|
+
- **Deklarativ:** Nur Schemata mit Beispielsätzen definieren, `compile()`, fertig.
|
|
49
|
+
|
|
50
|
+
## Installation
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
uv add klix-engine
|
|
54
|
+
# oder
|
|
55
|
+
pip install klix-engine
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Beim ersten Aufruf lädt FastEmbed das Modell `paraphrase-multilingual-MiniLM-L12-v2`
|
|
59
|
+
(~120 MB, einmalig, danach lokal gecacht). Danach läuft alles offline.
|
|
60
|
+
|
|
61
|
+
## Schnellstart
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from klix import DecisionEngine, Choice, Score, Flag
|
|
65
|
+
|
|
66
|
+
engine = DecisionEngine()
|
|
67
|
+
|
|
68
|
+
engine.add_head(
|
|
69
|
+
Choice(
|
|
70
|
+
name="target",
|
|
71
|
+
options={
|
|
72
|
+
"it_ops": ["VPN abgerissen", "Server down", "Rechner bootet nicht"],
|
|
73
|
+
"ot_plant": ["Roboterzelle steht", "SPS Fehler", "Taktzeit deviation"],
|
|
74
|
+
"finance": ["KST 4210 über Budget", "Rechnung freigeben"],
|
|
75
|
+
"facility": ["Öllache Halle 2", "Heizung defekt"],
|
|
76
|
+
},
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
engine.add_head(
|
|
81
|
+
Score(
|
|
82
|
+
name="urgency",
|
|
83
|
+
low_anchors=["Routine-Wartung", "Informelle Frage"],
|
|
84
|
+
high_anchors=["Notfall sofort", "Produktionsstillstand", "Akute Gefahr"],
|
|
85
|
+
min_val=0.0,
|
|
86
|
+
max_val=3.0,
|
|
87
|
+
)
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
engine.add_head(
|
|
91
|
+
Flag(
|
|
92
|
+
name="is_security",
|
|
93
|
+
true_anchors=["Hackerangriff", "Ransomware Befall", "Datenabfluss"],
|
|
94
|
+
false_anchors=["Hardware kaputt", "Netzwerkstörung", "Alltägliche Anfrage"],
|
|
95
|
+
threshold=0.5,
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
engine.compile()
|
|
100
|
+
|
|
101
|
+
res = engine.decide("plc-34 meldet fehler, förderband steht sofort!")
|
|
102
|
+
|
|
103
|
+
print(res) # <DecisionResult (11 ms): target=ot_plant, urgency=2.9, is_security=False>
|
|
104
|
+
print(res.target) # 'ot_plant'
|
|
105
|
+
print(res.urgency) # 2.87
|
|
106
|
+
print(res.is_security) # False
|
|
107
|
+
print(res.details("is_security")) # {'value': False, 'probability': 0.03}
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Eigene Köpfe
|
|
111
|
+
|
|
112
|
+
Von `BaseHead` erben und `evaluate(encoded)` implementieren:
|
|
113
|
+
|
|
114
|
+
```python
|
|
115
|
+
import re
|
|
116
|
+
from klix import BaseHead
|
|
117
|
+
|
|
118
|
+
class RegExExtractionHead(BaseHead):
|
|
119
|
+
def __init__(self, name: str, pattern: str):
|
|
120
|
+
super().__init__(name)
|
|
121
|
+
self.re = re.compile(pattern)
|
|
122
|
+
|
|
123
|
+
def get_reference_texts(self) -> list[str]:
|
|
124
|
+
return [] # keine Referenztexte nötig
|
|
125
|
+
|
|
126
|
+
def fit(self, backbone) -> None:
|
|
127
|
+
pass
|
|
128
|
+
|
|
129
|
+
def evaluate(self, encoded) -> dict:
|
|
130
|
+
match = self.re.search(encoded.text)
|
|
131
|
+
return {"value": match.group(0) if match else None}
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
## Entwicklung
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
uv sync # Abhängigkeiten installieren
|
|
138
|
+
uv run pytest # Tests
|
|
139
|
+
uv run python examples/demo.py
|
|
140
|
+
uv build # PyPI-Artefakte (wheel + sdist) nach dist/
|
|
141
|
+
uv publish # Hochladen (erfordert Token/Account)
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
## Lizenz
|
|
145
|
+
|
|
146
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
klix/__init__.py,sha256=YXd4B5-tKilvsmm2SBFoFmKfB9D87TPc1egtgbdZtBs,355
|
|
2
|
+
klix/backbone.py,sha256=fZH9gdr8PXcXWs1mnKweqtLVVThrArN5IrZakVDsK4c,2479
|
|
3
|
+
klix/engine.py,sha256=H6YbYf41zAFd4PCwtwzrmSA98e5427BFbKOqtD_7KAk,2889
|
|
4
|
+
klix/heads.py,sha256=QgkR9Gj5sw_L61jA2kjX8IsbUvlcArX-Jz8oax5lJ_o,9348
|
|
5
|
+
klix_engine-0.1.0.dist-info/METADATA,sha256=3l7f2mt3KezDHjZFhuytfCyKOcp56x2sDlaCaJ6koho,4781
|
|
6
|
+
klix_engine-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
7
|
+
klix_engine-0.1.0.dist-info/licenses/LICENSE,sha256=ExuIc59574ZQRqmBXXSY2AZrC8g4rzQ7syTDH6BKWvE,1076
|
|
8
|
+
klix_engine-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Max-Christoph Hänel
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|