cine-rec-engine 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ """cine-rec-engine — content-based movie & series recommendations from your own PostgreSQL."""
2
+
3
+ from .service import RecommendationService
4
+
5
+ __version__ = "0.4.0"
6
+ __all__ = ["RecommendationService", "__version__"]
@@ -0,0 +1,115 @@
1
+ """Cache layer for the engine: Redis when available, in-process TTL otherwise.
2
+
3
+ The engine touches the cache through a tiny async surface —
4
+ ``get / set / set_nx / exists / delete / is_connected`` — so any client
5
+ implementing it plugs in (we ship one below). If a Redis URL is
6
+ configured via ``CINE_REC_REDIS_URL`` we lazily connect with ``redis``
7
+ (asyncio); otherwise a per-process TTL dict keeps everything working
8
+ with zero infrastructure.
9
+
10
+ Every cache call in the engine is wrapped in ``try/except`` and degrades
11
+ on failure — a cache outage must never break a recommendation.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import asyncio
17
+ import json
18
+ import os
19
+ import time
20
+ from typing import Any, Optional
21
+
22
+ _client: Any = None
23
+ _client_checked = False
24
+ _fallback: Optional["_TTLCache"] = None
25
+
26
+
27
+ class _TTLCache:
28
+ """Minimal async cache with TTL + NX semantics (single-process)."""
29
+
30
+ def __init__(self) -> None:
31
+ self._data: dict = {}
32
+ self._lock = asyncio.Lock()
33
+
34
+ def is_connected(self) -> bool:
35
+ return True
36
+
37
+ async def get(self, key: str) -> Any:
38
+ async with self._lock:
39
+ item = self._data.get(key)
40
+ if item is None:
41
+ return None
42
+ value, expires = item
43
+ if expires is not None and expires < time.monotonic():
44
+ del self._data[key]
45
+ return None
46
+ # JSON round-trip so callers see the same shape as Redis gives
47
+ try:
48
+ return json.loads(value)
49
+ except (TypeError, ValueError):
50
+ return value
51
+
52
+ async def set(self, key: str, value: Any, ttl: Optional[int] = None) -> Any:
53
+ async with self._lock:
54
+ serialized = json.dumps(value, default=str)
55
+ self._data[key] = (serialized, time.monotonic() + ttl if ttl is not None else None)
56
+ return True
57
+
58
+ async def set_nx(self, key: str, value: Any, ttl: Optional[int] = None) -> bool:
59
+ async with self._lock:
60
+ item = self._data.get(key)
61
+ if item is not None:
62
+ expires = item[1]
63
+ if expires is None or expires >= time.monotonic():
64
+ return False
65
+ del self._data[key]
66
+ self._data[key] = (json.dumps(value, default=str),
67
+ time.monotonic() + ttl if ttl is not None else None)
68
+ return True
69
+
70
+ async def exists(self, key: str) -> bool:
71
+ return await self.get(key) is not None
72
+
73
+ async def delete(self, key: str) -> None:
74
+ async with self._lock:
75
+ self._data.pop(key, None)
76
+
77
+
78
+ async def get_cache() -> Any:
79
+ """Return a cache client, or None if nothing is available.
80
+
81
+ Priority: Redis (``CINE_REC_REDIS_URL``) → in-process TTL cache.
82
+ Redis is probed once per process; afterwards the decision sticks
83
+ unless a live connection errors out.
84
+ """
85
+ global _client, _client_checked, _fallback
86
+
87
+ if not _client_checked:
88
+ _client_checked = True
89
+ url = os.getenv("CINE_REC_REDIS_URL")
90
+ if url:
91
+ try:
92
+ import redis.asyncio as aioredis # type: ignore
93
+
94
+ _client = aioredis.from_url(
95
+ url, decode_responses=True, socket_timeout=2,
96
+ socket_connect_timeout=2,
97
+ )
98
+ await _client.ping()
99
+ except Exception:
100
+ _client = None
101
+
102
+ if _client is not None:
103
+ try:
104
+ if await _client.ping():
105
+ return _client
106
+ except Exception:
107
+ _client = None # Redis died mid-flight — fall through to local
108
+ elif _fallback is not None:
109
+ # Redis was already ruled out this process; stay local without
110
+ # re-probing on every call (a recommendation path must not ping).
111
+ return _fallback
112
+
113
+ if _fallback is None:
114
+ _fallback = _TTLCache()
115
+ return _fallback
@@ -0,0 +1,221 @@
1
+ """Scoring configuration for the recommendation engine.
2
+
3
+ Tuned feature weights for the similarity scorer (see weights.json provenance).
4
+ """
5
+
6
+ # Scoring weights for recommendation components
7
+ COMPANY_SIMILARITY_WEIGHT: float = 0.75 # Production company overlap — a weak taste signal
8
+ # (Warner Bros produced half the action catalog; at 3.0 it drowned semantics)
9
+ COLLECTION_MATCH_BONUS: float = 2.0 # Same franchise/collection (flat +2)
10
+ NETWORK_MATCH_BONUS: float = 2.0 # Shared TV network (flat +2, TV only)
11
+ # Director DNA — person-id match from tmdb_crew (name-string match kept as
12
+ # fallback for rows the crew backfill hasn't reached yet). The channel bonus
13
+ # rides on top for candidates the director channel recalled: by construction
14
+ # they share the seed's director person_id, the strongest auteur signal the
15
+ # overview vectors cannot see (Coen brothers never match by name string).
16
+ DIRECTOR_MATCH_BONUS: float = 3.0
17
+ # Modest on purpose: writer/cast/behavioral signals carry the rest of the
18
+ # auteur signal, so the channel bonus only nudges.
19
+ DIRECTOR_CHANNEL_BONUS: float = 3.0
20
+ # Writer DNA — a shared screenwriter links taste clusters the director
21
+ # signal can't see (Taylor Sheridan's Sicario / Hell or High Water / Wind
22
+ # River are different directors, one voice). Weaker than an auteur stamp,
23
+ # so it starts lower and applies once per pair (match or channel recall).
24
+ WRITER_BONUS: float = 3.5
25
+ # Behavioral signal — a title TMDB's own users ranked in the seed's top
26
+ # recommendations ("people also watched"). Today it only escapes the
27
+ # genre gate; this gives the rank-1-3 recalls real points.
28
+ TMDB_REC_BONUS: float = 3.5
29
+ # Auteur diversification — geometric decay per additional same-director
30
+ # title in the final list (first keeps full score). (1.0, 1.0, 1.0) = off;
31
+ # diversification is handled by MMR instead.
32
+ AUTEUR_DECAY_FACTORS: tuple = (1.0, 1.0, 1.0)
33
+ # Genre-conditional auteur decay (7c): the second-plus same-director title
34
+ # whose genres share NOTHING with the seed decays by this factor — Sicario
35
+ # (Crime) under Arrival (Sci-Fi/Drama) is director brand without a thematic
36
+ # bridge, while Dune (Sci-Fi) keeps its full auteur score.
37
+ GENRE_MISMATCH_AUTEUR_FACTOR: float = 0.5
38
+ # --- MMR selection (step 7d) ------------------------------------------
39
+ # Greedy Maximal Marginal Relevance over the ranked list before the final
40
+ # limit: score(c) = λ·relevance_norm − (1−λ)·max_sim(c, selected).
41
+ # Relevance is min-max normalized per list so both terms live in [0,1]
42
+ # and λ has true 70/30 meaning. Similarity is graded: same collection 1.0
43
+ # (sequel pile-up), shared director 0.85 (a second masterpiece can still
44
+ # surface), genre Jaccard capped at 0.5 (genre overlap must not erase
45
+ # good picks). Replaces the point-penalty decays as THE diversifier.
46
+ MMR_ENABLED: bool = True
47
+ MMR_LAMBDA: float = 0.8 # tuned for a 70/30 relevance/diversity balance
48
+ # Pop-action leak damper — mega-vote titles (>10k votes) whose semantic
49
+ # similarity to the seed is below the floor lose POP_ACTION_PENALTY:
50
+ # Transformers has genre/company connectivity with Arrival but zero
51
+ # thematic overlap, and the light −1.0 popularity damping never stopped it.
52
+ POP_ACTION_SIM_FLOOR: float = 0.30
53
+ # 0.0 — off; the similarity floor above does the gating.
54
+ POP_ACTION_PENALTY: float = 0.0
55
+
56
+ # Genre priority weights — uniform 1.0 for most genres, with selective
57
+ # boosts for strong secondary-genre signals.
58
+ # - Music 1.5: when someone searches for a musical (telenovela, teen pop),
59
+ # other musicals should outrank non-musical matches in the same tone
60
+ # (a teen musical should steer toward other teen musicals).
61
+ # - Soap 1.3: telenovela format is a strong affinity signal.
62
+ # The old values (Mystery=1.6 vs Family=0.7) actively suppressed light
63
+ # content: a user searching for a teen musical got dark fantasy because
64
+ # dark genres "scored more" per match. Genre matching should reflect
65
+ # similarity, not prestige. Only truly niche formats get < 1.0.
66
+ GENRE_PRIORITY: dict[str, float] = {
67
+ "Mystery": 1.0,
68
+ "Psychological": 1.0,
69
+ "Science Fiction": 1.0,
70
+ "Horror": 1.0,
71
+ "Thriller": 1.0,
72
+ "Crime": 1.0,
73
+ "Fantasy": 1.0,
74
+ "Action": 1.0,
75
+ "War": 1.0,
76
+ "History": 1.0,
77
+ "Western": 1.0,
78
+ "Adventure": 1.0,
79
+ "Drama": 1.0,
80
+ "Comedy": 1.0,
81
+ "Romance": 1.0,
82
+ "Animation": 1.0,
83
+ "Family": 1.0,
84
+ "Music": 1.5,
85
+ "Soap": 1.3,
86
+ "Documentary": 0.5,
87
+ }
88
+
89
+ # Keyword → style mapping for style matching (+3 points)
90
+ KEYWORDS_STYLE: dict[str, set[str]] = {
91
+ "mind-bending": {
92
+ "dream",
93
+ "subconscious",
94
+ "lucid dream",
95
+ "virtual reality",
96
+ "simulation",
97
+ "manipulation",
98
+ "memory loss",
99
+ "amnesia",
100
+ "alternate reality",
101
+ "hallucination",
102
+ "unreliable narrator",
103
+ "mind game",
104
+ "twist ending",
105
+ "psychological thriller",
106
+ },
107
+ "cyberpunk/dystopian": {
108
+ "cyberpunk",
109
+ "dystopia",
110
+ "artificial intelligence",
111
+ "android",
112
+ "transhumanism",
113
+ "megacorporation",
114
+ "post-apocalyptic",
115
+ },
116
+ "time-travel": {
117
+ "time travel",
118
+ "time loop",
119
+ "temporal paradox",
120
+ "parallel timeline",
121
+ "multiverse",
122
+ },
123
+ "noir": {
124
+ "neo-noir",
125
+ "femme fatale",
126
+ "cynical detective",
127
+ "dark city",
128
+ "gritty",
129
+ "conspiracy",
130
+ },
131
+ "buddy cop": {"buddy cop", "bromance", "police partners"},
132
+ "parody": {"parody", "satire", "spoof"},
133
+ "slapstick": {"absurd", "slapstick", "over-the-top"},
134
+ }
135
+
136
+ # Semantic (pgvector KNN) candidate recall: top-N nearest overview
137
+ # embeddings fetched per seed IN ADDITION to the genre+popularity candidates.
138
+ # Surfaces titles that are close in spirit but would never pass the
139
+ # genre-overlap + popularity ordering of the classic recall path.
140
+ KNN_CANDIDATES_PER_SEED: int = 60
141
+
142
+ # Local catalog stores Hebrew genre labels ('מותחן', 'מסתורין') while the
143
+ # scoring tables (GENRE_PRIORITY, style matching) use canonical English.
144
+ # Map every label found in tmdb_genres so priority weighting actually fires.
145
+ GENRE_LABELS_EN: dict[str, str] = {
146
+ "אימה": "Horror",
147
+ "אנימציה": "Animation",
148
+ "אקשן": "Action",
149
+ "אקשן והרפתקאות": "Action",
150
+ "דוקומנטרי": "Documentary",
151
+ "דיבורים": "Talk Show",
152
+ "דרמה": "Drama",
153
+ "הסטוריה": "History",
154
+ "הרפתקאות": "Adventure",
155
+ "חדשות": "News",
156
+ "ילדים": "Family",
157
+ "מדע בדיוני": "Science Fiction",
158
+ "מדע בדיוני ופנטזיה": "Science Fiction",
159
+ "מוסיקה": "Music",
160
+ "מותחן": "Thriller",
161
+ "מלחמה": "War",
162
+ "מלחמה ופוליטיקה": "War",
163
+ "מסתורין": "Mystery",
164
+ "מערבון": "Western",
165
+ "משפחה": "Family",
166
+ "סבון": "Soap",
167
+ "סרט טלויזיה": "TV Movie",
168
+ "פנטזיה": "Fantasy",
169
+ "פשע": "Crime",
170
+ "קומדיה": "Comedy",
171
+ "רומנטי": "Romance",
172
+ "ריאליטי": "Reality",
173
+ }
174
+
175
+ # Candidates below this many TMDB votes are dropped during pre-filtering —
176
+ # thin-voted oddities (old/niche titles with a lucky 6.5) should not crowd
177
+ # proven classics out of the shortlist.
178
+ MIN_VOTE_COUNT: int = 500
179
+
180
+ # Media-type-aware vote floors (channels AND the pre-filter). Measured on
181
+ # a lower TV floor (100/200) widens series recall
182
+ # but nets -0.2 (noise outweighs hits), so both stay 500 for now — the
183
+ # plumbing remains so the split can be revisited with better tv embeddings.
184
+ VOTE_FLOOR_MOVIE: int = 500
185
+ VOTE_FLOOR_TV: int = 500
186
+
187
+
188
+ # ----------------------------------------------------------------------------
189
+ # Embedding space — "original" (384-dim MiniLM, column `embedding`) or
190
+ # "v4b" (768-dim fine-tuned bge, column `embedding_v4`). All engine reads
191
+ # (both KNN channels + the cosine feature) follow this switch; embeddings
192
+ # are written by scripts/eval/embed_catalog_v4.py with the canonical text.
193
+ # ----------------------------------------------------------------------------
194
+ import os
195
+
196
+ REC_EMBEDDING_SPACE = os.getenv("CINE_REC_EMBEDDING", os.getenv("REC_EMBEDDING", "original"))
197
+ # Bake-off spaces map to their embedding_<name> columns; "v4b" keeps its
198
+ # historical name and "original" the legacy 384-dim column.
199
+ EMBEDDING_COLUMN = {
200
+ "v4b": "embedding_v4",
201
+ "original": "embedding",
202
+ }.get(REC_EMBEDDING_SPACE, f"embedding_{REC_EMBEDDING_SPACE}")
203
+
204
+
205
+ # Optional cosine blend across several embedding columns:
206
+ # REC_COSINE_BLEND="embedding_mpnetae:32.0,embedding_e5e:8.0,embedding_mpnetan:4.0"
207
+ # The final-score cosine becomes sum(w_i/sum(w) * cos_i) * cosine_sim_weight.
208
+ # Retrieval (KNN) stays on EMBEDDING_COLUMN — the blend rescores candidates.
209
+ def _parse_cosine_blend(raw: str) -> dict:
210
+ blend = {}
211
+ for part in (raw or "").split(","):
212
+ if ":" in part:
213
+ col, w = part.rsplit(":", 1)
214
+ try:
215
+ blend[col.strip()] = float(w)
216
+ except ValueError:
217
+ continue
218
+ return {c: w for c, w in blend.items() if w > 0}
219
+
220
+
221
+ COSINE_BLEND: dict = _parse_cosine_blend(os.getenv("CINE_REC_COSINE_BLEND", os.getenv("REC_COSINE_BLEND", "")))
@@ -0,0 +1,61 @@
1
+ """Embedding space registry — the seam between engine and models.
2
+
3
+ The engine's KNN recall channel and cosine-blend feature read OVERVIEW
4
+ EMBEDDINGS from columns on ``tmdb_media``. Which column(s) exist is a
5
+ property of YOUR database: you generate embeddings with whichever
6
+ sentence-encoder you like (see ``models/README.md``), store them as
7
+ ``vector`` columns, and register them here.
8
+
9
+ Each entry names:
10
+ - ``column``: the tmdb_media column holding the vectors (None = the
11
+ tuned multi-model ensemble driven by config.COSINE_BLEND).
12
+ - ``cosine_weight``: the cosine_sim feature weight to use when this
13
+ space is selected solo (overrides the tuned default).
14
+
15
+ The published ``weights.json`` was tuned on the ensemble space; solo
16
+ weights below are safe defaults, not fitted constants.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from typing import Optional
22
+
23
+ REC_MODELS: dict = {
24
+ "ensemble": {
25
+ "column": None,
26
+ "cosine_weight": None,
27
+ },
28
+ "e5e": {
29
+ "column": "embedding_e5e",
30
+ "cosine_weight": 26.0,
31
+ },
32
+ "mpnetae": {
33
+ "column": "embedding_mpnetae",
34
+ "cosine_weight": 26.0,
35
+ },
36
+ "mpnetan": {
37
+ "column": "embedding_mpnetan",
38
+ "cosine_weight": 26.0,
39
+ },
40
+ "v4": {
41
+ "column": "embedding_v4",
42
+ "cosine_weight": 10.0,
43
+ },
44
+ }
45
+
46
+
47
+ def normalize_model(name: Optional[str]) -> Optional[str]:
48
+ """Case-insensitive match against REC_MODELS, or None."""
49
+ if not name:
50
+ return None
51
+ low = name.strip().lower()
52
+ for key in REC_MODELS:
53
+ if key.lower() == low:
54
+ return key
55
+ return None
56
+
57
+
58
+ def is_solo(model_key: str) -> bool:
59
+ """True when the key selects a single embedding column."""
60
+ spec = REC_MODELS.get(model_key)
61
+ return bool(spec and spec["column"])