cine-rec-engine 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cine_rec_engine-0.4.0/LICENSE +21 -0
- cine_rec_engine-0.4.0/PKG-INFO +237 -0
- cine_rec_engine-0.4.0/README.md +205 -0
- cine_rec_engine-0.4.0/cine_rec_engine/__init__.py +6 -0
- cine_rec_engine-0.4.0/cine_rec_engine/cache.py +115 -0
- cine_rec_engine-0.4.0/cine_rec_engine/config.py +221 -0
- cine_rec_engine-0.4.0/cine_rec_engine/model_spaces.py +61 -0
- cine_rec_engine-0.4.0/cine_rec_engine/queries.py +1057 -0
- cine_rec_engine-0.4.0/cine_rec_engine/scoring.py +459 -0
- cine_rec_engine-0.4.0/cine_rec_engine/service.py +1768 -0
- cine_rec_engine-0.4.0/cine_rec_engine/tmdb_client.py +62 -0
- cine_rec_engine-0.4.0/cine_rec_engine/tmdb_recs.py +303 -0
- cine_rec_engine-0.4.0/cine_rec_engine/user_stats.py +440 -0
- cine_rec_engine-0.4.0/cine_rec_engine/user_vector.py +175 -0
- cine_rec_engine-0.4.0/cine_rec_engine/user_weights.py +180 -0
- cine_rec_engine-0.4.0/cine_rec_engine/watched.py +230 -0
- cine_rec_engine-0.4.0/cine_rec_engine/weights.json +53 -0
- cine_rec_engine-0.4.0/cine_rec_engine.egg-info/PKG-INFO +237 -0
- cine_rec_engine-0.4.0/cine_rec_engine.egg-info/SOURCES.txt +36 -0
- cine_rec_engine-0.4.0/cine_rec_engine.egg-info/dependency_links.txt +1 -0
- cine_rec_engine-0.4.0/cine_rec_engine.egg-info/requires.txt +14 -0
- cine_rec_engine-0.4.0/cine_rec_engine.egg-info/top_level.txt +2 -0
- cine_rec_engine-0.4.0/ingest/__init__.py +17 -0
- cine_rec_engine-0.4.0/ingest/cli.py +125 -0
- cine_rec_engine-0.4.0/ingest/exports.py +109 -0
- cine_rec_engine-0.4.0/ingest/loader.py +576 -0
- cine_rec_engine-0.4.0/ingest/rate.py +41 -0
- cine_rec_engine-0.4.0/pyproject.toml +72 -0
- cine_rec_engine-0.4.0/setup.cfg +4 -0
- cine_rec_engine-0.4.0/tests/test_cache.py +39 -0
- cine_rec_engine-0.4.0/tests/test_engine_smoke.py +50 -0
- cine_rec_engine-0.4.0/tests/test_fast_path_parity.py +99 -0
- cine_rec_engine-0.4.0/tests/test_feature_alignment.py +42 -0
- cine_rec_engine-0.4.0/tests/test_ingest.py +266 -0
- cine_rec_engine-0.4.0/tests/test_personalization.py +188 -0
- cine_rec_engine-0.4.0/tests/test_rate.py +29 -0
- cine_rec_engine-0.4.0/tests/test_scoring.py +232 -0
- cine_rec_engine-0.4.0/tests/test_weights_blending.py +57 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DovBer Kaplan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,237 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: cine-rec-engine
|
|
3
|
+
Version: 0.4.0
|
|
4
|
+
Summary: Content-based movie & series recommendation engine over your own PostgreSQL TMDB mirror
|
|
5
|
+
Author: DovBerKaplan
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/DovBerKaplan/cine-rec-engine
|
|
8
|
+
Project-URL: Issues, https://github.com/DovBerKaplan/cine-rec-engine/issues
|
|
9
|
+
Keywords: recommendations,tmdb,movies,postgres,pgvector,content-based
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Topic :: Database
|
|
15
|
+
Classifier: Topic :: Multimedia :: Video
|
|
16
|
+
Classifier: Framework :: AsyncIO
|
|
17
|
+
Requires-Python: >=3.10
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
License-File: LICENSE
|
|
20
|
+
Requires-Dist: asyncpg>=0.29
|
|
21
|
+
Requires-Dist: aiohttp>=3.9
|
|
22
|
+
Requires-Dist: loguru>=0.7
|
|
23
|
+
Provides-Extra: redis
|
|
24
|
+
Requires-Dist: redis>=5.0; extra == "redis"
|
|
25
|
+
Provides-Extra: pg
|
|
26
|
+
Requires-Dist: pgvector>=0.3; extra == "pg"
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
29
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
|
|
30
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
31
|
+
Dynamic: license-file
|
|
32
|
+
|
|
33
|
+
# Cine Rec Engine
|
|
34
|
+
|
|
35
|
+
[](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml)
|
|
36
|
+
[](LICENSE)
|
|
37
|
+
[](pyproject.toml)
|
|
38
|
+
|
|
39
|
+
> **Recommendations from your own database. No black-box API, no rented taste.**
|
|
40
|
+
|
|
41
|
+

|
|
42
|
+
|
|
43
|
+
**PostgreSQL in · ranked titles out · zero external calls on the hot path.**
|
|
44
|
+
|
|
45
|
+
## Try it in 60 seconds — no API key
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
git clone https://github.com/DovBerKaplan/cine-rec-engine && cd cine-rec-engine/demo
|
|
49
|
+
docker compose up # Postgres + 400 real titles + recommendations
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
That runs the full stack against a bundled catalog (TMDB data, en-US,
|
|
53
|
+
attribution below) and prints, for The Dark Knight and Breaking Bad:
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
Because you watched The Dark Knight (2008)
|
|
57
|
+
The Dark Knight Rises (2012) why: same saga · same director
|
|
58
|
+
The Batman (2022) why: same style tags · shared keywords
|
|
59
|
+
Memento (2000) why: same director · year window
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
A content-based recommendation engine for movies & series: give it one title
|
|
63
|
+
(or a user's whole watch history) and it returns similar titles, ranked by a
|
|
64
|
+
22-feature scorer whose weights were learned — not guessed.
|
|
65
|
+
|
|
66
|
+
```
|
|
67
|
+
your data (TMDB mirror + user events) what you get back
|
|
68
|
+
┌───────────────────────────────┐ ┌───────────────────────┐
|
|
69
|
+
│ PostgreSQL │ │ ranked similar titles │
|
|
70
|
+
│ · tmdb_media + satellites │ │ · score + why │
|
|
71
|
+
│ · (optional) embeddings │──engine──► │ · movies/series mix │
|
|
72
|
+
│ · (optional) TMDB rec cache │ │ · per-user filtering │
|
|
73
|
+
│ · your watch/rating events │ │ · saga advancement │
|
|
74
|
+
└───────────────────────────────┘ └───────────────────────┘
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
## Quick start
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install cine-rec-engine # PyPI (or: pip install -e ".[pg,redis]")
|
|
81
|
+
psql -d yourdb -f docs/schema.sql # the tables it expects
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
import asyncpg
|
|
86
|
+
from cine_rec_engine import RecommendationService
|
|
87
|
+
|
|
88
|
+
pool = await asyncpg.create_pool("postgresql://user:pw@localhost/yourdb")
|
|
89
|
+
rec = RecommendationService()
|
|
90
|
+
await rec.initialize(pool)
|
|
91
|
+
|
|
92
|
+
# Similar to one title
|
|
93
|
+
results = await rec.find_similar(155) # The Dark Knight → [Batman Begins, …]
|
|
94
|
+
|
|
95
|
+
# Or the titles a user loved (movies AND series, auto-balanced)
|
|
96
|
+
results = await rec.find_similar(
|
|
97
|
+
tmdb_id=[155, 27205, 1396], # Dark Knight, Inception, Breaking Bad
|
|
98
|
+
limit=60,
|
|
99
|
+
user_id=42, # filters watched titles, advances sagas
|
|
100
|
+
)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Each result carries its evidence — score, matched features, media type —
|
|
104
|
+
so your UI can explain *why* it recommended something.
|
|
105
|
+
|
|
106
|
+
## Why another recommender
|
|
107
|
+
|
|
108
|
+
| | Hosted rec APIs | Collaborative-filtering stacks | **Cine Rec Engine** |
|
|
109
|
+
|---|---|---|---|
|
|
110
|
+
| Works on day one (no user base) | ✅ | ❌ cold start | ✅ content-based |
|
|
111
|
+
| Your data stays yours | ❌ | ✅ | ✅ |
|
|
112
|
+
| Needs users' rating matrix | ❌ | ✅ | ❌ — metadata + your event table |
|
|
113
|
+
| Explainable per-title | partial | partial | ✅ 22 named features |
|
|
114
|
+
| Runs offline / on-prem | ❌ | ✅ | ✅ one Postgres |
|
|
115
|
+
| Embedding models included | — | — | ❌ **bring your own** (see below) |
|
|
116
|
+
|
|
117
|
+
The engine separates **recall** (SQL: genres, keywords, cast, crew,
|
|
118
|
+
companies, networks, collections, cached TMDB behavioral recs, optional
|
|
119
|
+
pgvector KNN) from **ranking** (a logistic scorer over 22 features with
|
|
120
|
+
published, fitted weights) from **models** (your embedding encoders —
|
|
121
|
+
deliberately not shipped).
|
|
122
|
+
|
|
123
|
+
## Measured against baselines
|
|
124
|
+
|
|
125
|
+
On the bundled 400-title demo pool with hand-curated adjacency judgments
|
|
126
|
+
(`eval/judgments.jsonl`, `eval/eval.py`):
|
|
127
|
+
|
|
128
|
+
| method | pairwise acc. | NDCG@10 |
|
|
129
|
+
|---|---|---|
|
|
130
|
+
| TMDB similar (behavioral graph) | 0.09 | 0.13 |
|
|
131
|
+
| cosine over overviews | 0.63 | 0.04 |
|
|
132
|
+
| **this engine (learned 22-feature scorer)** | **0.83** | **0.36** |
|
|
133
|
+
|
|
134
|
+
Honest caveats: the pool is small (the TMDB graph mostly points outside
|
|
135
|
+
it, hence its floor), and the judgments are one curator's. Bring your own
|
|
136
|
+
judgments file — the harness is in the repo.
|
|
137
|
+
|
|
138
|
+
## Feature highlights
|
|
139
|
+
|
|
140
|
+
- **Multi-seed blending** — one title or a thousand; per-seed weights;
|
|
141
|
+
movie/series ratio follows the seed mix (90/10 cap so a minority is
|
|
142
|
+
never silenced).
|
|
143
|
+
- **Saga-aware** — collection members chain: watched *Rocky I* → recommends
|
|
144
|
+
*Rocky II*, not *Rocky I* again. Whole-saga watchers graduate out.
|
|
145
|
+
- **Per-user personalization** — `recommend_for_user()`: raw watch
|
|
146
|
+
events → per-title weights (completion, series depth, recency,
|
|
147
|
+
engagement) → a normalized user vector → an ANN recall channel through
|
|
148
|
+
the same LTR scorer, with watched/rated/disliked hard-filtered.
|
|
149
|
+
See `docs/personalization.md`.
|
|
150
|
+
- **Auteur recall** — director/writer/composer/DP channels with decay
|
|
151
|
+
(someone's 8th film matters less than their 2nd).
|
|
152
|
+
- **Popularity guardrails** — vote floors per media type kill
|
|
153
|
+
"high rating, 12 votes" noise; an action-popularity leak term keeps
|
|
154
|
+
Marvel out of every list.
|
|
155
|
+
- **MMR diversification** — optional re-rank so one franchise doesn't
|
|
156
|
+
take five consecutive slots.
|
|
157
|
+
- **Deterministic or noisy** — `randomness=0.0` is reproducible; dial it
|
|
158
|
+
up for exploration without burying the top-3.
|
|
159
|
+
- **Degrades gracefully** — no embeddings? no TMDB rec cache? no Redis?
|
|
160
|
+
Those channels switch off; the rest of the engine keeps working.
|
|
161
|
+
|
|
162
|
+
## The published weights
|
|
163
|
+
|
|
164
|
+
`cine_rec_engine/weights.json` — 22 features fitted with pairwise
|
|
165
|
+
logistic regression (6,650 training pairs). Opt in with
|
|
166
|
+
`CINE_REC_SCORER=learned`; features the artifact doesn't cover keep
|
|
167
|
+
their heuristic coefficients (partial application). A few, to set the
|
|
168
|
+
scale:
|
|
169
|
+
|
|
170
|
+
| Feature | Weight |
|
|
171
|
+
|---|---|
|
|
172
|
+
| `tmdb_rec_decay` (behavioral signal, rank-decayed) | 13.74 |
|
|
173
|
+
| `composer_match` | 5.54 |
|
|
174
|
+
| `cosine_sim` (overview embedding similarity) | 5.13 |
|
|
175
|
+
| `keyword_sim` | 4.35 |
|
|
176
|
+
| `writer_match` | 4.08 |
|
|
177
|
+
| `director_match` | 3.50 |
|
|
178
|
+
| `shared_collection` | 2.51 |
|
|
179
|
+
| `medium_mismatch` (movie↔tv penalty) | −0.52 |
|
|
180
|
+
|
|
181
|
+
Also in the repo: the evaluation harness and the demo judgments. Not
|
|
182
|
+
included: the larger labeled training sets the weights were fitted on and
|
|
183
|
+
the optional narrative-tag enrichment — the fitted coefficients
|
|
184
|
+
themselves are published in full.
|
|
185
|
+
|
|
186
|
+
## Feeding it data
|
|
187
|
+
|
|
188
|
+
Bring a local TMDB mirror — the built-in `ingest/` loader builds it from
|
|
189
|
+
TMDB's official daily ID exports (`python -m ingest.cli bootstrap`, then a
|
|
190
|
+
daily `refresh` off `/changes`; `en-US` only, adult-filtered, one API call
|
|
191
|
+
per title, upserts by key). The catalog schema splits movies and TV into
|
|
192
|
+
two fact tables with independent id spaces — exactly like TMDB — with
|
|
193
|
+
compatibility views serving the engine unchanged. `docs/data.md` has the
|
|
194
|
+
full contract. For user data, feed `user_watch_events` from your player
|
|
195
|
+
(`docs/personalization.md`) — or point `cine_rec_engine/watched.py` at
|
|
196
|
+
whatever events table you already have, one SQL string away.
|
|
197
|
+
|
|
198
|
+
## Repo layout
|
|
199
|
+
|
|
200
|
+
```
|
|
201
|
+
cine_rec_engine/ the engine (recall · scoring · ranking · weights)
|
|
202
|
+
ingest/ built-in TMDB mirror loader (bootstrap + daily refresh)
|
|
203
|
+
demo/ one-command demo: 400 bundled titles, no API key
|
|
204
|
+
eval/ pairwise/NDCG harness + the demo judgments
|
|
205
|
+
models/ embedding sidecar — YOUR encoders plug in here
|
|
206
|
+
docs/schema.sql canonical split schema (movies | tv) + engine views
|
|
207
|
+
docs/user_data.sql user layer: events, feedback, stats, vectors
|
|
208
|
+
docs/data.md one-call ingest, filters, rate limits, acceptance rule
|
|
209
|
+
docs/personalization.md w_i formulas, user vectors, recommend_for_user
|
|
210
|
+
examples/ runnable snippets
|
|
211
|
+
tests/ offline unit tests (no DB needed)
|
|
212
|
+
```
|
|
213
|
+
|
|
214
|
+
## Requirements & support
|
|
215
|
+
|
|
216
|
+
- Python ≥ 3.10, PostgreSQL ≥ 14, asyncpg
|
|
217
|
+
- Optional extras: `pip install ".[pg]"` (pgvector — KNN recall),
|
|
218
|
+
`".[redis]"` (result + history caching), `TMDB_API_KEY` (behavioral
|
|
219
|
+
rec sync), sentence-embedding encoders (KNN + cosine features)
|
|
220
|
+
|
|
221
|
+
## Performance
|
|
222
|
+
|
|
223
|
+
Measured on the synthetic 25k-title benchmark (`benchmarks/bench_e2e.py`,
|
|
224
|
+
PostgreSQL 16, 2-core container): **cold single-seed ≈ 45 ms · warm
|
|
225
|
+
(cache) 0.4 ms · 5-seed 322 ms**. The scorer runs ~37k (seed, candidate)
|
|
226
|
+
pairs/second/core with outputs within 1 float ULP of the reference
|
|
227
|
+
implementation (golden-vector tests). `benchmarks/bench_scoring.py`
|
|
228
|
+
reproduces the hot-loop number without any database.
|
|
229
|
+
|
|
230
|
+
## Roadmap
|
|
231
|
+
|
|
232
|
+
See [ROADMAP.md](ROADMAP.md) — history-based personalization is next.
|
|
233
|
+
|
|
234
|
+
## License
|
|
235
|
+
|
|
236
|
+
[MIT](LICENSE) — engine code and the fitted weights.
|
|
237
|
+
TMDB metadata itself is © TMDb — the engine never redistributes it.
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
# Cine Rec Engine
|
|
2
|
+
|
|
3
|
+
[](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml)
|
|
4
|
+
[](LICENSE)
|
|
5
|
+
[](pyproject.toml)
|
|
6
|
+
|
|
7
|
+
> **Recommendations from your own database. No black-box API, no rented taste.**
|
|
8
|
+
|
|
9
|
+

|
|
10
|
+
|
|
11
|
+
**PostgreSQL in · ranked titles out · zero external calls on the hot path.**
|
|
12
|
+
|
|
13
|
+
## Try it in 60 seconds — no API key
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
git clone https://github.com/DovBerKaplan/cine-rec-engine && cd cine-rec-engine/demo
|
|
17
|
+
docker compose up # Postgres + 400 real titles + recommendations
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
That runs the full stack against a bundled catalog (TMDB data, en-US,
|
|
21
|
+
attribution below) and prints, for The Dark Knight and Breaking Bad:
|
|
22
|
+
|
|
23
|
+
```
|
|
24
|
+
Because you watched The Dark Knight (2008)
|
|
25
|
+
The Dark Knight Rises (2012) why: same saga · same director
|
|
26
|
+
The Batman (2022) why: same style tags · shared keywords
|
|
27
|
+
Memento (2000) why: same director · year window
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
A content-based recommendation engine for movies & series: give it one title
|
|
31
|
+
(or a user's whole watch history) and it returns similar titles, ranked by a
|
|
32
|
+
22-feature scorer whose weights were learned — not guessed.
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
your data (TMDB mirror + user events) what you get back
|
|
36
|
+
┌───────────────────────────────┐ ┌───────────────────────┐
|
|
37
|
+
│ PostgreSQL │ │ ranked similar titles │
|
|
38
|
+
│ · tmdb_media + satellites │ │ · score + why │
|
|
39
|
+
│ · (optional) embeddings │──engine──► │ · movies/series mix │
|
|
40
|
+
│ · (optional) TMDB rec cache │ │ · per-user filtering │
|
|
41
|
+
│ · your watch/rating events │ │ · saga advancement │
|
|
42
|
+
└───────────────────────────────┘ └───────────────────────┘
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## Quick start
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install cine-rec-engine # PyPI (or: pip install -e ".[pg,redis]")
|
|
49
|
+
psql -d yourdb -f docs/schema.sql # the tables it expects
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
import asyncpg
|
|
54
|
+
from cine_rec_engine import RecommendationService
|
|
55
|
+
|
|
56
|
+
pool = await asyncpg.create_pool("postgresql://user:pw@localhost/yourdb")
|
|
57
|
+
rec = RecommendationService()
|
|
58
|
+
await rec.initialize(pool)
|
|
59
|
+
|
|
60
|
+
# Similar to one title
|
|
61
|
+
results = await rec.find_similar(155) # The Dark Knight → [Batman Begins, …]
|
|
62
|
+
|
|
63
|
+
# Or the titles a user loved (movies AND series, auto-balanced)
|
|
64
|
+
results = await rec.find_similar(
|
|
65
|
+
tmdb_id=[155, 27205, 1396], # Dark Knight, Inception, Breaking Bad
|
|
66
|
+
limit=60,
|
|
67
|
+
user_id=42, # filters watched titles, advances sagas
|
|
68
|
+
)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Each result carries its evidence — score, matched features, media type —
|
|
72
|
+
so your UI can explain *why* it recommended something.
|
|
73
|
+
|
|
74
|
+
## Why another recommender
|
|
75
|
+
|
|
76
|
+
| | Hosted rec APIs | Collaborative-filtering stacks | **Cine Rec Engine** |
|
|
77
|
+
|---|---|---|---|
|
|
78
|
+
| Works on day one (no user base) | ✅ | ❌ cold start | ✅ content-based |
|
|
79
|
+
| Your data stays yours | ❌ | ✅ | ✅ |
|
|
80
|
+
| Needs users' rating matrix | ❌ | ✅ | ❌ — metadata + your event table |
|
|
81
|
+
| Explainable per-title | partial | partial | ✅ 22 named features |
|
|
82
|
+
| Runs offline / on-prem | ❌ | ✅ | ✅ one Postgres |
|
|
83
|
+
| Embedding models included | — | — | ❌ **bring your own** (see below) |
|
|
84
|
+
|
|
85
|
+
The engine separates **recall** (SQL: genres, keywords, cast, crew,
|
|
86
|
+
companies, networks, collections, cached TMDB behavioral recs, optional
|
|
87
|
+
pgvector KNN) from **ranking** (a logistic scorer over 22 features with
|
|
88
|
+
published, fitted weights) from **models** (your embedding encoders —
|
|
89
|
+
deliberately not shipped).
|
|
90
|
+
|
|
91
|
+
## Measured against baselines
|
|
92
|
+
|
|
93
|
+
On the bundled 400-title demo pool with hand-curated adjacency judgments
|
|
94
|
+
(`eval/judgments.jsonl`, `eval/eval.py`):
|
|
95
|
+
|
|
96
|
+
| method | pairwise acc. | NDCG@10 |
|
|
97
|
+
|---|---|---|
|
|
98
|
+
| TMDB similar (behavioral graph) | 0.09 | 0.13 |
|
|
99
|
+
| cosine over overviews | 0.63 | 0.04 |
|
|
100
|
+
| **this engine (learned 22-feature scorer)** | **0.83** | **0.36** |
|
|
101
|
+
|
|
102
|
+
Honest caveats: the pool is small (the TMDB graph mostly points outside
|
|
103
|
+
it, hence its floor), and the judgments are one curator's. Bring your own
|
|
104
|
+
judgments file — the harness is in the repo.
|
|
105
|
+
|
|
106
|
+
## Feature highlights
|
|
107
|
+
|
|
108
|
+
- **Multi-seed blending** — one title or a thousand; per-seed weights;
|
|
109
|
+
movie/series ratio follows the seed mix (90/10 cap so a minority is
|
|
110
|
+
never silenced).
|
|
111
|
+
- **Saga-aware** — collection members chain: watched *Rocky I* → recommends
|
|
112
|
+
*Rocky II*, not *Rocky I* again. Whole-saga watchers graduate out.
|
|
113
|
+
- **Per-user personalization** — `recommend_for_user()`: raw watch
|
|
114
|
+
events → per-title weights (completion, series depth, recency,
|
|
115
|
+
engagement) → a normalized user vector → an ANN recall channel through
|
|
116
|
+
the same LTR scorer, with watched/rated/disliked hard-filtered.
|
|
117
|
+
See `docs/personalization.md`.
|
|
118
|
+
- **Auteur recall** — director/writer/composer/DP channels with decay
|
|
119
|
+
(someone's 8th film matters less than their 2nd).
|
|
120
|
+
- **Popularity guardrails** — vote floors per media type kill
|
|
121
|
+
"high rating, 12 votes" noise; an action-popularity leak term keeps
|
|
122
|
+
Marvel out of every list.
|
|
123
|
+
- **MMR diversification** — optional re-rank so one franchise doesn't
|
|
124
|
+
take five consecutive slots.
|
|
125
|
+
- **Deterministic or noisy** — `randomness=0.0` is reproducible; dial it
|
|
126
|
+
up for exploration without burying the top-3.
|
|
127
|
+
- **Degrades gracefully** — no embeddings? no TMDB rec cache? no Redis?
|
|
128
|
+
Those channels switch off; the rest of the engine keeps working.
|
|
129
|
+
|
|
130
|
+
## The published weights
|
|
131
|
+
|
|
132
|
+
`cine_rec_engine/weights.json` — 22 features fitted with pairwise
|
|
133
|
+
logistic regression (6,650 training pairs). Opt in with
|
|
134
|
+
`CINE_REC_SCORER=learned`; features the artifact doesn't cover keep
|
|
135
|
+
their heuristic coefficients (partial application). A few, to set the
|
|
136
|
+
scale:
|
|
137
|
+
|
|
138
|
+
| Feature | Weight |
|
|
139
|
+
|---|---|
|
|
140
|
+
| `tmdb_rec_decay` (behavioral signal, rank-decayed) | 13.74 |
|
|
141
|
+
| `composer_match` | 5.54 |
|
|
142
|
+
| `cosine_sim` (overview embedding similarity) | 5.13 |
|
|
143
|
+
| `keyword_sim` | 4.35 |
|
|
144
|
+
| `writer_match` | 4.08 |
|
|
145
|
+
| `director_match` | 3.50 |
|
|
146
|
+
| `shared_collection` | 2.51 |
|
|
147
|
+
| `medium_mismatch` (movie↔tv penalty) | −0.52 |
|
|
148
|
+
|
|
149
|
+
Also in the repo: the evaluation harness and the demo judgments. Not
|
|
150
|
+
included: the larger labeled training sets the weights were fitted on and
|
|
151
|
+
the optional narrative-tag enrichment — the fitted coefficients
|
|
152
|
+
themselves are published in full.
|
|
153
|
+
|
|
154
|
+
## Feeding it data
|
|
155
|
+
|
|
156
|
+
Bring a local TMDB mirror — the built-in `ingest/` loader builds it from
|
|
157
|
+
TMDB's official daily ID exports (`python -m ingest.cli bootstrap`, then a
|
|
158
|
+
daily `refresh` off `/changes`; `en-US` only, adult-filtered, one API call
|
|
159
|
+
per title, upserts by key). The catalog schema splits movies and TV into
|
|
160
|
+
two fact tables with independent id spaces — exactly like TMDB — with
|
|
161
|
+
compatibility views serving the engine unchanged. `docs/data.md` has the
|
|
162
|
+
full contract. For user data, feed `user_watch_events` from your player
|
|
163
|
+
(`docs/personalization.md`) — or point `cine_rec_engine/watched.py` at
|
|
164
|
+
whatever events table you already have, one SQL string away.
|
|
165
|
+
|
|
166
|
+
## Repo layout
|
|
167
|
+
|
|
168
|
+
```
|
|
169
|
+
cine_rec_engine/ the engine (recall · scoring · ranking · weights)
|
|
170
|
+
ingest/ built-in TMDB mirror loader (bootstrap + daily refresh)
|
|
171
|
+
demo/ one-command demo: 400 bundled titles, no API key
|
|
172
|
+
eval/ pairwise/NDCG harness + the demo judgments
|
|
173
|
+
models/ embedding sidecar — YOUR encoders plug in here
|
|
174
|
+
docs/schema.sql canonical split schema (movies | tv) + engine views
|
|
175
|
+
docs/user_data.sql user layer: events, feedback, stats, vectors
|
|
176
|
+
docs/data.md one-call ingest, filters, rate limits, acceptance rule
|
|
177
|
+
docs/personalization.md w_i formulas, user vectors, recommend_for_user
|
|
178
|
+
examples/ runnable snippets
|
|
179
|
+
tests/ offline unit tests (no DB needed)
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
## Requirements & support
|
|
183
|
+
|
|
184
|
+
- Python ≥ 3.10, PostgreSQL ≥ 14, asyncpg
|
|
185
|
+
- Optional extras: `pip install ".[pg]"` (pgvector — KNN recall),
|
|
186
|
+
`".[redis]"` (result + history caching), `TMDB_API_KEY` (behavioral
|
|
187
|
+
rec sync), sentence-embedding encoders (KNN + cosine features)
|
|
188
|
+
|
|
189
|
+
## Performance
|
|
190
|
+
|
|
191
|
+
Measured on the synthetic 25k-title benchmark (`benchmarks/bench_e2e.py`,
|
|
192
|
+
PostgreSQL 16, 2-core container): **cold single-seed ≈ 45 ms · warm
|
|
193
|
+
(cache) 0.4 ms · 5-seed 322 ms**. The scorer runs ~37k (seed, candidate)
|
|
194
|
+
pairs/second/core with outputs within 1 float ULP of the reference
|
|
195
|
+
implementation (golden-vector tests). `benchmarks/bench_scoring.py`
|
|
196
|
+
reproduces the hot-loop number without any database.
|
|
197
|
+
|
|
198
|
+
## Roadmap
|
|
199
|
+
|
|
200
|
+
See [ROADMAP.md](ROADMAP.md) — history-based personalization is next.
|
|
201
|
+
|
|
202
|
+
## License
|
|
203
|
+
|
|
204
|
+
[MIT](LICENSE) — engine code and the fitted weights.
|
|
205
|
+
TMDB metadata itself is © TMDb — the engine never redistributes it.
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Cache layer for the engine: Redis when available, in-process TTL otherwise.
|
|
2
|
+
|
|
3
|
+
The engine touches the cache through a tiny async surface —
|
|
4
|
+
``get / set / set_nx / exists / delete / is_connected`` — so any client
|
|
5
|
+
implementing it plugs in (we ship one below). If a Redis URL is
|
|
6
|
+
configured via ``CINE_REC_REDIS_URL`` we lazily connect with ``redis``
|
|
7
|
+
(asyncio); otherwise a per-process TTL dict keeps everything working
|
|
8
|
+
with zero infrastructure.
|
|
9
|
+
|
|
10
|
+
Every cache call in the engine is wrapped in ``try/except`` and degrades
|
|
11
|
+
on failure — a cache outage must never break a recommendation.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import asyncio
|
|
17
|
+
import json
|
|
18
|
+
import os
|
|
19
|
+
import time
|
|
20
|
+
from typing import Any, Optional
|
|
21
|
+
|
|
22
|
+
_client: Any = None
|
|
23
|
+
_client_checked = False
|
|
24
|
+
_fallback: Optional["_TTLCache"] = None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class _TTLCache:
|
|
28
|
+
"""Minimal async cache with TTL + NX semantics (single-process)."""
|
|
29
|
+
|
|
30
|
+
def __init__(self) -> None:
|
|
31
|
+
self._data: dict = {}
|
|
32
|
+
self._lock = asyncio.Lock()
|
|
33
|
+
|
|
34
|
+
def is_connected(self) -> bool:
|
|
35
|
+
return True
|
|
36
|
+
|
|
37
|
+
async def get(self, key: str) -> Any:
|
|
38
|
+
async with self._lock:
|
|
39
|
+
item = self._data.get(key)
|
|
40
|
+
if item is None:
|
|
41
|
+
return None
|
|
42
|
+
value, expires = item
|
|
43
|
+
if expires is not None and expires < time.monotonic():
|
|
44
|
+
del self._data[key]
|
|
45
|
+
return None
|
|
46
|
+
# JSON round-trip so callers see the same shape as Redis gives
|
|
47
|
+
try:
|
|
48
|
+
return json.loads(value)
|
|
49
|
+
except (TypeError, ValueError):
|
|
50
|
+
return value
|
|
51
|
+
|
|
52
|
+
async def set(self, key: str, value: Any, ttl: Optional[int] = None) -> Any:
|
|
53
|
+
async with self._lock:
|
|
54
|
+
serialized = json.dumps(value, default=str)
|
|
55
|
+
self._data[key] = (serialized, time.monotonic() + ttl if ttl is not None else None)
|
|
56
|
+
return True
|
|
57
|
+
|
|
58
|
+
async def set_nx(self, key: str, value: Any, ttl: Optional[int] = None) -> bool:
|
|
59
|
+
async with self._lock:
|
|
60
|
+
item = self._data.get(key)
|
|
61
|
+
if item is not None:
|
|
62
|
+
expires = item[1]
|
|
63
|
+
if expires is None or expires >= time.monotonic():
|
|
64
|
+
return False
|
|
65
|
+
del self._data[key]
|
|
66
|
+
self._data[key] = (json.dumps(value, default=str),
|
|
67
|
+
time.monotonic() + ttl if ttl is not None else None)
|
|
68
|
+
return True
|
|
69
|
+
|
|
70
|
+
async def exists(self, key: str) -> bool:
|
|
71
|
+
return await self.get(key) is not None
|
|
72
|
+
|
|
73
|
+
async def delete(self, key: str) -> None:
|
|
74
|
+
async with self._lock:
|
|
75
|
+
self._data.pop(key, None)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
async def get_cache() -> Any:
|
|
79
|
+
"""Return a cache client, or None if nothing is available.
|
|
80
|
+
|
|
81
|
+
Priority: Redis (``CINE_REC_REDIS_URL``) → in-process TTL cache.
|
|
82
|
+
Redis is probed once per process; afterwards the decision sticks
|
|
83
|
+
unless a live connection errors out.
|
|
84
|
+
"""
|
|
85
|
+
global _client, _client_checked, _fallback
|
|
86
|
+
|
|
87
|
+
if not _client_checked:
|
|
88
|
+
_client_checked = True
|
|
89
|
+
url = os.getenv("CINE_REC_REDIS_URL")
|
|
90
|
+
if url:
|
|
91
|
+
try:
|
|
92
|
+
import redis.asyncio as aioredis # type: ignore
|
|
93
|
+
|
|
94
|
+
_client = aioredis.from_url(
|
|
95
|
+
url, decode_responses=True, socket_timeout=2,
|
|
96
|
+
socket_connect_timeout=2,
|
|
97
|
+
)
|
|
98
|
+
await _client.ping()
|
|
99
|
+
except Exception:
|
|
100
|
+
_client = None
|
|
101
|
+
|
|
102
|
+
if _client is not None:
|
|
103
|
+
try:
|
|
104
|
+
if await _client.ping():
|
|
105
|
+
return _client
|
|
106
|
+
except Exception:
|
|
107
|
+
_client = None # Redis died mid-flight — fall through to local
|
|
108
|
+
elif _fallback is not None:
|
|
109
|
+
# Redis was already ruled out this process; stay local without
|
|
110
|
+
# re-probing on every call (a recommendation path must not ping).
|
|
111
|
+
return _fallback
|
|
112
|
+
|
|
113
|
+
if _fallback is None:
|
|
114
|
+
_fallback = _TTLCache()
|
|
115
|
+
return _fallback
|