cine-rec-engine 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. cine_rec_engine-0.4.0/LICENSE +21 -0
  2. cine_rec_engine-0.4.0/PKG-INFO +237 -0
  3. cine_rec_engine-0.4.0/README.md +205 -0
  4. cine_rec_engine-0.4.0/cine_rec_engine/__init__.py +6 -0
  5. cine_rec_engine-0.4.0/cine_rec_engine/cache.py +115 -0
  6. cine_rec_engine-0.4.0/cine_rec_engine/config.py +221 -0
  7. cine_rec_engine-0.4.0/cine_rec_engine/model_spaces.py +61 -0
  8. cine_rec_engine-0.4.0/cine_rec_engine/queries.py +1057 -0
  9. cine_rec_engine-0.4.0/cine_rec_engine/scoring.py +459 -0
  10. cine_rec_engine-0.4.0/cine_rec_engine/service.py +1768 -0
  11. cine_rec_engine-0.4.0/cine_rec_engine/tmdb_client.py +62 -0
  12. cine_rec_engine-0.4.0/cine_rec_engine/tmdb_recs.py +303 -0
  13. cine_rec_engine-0.4.0/cine_rec_engine/user_stats.py +440 -0
  14. cine_rec_engine-0.4.0/cine_rec_engine/user_vector.py +175 -0
  15. cine_rec_engine-0.4.0/cine_rec_engine/user_weights.py +180 -0
  16. cine_rec_engine-0.4.0/cine_rec_engine/watched.py +230 -0
  17. cine_rec_engine-0.4.0/cine_rec_engine/weights.json +53 -0
  18. cine_rec_engine-0.4.0/cine_rec_engine.egg-info/PKG-INFO +237 -0
  19. cine_rec_engine-0.4.0/cine_rec_engine.egg-info/SOURCES.txt +36 -0
  20. cine_rec_engine-0.4.0/cine_rec_engine.egg-info/dependency_links.txt +1 -0
  21. cine_rec_engine-0.4.0/cine_rec_engine.egg-info/requires.txt +14 -0
  22. cine_rec_engine-0.4.0/cine_rec_engine.egg-info/top_level.txt +2 -0
  23. cine_rec_engine-0.4.0/ingest/__init__.py +17 -0
  24. cine_rec_engine-0.4.0/ingest/cli.py +125 -0
  25. cine_rec_engine-0.4.0/ingest/exports.py +109 -0
  26. cine_rec_engine-0.4.0/ingest/loader.py +576 -0
  27. cine_rec_engine-0.4.0/ingest/rate.py +41 -0
  28. cine_rec_engine-0.4.0/pyproject.toml +72 -0
  29. cine_rec_engine-0.4.0/setup.cfg +4 -0
  30. cine_rec_engine-0.4.0/tests/test_cache.py +39 -0
  31. cine_rec_engine-0.4.0/tests/test_engine_smoke.py +50 -0
  32. cine_rec_engine-0.4.0/tests/test_fast_path_parity.py +99 -0
  33. cine_rec_engine-0.4.0/tests/test_feature_alignment.py +42 -0
  34. cine_rec_engine-0.4.0/tests/test_ingest.py +266 -0
  35. cine_rec_engine-0.4.0/tests/test_personalization.py +188 -0
  36. cine_rec_engine-0.4.0/tests/test_rate.py +29 -0
  37. cine_rec_engine-0.4.0/tests/test_scoring.py +232 -0
  38. cine_rec_engine-0.4.0/tests/test_weights_blending.py +57 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DovBer Kaplan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,237 @@
1
+ Metadata-Version: 2.4
2
+ Name: cine-rec-engine
3
+ Version: 0.4.0
4
+ Summary: Content-based movie & series recommendation engine over your own PostgreSQL TMDB mirror
5
+ Author: DovBerKaplan
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/DovBerKaplan/cine-rec-engine
8
+ Project-URL: Issues, https://github.com/DovBerKaplan/cine-rec-engine/issues
9
+ Keywords: recommendations,tmdb,movies,postgres,pgvector,content-based
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3.11
14
+ Classifier: Topic :: Database
15
+ Classifier: Topic :: Multimedia :: Video
16
+ Classifier: Framework :: AsyncIO
17
+ Requires-Python: >=3.10
18
+ Description-Content-Type: text/markdown
19
+ License-File: LICENSE
20
+ Requires-Dist: asyncpg>=0.29
21
+ Requires-Dist: aiohttp>=3.9
22
+ Requires-Dist: loguru>=0.7
23
+ Provides-Extra: redis
24
+ Requires-Dist: redis>=5.0; extra == "redis"
25
+ Provides-Extra: pg
26
+ Requires-Dist: pgvector>=0.3; extra == "pg"
27
+ Provides-Extra: dev
28
+ Requires-Dist: pytest>=8.0; extra == "dev"
29
+ Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
30
+ Requires-Dist: ruff>=0.5; extra == "dev"
31
+ Dynamic: license-file
32
+
33
+ # Cine Rec Engine
34
+
35
+ [![CI](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml/badge.svg)](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml)
36
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
37
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg)](pyproject.toml)
38
+
39
+ > **Recommendations from your own database. No black-box API, no rented taste.**
40
+
41
+ ![demo](docs/demo.gif)
42
+
43
+ **PostgreSQL in · ranked titles out · zero external calls on the hot path.**
44
+
45
+ ## Try it in 60 seconds — no API key
46
+
47
+ ```bash
48
+ git clone https://github.com/DovBerKaplan/cine-rec-engine && cd cine-rec-engine/demo
49
+ docker compose up # Postgres + 400 real titles + recommendations
50
+ ```
51
+
52
+ That runs the full stack against a bundled catalog (TMDB data, en-US,
53
+ attribution below) and prints, for The Dark Knight and Breaking Bad:
54
+
55
+ ```
56
+ Because you watched The Dark Knight (2008)
57
+ The Dark Knight Rises (2012) why: same saga · same director
58
+ The Batman (2022) why: same style tags · shared keywords
59
+ Memento (2000) why: same director · year window
60
+ ```
61
+
62
+ A content-based recommendation engine for movies & series: give it one title
63
+ (or a user's whole watch history) and it returns similar titles, ranked by a
64
+ 22-feature scorer whose weights were learned — not guessed.
65
+
66
+ ```
67
+ your data (TMDB mirror + user events) what you get back
68
+ ┌───────────────────────────────┐ ┌───────────────────────┐
69
+ │ PostgreSQL │ │ ranked similar titles │
70
+ │ · tmdb_media + satellites │ │ · score + why │
71
+ │ · (optional) embeddings │──engine──► │ · movies/series mix │
72
+ │ · (optional) TMDB rec cache │ │ · per-user filtering │
73
+ │ · your watch/rating events │ │ · saga advancement │
74
+ └───────────────────────────────┘ └───────────────────────┘
75
+ ```
76
+
77
+ ## Quick start
78
+
79
+ ```bash
80
+ pip install cine-rec-engine # PyPI (or: pip install -e ".[pg,redis]")
81
+ psql -d yourdb -f docs/schema.sql # the tables it expects
82
+ ```
83
+
84
+ ```python
85
+ import asyncpg
86
+ from cine_rec_engine import RecommendationService
87
+
88
+ pool = await asyncpg.create_pool("postgresql://user:pw@localhost/yourdb")
89
+ rec = RecommendationService()
90
+ await rec.initialize(pool)
91
+
92
+ # Similar to one title
93
+ results = await rec.find_similar(155) # The Dark Knight → [Batman Begins, …]
94
+
95
+ # Or the titles a user loved (movies AND series, auto-balanced)
96
+ results = await rec.find_similar(
97
+ tmdb_id=[155, 27205, 1396], # Dark Knight, Inception, Breaking Bad
98
+ limit=60,
99
+ user_id=42, # filters watched titles, advances sagas
100
+ )
101
+ ```
102
+
103
+ Each result carries its evidence — score, matched features, media type —
104
+ so your UI can explain *why* it recommended something.
105
+
106
+ ## Why another recommender
107
+
108
+ | | Hosted rec APIs | Collaborative-filtering stacks | **Cine Rec Engine** |
109
+ |---|---|---|---|
110
+ | Works on day one (no user base) | ✅ | ❌ cold start | ✅ content-based |
111
+ | Your data stays yours | ❌ | ✅ | ✅ |
112
+ | Needs users' rating matrix | ❌ | ✅ | ❌ — metadata + your event table |
113
+ | Explainable per-title | partial | partial | ✅ 22 named features |
114
+ | Runs offline / on-prem | ❌ | ✅ | ✅ one Postgres |
115
+ | Embedding models included | — | — | ❌ **bring your own** (see below) |
116
+
117
+ The engine separates **recall** (SQL: genres, keywords, cast, crew,
118
+ companies, networks, collections, cached TMDB behavioral recs, optional
119
+ pgvector KNN) from **ranking** (a logistic scorer over 22 features with
120
+ published, fitted weights) from **models** (your embedding encoders —
121
+ deliberately not shipped).
122
+
123
+ ## Measured against baselines
124
+
125
+ On the bundled 400-title demo pool with hand-curated adjacency judgments
126
+ (`eval/judgments.jsonl`, `eval/eval.py`):
127
+
128
+ | method | pairwise acc. | NDCG@10 |
129
+ |---|---|---|
130
+ | TMDB similar (behavioral graph) | 0.09 | 0.13 |
131
+ | cosine over overviews | 0.63 | 0.04 |
132
+ | **this engine (learned 22-feature scorer)** | **0.83** | **0.36** |
133
+
134
+ Honest caveats: the pool is small (the TMDB graph mostly points outside
135
+ it, hence its floor), and the judgments are one curator's. Bring your own
136
+ judgments file — the harness is in the repo.
137
+
138
+ ## Feature highlights
139
+
140
+ - **Multi-seed blending** — one title or a thousand; per-seed weights;
141
+ movie/series ratio follows the seed mix (90/10 cap so a minority is
142
+ never silenced).
143
+ - **Saga-aware** — collection members chain: watched *Rocky I* → recommends
144
+ *Rocky II*, not *Rocky I* again. Whole-saga watchers graduate out.
145
+ - **Per-user personalization** — `recommend_for_user()`: raw watch
146
+ events → per-title weights (completion, series depth, recency,
147
+ engagement) → a normalized user vector → an ANN recall channel through
148
+ the same LTR scorer, with watched/rated/disliked hard-filtered.
149
+ See `docs/personalization.md`.
150
+ - **Auteur recall** — director/writer/composer/DP channels with decay
151
+ (someone's 8th film matters less than their 2nd).
152
+ - **Popularity guardrails** — vote floors per media type kill
153
+ "high rating, 12 votes" noise; an action-popularity leak term keeps
154
+ Marvel out of every list.
155
+ - **MMR diversification** — optional re-rank so one franchise doesn't
156
+ take five consecutive slots.
157
+ - **Deterministic or noisy** — `randomness=0.0` is reproducible; dial it
158
+ up for exploration without burying the top-3.
159
+ - **Degrades gracefully** — no embeddings? no TMDB rec cache? no Redis?
160
+ Those channels switch off; the rest of the engine keeps working.
161
+
162
+ ## The published weights
163
+
164
+ `cine_rec_engine/weights.json` — 22 features fitted with pairwise
165
+ logistic regression (6,650 training pairs). Opt in with
166
+ `CINE_REC_SCORER=learned`; features the artifact doesn't cover keep
167
+ their heuristic coefficients (partial application). A few, to set the
168
+ scale:
169
+
170
+ | Feature | Weight |
171
+ |---|---|
172
+ | `tmdb_rec_decay` (behavioral signal, rank-decayed) | 13.74 |
173
+ | `composer_match` | 5.54 |
174
+ | `cosine_sim` (overview embedding similarity) | 5.13 |
175
+ | `keyword_sim` | 4.35 |
176
+ | `writer_match` | 4.08 |
177
+ | `director_match` | 3.50 |
178
+ | `shared_collection` | 2.51 |
179
+ | `medium_mismatch` (movie↔tv penalty) | −0.52 |
180
+
181
+ Also in the repo: the evaluation harness and the demo judgments. Not
182
+ included: the larger labeled training sets the weights were fitted on and
183
+ the optional narrative-tag enrichment — the fitted coefficients
184
+ themselves are published in full.
185
+
186
+ ## Feeding it data
187
+
188
+ Bring a local TMDB mirror — the built-in `ingest/` loader builds it from
189
+ TMDB's official daily ID exports (`python -m ingest.cli bootstrap`, then a
190
+ daily `refresh` off `/changes`; `en-US` only, adult-filtered, one API call
191
+ per title, upserts by key). The catalog schema splits movies and TV into
192
+ two fact tables with independent id spaces — exactly like TMDB — with
193
+ compatibility views serving the engine unchanged. `docs/data.md` has the
194
+ full contract. For user data, feed `user_watch_events` from your player
195
+ (`docs/personalization.md`) — or point `cine_rec_engine/watched.py` at
196
+ whatever events table you already have, one SQL string away.
197
+
198
+ ## Repo layout
199
+
200
+ ```
201
+ cine_rec_engine/ the engine (recall · scoring · ranking · weights)
202
+ ingest/ built-in TMDB mirror loader (bootstrap + daily refresh)
203
+ demo/ one-command demo: 400 bundled titles, no API key
204
+ eval/ pairwise/NDCG harness + the demo judgments
205
+ models/ embedding sidecar — YOUR encoders plug in here
206
+ docs/schema.sql canonical split schema (movies | tv) + engine views
207
+ docs/user_data.sql user layer: events, feedback, stats, vectors
208
+ docs/data.md one-call ingest, filters, rate limits, acceptance rule
209
+ docs/personalization.md w_i formulas, user vectors, recommend_for_user
210
+ examples/ runnable snippets
211
+ tests/ offline unit tests (no DB needed)
212
+ ```
213
+
214
+ ## Requirements & support
215
+
216
+ - Python ≥ 3.10, PostgreSQL ≥ 14, asyncpg
217
+ - Optional extras: `pip install ".[pg]"` (pgvector — KNN recall),
218
+ `".[redis]"` (result + history caching), `TMDB_API_KEY` (behavioral
219
+ rec sync), sentence-embedding encoders (KNN + cosine features)
220
+
221
+ ## Performance
222
+
223
+ Measured on the synthetic 25k-title benchmark (`benchmarks/bench_e2e.py`,
224
+ PostgreSQL 16, 2-core container): **cold single-seed ≈ 45 ms · warm
225
+ (cache) 0.4 ms · 5-seed 322 ms**. The scorer runs ~37k (seed, candidate)
226
+ pairs/second/core with outputs within 1 float ULP of the reference
227
+ implementation (golden-vector tests). `benchmarks/bench_scoring.py`
228
+ reproduces the hot-loop number without any database.
229
+
230
+ ## Roadmap
231
+
232
+ See [ROADMAP.md](ROADMAP.md) — history-based personalization is next.
233
+
234
+ ## License
235
+
236
+ [MIT](LICENSE) — engine code and the fitted weights.
237
+ TMDB metadata itself is © TMDb — the engine never redistributes it.
@@ -0,0 +1,205 @@
1
+ # Cine Rec Engine
2
+
3
+ [![CI](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml/badge.svg)](https://github.com/DovBerKaplan/cine-rec-engine/actions/workflows/ci.yml)
4
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
5
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg)](pyproject.toml)
6
+
7
+ > **Recommendations from your own database. No black-box API, no rented taste.**
8
+
9
+ ![demo](docs/demo.gif)
10
+
11
+ **PostgreSQL in · ranked titles out · zero external calls on the hot path.**
12
+
13
+ ## Try it in 60 seconds — no API key
14
+
15
+ ```bash
16
+ git clone https://github.com/DovBerKaplan/cine-rec-engine && cd cine-rec-engine/demo
17
+ docker compose up # Postgres + 400 real titles + recommendations
18
+ ```
19
+
20
+ That runs the full stack against a bundled catalog (TMDB data, en-US,
21
+ attribution below) and prints, for The Dark Knight and Breaking Bad:
22
+
23
+ ```
24
+ Because you watched The Dark Knight (2008)
25
+ The Dark Knight Rises (2012) why: same saga · same director
26
+ The Batman (2022) why: same style tags · shared keywords
27
+ Memento (2000) why: same director · year window
28
+ ```
29
+
30
+ A content-based recommendation engine for movies & series: give it one title
31
+ (or a user's whole watch history) and it returns similar titles, ranked by a
32
+ 22-feature scorer whose weights were learned — not guessed.
33
+
34
+ ```
35
+ your data (TMDB mirror + user events) what you get back
36
+ ┌───────────────────────────────┐ ┌───────────────────────┐
37
+ │ PostgreSQL │ │ ranked similar titles │
38
+ │ · tmdb_media + satellites │ │ · score + why │
39
+ │ · (optional) embeddings │──engine──► │ · movies/series mix │
40
+ │ · (optional) TMDB rec cache │ │ · per-user filtering │
41
+ │ · your watch/rating events │ │ · saga advancement │
42
+ └───────────────────────────────┘ └───────────────────────┘
43
+ ```
44
+
45
+ ## Quick start
46
+
47
+ ```bash
48
+ pip install cine-rec-engine # PyPI (or: pip install -e ".[pg,redis]")
49
+ psql -d yourdb -f docs/schema.sql # the tables it expects
50
+ ```
51
+
52
+ ```python
53
+ import asyncpg
54
+ from cine_rec_engine import RecommendationService
55
+
56
+ pool = await asyncpg.create_pool("postgresql://user:pw@localhost/yourdb")
57
+ rec = RecommendationService()
58
+ await rec.initialize(pool)
59
+
60
+ # Similar to one title
61
+ results = await rec.find_similar(155) # The Dark Knight → [Batman Begins, …]
62
+
63
+ # Or the titles a user loved (movies AND series, auto-balanced)
64
+ results = await rec.find_similar(
65
+ tmdb_id=[155, 27205, 1396], # Dark Knight, Inception, Breaking Bad
66
+ limit=60,
67
+ user_id=42, # filters watched titles, advances sagas
68
+ )
69
+ ```
70
+
71
+ Each result carries its evidence — score, matched features, media type —
72
+ so your UI can explain *why* it recommended something.
73
+
74
+ ## Why another recommender
75
+
76
+ | | Hosted rec APIs | Collaborative-filtering stacks | **Cine Rec Engine** |
77
+ |---|---|---|---|
78
+ | Works on day one (no user base) | ✅ | ❌ cold start | ✅ content-based |
79
+ | Your data stays yours | ❌ | ✅ | ✅ |
80
+ | Needs users' rating matrix | ❌ | ✅ | ❌ — metadata + your event table |
81
+ | Explainable per-title | partial | partial | ✅ 22 named features |
82
+ | Runs offline / on-prem | ❌ | ✅ | ✅ one Postgres |
83
+ | Embedding models included | — | — | ❌ **bring your own** (see below) |
84
+
85
+ The engine separates **recall** (SQL: genres, keywords, cast, crew,
86
+ companies, networks, collections, cached TMDB behavioral recs, optional
87
+ pgvector KNN) from **ranking** (a logistic scorer over 22 features with
88
+ published, fitted weights) from **models** (your embedding encoders —
89
+ deliberately not shipped).
90
+
91
+ ## Measured against baselines
92
+
93
+ On the bundled 400-title demo pool with hand-curated adjacency judgments
94
+ (`eval/judgments.jsonl`, `eval/eval.py`):
95
+
96
+ | method | pairwise acc. | NDCG@10 |
97
+ |---|---|---|
98
+ | TMDB similar (behavioral graph) | 0.09 | 0.13 |
99
+ | cosine over overviews | 0.63 | 0.04 |
100
+ | **this engine (learned 22-feature scorer)** | **0.83** | **0.36** |
101
+
102
+ Honest caveats: the pool is small (the TMDB graph mostly points outside
103
+ it, hence its floor), and the judgments are one curator's. Bring your own
104
+ judgments file — the harness is in the repo.
105
+
106
+ ## Feature highlights
107
+
108
+ - **Multi-seed blending** — one title or a thousand; per-seed weights;
109
+ movie/series ratio follows the seed mix (90/10 cap so a minority is
110
+ never silenced).
111
+ - **Saga-aware** — collection members chain: watched *Rocky I* → recommends
112
+ *Rocky II*, not *Rocky I* again. Whole-saga watchers graduate out.
113
+ - **Per-user personalization** — `recommend_for_user()`: raw watch
114
+ events → per-title weights (completion, series depth, recency,
115
+ engagement) → a normalized user vector → an ANN recall channel through
116
+ the same LTR scorer, with watched/rated/disliked hard-filtered.
117
+ See `docs/personalization.md`.
118
+ - **Auteur recall** — director/writer/composer/DP channels with decay
119
+ (someone's 8th film matters less than their 2nd).
120
+ - **Popularity guardrails** — vote floors per media type kill
121
+ "high rating, 12 votes" noise; an action-popularity leak term keeps
122
+ Marvel out of every list.
123
+ - **MMR diversification** — optional re-rank so one franchise doesn't
124
+ take five consecutive slots.
125
+ - **Deterministic or noisy** — `randomness=0.0` is reproducible; dial it
126
+ up for exploration without burying the top-3.
127
+ - **Degrades gracefully** — no embeddings? no TMDB rec cache? no Redis?
128
+ Those channels switch off; the rest of the engine keeps working.
129
+
130
+ ## The published weights
131
+
132
+ `cine_rec_engine/weights.json` — 22 features fitted with pairwise
133
+ logistic regression (6,650 training pairs). Opt in with
134
+ `CINE_REC_SCORER=learned`; features the artifact doesn't cover keep
135
+ their heuristic coefficients (partial application). A few, to set the
136
+ scale:
137
+
138
+ | Feature | Weight |
139
+ |---|---|
140
+ | `tmdb_rec_decay` (behavioral signal, rank-decayed) | 13.74 |
141
+ | `composer_match` | 5.54 |
142
+ | `cosine_sim` (overview embedding similarity) | 5.13 |
143
+ | `keyword_sim` | 4.35 |
144
+ | `writer_match` | 4.08 |
145
+ | `director_match` | 3.50 |
146
+ | `shared_collection` | 2.51 |
147
+ | `medium_mismatch` (movie↔tv penalty) | −0.52 |
148
+
149
+ Also in the repo: the evaluation harness and the demo judgments. Not
150
+ included: the larger labeled training sets the weights were fitted on and
151
+ the optional narrative-tag enrichment — the fitted coefficients
152
+ themselves are published in full.
153
+
154
+ ## Feeding it data
155
+
156
+ Bring a local TMDB mirror — the built-in `ingest/` loader builds it from
157
+ TMDB's official daily ID exports (`python -m ingest.cli bootstrap`, then a
158
+ daily `refresh` off `/changes`; `en-US` only, adult-filtered, one API call
159
+ per title, upserts by key). The catalog schema splits movies and TV into
160
+ two fact tables with independent id spaces — exactly like TMDB — with
161
+ compatibility views serving the engine unchanged. `docs/data.md` has the
162
+ full contract. For user data, feed `user_watch_events` from your player
163
+ (`docs/personalization.md`) — or point `cine_rec_engine/watched.py` at
164
+ whatever events table you already have, one SQL string away.
165
+
166
+ ## Repo layout
167
+
168
+ ```
169
+ cine_rec_engine/ the engine (recall · scoring · ranking · weights)
170
+ ingest/ built-in TMDB mirror loader (bootstrap + daily refresh)
171
+ demo/ one-command demo: 400 bundled titles, no API key
172
+ eval/ pairwise/NDCG harness + the demo judgments
173
+ models/ embedding sidecar — YOUR encoders plug in here
174
+ docs/schema.sql canonical split schema (movies | tv) + engine views
175
+ docs/user_data.sql user layer: events, feedback, stats, vectors
176
+ docs/data.md one-call ingest, filters, rate limits, acceptance rule
177
+ docs/personalization.md w_i formulas, user vectors, recommend_for_user
178
+ examples/ runnable snippets
179
+ tests/ offline unit tests (no DB needed)
180
+ ```
181
+
182
+ ## Requirements & support
183
+
184
+ - Python ≥ 3.10, PostgreSQL ≥ 14, asyncpg
185
+ - Optional extras: `pip install ".[pg]"` (pgvector — KNN recall),
186
+ `".[redis]"` (result + history caching), `TMDB_API_KEY` (behavioral
187
+ rec sync), sentence-embedding encoders (KNN + cosine features)
188
+
189
+ ## Performance
190
+
191
+ Measured on the synthetic 25k-title benchmark (`benchmarks/bench_e2e.py`,
192
+ PostgreSQL 16, 2-core container): **cold single-seed ≈ 45 ms · warm
193
+ (cache) 0.4 ms · 5-seed 322 ms**. The scorer runs ~37k (seed, candidate)
194
+ pairs/second/core with outputs within 1 float ULP of the reference
195
+ implementation (golden-vector tests). `benchmarks/bench_scoring.py`
196
+ reproduces the hot-loop number without any database.
197
+
198
+ ## Roadmap
199
+
200
+ See [ROADMAP.md](ROADMAP.md) — history-based personalization is next.
201
+
202
+ ## License
203
+
204
+ [MIT](LICENSE) — engine code and the fitted weights.
205
+ TMDB metadata itself is © TMDb — the engine never redistributes it.
@@ -0,0 +1,6 @@
1
+ """cine-rec-engine — content-based movie & series recommendations from your own PostgreSQL."""
2
+
3
+ from .service import RecommendationService
4
+
5
+ __version__ = "0.4.0"
6
+ __all__ = ["RecommendationService", "__version__"]
@@ -0,0 +1,115 @@
1
+ """Cache layer for the engine: Redis when available, in-process TTL otherwise.
2
+
3
+ The engine touches the cache through a tiny async surface —
4
+ ``get / set / set_nx / exists / delete / is_connected`` — so any client
5
+ implementing it plugs in (we ship one below). If a Redis URL is
6
+ configured via ``CINE_REC_REDIS_URL`` we lazily connect with ``redis``
7
+ (asyncio); otherwise a per-process TTL dict keeps everything working
8
+ with zero infrastructure.
9
+
10
+ Every cache call in the engine is wrapped in ``try/except`` and degrades
11
+ on failure — a cache outage must never break a recommendation.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import asyncio
17
+ import json
18
+ import os
19
+ import time
20
+ from typing import Any, Optional
21
+
22
+ _client: Any = None
23
+ _client_checked = False
24
+ _fallback: Optional["_TTLCache"] = None
25
+
26
+
27
+ class _TTLCache:
28
+ """Minimal async cache with TTL + NX semantics (single-process)."""
29
+
30
+ def __init__(self) -> None:
31
+ self._data: dict = {}
32
+ self._lock = asyncio.Lock()
33
+
34
+ def is_connected(self) -> bool:
35
+ return True
36
+
37
+ async def get(self, key: str) -> Any:
38
+ async with self._lock:
39
+ item = self._data.get(key)
40
+ if item is None:
41
+ return None
42
+ value, expires = item
43
+ if expires is not None and expires < time.monotonic():
44
+ del self._data[key]
45
+ return None
46
+ # JSON round-trip so callers see the same shape as Redis gives
47
+ try:
48
+ return json.loads(value)
49
+ except (TypeError, ValueError):
50
+ return value
51
+
52
+ async def set(self, key: str, value: Any, ttl: Optional[int] = None) -> Any:
53
+ async with self._lock:
54
+ serialized = json.dumps(value, default=str)
55
+ self._data[key] = (serialized, time.monotonic() + ttl if ttl is not None else None)
56
+ return True
57
+
58
+ async def set_nx(self, key: str, value: Any, ttl: Optional[int] = None) -> bool:
59
+ async with self._lock:
60
+ item = self._data.get(key)
61
+ if item is not None:
62
+ expires = item[1]
63
+ if expires is None or expires >= time.monotonic():
64
+ return False
65
+ del self._data[key]
66
+ self._data[key] = (json.dumps(value, default=str),
67
+ time.monotonic() + ttl if ttl is not None else None)
68
+ return True
69
+
70
+ async def exists(self, key: str) -> bool:
71
+ return await self.get(key) is not None
72
+
73
+ async def delete(self, key: str) -> None:
74
+ async with self._lock:
75
+ self._data.pop(key, None)
76
+
77
+
78
+ async def get_cache() -> Any:
79
+ """Return a cache client, or None if nothing is available.
80
+
81
+ Priority: Redis (``CINE_REC_REDIS_URL``) → in-process TTL cache.
82
+ Redis is probed once per process; afterwards the decision sticks
83
+ unless a live connection errors out.
84
+ """
85
+ global _client, _client_checked, _fallback
86
+
87
+ if not _client_checked:
88
+ _client_checked = True
89
+ url = os.getenv("CINE_REC_REDIS_URL")
90
+ if url:
91
+ try:
92
+ import redis.asyncio as aioredis # type: ignore
93
+
94
+ _client = aioredis.from_url(
95
+ url, decode_responses=True, socket_timeout=2,
96
+ socket_connect_timeout=2,
97
+ )
98
+ await _client.ping()
99
+ except Exception:
100
+ _client = None
101
+
102
+ if _client is not None:
103
+ try:
104
+ if await _client.ping():
105
+ return _client
106
+ except Exception:
107
+ _client = None # Redis died mid-flight — fall through to local
108
+ elif _fallback is not None:
109
+ # Redis was already ruled out this process; stay local without
110
+ # re-probing on every call (a recommendation path must not ping).
111
+ return _fallback
112
+
113
+ if _fallback is None:
114
+ _fallback = _TTLCache()
115
+ return _fallback