TriCacheLLM-MMA 0.1.4__tar.gz → 0.1.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/PKG-INFO +27 -25
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/README.md +27 -25
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_bgWorkers/portable_cache_celery_conf.py +1 -1
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_bgWorkers/portable_cache_workers.py +0 -2
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_main.py +1 -1
- tricachellm_mma-0.1.6/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py +31 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/PKG-INFO +27 -25
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/pyproject.toml +1 -1
- tricachellm_mma-0.1.4/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py +0 -7
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/LICENSE +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/__init__.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_Ai/portable_cache_rerankAi.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_dbSchema.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_redis.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_dbBase.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_dbConf.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_schemas.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA/portable_cache_utils/protable_cache_DynamicEnv_maker.py +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/SOURCES.txt +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/dependency_links.txt +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/requires.txt +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/top_level.txt +0 -0
- {tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: TriCacheLLM_MMA
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.6
|
|
4
4
|
Summary: A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap.
|
|
5
5
|
Author-email: Mohib Ashfaq Butt <inboxmohib@gmail.com>
|
|
6
6
|
License-Expression: LGPL-3.0-only
|
|
@@ -76,7 +76,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
76
76
|
|
|
77
77
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
78
78
|
|
|
79
|
-
|
|
80
79
|
---
|
|
81
80
|
|
|
82
81
|
# Architecture
|
|
@@ -172,7 +171,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
172
171
|
|
|
173
172
|
Tier 3 is the persistent cache layer.
|
|
174
173
|
|
|
175
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
174
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
176
175
|
|
|
177
176
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
178
177
|
|
|
@@ -293,8 +292,6 @@ Persistent cache seeded
|
|
|
293
292
|
|
|
294
293
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
295
294
|
|
|
296
|
-
|
|
297
|
-
|
|
298
295
|
`populate_cache()` seeds the persistent cache VDB.
|
|
299
296
|
|
|
300
297
|
It does **not** directly populate every cache tier.
|
|
@@ -427,7 +424,7 @@ Or install the development version directly from the repository:
|
|
|
427
424
|
pip install .
|
|
428
425
|
```
|
|
429
426
|
|
|
430
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
427
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
431
428
|
|
|
432
429
|
---
|
|
433
430
|
|
|
@@ -459,7 +456,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
459
456
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
460
457
|
|
|
461
458
|
```bash
|
|
462
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
459
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
463
460
|
```
|
|
464
461
|
|
|
465
462
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -511,18 +508,31 @@ from TriCacheLLM_MMA import (
|
|
|
511
508
|
check_tenant_creation_status,
|
|
512
509
|
)
|
|
513
510
|
|
|
514
|
-
|
|
511
|
+
|
|
512
|
+
@asynccontextmanager
|
|
513
|
+
async def lifespan(app: FastAPI):
|
|
514
|
+
yield
|
|
515
|
+
await close_cache_system()
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
app = FastAPI(
|
|
519
|
+
title="TriCacheLLM_MMA V1 Route Test",
|
|
520
|
+
lifespan=lifespan,
|
|
521
|
+
)
|
|
522
|
+
|
|
515
523
|
|
|
516
524
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
517
|
-
@app.
|
|
518
|
-
async def
|
|
519
|
-
|
|
525
|
+
@app.post("/api/consumer/init")
|
|
526
|
+
async def initialize_consumer():
|
|
527
|
+
user_id: int = 0
|
|
528
|
+
result = await create_cache_system(
|
|
520
529
|
redis_url="redis://localhost:6379/0",
|
|
521
|
-
cohere_api_key="
|
|
530
|
+
cohere_api_key="KEY_HERE",
|
|
522
531
|
chroma_db_dir="./chroma_db",
|
|
523
532
|
db_path="./cache.db",
|
|
524
|
-
user_id=
|
|
533
|
+
user_id=user_id,
|
|
525
534
|
)
|
|
535
|
+
return {"status": "success", "details": result}
|
|
526
536
|
|
|
527
537
|
|
|
528
538
|
# 2. TENANT ENDPOINT (Zero boilerplate for subsequent users)
|
|
@@ -552,18 +562,18 @@ async def ask_question(
|
|
|
552
562
|
)
|
|
553
563
|
|
|
554
564
|
return {"source": "ai", "response": llm_response}
|
|
565
|
+
|
|
555
566
|
```
|
|
556
567
|
The core cache operations are asynchronous Python APIs and are not inherently tied to FastAPI.
|
|
557
568
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
558
569
|
|
|
559
570
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
560
|
-
***NOTE: When you import files they'll send request to hugging face to get embedding model weights but only once! untill you reload server***
|
|
561
|
-
|
|
562
|
-
---
|
|
563
|
-
|
|
564
571
|
|
|
565
572
|
|
|
573
|
+
> **Note on Model Initialization:**
|
|
574
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
566
575
|
|
|
576
|
+
---
|
|
567
577
|
|
|
568
578
|
# Cache Data Model
|
|
569
579
|
|
|
@@ -599,9 +609,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
599
609
|
|
|
600
610
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
601
611
|
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
612
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
606
613
|
|
|
607
614
|
---
|
|
@@ -612,8 +619,6 @@ Additional metadata can be added to the cache payload.
|
|
|
612
619
|
|
|
613
620
|
The cache population path constructs metadata similar to:
|
|
614
621
|
|
|
615
|
-
|
|
616
|
-
|
|
617
622
|
For example:
|
|
618
623
|
|
|
619
624
|
```python
|
|
@@ -866,8 +871,6 @@ These can be considered for future versions.
|
|
|
866
871
|
|
|
867
872
|
---
|
|
868
873
|
|
|
869
|
-
|
|
870
|
-
|
|
871
874
|
# Consumer Responsibility
|
|
872
875
|
|
|
873
876
|
The consuming application is responsible for:
|
|
@@ -924,7 +927,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
924
927
|
|
|
925
928
|
---
|
|
926
929
|
|
|
927
|
-
|
|
928
930
|
# License
|
|
929
931
|
|
|
930
932
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -34,7 +34,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
34
34
|
|
|
35
35
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
36
36
|
|
|
37
|
-
|
|
38
37
|
---
|
|
39
38
|
|
|
40
39
|
# Architecture
|
|
@@ -130,7 +129,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
130
129
|
|
|
131
130
|
Tier 3 is the persistent cache layer.
|
|
132
131
|
|
|
133
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
132
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
134
133
|
|
|
135
134
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
136
135
|
|
|
@@ -251,8 +250,6 @@ Persistent cache seeded
|
|
|
251
250
|
|
|
252
251
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
253
252
|
|
|
254
|
-
|
|
255
|
-
|
|
256
253
|
`populate_cache()` seeds the persistent cache VDB.
|
|
257
254
|
|
|
258
255
|
It does **not** directly populate every cache tier.
|
|
@@ -385,7 +382,7 @@ Or install the development version directly from the repository:
|
|
|
385
382
|
pip install .
|
|
386
383
|
```
|
|
387
384
|
|
|
388
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
385
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
389
386
|
|
|
390
387
|
---
|
|
391
388
|
|
|
@@ -417,7 +414,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
417
414
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
418
415
|
|
|
419
416
|
```bash
|
|
420
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
417
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
421
418
|
```
|
|
422
419
|
|
|
423
420
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -469,18 +466,31 @@ from TriCacheLLM_MMA import (
|
|
|
469
466
|
check_tenant_creation_status,
|
|
470
467
|
)
|
|
471
468
|
|
|
472
|
-
|
|
469
|
+
|
|
470
|
+
@asynccontextmanager
|
|
471
|
+
async def lifespan(app: FastAPI):
|
|
472
|
+
yield
|
|
473
|
+
await close_cache_system()
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
app = FastAPI(
|
|
477
|
+
title="TriCacheLLM_MMA V1 Route Test",
|
|
478
|
+
lifespan=lifespan,
|
|
479
|
+
)
|
|
480
|
+
|
|
473
481
|
|
|
474
482
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
475
|
-
@app.
|
|
476
|
-
async def
|
|
477
|
-
|
|
483
|
+
@app.post("/api/consumer/init")
|
|
484
|
+
async def initialize_consumer():
|
|
485
|
+
user_id: int = 0
|
|
486
|
+
result = await create_cache_system(
|
|
478
487
|
redis_url="redis://localhost:6379/0",
|
|
479
|
-
cohere_api_key="
|
|
488
|
+
cohere_api_key="KEY_HERE",
|
|
480
489
|
chroma_db_dir="./chroma_db",
|
|
481
490
|
db_path="./cache.db",
|
|
482
|
-
user_id=
|
|
491
|
+
user_id=user_id,
|
|
483
492
|
)
|
|
493
|
+
return {"status": "success", "details": result}
|
|
484
494
|
|
|
485
495
|
|
|
486
496
|
# 2. TENANT ENDPOINT (Zero boilerplate for subsequent users)
|
|
@@ -510,18 +520,18 @@ async def ask_question(
|
|
|
510
520
|
)
|
|
511
521
|
|
|
512
522
|
return {"source": "ai", "response": llm_response}
|
|
523
|
+
|
|
513
524
|
```
|
|
514
525
|
The core cache operations are asynchronous Python APIs and are not inherently tied to FastAPI.
|
|
515
526
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
516
527
|
|
|
517
528
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
518
|
-
***NOTE: When you import files they'll send request to hugging face to get embedding model weights but only once! untill you reload server***
|
|
519
|
-
|
|
520
|
-
---
|
|
521
|
-
|
|
522
529
|
|
|
523
530
|
|
|
531
|
+
> **Note on Model Initialization:**
|
|
532
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
524
533
|
|
|
534
|
+
---
|
|
525
535
|
|
|
526
536
|
# Cache Data Model
|
|
527
537
|
|
|
@@ -557,9 +567,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
557
567
|
|
|
558
568
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
559
569
|
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
570
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
564
571
|
|
|
565
572
|
---
|
|
@@ -570,8 +577,6 @@ Additional metadata can be added to the cache payload.
|
|
|
570
577
|
|
|
571
578
|
The cache population path constructs metadata similar to:
|
|
572
579
|
|
|
573
|
-
|
|
574
|
-
|
|
575
580
|
For example:
|
|
576
581
|
|
|
577
582
|
```python
|
|
@@ -824,8 +829,6 @@ These can be considered for future versions.
|
|
|
824
829
|
|
|
825
830
|
---
|
|
826
831
|
|
|
827
|
-
|
|
828
|
-
|
|
829
832
|
# Consumer Responsibility
|
|
830
833
|
|
|
831
834
|
The consuming application is responsible for:
|
|
@@ -882,7 +885,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
882
885
|
|
|
883
886
|
---
|
|
884
887
|
|
|
885
|
-
|
|
886
888
|
# License
|
|
887
889
|
|
|
888
890
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -915,4 +917,4 @@ with the goal of making expensive LLM inference the **last resort rather than th
|
|
|
915
917
|
---
|
|
916
918
|
|
|
917
919
|
# Disclaimer:
|
|
918
|
-
This software is provided "as is" without warranty of any kind. The author is not responsible for any data loss, system failures, or damages arising from its use.
|
|
920
|
+
This software is provided "as is" without warranty of any kind. The author is not responsible for any data loss, system failures, or damages arising from its use.
|
|
@@ -46,7 +46,6 @@ async def create_cache_vdb_async(task_instance: Any, user_id: int):
|
|
|
46
46
|
parents=True,
|
|
47
47
|
exist_ok=True,
|
|
48
48
|
)
|
|
49
|
-
|
|
50
49
|
await asyncio.to_thread(
|
|
51
50
|
lambda: Chroma(
|
|
52
51
|
collection_name=f"question_cache_{user_id}",
|
|
@@ -142,7 +141,6 @@ async def push_response_in_cache_async(task_instance: Any, user_id, question: st
|
|
|
142
141
|
)
|
|
143
142
|
|
|
144
143
|
user_cache_vdb_path = Path(cache_res.vdb_path)
|
|
145
|
-
|
|
146
144
|
user_cache_vdb = await asyncio.to_thread(
|
|
147
145
|
lambda: Chroma(
|
|
148
146
|
collection_name=f"question_cache_{user_id}",
|
|
@@ -524,7 +524,7 @@ async def populate_cache(to_cache_answer, to_cache_question, user_id: Any = 0) -
|
|
|
524
524
|
async def close_cache_system():
|
|
525
525
|
print("SHUTTING_DOWN_PORTABLE_CACHE: Cleaning up resources...")
|
|
526
526
|
|
|
527
|
-
from portable_cache_schemas.portable_cache_dbConf import db_manager
|
|
527
|
+
from .portable_cache_schemas.portable_cache_dbConf import db_manager
|
|
528
528
|
|
|
529
529
|
if db_manager.norma_engine is not None:
|
|
530
530
|
try:
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# File: portable_cache_utils/portable_cache_embedding_model.py
|
|
2
|
+
|
|
3
|
+
#on each --reload this reads weights
|
|
4
|
+
from langchain_huggingface import HuggingFaceEmbeddings
|
|
5
|
+
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
6
|
+
|
|
7
|
+
embedding_model = HuggingFaceEmbeddings(
|
|
8
|
+
model_name=get_settings().portable_cache_cache_proj_embedding_model,
|
|
9
|
+
cache_folder=get_settings().portable_cache_chroma_db_dir
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
#lazy load, on --reload it will read weights but not at the start but when needed (un-comment me if u need lazyload)
|
|
15
|
+
"""
|
|
16
|
+
from langchain_huggingface import HuggingFaceEmbeddings
|
|
17
|
+
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
18
|
+
|
|
19
|
+
class EmbeddingModel:
|
|
20
|
+
def __init__(self):
|
|
21
|
+
self.model = None
|
|
22
|
+
|
|
23
|
+
def get_model(self):
|
|
24
|
+
if self.model is None:
|
|
25
|
+
self.model = HuggingFaceEmbeddings(
|
|
26
|
+
model_name=get_settings().portable_cache_cache_proj_embedding_model,
|
|
27
|
+
cache_folder=get_settings().portable_cache_chroma_db_dir,
|
|
28
|
+
)
|
|
29
|
+
return self.model
|
|
30
|
+
embedding_manager = EmbeddingModel()
|
|
31
|
+
"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: TriCacheLLM_MMA
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.6
|
|
4
4
|
Summary: A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap.
|
|
5
5
|
Author-email: Mohib Ashfaq Butt <inboxmohib@gmail.com>
|
|
6
6
|
License-Expression: LGPL-3.0-only
|
|
@@ -76,7 +76,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
76
76
|
|
|
77
77
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
78
78
|
|
|
79
|
-
|
|
80
79
|
---
|
|
81
80
|
|
|
82
81
|
# Architecture
|
|
@@ -172,7 +171,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
172
171
|
|
|
173
172
|
Tier 3 is the persistent cache layer.
|
|
174
173
|
|
|
175
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
174
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
176
175
|
|
|
177
176
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
178
177
|
|
|
@@ -293,8 +292,6 @@ Persistent cache seeded
|
|
|
293
292
|
|
|
294
293
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
295
294
|
|
|
296
|
-
|
|
297
|
-
|
|
298
295
|
`populate_cache()` seeds the persistent cache VDB.
|
|
299
296
|
|
|
300
297
|
It does **not** directly populate every cache tier.
|
|
@@ -427,7 +424,7 @@ Or install the development version directly from the repository:
|
|
|
427
424
|
pip install .
|
|
428
425
|
```
|
|
429
426
|
|
|
430
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
427
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
431
428
|
|
|
432
429
|
---
|
|
433
430
|
|
|
@@ -459,7 +456,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
459
456
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
460
457
|
|
|
461
458
|
```bash
|
|
462
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
459
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
463
460
|
```
|
|
464
461
|
|
|
465
462
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -511,18 +508,31 @@ from TriCacheLLM_MMA import (
|
|
|
511
508
|
check_tenant_creation_status,
|
|
512
509
|
)
|
|
513
510
|
|
|
514
|
-
|
|
511
|
+
|
|
512
|
+
@asynccontextmanager
|
|
513
|
+
async def lifespan(app: FastAPI):
|
|
514
|
+
yield
|
|
515
|
+
await close_cache_system()
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
app = FastAPI(
|
|
519
|
+
title="TriCacheLLM_MMA V1 Route Test",
|
|
520
|
+
lifespan=lifespan,
|
|
521
|
+
)
|
|
522
|
+
|
|
515
523
|
|
|
516
524
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
517
|
-
@app.
|
|
518
|
-
async def
|
|
519
|
-
|
|
525
|
+
@app.post("/api/consumer/init")
|
|
526
|
+
async def initialize_consumer():
|
|
527
|
+
user_id: int = 0
|
|
528
|
+
result = await create_cache_system(
|
|
520
529
|
redis_url="redis://localhost:6379/0",
|
|
521
|
-
cohere_api_key="
|
|
530
|
+
cohere_api_key="KEY_HERE",
|
|
522
531
|
chroma_db_dir="./chroma_db",
|
|
523
532
|
db_path="./cache.db",
|
|
524
|
-
user_id=
|
|
533
|
+
user_id=user_id,
|
|
525
534
|
)
|
|
535
|
+
return {"status": "success", "details": result}
|
|
526
536
|
|
|
527
537
|
|
|
528
538
|
# 2. TENANT ENDPOINT (Zero boilerplate for subsequent users)
|
|
@@ -552,18 +562,18 @@ async def ask_question(
|
|
|
552
562
|
)
|
|
553
563
|
|
|
554
564
|
return {"source": "ai", "response": llm_response}
|
|
565
|
+
|
|
555
566
|
```
|
|
556
567
|
The core cache operations are asynchronous Python APIs and are not inherently tied to FastAPI.
|
|
557
568
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
558
569
|
|
|
559
570
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
560
|
-
***NOTE: When you import files they'll send request to hugging face to get embedding model weights but only once! untill you reload server***
|
|
561
|
-
|
|
562
|
-
---
|
|
563
|
-
|
|
564
571
|
|
|
565
572
|
|
|
573
|
+
> **Note on Model Initialization:**
|
|
574
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
566
575
|
|
|
576
|
+
---
|
|
567
577
|
|
|
568
578
|
# Cache Data Model
|
|
569
579
|
|
|
@@ -599,9 +609,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
599
609
|
|
|
600
610
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
601
611
|
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
612
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
606
613
|
|
|
607
614
|
---
|
|
@@ -612,8 +619,6 @@ Additional metadata can be added to the cache payload.
|
|
|
612
619
|
|
|
613
620
|
The cache population path constructs metadata similar to:
|
|
614
621
|
|
|
615
|
-
|
|
616
|
-
|
|
617
622
|
For example:
|
|
618
623
|
|
|
619
624
|
```python
|
|
@@ -866,8 +871,6 @@ These can be considered for future versions.
|
|
|
866
871
|
|
|
867
872
|
---
|
|
868
873
|
|
|
869
|
-
|
|
870
|
-
|
|
871
874
|
# Consumer Responsibility
|
|
872
875
|
|
|
873
876
|
The consuming application is responsible for:
|
|
@@ -924,7 +927,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
924
927
|
|
|
925
928
|
---
|
|
926
929
|
|
|
927
|
-
|
|
928
930
|
# License
|
|
929
931
|
|
|
930
932
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "TriCacheLLM_MMA"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.6"
|
|
8
8
|
description = "A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
tricachellm_mma-0.1.4/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py
DELETED
|
@@ -1,7 +0,0 @@
|
|
|
1
|
-
# # File: portable_cache_utils/portable_cache_embedding_model.py
|
|
2
|
-
from langchain_huggingface import HuggingFaceEmbeddings
|
|
3
|
-
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
4
|
-
|
|
5
|
-
embedding_model = HuggingFaceEmbeddings(
|
|
6
|
-
model_name=get_settings().portable_cache_cache_proj_embedding_model
|
|
7
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{tricachellm_mma-0.1.4 → tricachellm_mma-0.1.6}/TriCacheLLM_MMA.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|