TriCacheLLM-MMA 0.1.5__tar.gz → 0.1.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/PKG-INFO +11 -19
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/README.md +11 -19
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_bgWorkers/portable_cache_celery_conf.py +1 -1
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_bgWorkers/portable_cache_workers.py +0 -2
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_main.py +45 -2
- tricachellm_mma-0.1.7/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py +31 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/PKG-INFO +11 -19
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/pyproject.toml +1 -1
- tricachellm_mma-0.1.5/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py +0 -8
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/LICENSE +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/__init__.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_Ai/portable_cache_rerankAi.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_dbSchema.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_redis.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_dbBase.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_dbConf.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_schemas/portable_cache_schemas.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA/portable_cache_utils/protable_cache_DynamicEnv_maker.py +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/SOURCES.txt +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/dependency_links.txt +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/requires.txt +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/top_level.txt +0 -0
- {tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: TriCacheLLM_MMA
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap.
|
|
5
5
|
Author-email: Mohib Ashfaq Butt <inboxmohib@gmail.com>
|
|
6
6
|
License-Expression: LGPL-3.0-only
|
|
@@ -76,7 +76,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
76
76
|
|
|
77
77
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
78
78
|
|
|
79
|
-
|
|
80
79
|
---
|
|
81
80
|
|
|
82
81
|
# Architecture
|
|
@@ -172,7 +171,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
172
171
|
|
|
173
172
|
Tier 3 is the persistent cache layer.
|
|
174
173
|
|
|
175
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
174
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
176
175
|
|
|
177
176
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
178
177
|
|
|
@@ -293,8 +292,6 @@ Persistent cache seeded
|
|
|
293
292
|
|
|
294
293
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
295
294
|
|
|
296
|
-
|
|
297
|
-
|
|
298
295
|
`populate_cache()` seeds the persistent cache VDB.
|
|
299
296
|
|
|
300
297
|
It does **not** directly populate every cache tier.
|
|
@@ -427,7 +424,7 @@ Or install the development version directly from the repository:
|
|
|
427
424
|
pip install .
|
|
428
425
|
```
|
|
429
426
|
|
|
430
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
427
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
431
428
|
|
|
432
429
|
---
|
|
433
430
|
|
|
@@ -459,7 +456,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
459
456
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
460
457
|
|
|
461
458
|
```bash
|
|
462
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
459
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
463
460
|
```
|
|
464
461
|
|
|
465
462
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -524,6 +521,10 @@ app = FastAPI(
|
|
|
524
521
|
)
|
|
525
522
|
|
|
526
523
|
|
|
524
|
+
# NOTE: It is strongly recommended—if not required—to run create_cache_system(user_id=0) for initialization before starting Celery.
|
|
525
|
+
# In upcoming versions, this process will be automated, so you won't need to manually run the Celery command.
|
|
526
|
+
|
|
527
|
+
|
|
527
528
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
528
529
|
@app.post("/api/consumer/init")
|
|
529
530
|
async def initialize_consumer():
|
|
@@ -571,13 +572,12 @@ The core cache operations are asynchronous Python APIs and are not inherently ti
|
|
|
571
572
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
572
573
|
|
|
573
574
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
574
|
-
***NOTE: On first initialization, the embedding model may contact Hugging Face to obtain the model if it is not already available in the local cache. Subsequent process reloads and reuse the locally cached model files Blazing fast.***
|
|
575
|
-
|
|
576
|
-
---
|
|
577
|
-
|
|
578
575
|
|
|
579
576
|
|
|
577
|
+
> **Note on Model Initialization:**
|
|
578
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
580
579
|
|
|
580
|
+
---
|
|
581
581
|
|
|
582
582
|
# Cache Data Model
|
|
583
583
|
|
|
@@ -613,9 +613,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
613
613
|
|
|
614
614
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
615
615
|
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
616
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
620
617
|
|
|
621
618
|
---
|
|
@@ -626,8 +623,6 @@ Additional metadata can be added to the cache payload.
|
|
|
626
623
|
|
|
627
624
|
The cache population path constructs metadata similar to:
|
|
628
625
|
|
|
629
|
-
|
|
630
|
-
|
|
631
626
|
For example:
|
|
632
627
|
|
|
633
628
|
```python
|
|
@@ -880,8 +875,6 @@ These can be considered for future versions.
|
|
|
880
875
|
|
|
881
876
|
---
|
|
882
877
|
|
|
883
|
-
|
|
884
|
-
|
|
885
878
|
# Consumer Responsibility
|
|
886
879
|
|
|
887
880
|
The consuming application is responsible for:
|
|
@@ -938,7 +931,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
938
931
|
|
|
939
932
|
---
|
|
940
933
|
|
|
941
|
-
|
|
942
934
|
# License
|
|
943
935
|
|
|
944
936
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -34,7 +34,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
34
34
|
|
|
35
35
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
36
36
|
|
|
37
|
-
|
|
38
37
|
---
|
|
39
38
|
|
|
40
39
|
# Architecture
|
|
@@ -130,7 +129,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
130
129
|
|
|
131
130
|
Tier 3 is the persistent cache layer.
|
|
132
131
|
|
|
133
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
132
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
134
133
|
|
|
135
134
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
136
135
|
|
|
@@ -251,8 +250,6 @@ Persistent cache seeded
|
|
|
251
250
|
|
|
252
251
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
253
252
|
|
|
254
|
-
|
|
255
|
-
|
|
256
253
|
`populate_cache()` seeds the persistent cache VDB.
|
|
257
254
|
|
|
258
255
|
It does **not** directly populate every cache tier.
|
|
@@ -385,7 +382,7 @@ Or install the development version directly from the repository:
|
|
|
385
382
|
pip install .
|
|
386
383
|
```
|
|
387
384
|
|
|
388
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
385
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
389
386
|
|
|
390
387
|
---
|
|
391
388
|
|
|
@@ -417,7 +414,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
417
414
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
418
415
|
|
|
419
416
|
```bash
|
|
420
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
417
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
421
418
|
```
|
|
422
419
|
|
|
423
420
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -482,6 +479,10 @@ app = FastAPI(
|
|
|
482
479
|
)
|
|
483
480
|
|
|
484
481
|
|
|
482
|
+
# NOTE: It is strongly recommended—if not required—to run create_cache_system(user_id=0) for initialization before starting Celery.
|
|
483
|
+
# In upcoming versions, this process will be automated, so you won't need to manually run the Celery command.
|
|
484
|
+
|
|
485
|
+
|
|
485
486
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
486
487
|
@app.post("/api/consumer/init")
|
|
487
488
|
async def initialize_consumer():
|
|
@@ -529,13 +530,12 @@ The core cache operations are asynchronous Python APIs and are not inherently ti
|
|
|
529
530
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
530
531
|
|
|
531
532
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
532
|
-
***NOTE: On first initialization, the embedding model may contact Hugging Face to obtain the model if it is not already available in the local cache. Subsequent process reloads and reuse the locally cached model files Blazing fast.***
|
|
533
|
-
|
|
534
|
-
---
|
|
535
|
-
|
|
536
533
|
|
|
537
534
|
|
|
535
|
+
> **Note on Model Initialization:**
|
|
536
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
538
537
|
|
|
538
|
+
---
|
|
539
539
|
|
|
540
540
|
# Cache Data Model
|
|
541
541
|
|
|
@@ -571,9 +571,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
571
571
|
|
|
572
572
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
573
573
|
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
574
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
578
575
|
|
|
579
576
|
---
|
|
@@ -584,8 +581,6 @@ Additional metadata can be added to the cache payload.
|
|
|
584
581
|
|
|
585
582
|
The cache population path constructs metadata similar to:
|
|
586
583
|
|
|
587
|
-
|
|
588
|
-
|
|
589
584
|
For example:
|
|
590
585
|
|
|
591
586
|
```python
|
|
@@ -838,8 +833,6 @@ These can be considered for future versions.
|
|
|
838
833
|
|
|
839
834
|
---
|
|
840
835
|
|
|
841
|
-
|
|
842
|
-
|
|
843
836
|
# Consumer Responsibility
|
|
844
837
|
|
|
845
838
|
The consuming application is responsible for:
|
|
@@ -896,7 +889,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
896
889
|
|
|
897
890
|
---
|
|
898
891
|
|
|
899
|
-
|
|
900
892
|
# License
|
|
901
893
|
|
|
902
894
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -929,4 +921,4 @@ with the goal of making expensive LLM inference the **last resort rather than th
|
|
|
929
921
|
---
|
|
930
922
|
|
|
931
923
|
# Disclaimer:
|
|
932
|
-
This software is provided "as is" without warranty of any kind. The author is not responsible for any data loss, system failures, or damages arising from its use.
|
|
924
|
+
This software is provided "as is" without warranty of any kind. The author is not responsible for any data loss, system failures, or damages arising from its use.
|
|
@@ -46,7 +46,6 @@ async def create_cache_vdb_async(task_instance: Any, user_id: int):
|
|
|
46
46
|
parents=True,
|
|
47
47
|
exist_ok=True,
|
|
48
48
|
)
|
|
49
|
-
|
|
50
49
|
await asyncio.to_thread(
|
|
51
50
|
lambda: Chroma(
|
|
52
51
|
collection_name=f"question_cache_{user_id}",
|
|
@@ -142,7 +141,6 @@ async def push_response_in_cache_async(task_instance: Any, user_id, question: st
|
|
|
142
141
|
)
|
|
143
142
|
|
|
144
143
|
user_cache_vdb_path = Path(cache_res.vdb_path)
|
|
145
|
-
|
|
146
144
|
user_cache_vdb = await asyncio.to_thread(
|
|
147
145
|
lambda: Chroma(
|
|
148
146
|
collection_name=f"question_cache_{user_id}",
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
# File: portable_cache_main.py
|
|
2
2
|
import asyncio
|
|
3
|
+
from asyncio import subprocess
|
|
3
4
|
import hashlib
|
|
4
5
|
import json
|
|
6
|
+
import os
|
|
5
7
|
from pathlib import Path
|
|
6
8
|
from typing import Any
|
|
7
9
|
from .portable_cache_schemas.portable_cache_dbConf import init_cache_database, init_db_tables, db_manager
|
|
@@ -29,6 +31,8 @@ from .portable_cache_redis import get_redis
|
|
|
29
31
|
from .portable_cache_redis import redis_manager
|
|
30
32
|
from .portable_cache_utils.protable_cache_DynamicEnv_maker import system_key
|
|
31
33
|
from .portable_cache_dbSchema import Paths
|
|
34
|
+
import sys
|
|
35
|
+
|
|
32
36
|
|
|
33
37
|
def create_vector_index_schema(dim: int, distance_metric: str = "COSINE", m: int = 16, ef_construction: int = 200, ef_runtime: int = 10) -> list:
|
|
34
38
|
return [
|
|
@@ -439,10 +443,49 @@ async def create_cache_system(
|
|
|
439
443
|
if not key_res["success"]:
|
|
440
444
|
raise ValueError("Failed to initialize system keys. Verify if the keys are correct.")
|
|
441
445
|
|
|
446
|
+
|
|
447
|
+
# This bit is for 0.1.8 (it being after sysetm_key is on purpose)
|
|
448
|
+
"""
|
|
449
|
+
is_in_venv = sys.prefix != sys.base_prefix
|
|
450
|
+
celery_name = "celery.exe" if os.name == "nt" else "celery"
|
|
451
|
+
celery_executable = Path(sys.executable).parent / celery_name
|
|
452
|
+
|
|
453
|
+
if is_in_venv and celery_executable.exists():
|
|
454
|
+
celery_cmd = [
|
|
455
|
+
str(celery_executable),
|
|
456
|
+
"-A",
|
|
457
|
+
"TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app",
|
|
458
|
+
"worker",
|
|
459
|
+
"--loglevel=info",
|
|
460
|
+
"-Q",
|
|
461
|
+
"ai"
|
|
462
|
+
]
|
|
463
|
+
|
|
464
|
+
# Check OS and launch accordingly
|
|
465
|
+
if os.name == "nt":
|
|
466
|
+
subprocess.Popen(celery_cmd, creationflags=subprocess.CREATE_NEW_CONSOLE)
|
|
467
|
+
print("Don't worry, TriCacheLLM_MMA is running celery background workers automatically")
|
|
468
|
+
|
|
469
|
+
elif sys.platform == "darwin":
|
|
470
|
+
try:
|
|
471
|
+
cmd_str = " ".join(celery_cmd)
|
|
472
|
+
subprocess.Popen(["osascript", "-e", f'tell application "Terminal" to do script "{cmd_str}"'])
|
|
473
|
+
print("Don't worry, TriCacheLLM_MMA is running celery background workers automatically")
|
|
474
|
+
except Exception:
|
|
475
|
+
subprocess.Popen(celery_cmd)
|
|
476
|
+
print("Don't worry, TriCacheLLM_MMA is running celery background workers automatically")
|
|
477
|
+
else:
|
|
478
|
+
# Linux / Unix fallback (runs as a background child process)
|
|
479
|
+
subprocess.Popen(celery_cmd)
|
|
480
|
+
print("Don't worry, TriCacheLLM_MMA is running celery background workers automatically")
|
|
481
|
+
else:
|
|
482
|
+
print("---WARNING--- auto celery start failed manually run this command:\ncelery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai")
|
|
483
|
+
|
|
484
|
+
"""
|
|
485
|
+
|
|
442
486
|
return #user_id=0 doesnt need its own vdb! if you want to be it user-self be user_id=1 -> single user
|
|
443
487
|
#if you want multi-tanent then keep sending in user_ids lol
|
|
444
488
|
|
|
445
|
-
|
|
446
489
|
async with db_manager.async_session() as db:
|
|
447
490
|
ans: None | dict = await create_cache_vdb_inishiator(user_id=user_id, db=db)
|
|
448
491
|
|
|
@@ -524,7 +567,7 @@ async def populate_cache(to_cache_answer, to_cache_question, user_id: Any = 0) -
|
|
|
524
567
|
async def close_cache_system():
|
|
525
568
|
print("SHUTTING_DOWN_PORTABLE_CACHE: Cleaning up resources...")
|
|
526
569
|
|
|
527
|
-
from portable_cache_schemas.portable_cache_dbConf import db_manager
|
|
570
|
+
from .portable_cache_schemas.portable_cache_dbConf import db_manager
|
|
528
571
|
|
|
529
572
|
if db_manager.norma_engine is not None:
|
|
530
573
|
try:
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# File: portable_cache_utils/portable_cache_embedding_model.py
|
|
2
|
+
|
|
3
|
+
#on each --reload this reads weights
|
|
4
|
+
from langchain_huggingface import HuggingFaceEmbeddings
|
|
5
|
+
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
6
|
+
|
|
7
|
+
embedding_model = HuggingFaceEmbeddings(
|
|
8
|
+
model_name=get_settings().portable_cache_cache_proj_embedding_model,
|
|
9
|
+
cache_folder=get_settings().portable_cache_chroma_db_dir
|
|
10
|
+
)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
#lazy load, on --reload it will read weights but not at the start but when needed (un-comment me if u need lazyload)
|
|
15
|
+
"""
|
|
16
|
+
from langchain_huggingface import HuggingFaceEmbeddings
|
|
17
|
+
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
18
|
+
|
|
19
|
+
class EmbeddingModel:
|
|
20
|
+
def __init__(self):
|
|
21
|
+
self.model = None
|
|
22
|
+
|
|
23
|
+
def get_model(self):
|
|
24
|
+
if self.model is None:
|
|
25
|
+
self.model = HuggingFaceEmbeddings(
|
|
26
|
+
model_name=get_settings().portable_cache_cache_proj_embedding_model,
|
|
27
|
+
cache_folder=get_settings().portable_cache_chroma_db_dir,
|
|
28
|
+
)
|
|
29
|
+
return self.model
|
|
30
|
+
embedding_manager = EmbeddingModel()
|
|
31
|
+
"""
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: TriCacheLLM_MMA
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap.
|
|
5
5
|
Author-email: Mohib Ashfaq Butt <inboxmohib@gmail.com>
|
|
6
6
|
License-Expression: LGPL-3.0-only
|
|
@@ -76,7 +76,6 @@ The internal infrastructure handles the multi-tier lookup and promotion logic.
|
|
|
76
76
|
|
|
77
77
|
> **💡 Note for PyPI Visitors:** If you are reading this on PyPI, please check out the [TriCacheLLM_MMA GitHub Repository](https://github.com/mohib-ash/TriCacheLLM_MMA) for the most up-to-date documentation, integration guides, and advanced examples!
|
|
78
78
|
|
|
79
|
-
|
|
80
79
|
---
|
|
81
80
|
|
|
82
81
|
# Architecture
|
|
@@ -172,7 +171,7 @@ Tier 2 is intentionally optimized for fast semantic cache retrieval.
|
|
|
172
171
|
|
|
173
172
|
Tier 3 is the persistent cache layer.
|
|
174
173
|
|
|
175
|
-
The persistent vector database acts as the long-term backing store for cached Q
|
|
174
|
+
The persistent vector database acts as the long-term backing store for cached Q\&A entries.
|
|
176
175
|
|
|
177
176
|
When Tier 1 and Tier 2 miss, Tier 3 performs the persistent lookup.
|
|
178
177
|
|
|
@@ -293,8 +292,6 @@ Persistent cache seeded
|
|
|
293
292
|
|
|
294
293
|
`populate_cache()` and `check_cache()` have intentionally different responsibilities.
|
|
295
294
|
|
|
296
|
-
|
|
297
|
-
|
|
298
295
|
`populate_cache()` seeds the persistent cache VDB.
|
|
299
296
|
|
|
300
297
|
It does **not** directly populate every cache tier.
|
|
@@ -427,7 +424,7 @@ Or install the development version directly from the repository:
|
|
|
427
424
|
pip install .
|
|
428
425
|
```
|
|
429
426
|
|
|
430
|
-
PyPI link: https://pypi.org/project/TriCacheLLM-MMA/0.1.
|
|
427
|
+
PyPI link: [https://pypi.org/project/TriCacheLLM-MMA/0.1.6/](https://pypi.org/project/TriCacheLLM-MMA/0.1.6/)
|
|
431
428
|
|
|
432
429
|
---
|
|
433
430
|
|
|
@@ -459,7 +456,7 @@ TriCacheLLM_MMA uses Celery for background operations such as:
|
|
|
459
456
|
After installing the package, start the cache worker in a **separate terminal**:
|
|
460
457
|
|
|
461
458
|
```bash
|
|
462
|
-
celery -A portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
459
|
+
celery -A TriCacheLLM_MMA.portable_cache_bgWorkers.portable_cache_celery_conf.celery_app worker --loglevel=info -Q ai
|
|
463
460
|
```
|
|
464
461
|
|
|
465
462
|
You do **not** need to navigate into the package's `site-packages` directory.
|
|
@@ -524,6 +521,10 @@ app = FastAPI(
|
|
|
524
521
|
)
|
|
525
522
|
|
|
526
523
|
|
|
524
|
+
# NOTE: It is strongly recommended—if not required—to run create_cache_system(user_id=0) for initialization before starting Celery.
|
|
525
|
+
# In upcoming versions, this process will be automated, so you won't need to manually run the Celery command.
|
|
526
|
+
|
|
527
|
+
|
|
527
528
|
# 1. ADMIN SETUP (Run once on startup / first deploy with user_id=0)
|
|
528
529
|
@app.post("/api/consumer/init")
|
|
529
530
|
async def initialize_consumer():
|
|
@@ -571,13 +572,12 @@ The core cache operations are asynchronous Python APIs and are not inherently ti
|
|
|
571
572
|
The included integration example uses FastAPI because it provides a convenient demonstration of multi-user request handling.
|
|
572
573
|
|
|
573
574
|
V1 is primarily demonstrated in a web-service architecture, while future versions aim to make standalone application integration equally straightforward.
|
|
574
|
-
***NOTE: On first initialization, the embedding model may contact Hugging Face to obtain the model if it is not already available in the local cache. Subsequent process reloads and reuse the locally cached model files Blazing fast.***
|
|
575
|
-
|
|
576
|
-
---
|
|
577
|
-
|
|
578
575
|
|
|
579
576
|
|
|
577
|
+
> **Note on Model Initialization:**
|
|
578
|
+
> On the first request or system initialization, the embedding model will load into RAM (fetching from Hugging Face if it's not already present in your local cache). Subsequent requests and code reloads reuse the cached instance instantly for blazing-fast performance.
|
|
580
579
|
|
|
580
|
+
---
|
|
581
581
|
|
|
582
582
|
# Cache Data Model
|
|
583
583
|
|
|
@@ -613,9 +613,6 @@ For example, you may want to restrict cache retrieval to:
|
|
|
613
613
|
|
|
614
614
|
The persistent retrieval logic can be extended around the VDB lookup.
|
|
615
615
|
|
|
616
|
-
|
|
617
|
-
|
|
618
|
-
|
|
619
616
|
This allows applications to combine semantic retrieval with deterministic metadata constraints.
|
|
620
617
|
|
|
621
618
|
---
|
|
@@ -626,8 +623,6 @@ Additional metadata can be added to the cache payload.
|
|
|
626
623
|
|
|
627
624
|
The cache population path constructs metadata similar to:
|
|
628
625
|
|
|
629
|
-
|
|
630
|
-
|
|
631
626
|
For example:
|
|
632
627
|
|
|
633
628
|
```python
|
|
@@ -880,8 +875,6 @@ These can be considered for future versions.
|
|
|
880
875
|
|
|
881
876
|
---
|
|
882
877
|
|
|
883
|
-
|
|
884
|
-
|
|
885
878
|
# Consumer Responsibility
|
|
886
879
|
|
|
887
880
|
The consuming application is responsible for:
|
|
@@ -938,7 +931,6 @@ The current V1 intentionally keeps the architecture close to the underlying impl
|
|
|
938
931
|
|
|
939
932
|
---
|
|
940
933
|
|
|
941
|
-
|
|
942
934
|
# License
|
|
943
935
|
|
|
944
936
|
This project is licensed under the **GNU Lesser General Public License v3.0 (LGPLv3)**.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "TriCacheLLM_MMA"
|
|
7
|
-
version = "0.1.
|
|
7
|
+
version = "0.1.7"
|
|
8
8
|
description = "A robust three-tier LLM-aware caching SDK using Redis, ChromaDB, and automatic SQLite bootstrap."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.11"
|
tricachellm_mma-0.1.5/TriCacheLLM_MMA/portable_cache_utils/portable_cache_embedding_model.py
DELETED
|
@@ -1,8 +0,0 @@
|
|
|
1
|
-
# # File: portable_cache_utils/portable_cache_embedding_model.py
|
|
2
|
-
from langchain_huggingface import HuggingFaceEmbeddings
|
|
3
|
-
from ..portable_cache_utils.protable_cache_DynamicEnv_maker import get_settings
|
|
4
|
-
|
|
5
|
-
embedding_model = HuggingFaceEmbeddings(
|
|
6
|
-
model_name=get_settings().portable_cache_cache_proj_embedding_model,
|
|
7
|
-
cache_folder=get_settings().portable_cache_chroma_db_dir
|
|
8
|
-
)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{tricachellm_mma-0.1.5 → tricachellm_mma-0.1.7}/TriCacheLLM_MMA.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|