optivision-rag 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- optivision_rag-0.1.1/LICENSE +21 -0
- optivision_rag-0.1.1/PACKAGE.md +90 -0
- optivision_rag-0.1.1/PKG-INFO +136 -0
- optivision_rag-0.1.1/README.md +349 -0
- optivision_rag-0.1.1/pyproject.toml +71 -0
- optivision_rag-0.1.1/setup.cfg +4 -0
- optivision_rag-0.1.1/src/optivision/__init__.py +44 -0
- optivision_rag-0.1.1/src/optivision/bench.py +587 -0
- optivision_rag-0.1.1/src/optivision/cli.py +344 -0
- optivision_rag-0.1.1/src/optivision/compression/__init__.py +170 -0
- optivision_rag-0.1.1/src/optivision/compression/binary.py +116 -0
- optivision_rag-0.1.1/src/optivision/compression/lloyd2.py +229 -0
- optivision_rag-0.1.1/src/optivision/config.py +200 -0
- optivision_rag-0.1.1/src/optivision/corpus.py +359 -0
- optivision_rag-0.1.1/src/optivision/demo.py +369 -0
- optivision_rag-0.1.1/src/optivision/encoders/__init__.py +33 -0
- optivision_rag-0.1.1/src/optivision/encoders/base.py +54 -0
- optivision_rag-0.1.1/src/optivision/encoders/colvlm.py +388 -0
- optivision_rag-0.1.1/src/optivision/encoders/synthetic.py +170 -0
- optivision_rag-0.1.1/src/optivision/index/__init__.py +51 -0
- optivision_rag-0.1.1/src/optivision/index/base.py +39 -0
- optivision_rag-0.1.1/src/optivision/index/numpy_index.py +272 -0
- optivision_rag-0.1.1/src/optivision/index/qdrant_index.py +261 -0
- optivision_rag-0.1.1/src/optivision/ingest.py +85 -0
- optivision_rag-0.1.1/src/optivision/metrics.py +172 -0
- optivision_rag-0.1.1/src/optivision/pipeline.py +286 -0
- optivision_rag-0.1.1/src/optivision/presets/colpali.yaml +27 -0
- optivision_rag-0.1.1/src/optivision/presets/colsmol.yaml +25 -0
- optivision_rag-0.1.1/src/optivision/presets/qdrant.yaml +18 -0
- optivision_rag-0.1.1/src/optivision/presets/synthetic.yaml +25 -0
- optivision_rag-0.1.1/src/optivision/pruning/__init__.py +130 -0
- optivision_rag-0.1.1/src/optivision/pruning/codebook.py +127 -0
- optivision_rag-0.1.1/src/optivision/pruning/redundancy.py +78 -0
- optivision_rag-0.1.1/src/optivision/pruning/saliency.py +86 -0
- optivision_rag-0.1.1/src/optivision/pruning/spatial.py +78 -0
- optivision_rag-0.1.1/src/optivision/types.py +146 -0
- optivision_rag-0.1.1/src/optivision/viz.py +100 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/PKG-INFO +136 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/SOURCES.txt +49 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/dependency_links.txt +1 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/entry_points.txt +2 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/requires.txt +26 -0
- optivision_rag-0.1.1/src/optivision_rag.egg-info/top_level.txt +1 -0
- optivision_rag-0.1.1/tests/test_compression.py +438 -0
- optivision_rag-0.1.1/tests/test_demo.py +251 -0
- optivision_rag-0.1.1/tests/test_index.py +206 -0
- optivision_rag-0.1.1/tests/test_pipeline.py +316 -0
- optivision_rag-0.1.1/tests/test_presets.py +27 -0
- optivision_rag-0.1.1/tests/test_pruning.py +152 -0
- optivision_rag-0.1.1/tests/test_qdrant.py +71 -0
- optivision_rag-0.1.1/tests/test_release_versions.py +13 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 T. Rithik Krishna, Amgovath Navanitha, Badavath Akhila
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# OptiVision RAG
|
|
2
|
+
|
|
3
|
+
**Extreme token compression for vision-language document retrieval.**
|
|
4
|
+
|
|
5
|
+
ColPali-style models search scanned documents as images, with no OCR. They store
|
|
6
|
+
one 128-dimensional float vector per image patch, so a single page costs about
|
|
7
|
+
512 KB of index and a million pages cost about half a terabyte.
|
|
8
|
+
|
|
9
|
+
OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
|
|
10
|
+
unmodified and retrieval is still late-interaction MaxSim. Only the number and the
|
|
11
|
+
size of the stored vectors change:
|
|
12
|
+
|
|
13
|
+
```
|
|
14
|
+
page image
|
|
15
|
+
├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
|
|
16
|
+
├─ 1. spatial pruning ──► drop patches on blank paper
|
|
17
|
+
├─ 2. redundancy prune ─► collapse near-duplicate patches
|
|
18
|
+
├─ 3. binary quantize ──► 128 floats (512 B) -> 128 bits (16 B)
|
|
19
|
+
└─ index ───────────────► Qdrant MaxSim, or an exact numpy index
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
| setting | encoder / corpus | compression | nDCG@5 retained |
|
|
23
|
+
|---|---|---|---|
|
|
24
|
+
| prune + binary | ColSmol-256M, generated pages | 113.5x | 86.7% |
|
|
25
|
+
| prune + int8 | ColSmol-256M, generated pages | 14.2x | 96.0% |
|
|
26
|
+
| prune + binary | ColPali-v1.3, ViDoRe (4 splits) | 53-60x | 94.6-103.4% |
|
|
27
|
+
|
|
28
|
+
Full results, the ablations and the analysis are in the
|
|
29
|
+
[repository](https://github.com/Daemon-VI/optivision-rag).
|
|
30
|
+
|
|
31
|
+
## Install
|
|
32
|
+
|
|
33
|
+
Python 3.10 or newer.
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install optivision-rag # core pipeline + CLI
|
|
37
|
+
pip install "optivision-rag[corpus]" # + synthetic test-corpus generator
|
|
38
|
+
pip install "optivision-rag[vlm]" # + real encoders (torch, colpali-engine)
|
|
39
|
+
pip install "optivision-rag[vlm,corpus,bench]"
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Or run the CLI through npm without touching pip yourself (it still needs Python
|
|
43
|
+
3.10+ on the machine):
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
npx optivision-rag --help
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Quick start (offline, no model download)
|
|
50
|
+
|
|
51
|
+
The `synthetic` preset uses a deterministic stand-in encoder, so the whole pipeline
|
|
52
|
+
runs in seconds on any laptop. Use it to try the tool, not for real results.
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
optivision make-corpus data/corpus --docs 10 --pages 2
|
|
56
|
+
optivision index data/corpus/pdfs -c synthetic
|
|
57
|
+
optivision search "renewal of vehicle insurance policy" -c synthetic
|
|
58
|
+
optivision stats -c synthetic
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## With a real model
|
|
62
|
+
|
|
63
|
+
Install the `vlm` extra, then switch the preset. `colsmol` is a 256M-parameter
|
|
64
|
+
model that runs on a CPU laptop; `colpali` wants a GPU.
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
optivision index path/to/your/pdfs -c colsmol
|
|
68
|
+
optivision search "fire safety audit memorandum" -c colsmol
|
|
69
|
+
optivision explain path/to/your/pdfs -c colsmol --out figures
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`-c` takes a bundled preset name (`synthetic`, `colsmol`, `colpali`, `qdrant`) or a
|
|
73
|
+
path to your own YAML file. `optivision init-config my.yaml` writes one pre-filled
|
|
74
|
+
with every default.
|
|
75
|
+
|
|
76
|
+
## Python API
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from optivision import Config, OptiVisionRAG
|
|
80
|
+
|
|
81
|
+
rag = OptiVisionRAG(Config.load("colsmol"))
|
|
82
|
+
report = rag.build("path/to/pdfs")
|
|
83
|
+
print(report.compression_ratio)
|
|
84
|
+
print(rag.search("office memorandum on fire safety audit").hits[0].ref.page_id)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
## Authors
|
|
88
|
+
|
|
89
|
+
T. Rithik Krishna, Amgovath Navanitha and Badavath Akhila, Mahatma Gandhi Institute
|
|
90
|
+
of Technology, Hyderabad. MIT licensed.
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: optivision-rag
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: OptiVision RAG: Extreme Token Compression for Vision-Language Models
|
|
5
|
+
Author: T. Rithik Krishna, Amgovath Navanitha, Badavath Akhila
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/Daemon-VI/optivision-rag
|
|
8
|
+
Project-URL: Repository, https://github.com/Daemon-VI/optivision-rag
|
|
9
|
+
Project-URL: Issues, https://github.com/Daemon-VI/optivision-rag/issues
|
|
10
|
+
Keywords: rag,colpali,colqwen,late-interaction,vision-language,retrieval,quantization,token-pruning,qdrant
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: numpy<3,>=1.26
|
|
25
|
+
Requires-Dist: pillow>=10
|
|
26
|
+
Requires-Dist: pypdfium2>=4
|
|
27
|
+
Requires-Dist: qdrant-client>=1.10
|
|
28
|
+
Requires-Dist: pyyaml>=6
|
|
29
|
+
Requires-Dist: typer>=0.12
|
|
30
|
+
Requires-Dist: rich>=13
|
|
31
|
+
Provides-Extra: vlm
|
|
32
|
+
Requires-Dist: torch>=2.2; extra == "vlm"
|
|
33
|
+
Requires-Dist: colpali-engine>=0.3.10; extra == "vlm"
|
|
34
|
+
Requires-Dist: transformers>=4.46; extra == "vlm"
|
|
35
|
+
Provides-Extra: corpus
|
|
36
|
+
Requires-Dist: reportlab>=4; extra == "corpus"
|
|
37
|
+
Provides-Extra: bench
|
|
38
|
+
Requires-Dist: reportlab>=4; extra == "bench"
|
|
39
|
+
Requires-Dist: datasets>=2.19; extra == "bench"
|
|
40
|
+
Provides-Extra: app
|
|
41
|
+
Requires-Dist: streamlit>=1.34; extra == "app"
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
44
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
45
|
+
Dynamic: license-file
|
|
46
|
+
|
|
47
|
+
# OptiVision RAG
|
|
48
|
+
|
|
49
|
+
**Extreme token compression for vision-language document retrieval.**
|
|
50
|
+
|
|
51
|
+
ColPali-style models search scanned documents as images, with no OCR. They store
|
|
52
|
+
one 128-dimensional float vector per image patch, so a single page costs about
|
|
53
|
+
512 KB of index and a million pages cost about half a terabyte.
|
|
54
|
+
|
|
55
|
+
OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
|
|
56
|
+
unmodified and retrieval is still late-interaction MaxSim. Only the number and the
|
|
57
|
+
size of the stored vectors change:
|
|
58
|
+
|
|
59
|
+
```
|
|
60
|
+
page image
|
|
61
|
+
├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
|
|
62
|
+
├─ 1. spatial pruning ──► drop patches on blank paper
|
|
63
|
+
├─ 2. redundancy prune ─► collapse near-duplicate patches
|
|
64
|
+
├─ 3. binary quantize ──► 128 floats (512 B) -> 128 bits (16 B)
|
|
65
|
+
└─ index ───────────────► Qdrant MaxSim, or an exact numpy index
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
| setting | encoder / corpus | compression | nDCG@5 retained |
|
|
69
|
+
|---|---|---|---|
|
|
70
|
+
| prune + binary | ColSmol-256M, generated pages | 113.5x | 86.7% |
|
|
71
|
+
| prune + int8 | ColSmol-256M, generated pages | 14.2x | 96.0% |
|
|
72
|
+
| prune + binary | ColPali-v1.3, ViDoRe (4 splits) | 53-60x | 94.6-103.4% |
|
|
73
|
+
|
|
74
|
+
Full results, the ablations and the analysis are in the
|
|
75
|
+
[repository](https://github.com/Daemon-VI/optivision-rag).
|
|
76
|
+
|
|
77
|
+
## Install
|
|
78
|
+
|
|
79
|
+
Python 3.10 or newer.
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install optivision-rag # core pipeline + CLI
|
|
83
|
+
pip install "optivision-rag[corpus]" # + synthetic test-corpus generator
|
|
84
|
+
pip install "optivision-rag[vlm]" # + real encoders (torch, colpali-engine)
|
|
85
|
+
pip install "optivision-rag[vlm,corpus,bench]"
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Or run the CLI through npm without touching pip yourself (it still needs Python
|
|
89
|
+
3.10+ on the machine):
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
npx optivision-rag --help
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Quick start (offline, no model download)
|
|
96
|
+
|
|
97
|
+
The `synthetic` preset uses a deterministic stand-in encoder, so the whole pipeline
|
|
98
|
+
runs in seconds on any laptop. Use it to try the tool, not for real results.
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
optivision make-corpus data/corpus --docs 10 --pages 2
|
|
102
|
+
optivision index data/corpus/pdfs -c synthetic
|
|
103
|
+
optivision search "renewal of vehicle insurance policy" -c synthetic
|
|
104
|
+
optivision stats -c synthetic
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
## With a real model
|
|
108
|
+
|
|
109
|
+
Install the `vlm` extra, then switch the preset. `colsmol` is a 256M-parameter
|
|
110
|
+
model that runs on a CPU laptop; `colpali` wants a GPU.
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
optivision index path/to/your/pdfs -c colsmol
|
|
114
|
+
optivision search "fire safety audit memorandum" -c colsmol
|
|
115
|
+
optivision explain path/to/your/pdfs -c colsmol --out figures
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
`-c` takes a bundled preset name (`synthetic`, `colsmol`, `colpali`, `qdrant`) or a
|
|
119
|
+
path to your own YAML file. `optivision init-config my.yaml` writes one pre-filled
|
|
120
|
+
with every default.
|
|
121
|
+
|
|
122
|
+
## Python API
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from optivision import Config, OptiVisionRAG
|
|
126
|
+
|
|
127
|
+
rag = OptiVisionRAG(Config.load("colsmol"))
|
|
128
|
+
report = rag.build("path/to/pdfs")
|
|
129
|
+
print(report.compression_ratio)
|
|
130
|
+
print(rag.search("office memorandum on fire safety audit").hits[0].ref.page_id)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
## Authors
|
|
134
|
+
|
|
135
|
+
T. Rithik Krishna, Amgovath Navanitha and Badavath Akhila, Mahatma Gandhi Institute
|
|
136
|
+
of Technology, Hyderabad. MIT licensed.
|
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
---
|
|
2
|
+
title: OptiVision RAG
|
|
3
|
+
emoji: 🗜️
|
|
4
|
+
colorFrom: indigo
|
|
5
|
+
colorTo: green
|
|
6
|
+
sdk: gradio
|
|
7
|
+
sdk_version: 6.24.0
|
|
8
|
+
python_version: "3.11"
|
|
9
|
+
app_file: app.py
|
|
10
|
+
pinned: false
|
|
11
|
+
license: mit
|
|
12
|
+
short_description: Visual-token pruning and binary quantization for document retrieval
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# OptiVision RAG
|
|
16
|
+
|
|
17
|
+
**Extreme token compression for vision-language document retrieval.**
|
|
18
|
+
|
|
19
|
+
> The YAML block above is Hugging Face Space metadata. It is ignored by GitHub
|
|
20
|
+
> and by every tool in this repo; see [docs/DEPLOY.md](docs/DEPLOY.md) for how
|
|
21
|
+
> the Space is set up.
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
## The problem
|
|
26
|
+
|
|
27
|
+
ColPali-style models search scanned documents *as images* — no OCR, so they work on
|
|
28
|
+
old copies, seals, stamps and handwriting that text pipelines fail on. They do it by
|
|
29
|
+
splitting each page into patches and keeping **one vector per patch**.
|
|
30
|
+
|
|
31
|
+
That is also the problem. A single page becomes ~800–1000 vectors of 128 float32
|
|
32
|
+
dimensions:
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
1 page = 1024 vectors x 128 dims x 4 bytes = 512 KB
|
|
36
|
+
1 million pages = 512 GB
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Half a terabyte of RAM-resident vector index for a corpus that a records office
|
|
40
|
+
would consider small. The accuracy is excellent; the storage is what stops it
|
|
41
|
+
being deployable.
|
|
42
|
+
|
|
43
|
+
## What this project does
|
|
44
|
+
|
|
45
|
+
OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
|
|
46
|
+
unmodified, the query path is untouched, and the retrieval is still late-interaction
|
|
47
|
+
MaxSim — only the number and the size of the stored vectors change.
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
page image
|
|
51
|
+
│
|
|
52
|
+
├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
|
|
53
|
+
│
|
|
54
|
+
├─ 1. spatial pruning ──► drop patches on blank paper
|
|
55
|
+
│
|
|
56
|
+
├─ 2. redundancy prune ─► collapse near-duplicate patches
|
|
57
|
+
│
|
|
58
|
+
├─ 3. binary quantize ──► 128 floats (512 B) → 128 bits (16 B)
|
|
59
|
+
│
|
|
60
|
+
└─ index ───────────────► Qdrant MaxSim, or an exact numpy index
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
**1. Spatial pruning.** Most of a document page is paper. Per patch we measure ink
|
|
64
|
+
density (how much is darker than the estimated paper background) and edge energy
|
|
65
|
+
(local contrast — glyph strokes, rules, stamp borders), and drop the cells that carry
|
|
66
|
+
neither. The estimate runs on the pixels in microseconds, before the model, and needs
|
|
67
|
+
no attention maps or second forward pass. The paper background is estimated per page
|
|
68
|
+
(95th percentile) rather than assumed white, so a yellowed photocopy is not read as a
|
|
69
|
+
fully inked page.
|
|
70
|
+
|
|
71
|
+
**2. Redundancy pruning.** Spatial pruning removes patches with *no* content;
|
|
72
|
+
this removes patches with *duplicate* content — the interior of a filled table cell,
|
|
73
|
+
the middle of a thick rule. They look salient to a pixel detector but add nothing to a
|
|
74
|
+
MaxSim score, because MaxSim already takes a max over near-identical vectors. A greedy
|
|
75
|
+
single pass collapses each cluster to its renormalised centroid.
|
|
76
|
+
|
|
77
|
+
**3. Binary quantization.** These vectors are L2-normalised and roughly zero-centred
|
|
78
|
+
per dimension, so the sign bits keep the orthant and discard only the within-orthant
|
|
79
|
+
position. Crucially, the distortion is nearly the same for every document vector, so
|
|
80
|
+
it moves scores much more than it moves *ranking*. Queries stay in float32 and are
|
|
81
|
+
scored against ±1 document codes (asymmetric scoring): a corpus has millions of
|
|
82
|
+
vectors, a query has twenty, so query precision is the cheapest thing to keep.
|
|
83
|
+
|
|
84
|
+
The three stages multiply — under E1, pruning ~3.5× × quantization 32× ≈ **100×+** off
|
|
85
|
+
the index. On real ViDoRe pages pruning buys less, and the product lands in the 53-60× range; both
|
|
86
|
+
figures are in *Results* below.
|
|
87
|
+
|
|
88
|
+
## Results
|
|
89
|
+
|
|
90
|
+
We ran the ablation twice, and the two runs disagree. That disagreement is the
|
|
91
|
+
result — read both tables before drawing a conclusion from either.
|
|
92
|
+
|
|
93
|
+
### E1 — ColSmol-256M, generated pages
|
|
94
|
+
|
|
95
|
+
60 pages / 72 queries on CPU. Full table and analysis in
|
|
96
|
+
[docs/RESULTS.md](docs/RESULTS.md); reproduce with `make bench`.
|
|
97
|
+
|
|
98
|
+
| variant | tok/pg | KB/pg | compression | nDCG@5 | retained | tau |
|
|
99
|
+
|---|---|---|---|---|---|---|
|
|
100
|
+
| baseline-float32 | 875.0 | 448.00 | 1.0x | 0.7823 | 100.0% | 1.000 |
|
|
101
|
+
| spatial-only | 356.1 | 182.32 | 2.5x | 0.7602 | 97.2% | 0.935 |
|
|
102
|
+
| spatial+redundancy | 246.8 | 126.34 | 3.5x | 0.7519 | 96.1% | 0.866 |
|
|
103
|
+
| int8-only | 875.0 | 112.00 | 4.0x | 0.7877 | 100.7% | 0.972 |
|
|
104
|
+
| **prune+int8** | **246.8** | **31.59** | **14.2x** | **0.7511** | **96.0%** | 0.864 |
|
|
105
|
+
| binary-only | 875.0 | 14.00 | 32.0x | 0.6875 | 87.9% | 0.585 |
|
|
106
|
+
| **optivision** | **246.8** | **3.95** | **113.5x** | **0.6782** | **86.7%** | 0.606 |
|
|
107
|
+
| optivision-aggressive | 186.3 | 2.98 | 150.3x | 0.6680 | 85.4% | 0.602 |
|
|
108
|
+
|
|
109
|
+
**Under this encoder, the two halves of the proposal do not contribute equally.**
|
|
110
|
+
Pruning is nearly free — 3.5x fewer vectors for 3.9% of nDCG@5.
|
|
111
|
+
Binary quantization is where the quality goes — 12.1% lost *without dropping a single
|
|
112
|
+
token*, while int8 gives 4x for 0.5%. Pruning harder barely matters once binary is in
|
|
113
|
+
play (keeping the top 10% of patches scores the same as keeping the top 50%).
|
|
114
|
+
|
|
115
|
+
So under E1 the pipeline has two defensible operating points:
|
|
116
|
+
|
|
117
|
+
- **smallest index** — prune + binary: **113.5x** at 86.7% of baseline nDCG@5
|
|
118
|
+
- **best quality per byte** — prune + int8: **14.2x** at 96.0%, with a Kendall tau of
|
|
119
|
+
0.864 against 0.866 for pruning alone, so int8 adds no measurable ranking distortion on
|
|
120
|
+
top of the pruning
|
|
121
|
+
|
|
122
|
+
448 KB/page becomes 3.95 KB/page: a million-page archive drops from ~448 GB to ~4 GB.
|
|
123
|
+
At the quality-first setting it is ~32 GB.
|
|
124
|
+
|
|
125
|
+
### E2 — ColPali-v1.3, real ViDoRe pages
|
|
126
|
+
|
|
127
|
+
**The reference model disagrees with every headline above.** The same ablation on
|
|
128
|
+
ColPali-v1.3 over four ViDoRe splits (1,780 pages, 1,325 queries) reaches 53-60x at
|
|
129
|
+
94.6-103.4% of baseline nDCG@5 — less compression, far less loss. Two of the three findings above do not transfer:
|
|
130
|
+
pruning buys only 1.7-1.9x on real pages rather than 3.5x, and binary quantization
|
|
131
|
+
costs 0-3.7% rather than 12.1%. Full table and analysis in
|
|
132
|
+
[docs/RESULTS.md](docs/RESULTS.md#the-same-table-on-colpali-3b-and-real-vidore-pages);
|
|
133
|
+
raw reports in [reports/](reports/); reproduce with `bash scripts/run_bench_gpu.sh` on
|
|
134
|
+
any free T4.
|
|
135
|
+
|
|
136
|
+
The takeaway is not "E2 supersedes E1". It is that **the per-stage attribution is a
|
|
137
|
+
property of the encoder and the corpus, not of the compression layer** — the same code
|
|
138
|
+
paid its quality in different places under the two. A third run separates those two
|
|
139
|
+
variables; see below. A compression ratio reported without naming the encoder and the
|
|
140
|
+
corpus it was measured on is not enough information to act on, which is what the paper
|
|
141
|
+
argues.
|
|
142
|
+
|
|
143
|
+
### E3 — ColPali-3B, generated pages
|
|
144
|
+
|
|
145
|
+
The missing cell: the reference encoder over E1's own corpus, regenerated at seed 7 so
|
|
146
|
+
the pages are the same ones. It splits the reversal in two.
|
|
147
|
+
|
|
148
|
+
| | corpus fixed, encoder swapped | encoder fixed, corpus swapped |
|
|
149
|
+
|---|---|---|
|
|
150
|
+
| one-bit codec costs | E1 → E3: **12.1 → 1.6 points** | E3 → E2: 1.6 → 3.7 points |
|
|
151
|
+
| pruning buys | E1 → E3: 3.55x → 4.20x | E3 → E2: **4.20x → 1.85x** |
|
|
152
|
+
|
|
153
|
+
**The corpus sets what pruning buys.** Run `python scripts/compare_regimes.py` to
|
|
154
|
+
print it from the benchmark files. E3 is also the best operating point in the project
|
|
155
|
+
(98.7% retention at 134.5x) and the least representative one.
|
|
156
|
+
|
|
157
|
+
The codec half is not an encoder property. Per query, a sign code adds the same ~3% of
|
|
158
|
+
score noise under both encoders, and it flips only queries whose float margin over
|
|
159
|
+
the best competitor is smaller than that noise — under 1% of 945 real ViDoRe queries
|
|
160
|
+
above twice the noise, against a third of those below half.
|
|
161
|
+
E1's precise queries are decided by three digits at a 0.05% margin, which is why the
|
|
162
|
+
small encoder looks fragile; ColPali cannot read those digits at 448 px and never wins
|
|
163
|
+
the queries it is credited with not losing (precise R@1 0.250, floor 0.200). Enlarge the
|
|
164
|
+
code and its one-bit cost appears. See [RESULTS.md](docs/RESULTS.md#e3---colpali-3b-on-the-generated-corpus)
|
|
165
|
+
and [REVIEW-2026-08-21.md](docs/REVIEW-2026-08-21.md); `scripts/tau_audit.py` explains why
|
|
166
|
+
E1's tau of 0.585 and E2's 0.527 were never the same measurement.
|
|
167
|
+
|
|
168
|
+
On dense pages, selecting tokens by embedding coverage beats pixel saliency below a 50%
|
|
169
|
+
budget — and 256 random probe directions do as well as the designed selector
|
|
170
|
+
([RESULTS.md](docs/RESULTS.md#token-selection-on-dense-pages---the-stage-ii-controls)).
|
|
171
|
+
|
|
172
|
+
### Memory at query time
|
|
173
|
+
|
|
174
|
+
This part holds under both experiments. A compressed index is worth nothing if
|
|
175
|
+
searching it expands the vectors back to float32, which costs 32x the index and
|
|
176
|
+
is what the obvious implementation does. Scoring here runs in blocks bounded by
|
|
177
|
+
a memory budget, so peak RAM is set by that budget rather than by the size of
|
|
178
|
+
the corpus — 239 MB at the 256 MB default, whether the index holds sixty pages
|
|
179
|
+
or a million. See [docs/IMPROVEMENTS.md](docs/IMPROVEMENTS.md).
|
|
180
|
+
|
|
181
|
+
### Reading either table
|
|
182
|
+
|
|
183
|
+
| column | what it means |
|
|
184
|
+
|---|---|
|
|
185
|
+
| `Tok/pg` | vectors actually stored per page |
|
|
186
|
+
| `KB/pg` | index bytes per page |
|
|
187
|
+
| `Compr.` | vs. the uncompressed float32 index |
|
|
188
|
+
| `nDCG@5`, `R@1` | absolute retrieval quality against ground truth |
|
|
189
|
+
| `Retain` | nDCG@5 as a fraction of the uncompressed baseline's |
|
|
190
|
+
| `Tau` | Kendall tau against the **baseline's own ranking** |
|
|
191
|
+
|
|
192
|
+
`Tau` is the honest measure of compression damage: it does not ask whether the model
|
|
193
|
+
was right, only whether compressing changed its mind.
|
|
194
|
+
|
|
195
|
+
## Demo app
|
|
196
|
+
|
|
197
|
+
A single-page Gradio demo (`app.py`) shows the compression happening on one
|
|
198
|
+
uploaded document — built for the Stage-I presentation, not as a production RAG
|
|
199
|
+
service.
|
|
200
|
+
|
|
201
|
+
Double-click `run_demo.bat`, or:
|
|
202
|
+
|
|
203
|
+
```bash
|
|
204
|
+
pip install -e ".[vlm,app]" && pip install gradio
|
|
205
|
+
python app.py # http://127.0.0.1:7860
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
Upload a PDF or scan, press **Compress Document**, and it reports the real token
|
|
209
|
+
counts, the real byte sizes, and renders the pipeline's actual keep-mask over the
|
|
210
|
+
page. Every figure is read back off the arrays the pipeline produced.
|
|
211
|
+
|
|
212
|
+
It runs **locally**: Hugging Face now requires a PRO subscription to host any live
|
|
213
|
+
Gradio Space, free CPU hardware included. `app.py` and `scripts/deploy_space.py`
|
|
214
|
+
are Space-ready for whenever that changes — see [docs/DEPLOY.md](docs/DEPLOY.md).
|
|
215
|
+
|
|
216
|
+
The demo refuses to run on the `synthetic` backend: showing a regression harness's
|
|
217
|
+
output as a demonstration result would misrepresent the project.
|
|
218
|
+
|
|
219
|
+
## Install
|
|
220
|
+
|
|
221
|
+
```bash
|
|
222
|
+
python -m venv .venv
|
|
223
|
+
.venv/Scripts/activate # Windows; source .venv/bin/activate elsewhere
|
|
224
|
+
pip install -e ".[vlm,bench,app,dev]"
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
`torch` is CPU-only by default. For a GPU box install the CUDA build first
|
|
228
|
+
(`pip install torch --index-url https://download.pytorch.org/whl/cu121`).
|
|
229
|
+
|
|
230
|
+
## Quick start
|
|
231
|
+
|
|
232
|
+
```bash
|
|
233
|
+
# 1. build a corpus of scanned-looking documents with ground-truth queries
|
|
234
|
+
optivision make-corpus data/corpus --docs 30 --pages 2
|
|
235
|
+
|
|
236
|
+
# 2. index it (colsmol.yaml = real 256M model, runs on a CPU laptop)
|
|
237
|
+
optivision index data/corpus/pdfs -c configs/colsmol.yaml
|
|
238
|
+
|
|
239
|
+
# 3. search
|
|
240
|
+
optivision search "renewal of vehicle insurance policy" -c configs/colsmol.yaml
|
|
241
|
+
|
|
242
|
+
# 4. see exactly which patches were dropped
|
|
243
|
+
optivision explain data/corpus/pdfs -c configs/colsmol.yaml --out reports/figures
|
|
244
|
+
|
|
245
|
+
# 5. run the full ablation table
|
|
246
|
+
# --cache stores the encode pass, so re-running to add or change a row
|
|
247
|
+
# takes seconds instead of re-encoding the whole corpus
|
|
248
|
+
optivision bench data/corpus/pdfs data/corpus/queries.json \
|
|
249
|
+
-c configs/colsmol.yaml --out reports/colsmol --sweep \
|
|
250
|
+
--cache data/cache/colsmol.npz
|
|
251
|
+
|
|
252
|
+
# 6. demo UI
|
|
253
|
+
streamlit run app/streamlit_app.py -- --config configs/colsmol.yaml
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
Point step 2 at your own folder of PDFs or scans to index real documents.
|
|
257
|
+
|
|
258
|
+
For the reference numbers — ColPali-v1.3 over ViDoRe splits — you need a GPU.
|
|
259
|
+
Two paths, same config and same code:
|
|
260
|
+
`notebooks/vidore_colpali_bench.ipynb` (Colab or Kaggle, interactive) and
|
|
261
|
+
`scripts/run_bench_gpu.sh` (any Ubuntu + CUDA box, unattended, all four splits).
|
|
262
|
+
See [docs/GPU_RUN.md](docs/GPU_RUN.md) — a rented GPU costs about $0.25 for the
|
|
263
|
+
whole benchmark, which is less trouble than the free tiers' quotas.
|
|
264
|
+
|
|
265
|
+
## Configurations
|
|
266
|
+
|
|
267
|
+
| config | encoder | needs | use for |
|
|
268
|
+
|---|---|---|---|
|
|
269
|
+
| `configs/synthetic.yaml` | hashed stand-in | nothing | tests, CI, first smoke run |
|
|
270
|
+
| `configs/colsmol.yaml` | ColSmol-256M | ~0.5 GB download, CPU ok | **default** — real results on a laptop |
|
|
271
|
+
| `configs/colqwen2.yaml`* | ColQwen2-2B | GPU | stronger quality |
|
|
272
|
+
| `configs/colpali.yaml` | ColPali-v1.3 | GPU (~6 GB) | reference model from the paper |
|
|
273
|
+
| `configs/qdrant.yaml` | ColSmol + Qdrant | optional server | deployment-shaped storage |
|
|
274
|
+
|
|
275
|
+
\* generate with `optivision init-config configs/colqwen2.yaml --backend colqwen2`.
|
|
276
|
+
|
|
277
|
+
The **synthetic** encoder deserves a warning: it makes the whole pipeline runnable
|
|
278
|
+
with no downloads and its retrieval genuinely works, so it is a real correctness
|
|
279
|
+
harness — but its hashed word vectors are near-orthogonal, which makes MaxSim behave
|
|
280
|
+
like exact matching. Quality metrics there saturate at 1.0 and **must not be reported
|
|
281
|
+
as results**. Use it to test plumbing; use ColSmol or ColPali for numbers.
|
|
282
|
+
|
|
283
|
+
## Layout
|
|
284
|
+
|
|
285
|
+
```
|
|
286
|
+
src/optivision/
|
|
287
|
+
types.py PageEncoding → PrunedPage → CompressedPage
|
|
288
|
+
config.py every experimental knob, YAML-loadable
|
|
289
|
+
encoders/
|
|
290
|
+
colvlm.py ColPali / ColQwen2 / ColSmol via colpali-engine
|
|
291
|
+
synthetic.py offline stand-in (tests only)
|
|
292
|
+
pruning/
|
|
293
|
+
saliency.py ink density + edge energy per patch
|
|
294
|
+
spatial.py keep-mask construction, dilation, budgets
|
|
295
|
+
redundancy.py greedy near-duplicate collapsing
|
|
296
|
+
compression/
|
|
297
|
+
binary.py bit packing, asymmetric/symmetric MaxSim
|
|
298
|
+
index/
|
|
299
|
+
numpy_index.py exact brute-force MaxSim (the reference)
|
|
300
|
+
qdrant_index.py Qdrant multivector + binary quantization
|
|
301
|
+
pipeline.py OptiVisionRAG.build() / .search()
|
|
302
|
+
bench.py encode-once, replay-every-variant ablation harness
|
|
303
|
+
corpus.py synthetic corpus generator + ViDoRe loader
|
|
304
|
+
metrics.py nDCG / recall / MRR / Kendall tau / storage
|
|
305
|
+
viz.py keep-mask and saliency figures
|
|
306
|
+
app/streamlit_app.py demo UI
|
|
307
|
+
docs/ architecture, results, viva notes
|
|
308
|
+
notebooks/ Colab / Kaggle runner for the ColPali benchmark
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
## Testing
|
|
312
|
+
|
|
313
|
+
```bash
|
|
314
|
+
pytest # 75 tests, no model download needed
|
|
315
|
+
ruff check src tests app
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
The suite covers saliency behaviour on blank/grey/inked pages, keep-mask budgets,
|
|
319
|
+
redundancy clustering invariants, bit-packing round-trips, MaxSim segment maths on
|
|
320
|
+
variable-length pages, index save/load, Qdrant multivector round-trips, and a full
|
|
321
|
+
index→search→evaluate loop.
|
|
322
|
+
|
|
323
|
+
## Improvements
|
|
324
|
+
|
|
325
|
+
[docs/IMPROVEMENTS.md](docs/IMPROVEMENTS.md) records five changes made after the
|
|
326
|
+
pipeline was working — bounded query-time memory, an int8 quantizer that uses
|
|
327
|
+
the range it pays for, a single-allocation decode path, a Qdrant stats fix, and
|
|
328
|
+
one quadratic loop — each with the measurement behind it, plus what was
|
|
329
|
+
reviewed and deliberately left alone.
|
|
330
|
+
|
|
331
|
+
## Honest limitations
|
|
332
|
+
|
|
333
|
+
- **Encoder speed on CPU.** ColSmol-256M takes ~30 s/page on this laptop (no GPU).
|
|
334
|
+
Pruning and quantization together take ~15 ms/page — the compression is free
|
|
335
|
+
relative to encoding, but building a large index needs a GPU.
|
|
336
|
+
- **The bundled corpus is generated**, not scanned. It has the right whitespace
|
|
337
|
+
profile and gives exact ground truth, but real scans have noise, skew and bleed-through.
|
|
338
|
+
`optivision fetch-vidore` pulls the real ViDoRe benchmark for that reason.
|
|
339
|
+
- **Redundancy merging changes the vectors**, not just their count. It is nearly free
|
|
340
|
+
under MaxSim but it is not lossless — the ablation reports both stages separately so
|
|
341
|
+
the cost is visible.
|
|
342
|
+
- **Qdrant local mode** is convenient but not a performance claim; latency numbers
|
|
343
|
+
come from the exact numpy index so they are not confounded by ANN recall.
|
|
344
|
+
|
|
345
|
+
## References
|
|
346
|
+
|
|
347
|
+
- Faysse et al., *ColPali: Efficient Document Retrieval with Vision Language Models* (2024)
|
|
348
|
+
- Khattab & Zaharia, *ColBERT* (2020) — late interaction / MaxSim
|
|
349
|
+
- Qdrant multivector + binary quantization documentation
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "optivision-rag"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "OptiVision RAG: Extreme Token Compression for Vision-Language Models"
|
|
9
|
+
readme = "PACKAGE.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = ["rag", "colpali", "colqwen", "late-interaction", "vision-language", "retrieval", "quantization", "token-pruning", "qdrant"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.10",
|
|
20
|
+
"Programming Language :: Python :: 3.11",
|
|
21
|
+
"Programming Language :: Python :: 3.12",
|
|
22
|
+
"Programming Language :: Python :: 3.13",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
24
|
+
"Topic :: Text Processing :: Indexing",
|
|
25
|
+
]
|
|
26
|
+
authors = [
|
|
27
|
+
{ name = "T. Rithik Krishna" },
|
|
28
|
+
{ name = "Amgovath Navanitha" },
|
|
29
|
+
{ name = "Badavath Akhila" },
|
|
30
|
+
]
|
|
31
|
+
dependencies = [
|
|
32
|
+
"numpy>=1.26,<3",
|
|
33
|
+
"pillow>=10",
|
|
34
|
+
"pypdfium2>=4",
|
|
35
|
+
"qdrant-client>=1.10",
|
|
36
|
+
"pyyaml>=6",
|
|
37
|
+
"typer>=0.12",
|
|
38
|
+
"rich>=13",
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
[project.optional-dependencies]
|
|
42
|
+
vlm = ["torch>=2.2", "colpali-engine>=0.3.10", "transformers>=4.46"]
|
|
43
|
+
corpus = ["reportlab>=4"]
|
|
44
|
+
bench = ["reportlab>=4", "datasets>=2.19"]
|
|
45
|
+
app = ["streamlit>=1.34"]
|
|
46
|
+
dev = ["pytest>=8", "ruff>=0.5"]
|
|
47
|
+
|
|
48
|
+
[project.urls]
|
|
49
|
+
Homepage = "https://github.com/Daemon-VI/optivision-rag"
|
|
50
|
+
Repository = "https://github.com/Daemon-VI/optivision-rag"
|
|
51
|
+
Issues = "https://github.com/Daemon-VI/optivision-rag/issues"
|
|
52
|
+
|
|
53
|
+
[project.scripts]
|
|
54
|
+
optivision = "optivision.cli:app"
|
|
55
|
+
|
|
56
|
+
[tool.setuptools.packages.find]
|
|
57
|
+
where = ["src"]
|
|
58
|
+
|
|
59
|
+
[tool.setuptools.package-data]
|
|
60
|
+
optivision = ["presets/*.yaml"]
|
|
61
|
+
|
|
62
|
+
[tool.setuptools.dynamic]
|
|
63
|
+
version = { attr = "optivision.__version__" }
|
|
64
|
+
|
|
65
|
+
[tool.pytest.ini_options]
|
|
66
|
+
testpaths = ["tests"]
|
|
67
|
+
filterwarnings = ["ignore::DeprecationWarning"]
|
|
68
|
+
|
|
69
|
+
[tool.ruff]
|
|
70
|
+
line-length = 100
|
|
71
|
+
target-version = "py310"
|