optivision-rag 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. optivision_rag-0.1.1/LICENSE +21 -0
  2. optivision_rag-0.1.1/PACKAGE.md +90 -0
  3. optivision_rag-0.1.1/PKG-INFO +136 -0
  4. optivision_rag-0.1.1/README.md +349 -0
  5. optivision_rag-0.1.1/pyproject.toml +71 -0
  6. optivision_rag-0.1.1/setup.cfg +4 -0
  7. optivision_rag-0.1.1/src/optivision/__init__.py +44 -0
  8. optivision_rag-0.1.1/src/optivision/bench.py +587 -0
  9. optivision_rag-0.1.1/src/optivision/cli.py +344 -0
  10. optivision_rag-0.1.1/src/optivision/compression/__init__.py +170 -0
  11. optivision_rag-0.1.1/src/optivision/compression/binary.py +116 -0
  12. optivision_rag-0.1.1/src/optivision/compression/lloyd2.py +229 -0
  13. optivision_rag-0.1.1/src/optivision/config.py +200 -0
  14. optivision_rag-0.1.1/src/optivision/corpus.py +359 -0
  15. optivision_rag-0.1.1/src/optivision/demo.py +369 -0
  16. optivision_rag-0.1.1/src/optivision/encoders/__init__.py +33 -0
  17. optivision_rag-0.1.1/src/optivision/encoders/base.py +54 -0
  18. optivision_rag-0.1.1/src/optivision/encoders/colvlm.py +388 -0
  19. optivision_rag-0.1.1/src/optivision/encoders/synthetic.py +170 -0
  20. optivision_rag-0.1.1/src/optivision/index/__init__.py +51 -0
  21. optivision_rag-0.1.1/src/optivision/index/base.py +39 -0
  22. optivision_rag-0.1.1/src/optivision/index/numpy_index.py +272 -0
  23. optivision_rag-0.1.1/src/optivision/index/qdrant_index.py +261 -0
  24. optivision_rag-0.1.1/src/optivision/ingest.py +85 -0
  25. optivision_rag-0.1.1/src/optivision/metrics.py +172 -0
  26. optivision_rag-0.1.1/src/optivision/pipeline.py +286 -0
  27. optivision_rag-0.1.1/src/optivision/presets/colpali.yaml +27 -0
  28. optivision_rag-0.1.1/src/optivision/presets/colsmol.yaml +25 -0
  29. optivision_rag-0.1.1/src/optivision/presets/qdrant.yaml +18 -0
  30. optivision_rag-0.1.1/src/optivision/presets/synthetic.yaml +25 -0
  31. optivision_rag-0.1.1/src/optivision/pruning/__init__.py +130 -0
  32. optivision_rag-0.1.1/src/optivision/pruning/codebook.py +127 -0
  33. optivision_rag-0.1.1/src/optivision/pruning/redundancy.py +78 -0
  34. optivision_rag-0.1.1/src/optivision/pruning/saliency.py +86 -0
  35. optivision_rag-0.1.1/src/optivision/pruning/spatial.py +78 -0
  36. optivision_rag-0.1.1/src/optivision/types.py +146 -0
  37. optivision_rag-0.1.1/src/optivision/viz.py +100 -0
  38. optivision_rag-0.1.1/src/optivision_rag.egg-info/PKG-INFO +136 -0
  39. optivision_rag-0.1.1/src/optivision_rag.egg-info/SOURCES.txt +49 -0
  40. optivision_rag-0.1.1/src/optivision_rag.egg-info/dependency_links.txt +1 -0
  41. optivision_rag-0.1.1/src/optivision_rag.egg-info/entry_points.txt +2 -0
  42. optivision_rag-0.1.1/src/optivision_rag.egg-info/requires.txt +26 -0
  43. optivision_rag-0.1.1/src/optivision_rag.egg-info/top_level.txt +1 -0
  44. optivision_rag-0.1.1/tests/test_compression.py +438 -0
  45. optivision_rag-0.1.1/tests/test_demo.py +251 -0
  46. optivision_rag-0.1.1/tests/test_index.py +206 -0
  47. optivision_rag-0.1.1/tests/test_pipeline.py +316 -0
  48. optivision_rag-0.1.1/tests/test_presets.py +27 -0
  49. optivision_rag-0.1.1/tests/test_pruning.py +152 -0
  50. optivision_rag-0.1.1/tests/test_qdrant.py +71 -0
  51. optivision_rag-0.1.1/tests/test_release_versions.py +13 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 T. Rithik Krishna, Amgovath Navanitha, Badavath Akhila
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,90 @@
1
+ # OptiVision RAG
2
+
3
+ **Extreme token compression for vision-language document retrieval.**
4
+
5
+ ColPali-style models search scanned documents as images, with no OCR. They store
6
+ one 128-dimensional float vector per image patch, so a single page costs about
7
+ 512 KB of index and a million pages cost about half a terabyte.
8
+
9
+ OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
10
+ unmodified and retrieval is still late-interaction MaxSim. Only the number and the
11
+ size of the stored vectors change:
12
+
13
+ ```
14
+ page image
15
+ ├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
16
+ ├─ 1. spatial pruning ──► drop patches on blank paper
17
+ ├─ 2. redundancy prune ─► collapse near-duplicate patches
18
+ ├─ 3. binary quantize ──► 128 floats (512 B) -> 128 bits (16 B)
19
+ └─ index ───────────────► Qdrant MaxSim, or an exact numpy index
20
+ ```
21
+
22
+ | setting | encoder / corpus | compression | nDCG@5 retained |
23
+ |---|---|---|---|
24
+ | prune + binary | ColSmol-256M, generated pages | 113.5x | 86.7% |
25
+ | prune + int8 | ColSmol-256M, generated pages | 14.2x | 96.0% |
26
+ | prune + binary | ColPali-v1.3, ViDoRe (4 splits) | 53-60x | 94.6-103.4% |
27
+
28
+ Full results, the ablations and the analysis are in the
29
+ [repository](https://github.com/Daemon-VI/optivision-rag).
30
+
31
+ ## Install
32
+
33
+ Python 3.10 or newer.
34
+
35
+ ```bash
36
+ pip install optivision-rag # core pipeline + CLI
37
+ pip install "optivision-rag[corpus]" # + synthetic test-corpus generator
38
+ pip install "optivision-rag[vlm]" # + real encoders (torch, colpali-engine)
39
+ pip install "optivision-rag[vlm,corpus,bench]"
40
+ ```
41
+
42
+ Or run the CLI through npm without touching pip yourself (it still needs Python
43
+ 3.10+ on the machine):
44
+
45
+ ```bash
46
+ npx optivision-rag --help
47
+ ```
48
+
49
+ ## Quick start (offline, no model download)
50
+
51
+ The `synthetic` preset uses a deterministic stand-in encoder, so the whole pipeline
52
+ runs in seconds on any laptop. Use it to try the tool, not for real results.
53
+
54
+ ```bash
55
+ optivision make-corpus data/corpus --docs 10 --pages 2
56
+ optivision index data/corpus/pdfs -c synthetic
57
+ optivision search "renewal of vehicle insurance policy" -c synthetic
58
+ optivision stats -c synthetic
59
+ ```
60
+
61
+ ## With a real model
62
+
63
+ Install the `vlm` extra, then switch the preset. `colsmol` is a 256M-parameter
64
+ model that runs on a CPU laptop; `colpali` wants a GPU.
65
+
66
+ ```bash
67
+ optivision index path/to/your/pdfs -c colsmol
68
+ optivision search "fire safety audit memorandum" -c colsmol
69
+ optivision explain path/to/your/pdfs -c colsmol --out figures
70
+ ```
71
+
72
+ `-c` takes a bundled preset name (`synthetic`, `colsmol`, `colpali`, `qdrant`) or a
73
+ path to your own YAML file. `optivision init-config my.yaml` writes one pre-filled
74
+ with every default.
75
+
76
+ ## Python API
77
+
78
+ ```python
79
+ from optivision import Config, OptiVisionRAG
80
+
81
+ rag = OptiVisionRAG(Config.load("colsmol"))
82
+ report = rag.build("path/to/pdfs")
83
+ print(report.compression_ratio)
84
+ print(rag.search("office memorandum on fire safety audit").hits[0].ref.page_id)
85
+ ```
86
+
87
+ ## Authors
88
+
89
+ T. Rithik Krishna, Amgovath Navanitha and Badavath Akhila, Mahatma Gandhi Institute
90
+ of Technology, Hyderabad. MIT licensed.
@@ -0,0 +1,136 @@
1
+ Metadata-Version: 2.4
2
+ Name: optivision-rag
3
+ Version: 0.1.1
4
+ Summary: OptiVision RAG: Extreme Token Compression for Vision-Language Models
5
+ Author: T. Rithik Krishna, Amgovath Navanitha, Badavath Akhila
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/Daemon-VI/optivision-rag
8
+ Project-URL: Repository, https://github.com/Daemon-VI/optivision-rag
9
+ Project-URL: Issues, https://github.com/Daemon-VI/optivision-rag/issues
10
+ Keywords: rag,colpali,colqwen,late-interaction,vision-language,retrieval,quantization,token-pruning,qdrant
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Text Processing :: Indexing
21
+ Requires-Python: >=3.10
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: numpy<3,>=1.26
25
+ Requires-Dist: pillow>=10
26
+ Requires-Dist: pypdfium2>=4
27
+ Requires-Dist: qdrant-client>=1.10
28
+ Requires-Dist: pyyaml>=6
29
+ Requires-Dist: typer>=0.12
30
+ Requires-Dist: rich>=13
31
+ Provides-Extra: vlm
32
+ Requires-Dist: torch>=2.2; extra == "vlm"
33
+ Requires-Dist: colpali-engine>=0.3.10; extra == "vlm"
34
+ Requires-Dist: transformers>=4.46; extra == "vlm"
35
+ Provides-Extra: corpus
36
+ Requires-Dist: reportlab>=4; extra == "corpus"
37
+ Provides-Extra: bench
38
+ Requires-Dist: reportlab>=4; extra == "bench"
39
+ Requires-Dist: datasets>=2.19; extra == "bench"
40
+ Provides-Extra: app
41
+ Requires-Dist: streamlit>=1.34; extra == "app"
42
+ Provides-Extra: dev
43
+ Requires-Dist: pytest>=8; extra == "dev"
44
+ Requires-Dist: ruff>=0.5; extra == "dev"
45
+ Dynamic: license-file
46
+
47
+ # OptiVision RAG
48
+
49
+ **Extreme token compression for vision-language document retrieval.**
50
+
51
+ ColPali-style models search scanned documents as images, with no OCR. They store
52
+ one 128-dimensional float vector per image patch, so a single page costs about
53
+ 512 KB of index and a million pages cost about half a terabyte.
54
+
55
+ OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
56
+ unmodified and retrieval is still late-interaction MaxSim. Only the number and the
57
+ size of the stored vectors change:
58
+
59
+ ```
60
+ page image
61
+ ├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
62
+ ├─ 1. spatial pruning ──► drop patches on blank paper
63
+ ├─ 2. redundancy prune ─► collapse near-duplicate patches
64
+ ├─ 3. binary quantize ──► 128 floats (512 B) -> 128 bits (16 B)
65
+ └─ index ───────────────► Qdrant MaxSim, or an exact numpy index
66
+ ```
67
+
68
+ | setting | encoder / corpus | compression | nDCG@5 retained |
69
+ |---|---|---|---|
70
+ | prune + binary | ColSmol-256M, generated pages | 113.5x | 86.7% |
71
+ | prune + int8 | ColSmol-256M, generated pages | 14.2x | 96.0% |
72
+ | prune + binary | ColPali-v1.3, ViDoRe (4 splits) | 53-60x | 94.6-103.4% |
73
+
74
+ Full results, the ablations and the analysis are in the
75
+ [repository](https://github.com/Daemon-VI/optivision-rag).
76
+
77
+ ## Install
78
+
79
+ Python 3.10 or newer.
80
+
81
+ ```bash
82
+ pip install optivision-rag # core pipeline + CLI
83
+ pip install "optivision-rag[corpus]" # + synthetic test-corpus generator
84
+ pip install "optivision-rag[vlm]" # + real encoders (torch, colpali-engine)
85
+ pip install "optivision-rag[vlm,corpus,bench]"
86
+ ```
87
+
88
+ Or run the CLI through npm without touching pip yourself (it still needs Python
89
+ 3.10+ on the machine):
90
+
91
+ ```bash
92
+ npx optivision-rag --help
93
+ ```
94
+
95
+ ## Quick start (offline, no model download)
96
+
97
+ The `synthetic` preset uses a deterministic stand-in encoder, so the whole pipeline
98
+ runs in seconds on any laptop. Use it to try the tool, not for real results.
99
+
100
+ ```bash
101
+ optivision make-corpus data/corpus --docs 10 --pages 2
102
+ optivision index data/corpus/pdfs -c synthetic
103
+ optivision search "renewal of vehicle insurance policy" -c synthetic
104
+ optivision stats -c synthetic
105
+ ```
106
+
107
+ ## With a real model
108
+
109
+ Install the `vlm` extra, then switch the preset. `colsmol` is a 256M-parameter
110
+ model that runs on a CPU laptop; `colpali` wants a GPU.
111
+
112
+ ```bash
113
+ optivision index path/to/your/pdfs -c colsmol
114
+ optivision search "fire safety audit memorandum" -c colsmol
115
+ optivision explain path/to/your/pdfs -c colsmol --out figures
116
+ ```
117
+
118
+ `-c` takes a bundled preset name (`synthetic`, `colsmol`, `colpali`, `qdrant`) or a
119
+ path to your own YAML file. `optivision init-config my.yaml` writes one pre-filled
120
+ with every default.
121
+
122
+ ## Python API
123
+
124
+ ```python
125
+ from optivision import Config, OptiVisionRAG
126
+
127
+ rag = OptiVisionRAG(Config.load("colsmol"))
128
+ report = rag.build("path/to/pdfs")
129
+ print(report.compression_ratio)
130
+ print(rag.search("office memorandum on fire safety audit").hits[0].ref.page_id)
131
+ ```
132
+
133
+ ## Authors
134
+
135
+ T. Rithik Krishna, Amgovath Navanitha and Badavath Akhila, Mahatma Gandhi Institute
136
+ of Technology, Hyderabad. MIT licensed.
@@ -0,0 +1,349 @@
1
+ ---
2
+ title: OptiVision RAG
3
+ emoji: 🗜️
4
+ colorFrom: indigo
5
+ colorTo: green
6
+ sdk: gradio
7
+ sdk_version: 6.24.0
8
+ python_version: "3.11"
9
+ app_file: app.py
10
+ pinned: false
11
+ license: mit
12
+ short_description: Visual-token pruning and binary quantization for document retrieval
13
+ ---
14
+
15
+ # OptiVision RAG
16
+
17
+ **Extreme token compression for vision-language document retrieval.**
18
+
19
+ > The YAML block above is Hugging Face Space metadata. It is ignored by GitHub
20
+ > and by every tool in this repo; see [docs/DEPLOY.md](docs/DEPLOY.md) for how
21
+ > the Space is set up.
22
+
23
+ ---
24
+
25
+ ## The problem
26
+
27
+ ColPali-style models search scanned documents *as images* — no OCR, so they work on
28
+ old copies, seals, stamps and handwriting that text pipelines fail on. They do it by
29
+ splitting each page into patches and keeping **one vector per patch**.
30
+
31
+ That is also the problem. A single page becomes ~800–1000 vectors of 128 float32
32
+ dimensions:
33
+
34
+ ```
35
+ 1 page = 1024 vectors x 128 dims x 4 bytes = 512 KB
36
+ 1 million pages = 512 GB
37
+ ```
38
+
39
+ Half a terabyte of RAM-resident vector index for a corpus that a records office
40
+ would consider small. The accuracy is excellent; the storage is what stops it
41
+ being deployable.
42
+
43
+ ## What this project does
44
+
45
+ OptiVision RAG shrinks the **index**, not the model. The published checkpoint runs
46
+ unmodified, the query path is untouched, and the retrieval is still late-interaction
47
+ MaxSim — only the number and the size of the stored vectors change.
48
+
49
+ ```
50
+ page image
51
+ │
52
+ ├─ VLM encoder ─────────► ~1000 patch vectors (model unchanged)
53
+ │
54
+ ├─ 1. spatial pruning ──► drop patches on blank paper
55
+ │
56
+ ├─ 2. redundancy prune ─► collapse near-duplicate patches
57
+ │
58
+ ├─ 3. binary quantize ──► 128 floats (512 B) → 128 bits (16 B)
59
+ │
60
+ └─ index ───────────────► Qdrant MaxSim, or an exact numpy index
61
+ ```
62
+
63
+ **1. Spatial pruning.** Most of a document page is paper. Per patch we measure ink
64
+ density (how much is darker than the estimated paper background) and edge energy
65
+ (local contrast — glyph strokes, rules, stamp borders), and drop the cells that carry
66
+ neither. The estimate runs on the pixels in microseconds, before the model, and needs
67
+ no attention maps or second forward pass. The paper background is estimated per page
68
+ (95th percentile) rather than assumed white, so a yellowed photocopy is not read as a
69
+ fully inked page.
70
+
71
+ **2. Redundancy pruning.** Spatial pruning removes patches with *no* content;
72
+ this removes patches with *duplicate* content — the interior of a filled table cell,
73
+ the middle of a thick rule. They look salient to a pixel detector but add nothing to a
74
+ MaxSim score, because MaxSim already takes a max over near-identical vectors. A greedy
75
+ single pass collapses each cluster to its renormalised centroid.
76
+
77
+ **3. Binary quantization.** These vectors are L2-normalised and roughly zero-centred
78
+ per dimension, so the sign bits keep the orthant and discard only the within-orthant
79
+ position. Crucially, the distortion is nearly the same for every document vector, so
80
+ it moves scores much more than it moves *ranking*. Queries stay in float32 and are
81
+ scored against ±1 document codes (asymmetric scoring): a corpus has millions of
82
+ vectors, a query has twenty, so query precision is the cheapest thing to keep.
83
+
84
+ The three stages multiply — under E1, pruning ~3.5× × quantization 32× ≈ **100×+** off
85
+ the index. On real ViDoRe pages pruning buys less, and the product lands in the 53-60× range; both
86
+ figures are in *Results* below.
87
+
88
+ ## Results
89
+
90
+ We ran the ablation twice, and the two runs disagree. That disagreement is the
91
+ result — read both tables before drawing a conclusion from either.
92
+
93
+ ### E1 — ColSmol-256M, generated pages
94
+
95
+ 60 pages / 72 queries on CPU. Full table and analysis in
96
+ [docs/RESULTS.md](docs/RESULTS.md); reproduce with `make bench`.
97
+
98
+ | variant | tok/pg | KB/pg | compression | nDCG@5 | retained | tau |
99
+ |---|---|---|---|---|---|---|
100
+ | baseline-float32 | 875.0 | 448.00 | 1.0x | 0.7823 | 100.0% | 1.000 |
101
+ | spatial-only | 356.1 | 182.32 | 2.5x | 0.7602 | 97.2% | 0.935 |
102
+ | spatial+redundancy | 246.8 | 126.34 | 3.5x | 0.7519 | 96.1% | 0.866 |
103
+ | int8-only | 875.0 | 112.00 | 4.0x | 0.7877 | 100.7% | 0.972 |
104
+ | **prune+int8** | **246.8** | **31.59** | **14.2x** | **0.7511** | **96.0%** | 0.864 |
105
+ | binary-only | 875.0 | 14.00 | 32.0x | 0.6875 | 87.9% | 0.585 |
106
+ | **optivision** | **246.8** | **3.95** | **113.5x** | **0.6782** | **86.7%** | 0.606 |
107
+ | optivision-aggressive | 186.3 | 2.98 | 150.3x | 0.6680 | 85.4% | 0.602 |
108
+
109
+ **Under this encoder, the two halves of the proposal do not contribute equally.**
110
+ Pruning is nearly free — 3.5x fewer vectors for 3.9% of nDCG@5.
111
+ Binary quantization is where the quality goes — 12.1% lost *without dropping a single
112
+ token*, while int8 gives 4x for 0.5%. Pruning harder barely matters once binary is in
113
+ play (keeping the top 10% of patches scores the same as keeping the top 50%).
114
+
115
+ So under E1 the pipeline has two defensible operating points:
116
+
117
+ - **smallest index** — prune + binary: **113.5x** at 86.7% of baseline nDCG@5
118
+ - **best quality per byte** — prune + int8: **14.2x** at 96.0%, with a Kendall tau of
119
+ 0.864 against 0.866 for pruning alone, so int8 adds no measurable ranking distortion on
120
+ top of the pruning
121
+
122
+ 448 KB/page becomes 3.95 KB/page: a million-page archive drops from ~448 GB to ~4 GB.
123
+ At the quality-first setting it is ~32 GB.
124
+
125
+ ### E2 — ColPali-v1.3, real ViDoRe pages
126
+
127
+ **The reference model disagrees with every headline above.** The same ablation on
128
+ ColPali-v1.3 over four ViDoRe splits (1,780 pages, 1,325 queries) reaches 53-60x at
129
+ 94.6-103.4% of baseline nDCG@5 — less compression, far less loss. Two of the three findings above do not transfer:
130
+ pruning buys only 1.7-1.9x on real pages rather than 3.5x, and binary quantization
131
+ costs 0-3.7% rather than 12.1%. Full table and analysis in
132
+ [docs/RESULTS.md](docs/RESULTS.md#the-same-table-on-colpali-3b-and-real-vidore-pages);
133
+ raw reports in [reports/](reports/); reproduce with `bash scripts/run_bench_gpu.sh` on
134
+ any free T4.
135
+
136
+ The takeaway is not "E2 supersedes E1". It is that **the per-stage attribution is a
137
+ property of the encoder and the corpus, not of the compression layer** — the same code
138
+ paid its quality in different places under the two. A third run separates those two
139
+ variables; see below. A compression ratio reported without naming the encoder and the
140
+ corpus it was measured on is not enough information to act on, which is what the paper
141
+ argues.
142
+
143
+ ### E3 — ColPali-3B, generated pages
144
+
145
+ The missing cell: the reference encoder over E1's own corpus, regenerated at seed 7 so
146
+ the pages are the same ones. It splits the reversal in two.
147
+
148
+ | | corpus fixed, encoder swapped | encoder fixed, corpus swapped |
149
+ |---|---|---|
150
+ | one-bit codec costs | E1 → E3: **12.1 → 1.6 points** | E3 → E2: 1.6 → 3.7 points |
151
+ | pruning buys | E1 → E3: 3.55x → 4.20x | E3 → E2: **4.20x → 1.85x** |
152
+
153
+ **The corpus sets what pruning buys.** Run `python scripts/compare_regimes.py` to
154
+ print it from the benchmark files. E3 is also the best operating point in the project
155
+ (98.7% retention at 134.5x) and the least representative one.
156
+
157
+ The codec half is not an encoder property. Per query, a sign code adds the same ~3% of
158
+ score noise under both encoders, and it flips only queries whose float margin over
159
+ the best competitor is smaller than that noise — under 1% of 945 real ViDoRe queries
160
+ above twice the noise, against a third of those below half.
161
+ E1's precise queries are decided by three digits at a 0.05% margin, which is why the
162
+ small encoder looks fragile; ColPali cannot read those digits at 448 px and never wins
163
+ the queries it is credited with not losing (precise R@1 0.250, floor 0.200). Enlarge the
164
+ code and its one-bit cost appears. See [RESULTS.md](docs/RESULTS.md#e3---colpali-3b-on-the-generated-corpus)
165
+ and [REVIEW-2026-08-21.md](docs/REVIEW-2026-08-21.md); `scripts/tau_audit.py` explains why
166
+ E1's tau of 0.585 and E2's 0.527 were never the same measurement.
167
+
168
+ On dense pages, selecting tokens by embedding coverage beats pixel saliency below a 50%
169
+ budget — and 256 random probe directions do as well as the designed selector
170
+ ([RESULTS.md](docs/RESULTS.md#token-selection-on-dense-pages---the-stage-ii-controls)).
171
+
172
+ ### Memory at query time
173
+
174
+ This part holds under both experiments. A compressed index is worth nothing if
175
+ searching it expands the vectors back to float32, which costs 32x the index and
176
+ is what the obvious implementation does. Scoring here runs in blocks bounded by
177
+ a memory budget, so peak RAM is set by that budget rather than by the size of
178
+ the corpus — 239 MB at the 256 MB default, whether the index holds sixty pages
179
+ or a million. See [docs/IMPROVEMENTS.md](docs/IMPROVEMENTS.md).
180
+
181
+ ### Reading either table
182
+
183
+ | column | what it means |
184
+ |---|---|
185
+ | `Tok/pg` | vectors actually stored per page |
186
+ | `KB/pg` | index bytes per page |
187
+ | `Compr.` | vs. the uncompressed float32 index |
188
+ | `nDCG@5`, `R@1` | absolute retrieval quality against ground truth |
189
+ | `Retain` | nDCG@5 as a fraction of the uncompressed baseline's |
190
+ | `Tau` | Kendall tau against the **baseline's own ranking** |
191
+
192
+ `Tau` is the honest measure of compression damage: it does not ask whether the model
193
+ was right, only whether compressing changed its mind.
194
+
195
+ ## Demo app
196
+
197
+ A single-page Gradio demo (`app.py`) shows the compression happening on one
198
+ uploaded document — built for the Stage-I presentation, not as a production RAG
199
+ service.
200
+
201
+ Double-click `run_demo.bat`, or:
202
+
203
+ ```bash
204
+ pip install -e ".[vlm,app]" && pip install gradio
205
+ python app.py # http://127.0.0.1:7860
206
+ ```
207
+
208
+ Upload a PDF or scan, press **Compress Document**, and it reports the real token
209
+ counts, the real byte sizes, and renders the pipeline's actual keep-mask over the
210
+ page. Every figure is read back off the arrays the pipeline produced.
211
+
212
+ It runs **locally**: Hugging Face now requires a PRO subscription to host any live
213
+ Gradio Space, free CPU hardware included. `app.py` and `scripts/deploy_space.py`
214
+ are Space-ready for whenever that changes — see [docs/DEPLOY.md](docs/DEPLOY.md).
215
+
216
+ The demo refuses to run on the `synthetic` backend: showing a regression harness's
217
+ output as a demonstration result would misrepresent the project.
218
+
219
+ ## Install
220
+
221
+ ```bash
222
+ python -m venv .venv
223
+ .venv/Scripts/activate # Windows; source .venv/bin/activate elsewhere
224
+ pip install -e ".[vlm,bench,app,dev]"
225
+ ```
226
+
227
+ `torch` is CPU-only by default. For a GPU box install the CUDA build first
228
+ (`pip install torch --index-url https://download.pytorch.org/whl/cu121`).
229
+
230
+ ## Quick start
231
+
232
+ ```bash
233
+ # 1. build a corpus of scanned-looking documents with ground-truth queries
234
+ optivision make-corpus data/corpus --docs 30 --pages 2
235
+
236
+ # 2. index it (colsmol.yaml = real 256M model, runs on a CPU laptop)
237
+ optivision index data/corpus/pdfs -c configs/colsmol.yaml
238
+
239
+ # 3. search
240
+ optivision search "renewal of vehicle insurance policy" -c configs/colsmol.yaml
241
+
242
+ # 4. see exactly which patches were dropped
243
+ optivision explain data/corpus/pdfs -c configs/colsmol.yaml --out reports/figures
244
+
245
+ # 5. run the full ablation table
246
+ # --cache stores the encode pass, so re-running to add or change a row
247
+ # takes seconds instead of re-encoding the whole corpus
248
+ optivision bench data/corpus/pdfs data/corpus/queries.json \
249
+ -c configs/colsmol.yaml --out reports/colsmol --sweep \
250
+ --cache data/cache/colsmol.npz
251
+
252
+ # 6. demo UI
253
+ streamlit run app/streamlit_app.py -- --config configs/colsmol.yaml
254
+ ```
255
+
256
+ Point step 2 at your own folder of PDFs or scans to index real documents.
257
+
258
+ For the reference numbers — ColPali-v1.3 over ViDoRe splits — you need a GPU.
259
+ Two paths, same config and same code:
260
+ `notebooks/vidore_colpali_bench.ipynb` (Colab or Kaggle, interactive) and
261
+ `scripts/run_bench_gpu.sh` (any Ubuntu + CUDA box, unattended, all four splits).
262
+ See [docs/GPU_RUN.md](docs/GPU_RUN.md) — a rented GPU costs about $0.25 for the
263
+ whole benchmark, which is less trouble than the free tiers' quotas.
264
+
265
+ ## Configurations
266
+
267
+ | config | encoder | needs | use for |
268
+ |---|---|---|---|
269
+ | `configs/synthetic.yaml` | hashed stand-in | nothing | tests, CI, first smoke run |
270
+ | `configs/colsmol.yaml` | ColSmol-256M | ~0.5 GB download, CPU ok | **default** — real results on a laptop |
271
+ | `configs/colqwen2.yaml`* | ColQwen2-2B | GPU | stronger quality |
272
+ | `configs/colpali.yaml` | ColPali-v1.3 | GPU (~6 GB) | reference model from the paper |
273
+ | `configs/qdrant.yaml` | ColSmol + Qdrant | optional server | deployment-shaped storage |
274
+
275
+ \* generate with `optivision init-config configs/colqwen2.yaml --backend colqwen2`.
276
+
277
+ The **synthetic** encoder deserves a warning: it makes the whole pipeline runnable
278
+ with no downloads and its retrieval genuinely works, so it is a real correctness
279
+ harness — but its hashed word vectors are near-orthogonal, which makes MaxSim behave
280
+ like exact matching. Quality metrics there saturate at 1.0 and **must not be reported
281
+ as results**. Use it to test plumbing; use ColSmol or ColPali for numbers.
282
+
283
+ ## Layout
284
+
285
+ ```
286
+ src/optivision/
287
+ types.py PageEncoding → PrunedPage → CompressedPage
288
+ config.py every experimental knob, YAML-loadable
289
+ encoders/
290
+ colvlm.py ColPali / ColQwen2 / ColSmol via colpali-engine
291
+ synthetic.py offline stand-in (tests only)
292
+ pruning/
293
+ saliency.py ink density + edge energy per patch
294
+ spatial.py keep-mask construction, dilation, budgets
295
+ redundancy.py greedy near-duplicate collapsing
296
+ compression/
297
+ binary.py bit packing, asymmetric/symmetric MaxSim
298
+ index/
299
+ numpy_index.py exact brute-force MaxSim (the reference)
300
+ qdrant_index.py Qdrant multivector + binary quantization
301
+ pipeline.py OptiVisionRAG.build() / .search()
302
+ bench.py encode-once, replay-every-variant ablation harness
303
+ corpus.py synthetic corpus generator + ViDoRe loader
304
+ metrics.py nDCG / recall / MRR / Kendall tau / storage
305
+ viz.py keep-mask and saliency figures
306
+ app/streamlit_app.py demo UI
307
+ docs/ architecture, results, viva notes
308
+ notebooks/ Colab / Kaggle runner for the ColPali benchmark
309
+ ```
310
+
311
+ ## Testing
312
+
313
+ ```bash
314
+ pytest # 75 tests, no model download needed
315
+ ruff check src tests app
316
+ ```
317
+
318
+ The suite covers saliency behaviour on blank/grey/inked pages, keep-mask budgets,
319
+ redundancy clustering invariants, bit-packing round-trips, MaxSim segment maths on
320
+ variable-length pages, index save/load, Qdrant multivector round-trips, and a full
321
+ index→search→evaluate loop.
322
+
323
+ ## Improvements
324
+
325
+ [docs/IMPROVEMENTS.md](docs/IMPROVEMENTS.md) records five changes made after the
326
+ pipeline was working — bounded query-time memory, an int8 quantizer that uses
327
+ the range it pays for, a single-allocation decode path, a Qdrant stats fix, and
328
+ one quadratic loop — each with the measurement behind it, plus what was
329
+ reviewed and deliberately left alone.
330
+
331
+ ## Honest limitations
332
+
333
+ - **Encoder speed on CPU.** ColSmol-256M takes ~30 s/page on this laptop (no GPU).
334
+ Pruning and quantization together take ~15 ms/page — the compression is free
335
+ relative to encoding, but building a large index needs a GPU.
336
+ - **The bundled corpus is generated**, not scanned. It has the right whitespace
337
+ profile and gives exact ground truth, but real scans have noise, skew and bleed-through.
338
+ `optivision fetch-vidore` pulls the real ViDoRe benchmark for that reason.
339
+ - **Redundancy merging changes the vectors**, not just their count. It is nearly free
340
+ under MaxSim but it is not lossless — the ablation reports both stages separately so
341
+ the cost is visible.
342
+ - **Qdrant local mode** is convenient but not a performance claim; latency numbers
343
+ come from the exact numpy index so they are not confounded by ANN recall.
344
+
345
+ ## References
346
+
347
+ - Faysse et al., *ColPali: Efficient Document Retrieval with Vision Language Models* (2024)
348
+ - Khattab & Zaharia, *ColBERT* (2020) — late interaction / MaxSim
349
+ - Qdrant multivector + binary quantization documentation
@@ -0,0 +1,71 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "optivision-rag"
7
+ dynamic = ["version"]
8
+ description = "OptiVision RAG: Extreme Token Compression for Vision-Language Models"
9
+ readme = "PACKAGE.md"
10
+ requires-python = ">=3.10"
11
+ license = "MIT"
12
+ license-files = ["LICENSE"]
13
+ keywords = ["rag", "colpali", "colqwen", "late-interaction", "vision-language", "retrieval", "quantization", "token-pruning", "qdrant"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Science/Research",
17
+ "Intended Audience :: Developers",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Programming Language :: Python :: 3.13",
23
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
24
+ "Topic :: Text Processing :: Indexing",
25
+ ]
26
+ authors = [
27
+ { name = "T. Rithik Krishna" },
28
+ { name = "Amgovath Navanitha" },
29
+ { name = "Badavath Akhila" },
30
+ ]
31
+ dependencies = [
32
+ "numpy>=1.26,<3",
33
+ "pillow>=10",
34
+ "pypdfium2>=4",
35
+ "qdrant-client>=1.10",
36
+ "pyyaml>=6",
37
+ "typer>=0.12",
38
+ "rich>=13",
39
+ ]
40
+
41
+ [project.optional-dependencies]
42
+ vlm = ["torch>=2.2", "colpali-engine>=0.3.10", "transformers>=4.46"]
43
+ corpus = ["reportlab>=4"]
44
+ bench = ["reportlab>=4", "datasets>=2.19"]
45
+ app = ["streamlit>=1.34"]
46
+ dev = ["pytest>=8", "ruff>=0.5"]
47
+
48
+ [project.urls]
49
+ Homepage = "https://github.com/Daemon-VI/optivision-rag"
50
+ Repository = "https://github.com/Daemon-VI/optivision-rag"
51
+ Issues = "https://github.com/Daemon-VI/optivision-rag/issues"
52
+
53
+ [project.scripts]
54
+ optivision = "optivision.cli:app"
55
+
56
+ [tool.setuptools.packages.find]
57
+ where = ["src"]
58
+
59
+ [tool.setuptools.package-data]
60
+ optivision = ["presets/*.yaml"]
61
+
62
+ [tool.setuptools.dynamic]
63
+ version = { attr = "optivision.__version__" }
64
+
65
+ [tool.pytest.ini_options]
66
+ testpaths = ["tests"]
67
+ filterwarnings = ["ignore::DeprecationWarning"]
68
+
69
+ [tool.ruff]
70
+ line-length = 100
71
+ target-version = "py310"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+