rag-ladder 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_ladder-0.1.0/PKG-INFO +9 -0
- rag_ladder-0.1.0/README.md +217 -0
- rag_ladder-0.1.0/pyproject.toml +21 -0
- rag_ladder-0.1.0/rag.py +547 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/PKG-INFO +9 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/SOURCES.txt +10 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/dependency_links.txt +1 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/entry_points.txt +2 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/requires.txt +5 -0
- rag_ladder-0.1.0/rag_ladder.egg-info/top_level.txt +2 -0
- rag_ladder-0.1.0/ragcli.py +661 -0
- rag_ladder-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rag-ladder
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: 6 RAG recipes, ladder-first: BM25 full-text, LLM query rewriting, hybrid retrieval on SQLite FTS5
|
|
5
|
+
Requires-Python: >=3.10
|
|
6
|
+
Requires-Dist: numpy>=1.26
|
|
7
|
+
Provides-Extra: docs
|
|
8
|
+
Requires-Dist: pypdf>=4; extra == "docs"
|
|
9
|
+
Requires-Dist: python-docx>=1; extra == "docs"
|
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
# rag-ladder — the 6 RAG recipe book, ladder-first
|
|
2
|
+
|
|
3
|
+
The article's thesis: most stacks jump to embeddings + vector DB + rerankers while
|
|
4
|
+
users just want the doc that says "how to reset my password". So this code
|
|
5
|
+
climbs the ladder and stops where the problem is already solved.
|
|
6
|
+
|
|
7
|
+
**Inspiration.** This code implements the recipe book from Rafael Pierre's
|
|
8
|
+
article ["RAG Is Simpler Than You Think"](https://www.lighthousenewsletter.com/p/rag-is-simpler-than-you-think)
|
|
9
|
+
(Lighthouse Newsletter, Jun 10, 2026). The six recipes — BM25, agentic query
|
|
10
|
+
rewriting, hybrid, on-the-fly, hot/cold, and full pre-embedding — and the
|
|
11
|
+
decision tree that picks between them come directly from that article. The
|
|
12
|
+
guiding principle: *don't build the 5% solution for a 60% problem.*
|
|
13
|
+
|
|
14
|
+
**Start here. Four commands do the whole job:**
|
|
15
|
+
|
|
16
|
+
rag-ladder init # choose your model backend — writes rag.json
|
|
17
|
+
rag-ladder add <folder> # index PDFs, DOCX, markdown, text, images (vision), video
|
|
18
|
+
rag-ladder serve # open the browser UI, ask questions, get cited answers
|
|
19
|
+
rag-ladder doctor # when something doesn't work
|
|
20
|
+
|
|
21
|
+
`rag-ladder ask "..."` answers from your documents with citations, and says
|
|
22
|
+
`NOT_FOUND_IN_CORPUS` when your documents don't contain the answer. It never
|
|
23
|
+
invents one — that is the whole point for legal and contract work.
|
|
24
|
+
|
|
25
|
+
## Install
|
|
26
|
+
|
|
27
|
+
uv sync # core: numpy only, stdlib search
|
|
28
|
+
uv sync --extra docs # adds pypdf + python-docx, for PDF/DOCX folders
|
|
29
|
+
uv tool install -e . # then `rag-ladder` is a standalone command
|
|
30
|
+
|
|
31
|
+
No new pip dependency for images: each photo is read through your LLM's vision
|
|
32
|
+
capability. Video needs the `ffmpeg` system binary (no pip) — see "Image and video".
|
|
33
|
+
|
|
34
|
+
Without `--extra docs`, `rag-ladder add` still works for `.txt .md .csv .json .html`
|
|
35
|
+
and tells you the exact install command when it meets a PDF.
|
|
36
|
+
|
|
37
|
+
## Image and video
|
|
38
|
+
|
|
39
|
+
Photos in a listing folder are usually the actual content. `rag-ladder add` sends each
|
|
40
|
+
image through `llm()` as an OpenAI `image_url` content block and stores the
|
|
41
|
+
transcription plus a scene description for search — the citation is the file path,
|
|
42
|
+
so an answer quoting a photo traces back to it. No extra pip dependency; your
|
|
43
|
+
`llm_model` just needs vision.
|
|
44
|
+
|
|
45
|
+
- **Images** (`.jpg .jpeg .png .webp .gif .bmp .tif .tiff`) — need a **vision-capable**
|
|
46
|
+
`llm_model` (e.g. `gpt-4o-mini`). A text-only model is skipped with a clear
|
|
47
|
+
message and never filled with guessed text.
|
|
48
|
+
- **Video** (`.mp4 .mov .avi .mkv .webm .flv .m4v`) — needs `ffmpeg`. Up to
|
|
49
|
+
`MAX_VIDEO_FRAMES` (12) evenly spaced frames are extracted, each described via
|
|
50
|
+
vision and joined. This is the most expensive part of `rag-ladder add`: one vision
|
|
51
|
+
call per frame.
|
|
52
|
+
- **Size** — a 20 MB per-image cap (`MAX_IMAGE_BYTES`); resize oversized files
|
|
53
|
+
before indexing.
|
|
54
|
+
|
|
55
|
+
`rag-ladder doctor` reports whether `ffmpeg` is present and which `llm_model` is resolved
|
|
56
|
+
— confirm that model has vision before adding photo folders.
|
|
57
|
+
|
|
58
|
+
## Which model to use
|
|
59
|
+
|
|
60
|
+
`rag-ladder init` offers four backends:
|
|
61
|
+
|
|
62
|
+
| preset | endpoint | model |
|
|
63
|
+
|---|---|---|
|
|
64
|
+
| `local` | `http://127.0.0.1:8888` | whatever the endpoint serves |
|
|
65
|
+
| `ollama` | `http://127.0.0.1:11434` | `llama3.1:8b` |
|
|
66
|
+
| `openrouter` | `https://openrouter.ai/api` | `openai/gpt-4o-mini` |
|
|
67
|
+
| `openai` | `https://api.openai.com` | `gpt-4o-mini` |
|
|
68
|
+
|
|
69
|
+
Anything OpenAI-compatible works: set `llm_base` and `llm_model` in `rag.json`.
|
|
70
|
+
Cloud providers need a key — `rag-ladder init --preset openrouter --key sk-or-...`
|
|
71
|
+
writes it into `rag.json` with mode `0600`, or export `RAG_API_KEY`.
|
|
72
|
+
|
|
73
|
+
`rag-ladder doctor` tells you what your endpoint actually serves. If it reports a
|
|
74
|
+
reasoning model, recipe 2 (query rewriting) costs far more than the article's
|
|
75
|
+
$0.001/query — point `llm_model` at a small non-reasoning model for the real
|
|
76
|
+
price. `rag-ladder ask` works with no model at all: it returns matching passages and
|
|
77
|
+
says so instead of guessing.
|
|
78
|
+
|
|
79
|
+
## CLI reference
|
|
80
|
+
|
|
81
|
+
rag-ladder add docs/ contracts.pdf # index files or folders (recursive)
|
|
82
|
+
rag-ladder add docs/ # re-run: unchanged files are skipped
|
|
83
|
+
rag-ladder ask "how long do refunds take?"
|
|
84
|
+
rag-ladder ask --mode legal "what is the notice period?" # quotes clauses, never infers
|
|
85
|
+
rag-ladder search "invoice #12345" # recipe 1, raw BM25 hits
|
|
86
|
+
rag-ladder hybrid "alternatives to X" # recipe 3/4
|
|
87
|
+
rag-ladder multi "read csv, clean, plot"
|
|
88
|
+
rag-ladder hot "invoice paid" # recipe 5
|
|
89
|
+
rag-ladder pre # recipe 6: build vectors
|
|
90
|
+
rag-ladder pre "reset password" # recipe 6: query them
|
|
91
|
+
rag-ladder stats # corpus and index sizes
|
|
92
|
+
rag-ladder glossary # write the glossary template
|
|
93
|
+
rag-ladder recommend --qpd 500 --churn 2 --complaint "can't find docs"
|
|
94
|
+
|
|
95
|
+
Add `--json` to `search ask hybrid multi hot pre stats` for scripting.
|
|
96
|
+
Exit codes: `0` ok, `1` runtime failure, `2` bad input or missing extractor,
|
|
97
|
+
`3` empty corpus (the message tells you to run `rag-ladder add`).
|
|
98
|
+
|
|
99
|
+
## Decision tree
|
|
100
|
+
|
|
101
|
+
`rag-ladder recommend` picks the next recipe for you. It needs two numbers and a
|
|
102
|
+
complaint:
|
|
103
|
+
|
|
104
|
+
rag-ladder recommend --qpd 500 --churn 2 --complaint "can't find docs"
|
|
105
|
+
|
|
106
|
+
| Flag | Meaning | How to estimate |
|
|
107
|
+
|---|---|---|
|
|
108
|
+
| `--qpd` | **Queries per day** — how many searches your users make | Count `rag-ladder ask`/`rag-ladder search` calls in a day, or estimate from traffic |
|
|
109
|
+
| `--churn` | **Corpus churn %/day** — how many documents change daily | Adding 50 docs to a 500-doc corpus = `--churn 10` |
|
|
110
|
+
| `--complaint` | What's wrong with the BM25 results | "can't find", "not great", "okay" |
|
|
111
|
+
|
|
112
|
+
The tree's answer tells you the next rung to build:
|
|
113
|
+
|
|
114
|
+
| Complaint | Condition | Next recipe |
|
|
115
|
+
|---|---|---|
|
|
116
|
+
| "can't find" | vocabulary mismatch | **2** — LLM query rewriting |
|
|
117
|
+
| "not great" / "okay" | low latency OK, low churn | **3** — hybrid BM25 + embedding rerank |
|
|
118
|
+
| "not great" | churn >10%/day | **4** — embed candidates on the fly |
|
|
119
|
+
| "not great" | Pareto access (hot 20%) | **5** — hot/cold tiers |
|
|
120
|
+
| "not great" | >100K docs, >10K qpd, ML team | **6** — full pre-embedding |
|
|
121
|
+
| (satisfied) | users happy | **stop** — ship features |
|
|
122
|
+
|
|
123
|
+
Run it after `rag-ladder add`, before you write any code. It returns `None` when
|
|
124
|
+
recipe 1 + 2 are enough — that's the target for ~60% of systems.
|
|
125
|
+
|
|
126
|
+
## Config
|
|
127
|
+
|
|
128
|
+
`rag-ladder init` writes `rag.json` (see `rag.example.json`). Every key is optional.
|
|
129
|
+
|
|
130
|
+
| key | default | what it does |
|
|
131
|
+
|---|---|---|
|
|
132
|
+
| `llm_base` | `http://127.0.0.1:8888` | OpenAI-compatible endpoint |
|
|
133
|
+
| `llm_model` | endpoint's own | model name |
|
|
134
|
+
| `llm_api_key` | — | sent as `Authorization: Bearer` |
|
|
135
|
+
| `llm_reasoning` | `auto` | `none` sends `reasoning_effort=none` — for endpoints running a reasoning model, where a trivial rewrite otherwise burns ~20s thinking |
|
|
136
|
+
| `embed_base` | — | `/embeddings` endpoint; blank = hashed fallback |
|
|
137
|
+
| `embed_model` | `hash-bow-512` | embedding model name |
|
|
138
|
+
| `embed_max_chars` | `4000` | truncation cap for long documents |
|
|
139
|
+
| `db` | `rag.db` | SQLite file |
|
|
140
|
+
| `corpus` | `.` | root folder document ids are relative to |
|
|
141
|
+
| `glossary` | `glossary.txt` | domain terms never rewritten |
|
|
142
|
+
| `answer_mode` | `general` | `general` or `legal` (stricter, quotes clauses) |
|
|
143
|
+
| `serve_host` / `serve_port` | `127.0.0.1` / `8765` | browser UI bind |
|
|
144
|
+
|
|
145
|
+
Resolution order everywhere: **CLI flag > env var > `rag.json` > default.**
|
|
146
|
+
Env vars: `RAG_CONFIG`, `RAG_LLM_BASE`, `RAG_LLM_MODEL`, `RAG_API_KEY`,
|
|
147
|
+
`RAG_LLM_REASONING`, `RAG_EMBED_BASE`, `RAG_EMBED_MODEL`, `RAG_EMBED_MAX_CHARS`,
|
|
148
|
+
`RAG_DB`, `RAG_CORPUS`, `RAG_GLOSSARY`, `RAG_SERVE_HOST`, `RAG_SERVE_PORT`.
|
|
149
|
+
`--no-llm` forces deterministic rewriting — useful when there's no model around.
|
|
150
|
+
|
|
151
|
+
## Browser UI
|
|
152
|
+
|
|
153
|
+
`rag-ladder serve` binds `127.0.0.1:8765` and serves a read-only page: ask box, answer
|
|
154
|
+
pane, citations, corpus count. Routes: `GET /`, `GET /api/ask?q=&k=`,
|
|
155
|
+
`GET /api/search?q=&k=`, `GET /api/stats`. No write endpoint, so there's no CSRF
|
|
156
|
+
surface. Binding a non-local `--host` prints a warning — that exposes the corpus
|
|
157
|
+
to your network.
|
|
158
|
+
|
|
159
|
+
## Architecture
|
|
160
|
+
|
|
161
|
+
Two modules split at the engine/app seam:
|
|
162
|
+
|
|
163
|
+
- `rag.py` — engine: BM25 retrieval, LLM/embed clients, whole-document storage,
|
|
164
|
+
answer synthesis, `recommend()`.
|
|
165
|
+
- `ragcli.py` — app layer: config, ingestion, CLI, browser UI.
|
|
166
|
+
- `test_rag.py` — 26 runnable checks, no pytest, no network. `uv run test_rag.py`.
|
|
167
|
+
|
|
168
|
+
One SQLite file. Documents are stored **whole** — no chunking, no chunk-size or
|
|
169
|
+
overlap decisions, no eval harness for chunks (recipe 1's whole point). A `meta`
|
|
170
|
+
table holds the content hash, so re-indexing skips unchanged files.
|
|
171
|
+
|
|
172
|
+
| Piece | Rung | Why |
|
|
173
|
+
|---|---|---|
|
|
174
|
+
| BM25 index | SQLite FTS5 (`bm25()`, `ORDER BY rank`) | stdlib, native, zero ops |
|
|
175
|
+
| LLM client | `urllib` → OpenAI-compatible `/chat/completions` | no SDK dependency |
|
|
176
|
+
| Embeddings | `urllib` → `/embeddings`, else hashed bag-of-words | recipes 3-6 runnable with zero setup |
|
|
177
|
+
| Vector math | numpy>=1.26 (declared dep) | rerank + brute-force cosine |
|
|
178
|
+
| ANN | brute-force scan over BLOBs | ponytail: O(n); faiss/hnsw past ~100K docs |
|
|
179
|
+
| UI | `http.server` + one inline page | stdlib, no framework, no build step |
|
|
180
|
+
|
|
181
|
+
Tables: `docs` (FTS5, id + full text), `vectors` (id, vec BLOB, model),
|
|
182
|
+
`access` (id, hit count — feeds the hot tier), `meta` (id, sha, mtime, chars).
|
|
183
|
+
|
|
184
|
+
## Recipe → code
|
|
185
|
+
|
|
186
|
+
| Recipe | Entry point | When |
|
|
187
|
+
|---|---|---|
|
|
188
|
+
| 1 BM25 | `Rag.search` | start here, always |
|
|
189
|
+
| 2 query rewriting | `Rag.rewrite`, `Rag.agentic_search` | vocabulary mismatch; results bad → edit the prompt, not the corpus |
|
|
190
|
+
| 3 hybrid | `Rag.hybrid` | "okay but not great"; 100-500ms is acceptable |
|
|
191
|
+
| 4 on-the-fly | `Rag.hybrid` (docs embedded per query) | churn >10%/day, or you're swapping embedding models |
|
|
192
|
+
| 5 hot/cold | `Rag.search_hot_cold`, `Rag.refresh_hot` | Pareto access pattern, 100K+ docs |
|
|
193
|
+
| 6 pre-embedding | `Rag.preembed`, `Rag.search_preembedded` | >10K q/day, <5% churn/month, ML team |
|
|
194
|
+
| multi-intent | `Rag.decompose`, `Rag.search_multi_intent` | one query carrying several intents; parallel → latency is max, not sum |
|
|
195
|
+
| cited answers | `Rag.answer` | every user-facing question; `answer_mode: legal` for contracts |
|
|
196
|
+
|
|
197
|
+
The decision tree from the article is `recommend()`; the CLI runs it.
|
|
198
|
+
|
|
199
|
+
## Deliberate corners
|
|
200
|
+
|
|
201
|
+
- Fallback embeddings are lexical (hashed BoW), **not semantic**. Point
|
|
202
|
+
`embed_base` at a real model for recipes 3-6 to mean what the article says.
|
|
203
|
+
- Documents are embedded whole but truncated to `embed_max_chars` — a 2048-token
|
|
204
|
+
model can't see a long document's tail. That's the chunking cost recipe 1
|
|
205
|
+
avoids; pay it only if truncation measurably hurts ranking.
|
|
206
|
+
- `agentic_search`'s quality signal is lexical coverage, not an LLM judge.
|
|
207
|
+
- ANN is a linear scan; hot tier is recomputed on read, not on a cron.
|
|
208
|
+
- `decompose`'s fallback splits on "and"/commas and invents no dependencies.
|
|
209
|
+
- `answer()` cites the whole document it retrieved, not a page number — page
|
|
210
|
+
offsets aren't tracked, because documents are stored whole.
|
|
211
|
+
- A reasoning model spends its budget on `reasoning_content` before emitting the
|
|
212
|
+
rewrite, so recipe 2 costs far more than the article's $0.001/query. `llm()`
|
|
213
|
+
retries such truncations with `reasoning_effort="none"` (llama.cpp-style
|
|
214
|
+
servers); `rag-ladder doctor` warns, and presets supply a cheap model.
|
|
215
|
+
|
|
216
|
+
Bottom line from the article, encoded as the default: don't build the 5%
|
|
217
|
+
solution for a 60% problem. 60% of systems stop at recipe 1 + recipe 2.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "rag-ladder"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "6 RAG recipes, ladder-first: BM25 full-text, LLM query rewriting, hybrid retrieval on SQLite FTS5"
|
|
9
|
+
requires-python = ">=3.10"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"numpy>=1.26",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
[project.optional-dependencies]
|
|
15
|
+
docs = ["pypdf>=4", "python-docx>=1"]
|
|
16
|
+
|
|
17
|
+
[tool.setuptools]
|
|
18
|
+
py-modules = ["rag", "ragcli"]
|
|
19
|
+
|
|
20
|
+
[project.scripts]
|
|
21
|
+
rag-ladder = "ragcli:main"
|