rag-your-code 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.6.0/src/rag_your_code.egg-info → rag_your_code-0.8.0}/PKG-INFO +114 -15
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/README.md +113 -14
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/pyproject.toml +1 -1
- {rag_your_code-0.6.0 → rag_your_code-0.8.0/src/rag_your_code.egg-info}/PKG-INFO +114 -15
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/SOURCES.txt +5 -1
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/__init__.py +1 -1
- rag_your_code-0.8.0/src/ragyourcode/agentic.py +119 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/cli.py +96 -129
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/config.py +77 -1
- rag_your_code-0.8.0/src/ragyourcode/embeddings.py +194 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/graph.py +3 -2
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/indexer.py +17 -6
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/models.py +23 -11
- rag_your_code-0.8.0/src/ragyourcode/providers.py +174 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/search.py +45 -4
- rag_your_code-0.8.0/src/ragyourcode/workflow.py +191 -0
- rag_your_code-0.8.0/tests/test_agentic.py +146 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_config.py +1 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_metadata.py +53 -2
- rag_your_code-0.8.0/tests/test_providers.py +363 -0
- rag_your_code-0.8.0/tests/test_workflow.py +127 -0
- rag_your_code-0.6.0/src/ragyourcode/agentic.py +0 -62
- rag_your_code-0.6.0/src/ragyourcode/embeddings.py +0 -82
- rag_your_code-0.6.0/tests/test_agentic.py +0 -29
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/LICENSE +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/setup.cfg +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_document.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_ranking.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_resilience.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.8.0}/tests/test_retrieval_correctness.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
330
368
|
one reply per line:
|
|
331
369
|
|
|
332
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
333
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
334
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
335
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -340,6 +379,14 @@ one reply per line:
|
|
|
340
379
|
{"action":"stats"}
|
|
341
380
|
```
|
|
342
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
343
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
344
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
345
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -360,14 +407,66 @@ it stopped.
|
|
|
360
407
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
361
408
|
delete to clear the cache.
|
|
362
409
|
|
|
363
|
-
##
|
|
410
|
+
## Bringing your own model
|
|
411
|
+
|
|
412
|
+
Everything above works with no model at all. If you would rather have real
|
|
413
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
414
|
+
which includes a model server on your own machine:
|
|
415
|
+
|
|
416
|
+
```toml
|
|
417
|
+
# rag-your-code.toml
|
|
418
|
+
[embedding]
|
|
419
|
+
provider = "openai-compatible"
|
|
420
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
421
|
+
model = "nomic-embed-text"
|
|
422
|
+
dimensions = 768 # must match the model
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
426
|
+
name of the environment variable holding your key:
|
|
427
|
+
|
|
428
|
+
```toml
|
|
429
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
430
|
+
```
|
|
364
431
|
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
432
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
433
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
434
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
435
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
436
|
+
machine is refused rather than warned about.
|
|
437
|
+
|
|
438
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
439
|
+
before you do:
|
|
440
|
+
|
|
441
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
442
|
+
the whole reason the local case is written first here.
|
|
443
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
444
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
445
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
446
|
+
reached, which is the one gap no amount of ranking closes:
|
|
447
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
448
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
449
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
450
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
451
|
+
on the meaningless cosine between them with full confidence.
|
|
452
|
+
|
|
453
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
454
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
455
|
+
no request at all.
|
|
456
|
+
|
|
457
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
458
|
+
much. No number here is from a real model — this project has no key, and a
|
|
459
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
460
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
461
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
462
|
+
for a hash that carries no meaning.
|
|
463
|
+
|
|
464
|
+
## Still not here
|
|
465
|
+
|
|
466
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
467
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
468
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
469
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
371
470
|
|
|
372
471
|
## Development
|
|
373
472
|
|
|
@@ -49,14 +49,23 @@ package itself on first use.
|
|
|
49
49
|
```bash
|
|
50
50
|
pip install rag-your-code
|
|
51
51
|
|
|
52
|
-
rag-your-code index
|
|
52
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
53
53
|
rag-your-code search "where are HTTP retries handled" --json
|
|
54
54
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
55
55
|
```
|
|
56
56
|
|
|
57
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
58
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
59
|
+
sentence the parser generated, which adds no word the source did not already
|
|
60
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
61
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
62
|
+
work. It reads the state rather than remembering a position, so running it
|
|
63
|
+
again after each round is how you make progress. `index` still exists and does
|
|
64
|
+
only the indexing.
|
|
65
|
+
|
|
57
66
|
The index is written under `.rag-your-code/`; your source files are never
|
|
58
|
-
modified. Later
|
|
59
|
-
|
|
67
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
68
|
+
`--compact`.
|
|
60
69
|
|
|
61
70
|
## How it works
|
|
62
71
|
|
|
@@ -108,16 +117,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
108
117
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
109
118
|
word either way.
|
|
110
119
|
|
|
111
|
-
Retrieval works regardless, because **
|
|
112
|
-
natural language** —
|
|
113
|
-
|
|
114
|
-
|
|
120
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
121
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
122
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
123
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
124
|
+
implemented and measured against all three rulers, with query and stored
|
|
125
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
126
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
127
|
+
discounts to nothing.
|
|
128
|
+
|
|
129
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
130
|
+
rest of the gap, and neither is a model:
|
|
115
131
|
|
|
116
132
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
117
133
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
118
134
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
119
135
|
into the index once instead of into every query.
|
|
120
136
|
|
|
137
|
+
### Seven attempts to make the vector half earn its place
|
|
138
|
+
|
|
139
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
140
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
141
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
142
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
143
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
144
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
145
|
+
|
|
146
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
147
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
148
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
149
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
150
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
151
|
+
|
|
152
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
153
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
154
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
155
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
156
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
157
|
+
scheme that sharpened it lost ground there.
|
|
158
|
+
|
|
121
159
|
## Agent-authored descriptions
|
|
122
160
|
|
|
123
161
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -303,6 +341,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
303
341
|
one reply per line:
|
|
304
342
|
|
|
305
343
|
```json
|
|
344
|
+
{"action":"bootstrap"}
|
|
306
345
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
307
346
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
308
347
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -313,6 +352,14 @@ one reply per line:
|
|
|
313
352
|
{"action":"stats"}
|
|
314
353
|
```
|
|
315
354
|
|
|
355
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
356
|
+
path, line range, signature, description, score and matched terms. The code
|
|
357
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
358
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
359
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
360
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
361
|
+
111,843 by serialising the same eight units three times over.
|
|
362
|
+
|
|
316
363
|
**No single request can end the session.** Numeric fields saturate at their
|
|
317
364
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
318
365
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -333,14 +380,66 @@ it stopped.
|
|
|
333
380
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
334
381
|
delete to clear the cache.
|
|
335
382
|
|
|
336
|
-
##
|
|
383
|
+
## Bringing your own model
|
|
384
|
+
|
|
385
|
+
Everything above works with no model at all. If you would rather have real
|
|
386
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
387
|
+
which includes a model server on your own machine:
|
|
388
|
+
|
|
389
|
+
```toml
|
|
390
|
+
# rag-your-code.toml
|
|
391
|
+
[embedding]
|
|
392
|
+
provider = "openai-compatible"
|
|
393
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
394
|
+
model = "nomic-embed-text"
|
|
395
|
+
dimensions = 768 # must match the model
|
|
396
|
+
```
|
|
397
|
+
|
|
398
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
399
|
+
name of the environment variable holding your key:
|
|
400
|
+
|
|
401
|
+
```toml
|
|
402
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
403
|
+
```
|
|
337
404
|
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
405
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
406
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
407
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
408
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
409
|
+
machine is refused rather than warned about.
|
|
410
|
+
|
|
411
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
412
|
+
before you do:
|
|
413
|
+
|
|
414
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
415
|
+
the whole reason the local case is written first here.
|
|
416
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
417
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
418
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
419
|
+
reached, which is the one gap no amount of ranking closes:
|
|
420
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
421
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
422
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
423
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
424
|
+
on the meaningless cosine between them with full confidence.
|
|
425
|
+
|
|
426
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
427
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
428
|
+
no request at all.
|
|
429
|
+
|
|
430
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
431
|
+
much. No number here is from a real model — this project has no key, and a
|
|
432
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
433
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
434
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
435
|
+
for a hash that carries no meaning.
|
|
436
|
+
|
|
437
|
+
## Still not here
|
|
438
|
+
|
|
439
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
440
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
441
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
442
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
344
443
|
|
|
345
444
|
## Development
|
|
346
445
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
330
368
|
one reply per line:
|
|
331
369
|
|
|
332
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
333
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
334
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
335
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -340,6 +379,14 @@ one reply per line:
|
|
|
340
379
|
{"action":"stats"}
|
|
341
380
|
```
|
|
342
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
343
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
344
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
345
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -360,14 +407,66 @@ it stopped.
|
|
|
360
407
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
361
408
|
delete to clear the cache.
|
|
362
409
|
|
|
363
|
-
##
|
|
410
|
+
## Bringing your own model
|
|
411
|
+
|
|
412
|
+
Everything above works with no model at all. If you would rather have real
|
|
413
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
414
|
+
which includes a model server on your own machine:
|
|
415
|
+
|
|
416
|
+
```toml
|
|
417
|
+
# rag-your-code.toml
|
|
418
|
+
[embedding]
|
|
419
|
+
provider = "openai-compatible"
|
|
420
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
421
|
+
model = "nomic-embed-text"
|
|
422
|
+
dimensions = 768 # must match the model
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
426
|
+
name of the environment variable holding your key:
|
|
427
|
+
|
|
428
|
+
```toml
|
|
429
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
430
|
+
```
|
|
364
431
|
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
432
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
433
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
434
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
435
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
436
|
+
machine is refused rather than warned about.
|
|
437
|
+
|
|
438
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
439
|
+
before you do:
|
|
440
|
+
|
|
441
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
442
|
+
the whole reason the local case is written first here.
|
|
443
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
444
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
445
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
446
|
+
reached, which is the one gap no amount of ranking closes:
|
|
447
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
448
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
449
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
450
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
451
|
+
on the meaningless cosine between them with full confidence.
|
|
452
|
+
|
|
453
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
454
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
455
|
+
no request at all.
|
|
456
|
+
|
|
457
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
458
|
+
much. No number here is from a real model — this project has no key, and a
|
|
459
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
460
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
461
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
462
|
+
for a hash that carries no meaning.
|
|
463
|
+
|
|
464
|
+
## Still not here
|
|
465
|
+
|
|
466
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
467
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
468
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
469
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
371
470
|
|
|
372
471
|
## Development
|
|
373
472
|
|
|
@@ -19,8 +19,10 @@ src/ragyourcode/graph.py
|
|
|
19
19
|
src/ragyourcode/indexer.py
|
|
20
20
|
src/ragyourcode/models.py
|
|
21
21
|
src/ragyourcode/parser.py
|
|
22
|
+
src/ragyourcode/providers.py
|
|
22
23
|
src/ragyourcode/py.typed
|
|
23
24
|
src/ragyourcode/search.py
|
|
25
|
+
src/ragyourcode/workflow.py
|
|
24
26
|
tests/test_agent_protocol.py
|
|
25
27
|
tests/test_agentic.py
|
|
26
28
|
tests/test_config.py
|
|
@@ -35,8 +37,10 @@ tests/test_large_repo.py
|
|
|
35
37
|
tests/test_metadata.py
|
|
36
38
|
tests/test_multilanguage.py
|
|
37
39
|
tests/test_parser_edges.py
|
|
40
|
+
tests/test_providers.py
|
|
38
41
|
tests/test_ragyourcode.py
|
|
39
42
|
tests/test_ranking.py
|
|
40
43
|
tests/test_repo_queries.py
|
|
41
44
|
tests/test_resilience.py
|
|
42
|
-
tests/test_retrieval_correctness.py
|
|
45
|
+
tests/test_retrieval_correctness.py
|
|
46
|
+
tests/test_workflow.py
|