rag-your-code 0.5.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.5.0/src/rag_your_code.egg-info → rag_your_code-0.7.0}/PKG-INFO +92 -15
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/README.md +91 -14
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/pyproject.toml +1 -1
- {rag_your_code-0.5.0 → rag_your_code-0.7.0/src/rag_your_code.egg-info}/PKG-INFO +92 -15
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/SOURCES.txt +4 -1
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/__init__.py +1 -1
- rag_your_code-0.7.0/src/ragyourcode/agentic.py +118 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/cli.py +74 -117
- rag_your_code-0.7.0/src/ragyourcode/models.py +135 -0
- rag_your_code-0.7.0/src/ragyourcode/search.py +278 -0
- rag_your_code-0.7.0/src/ragyourcode/workflow.py +184 -0
- rag_your_code-0.7.0/tests/test_agentic.py +146 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_metadata.py +26 -2
- rag_your_code-0.7.0/tests/test_ranking.py +202 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_repo_queries.py +40 -5
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_retrieval_correctness.py +4 -22
- rag_your_code-0.7.0/tests/test_workflow.py +127 -0
- rag_your_code-0.5.0/src/ragyourcode/agentic.py +0 -62
- rag_your_code-0.5.0/src/ragyourcode/models.py +0 -109
- rag_your_code-0.5.0/src/ragyourcode/search.py +0 -151
- rag_your_code-0.5.0/tests/test_agentic.py +0 -29
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/LICENSE +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/setup.cfg +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/config.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_config.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_document.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.7.0}/tests/test_resilience.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -198,19 +236,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
198
236
|
|
|
199
237
|
| | generated descriptions | agent-written |
|
|
200
238
|
|---|---|---|
|
|
201
|
-
| hit@1 | 0.
|
|
202
|
-
| hit@3 | 0.
|
|
203
|
-
| MRR | 0.
|
|
204
|
-
| answered with no shared word at all |
|
|
239
|
+
| hit@1 | 0.271 | **0.500** |
|
|
240
|
+
| hit@3 | 0.486 | **0.800** |
|
|
241
|
+
| MRR | 0.367 | **0.631** |
|
|
242
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
205
243
|
|
|
206
|
-
Roughly
|
|
207
|
-
|
|
208
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
244
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
245
|
+
is what makes the set usable for measuring the next change;
|
|
246
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
247
|
+
the written column beats the generated one.
|
|
209
248
|
|
|
210
249
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
211
250
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
212
251
|
stemming — exactly the limit documented above.
|
|
213
252
|
|
|
253
|
+
### Measured on a repository nobody here wrote
|
|
254
|
+
|
|
255
|
+
The table above is the warmest case this project supports: its own code, its
|
|
256
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
257
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
258
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
259
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
260
|
+
the words of the docstring that answers it
|
|
261
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
262
|
+
|
|
263
|
+
| | before 0.6.0 | now |
|
|
264
|
+
|---|---|---|
|
|
265
|
+
| hit@1 | 0.086 | **0.257** |
|
|
266
|
+
| hit@3 | 0.229 | **0.400** |
|
|
267
|
+
| MRR | 0.157 | **0.314** |
|
|
268
|
+
|
|
269
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
270
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
271
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
272
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
273
|
+
that repository came back in the top three for four questions out of six. It
|
|
274
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
275
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
276
|
+
buried in a body.
|
|
277
|
+
|
|
278
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
279
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
280
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
281
|
+
|
|
214
282
|
**What this is:** it moves the semantic work from query time to index time.
|
|
215
283
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
216
284
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -300,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
300
368
|
one reply per line:
|
|
301
369
|
|
|
302
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
303
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
304
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
305
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -310,6 +379,14 @@ one reply per line:
|
|
|
310
379
|
{"action":"stats"}
|
|
311
380
|
```
|
|
312
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
313
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
314
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
315
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -49,14 +49,23 @@ package itself on first use.
|
|
|
49
49
|
```bash
|
|
50
50
|
pip install rag-your-code
|
|
51
51
|
|
|
52
|
-
rag-your-code index
|
|
52
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
53
53
|
rag-your-code search "where are HTTP retries handled" --json
|
|
54
54
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
55
55
|
```
|
|
56
56
|
|
|
57
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
58
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
59
|
+
sentence the parser generated, which adds no word the source did not already
|
|
60
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
61
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
62
|
+
work. It reads the state rather than remembering a position, so running it
|
|
63
|
+
again after each round is how you make progress. `index` still exists and does
|
|
64
|
+
only the indexing.
|
|
65
|
+
|
|
57
66
|
The index is written under `.rag-your-code/`; your source files are never
|
|
58
|
-
modified. Later
|
|
59
|
-
|
|
67
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
68
|
+
`--compact`.
|
|
60
69
|
|
|
61
70
|
## How it works
|
|
62
71
|
|
|
@@ -108,16 +117,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
108
117
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
109
118
|
word either way.
|
|
110
119
|
|
|
111
|
-
Retrieval works regardless, because **
|
|
112
|
-
natural language** —
|
|
113
|
-
|
|
114
|
-
|
|
120
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
121
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
122
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
123
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
124
|
+
implemented and measured against all three rulers, with query and stored
|
|
125
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
126
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
127
|
+
discounts to nothing.
|
|
128
|
+
|
|
129
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
130
|
+
rest of the gap, and neither is a model:
|
|
115
131
|
|
|
116
132
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
117
133
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
118
134
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
119
135
|
into the index once instead of into every query.
|
|
120
136
|
|
|
137
|
+
### Seven attempts to make the vector half earn its place
|
|
138
|
+
|
|
139
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
140
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
141
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
142
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
143
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
144
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
145
|
+
|
|
146
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
147
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
148
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
149
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
150
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
151
|
+
|
|
152
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
153
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
154
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
155
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
156
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
157
|
+
scheme that sharpened it lost ground there.
|
|
158
|
+
|
|
121
159
|
## Agent-authored descriptions
|
|
122
160
|
|
|
123
161
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -171,19 +209,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
171
209
|
|
|
172
210
|
| | generated descriptions | agent-written |
|
|
173
211
|
|---|---|---|
|
|
174
|
-
| hit@1 | 0.
|
|
175
|
-
| hit@3 | 0.
|
|
176
|
-
| MRR | 0.
|
|
177
|
-
| answered with no shared word at all |
|
|
212
|
+
| hit@1 | 0.271 | **0.500** |
|
|
213
|
+
| hit@3 | 0.486 | **0.800** |
|
|
214
|
+
| MRR | 0.367 | **0.631** |
|
|
215
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
178
216
|
|
|
179
|
-
Roughly
|
|
180
|
-
|
|
181
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
217
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
218
|
+
is what makes the set usable for measuring the next change;
|
|
219
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
220
|
+
the written column beats the generated one.
|
|
182
221
|
|
|
183
222
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
184
223
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
185
224
|
stemming — exactly the limit documented above.
|
|
186
225
|
|
|
226
|
+
### Measured on a repository nobody here wrote
|
|
227
|
+
|
|
228
|
+
The table above is the warmest case this project supports: its own code, its
|
|
229
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
230
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
231
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
232
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
233
|
+
the words of the docstring that answers it
|
|
234
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
235
|
+
|
|
236
|
+
| | before 0.6.0 | now |
|
|
237
|
+
|---|---|---|
|
|
238
|
+
| hit@1 | 0.086 | **0.257** |
|
|
239
|
+
| hit@3 | 0.229 | **0.400** |
|
|
240
|
+
| MRR | 0.157 | **0.314** |
|
|
241
|
+
|
|
242
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
243
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
244
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
245
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
246
|
+
that repository came back in the top three for four questions out of six. It
|
|
247
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
248
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
249
|
+
buried in a body.
|
|
250
|
+
|
|
251
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
252
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
253
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
254
|
+
|
|
187
255
|
**What this is:** it moves the semantic work from query time to index time.
|
|
188
256
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
189
257
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -273,6 +341,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
273
341
|
one reply per line:
|
|
274
342
|
|
|
275
343
|
```json
|
|
344
|
+
{"action":"bootstrap"}
|
|
276
345
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
277
346
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
278
347
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -283,6 +352,14 @@ one reply per line:
|
|
|
283
352
|
{"action":"stats"}
|
|
284
353
|
```
|
|
285
354
|
|
|
355
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
356
|
+
path, line range, signature, description, score and matched terms. The code
|
|
357
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
358
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
359
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
360
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
361
|
+
111,843 by serialising the same eight units three times over.
|
|
362
|
+
|
|
286
363
|
**No single request can end the session.** Numeric fields saturate at their
|
|
287
364
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
288
365
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -198,19 +236,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
198
236
|
|
|
199
237
|
| | generated descriptions | agent-written |
|
|
200
238
|
|---|---|---|
|
|
201
|
-
| hit@1 | 0.
|
|
202
|
-
| hit@3 | 0.
|
|
203
|
-
| MRR | 0.
|
|
204
|
-
| answered with no shared word at all |
|
|
239
|
+
| hit@1 | 0.271 | **0.500** |
|
|
240
|
+
| hit@3 | 0.486 | **0.800** |
|
|
241
|
+
| MRR | 0.367 | **0.631** |
|
|
242
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
205
243
|
|
|
206
|
-
Roughly
|
|
207
|
-
|
|
208
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
244
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
245
|
+
is what makes the set usable for measuring the next change;
|
|
246
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
247
|
+
the written column beats the generated one.
|
|
209
248
|
|
|
210
249
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
211
250
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
212
251
|
stemming — exactly the limit documented above.
|
|
213
252
|
|
|
253
|
+
### Measured on a repository nobody here wrote
|
|
254
|
+
|
|
255
|
+
The table above is the warmest case this project supports: its own code, its
|
|
256
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
257
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
258
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
259
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
260
|
+
the words of the docstring that answers it
|
|
261
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
262
|
+
|
|
263
|
+
| | before 0.6.0 | now |
|
|
264
|
+
|---|---|---|
|
|
265
|
+
| hit@1 | 0.086 | **0.257** |
|
|
266
|
+
| hit@3 | 0.229 | **0.400** |
|
|
267
|
+
| MRR | 0.157 | **0.314** |
|
|
268
|
+
|
|
269
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
270
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
271
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
272
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
273
|
+
that repository came back in the top three for four questions out of six. It
|
|
274
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
275
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
276
|
+
buried in a body.
|
|
277
|
+
|
|
278
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
279
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
280
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
281
|
+
|
|
214
282
|
**What this is:** it moves the semantic work from query time to index time.
|
|
215
283
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
216
284
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -300,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
300
368
|
one reply per line:
|
|
301
369
|
|
|
302
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
303
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
304
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
305
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -310,6 +379,14 @@ one reply per line:
|
|
|
310
379
|
{"action":"stats"}
|
|
311
380
|
```
|
|
312
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
313
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
314
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
315
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -21,6 +21,7 @@ src/ragyourcode/models.py
|
|
|
21
21
|
src/ragyourcode/parser.py
|
|
22
22
|
src/ragyourcode/py.typed
|
|
23
23
|
src/ragyourcode/search.py
|
|
24
|
+
src/ragyourcode/workflow.py
|
|
24
25
|
tests/test_agent_protocol.py
|
|
25
26
|
tests/test_agentic.py
|
|
26
27
|
tests/test_config.py
|
|
@@ -36,6 +37,8 @@ tests/test_metadata.py
|
|
|
36
37
|
tests/test_multilanguage.py
|
|
37
38
|
tests/test_parser_edges.py
|
|
38
39
|
tests/test_ragyourcode.py
|
|
40
|
+
tests/test_ranking.py
|
|
39
41
|
tests/test_repo_queries.py
|
|
40
42
|
tests/test_resilience.py
|
|
41
|
-
tests/test_retrieval_correctness.py
|
|
43
|
+
tests/test_retrieval_correctness.py
|
|
44
|
+
tests/test_workflow.py
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Bounded, observable agentic retrieval (ARAG) orchestration."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .graph import CodeGraph, graph_search
|
|
6
|
+
from .models import CodeUnit, SearchResult
|
|
7
|
+
from .search import DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _result_ids(results: list[SearchResult]) -> set[str]:
|
|
11
|
+
"""Collects the unit identifiers out of a list of search results, so two
|
|
12
|
+
result sets can be compared for overlap or novelty. Used to decide
|
|
13
|
+
whether a second retrieval round actually surfaced anything new.
|
|
14
|
+
"""
|
|
15
|
+
return {result.unit.id for result in results}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# How far ahead of the runner-up the top result must be before a second
|
|
19
|
+
# retrieval step is skipped, as a fraction of the top score.
|
|
20
|
+
#
|
|
21
|
+
# This replaced an absolute `confidence_threshold` of 0.8, which was a number
|
|
22
|
+
# tied to a scoring scale. When ranking became BM25F the scale moved and the
|
|
23
|
+
# threshold quietly died: measured across 105 ruler questions, only 3% of
|
|
24
|
+
# queries reached 0.8 at all, so the early stop had stopped existing and every
|
|
25
|
+
# research call ran the graph expansion. A margin is a ratio between two scores
|
|
26
|
+
# from the same query, so it cannot drift when the scoring changes again.
|
|
27
|
+
#
|
|
28
|
+
# Measured over those 105 questions: top-1 is correct 42% of the time overall,
|
|
29
|
+
# and 68-73% of the time among the queries this fires on. Anywhere in 0.2-0.5
|
|
30
|
+
# behaves the same at this sample size; 0.30 is the middle of that band, not a
|
|
31
|
+
# measured optimum.
|
|
32
|
+
DEFAULT_DOMINANCE: float = 0.30
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _serialize(results: list[SearchResult]) -> list[dict]:
|
|
36
|
+
"""Converts search results into plain JSON-ready dictionaries for the agent
|
|
37
|
+
protocol reply.
|
|
38
|
+
"""
|
|
39
|
+
return [result.to_dict() for result in results]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _trace(results: list[SearchResult]) -> list[dict]:
|
|
43
|
+
"""One step of the search, as a trace rather than as a payload.
|
|
44
|
+
|
|
45
|
+
A step exists so the caller can see how the answer was reached: which
|
|
46
|
+
units each round reached and how strongly. That needs an identifier, a
|
|
47
|
+
score and the matching words -- not a second and third copy of every
|
|
48
|
+
unit's description, callees and imports. Reporting two steps in full is
|
|
49
|
+
what made a research reply three times the size of the answer inside it.
|
|
50
|
+
"""
|
|
51
|
+
return [
|
|
52
|
+
{"id": result.unit.id, "score": round(result.score, 6), "matched_terms": result.matched_terms}
|
|
53
|
+
for result in results
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def dominance(results: list[SearchResult]) -> float:
|
|
58
|
+
"""How far the top result is ahead of the runner-up, relative to the top.
|
|
59
|
+
|
|
60
|
+
One result alone is unopposed and scores 1.0. Nothing scores 0.0. The
|
|
61
|
+
quantity is scale-free by construction, which is the whole point: it
|
|
62
|
+
compares two numbers produced by the same query under the same scoring
|
|
63
|
+
rule, so no future change to that rule can silently recalibrate it.
|
|
64
|
+
"""
|
|
65
|
+
if not results:
|
|
66
|
+
return 0.0
|
|
67
|
+
top = results[0].score
|
|
68
|
+
if len(results) == 1:
|
|
69
|
+
return 1.0
|
|
70
|
+
return (top - results[1].score) / top if top else 0.0
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def research(
|
|
74
|
+
units: list[CodeUnit],
|
|
75
|
+
query: str,
|
|
76
|
+
limit: int = 8,
|
|
77
|
+
hops: int = 1,
|
|
78
|
+
max_steps: int = 2,
|
|
79
|
+
dominance_threshold: float = DEFAULT_DOMINANCE,
|
|
80
|
+
graph: CodeGraph | None = None,
|
|
81
|
+
search_index: SearchIndex | None = None,
|
|
82
|
+
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
83
|
+
max_chars: int = 12000,
|
|
84
|
+
) -> dict:
|
|
85
|
+
"""Run at most two deterministic retrieval steps and explain the stop.
|
|
86
|
+
|
|
87
|
+
This is deliberately bounded. A future LLM planner can replace the query
|
|
88
|
+
proposal, but the budget, evidence format, and no-progress stop remain
|
|
89
|
+
stable safety contracts.
|
|
90
|
+
|
|
91
|
+
The reply carries the code once, in a context block trimmed to
|
|
92
|
+
``max_chars``. Results are navigation only. Reporting two retrieval steps
|
|
93
|
+
used to mean serialising the same eight units three times with their
|
|
94
|
+
source attached, which is how one research answer reached 111,843
|
|
95
|
+
characters against a stated budget of 12,000.
|
|
96
|
+
"""
|
|
97
|
+
max_steps = min(2, max(1, max_steps))
|
|
98
|
+
steps: list[dict] = []
|
|
99
|
+
initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight)
|
|
100
|
+
steps.append({"action": "search", "query": query, "results": _trace(initial)})
|
|
101
|
+
if not initial:
|
|
102
|
+
return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
|
|
103
|
+
unopposed = dominance(initial) >= dominance_threshold and bool(initial[0].matched_terms)
|
|
104
|
+
if max_steps == 1 or unopposed:
|
|
105
|
+
kept = initial[:limit]
|
|
106
|
+
return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
|
|
107
|
+
|
|
108
|
+
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight)
|
|
109
|
+
steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
|
|
110
|
+
merged = {result.unit.id: result for result in initial}
|
|
111
|
+
for result in expanded:
|
|
112
|
+
current = merged.get(result.unit.id)
|
|
113
|
+
if current is None or result.score > current.score:
|
|
114
|
+
merged[result.unit.id] = result
|
|
115
|
+
final = sorted(merged.values(), key=lambda result: (-result.score, result.unit.id))[:limit]
|
|
116
|
+
new_ids = _result_ids(final) - _result_ids(initial)
|
|
117
|
+
reason = "new_graph_evidence" if new_ids else "no_new_evidence"
|
|
118
|
+
return {"query": query, "results": _serialize(final), "steps": steps, "stop_reason": reason, "context": context(within_budget(final, max_chars), max_chars)}
|