rag-your-code 0.6.0__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.6.0/src/rag_your_code.egg-info → rag_your_code-0.7.0}/PKG-INFO +55 -8
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/README.md +54 -7
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/pyproject.toml +1 -1
- {rag_your_code-0.6.0 → rag_your_code-0.7.0/src/rag_your_code.egg-info}/PKG-INFO +55 -8
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/SOURCES.txt +3 -1
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/__init__.py +1 -1
- rag_your_code-0.7.0/src/ragyourcode/agentic.py +118 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/cli.py +70 -118
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/models.py +23 -11
- rag_your_code-0.7.0/src/ragyourcode/workflow.py +184 -0
- rag_your_code-0.7.0/tests/test_agentic.py +146 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_metadata.py +26 -2
- rag_your_code-0.7.0/tests/test_workflow.py +127 -0
- rag_your_code-0.6.0/src/ragyourcode/agentic.py +0 -62
- rag_your_code-0.6.0/tests/test_agentic.py +0 -29
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/LICENSE +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/setup.cfg +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/config.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/search.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_config.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_document.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_ranking.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_resilience.py +0 -0
- {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_retrieval_correctness.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
330
368
|
one reply per line:
|
|
331
369
|
|
|
332
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
333
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
334
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
335
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -340,6 +379,14 @@ one reply per line:
|
|
|
340
379
|
{"action":"stats"}
|
|
341
380
|
```
|
|
342
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
343
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
344
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
345
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -49,14 +49,23 @@ package itself on first use.
|
|
|
49
49
|
```bash
|
|
50
50
|
pip install rag-your-code
|
|
51
51
|
|
|
52
|
-
rag-your-code index
|
|
52
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
53
53
|
rag-your-code search "where are HTTP retries handled" --json
|
|
54
54
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
55
55
|
```
|
|
56
56
|
|
|
57
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
58
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
59
|
+
sentence the parser generated, which adds no word the source did not already
|
|
60
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
61
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
62
|
+
work. It reads the state rather than remembering a position, so running it
|
|
63
|
+
again after each round is how you make progress. `index` still exists and does
|
|
64
|
+
only the indexing.
|
|
65
|
+
|
|
57
66
|
The index is written under `.rag-your-code/`; your source files are never
|
|
58
|
-
modified. Later
|
|
59
|
-
|
|
67
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
68
|
+
`--compact`.
|
|
60
69
|
|
|
61
70
|
## How it works
|
|
62
71
|
|
|
@@ -108,16 +117,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
108
117
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
109
118
|
word either way.
|
|
110
119
|
|
|
111
|
-
Retrieval works regardless, because **
|
|
112
|
-
natural language** —
|
|
113
|
-
|
|
114
|
-
|
|
120
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
121
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
122
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
123
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
124
|
+
implemented and measured against all three rulers, with query and stored
|
|
125
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
126
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
127
|
+
discounts to nothing.
|
|
128
|
+
|
|
129
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
130
|
+
rest of the gap, and neither is a model:
|
|
115
131
|
|
|
116
132
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
117
133
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
118
134
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
119
135
|
into the index once instead of into every query.
|
|
120
136
|
|
|
137
|
+
### Seven attempts to make the vector half earn its place
|
|
138
|
+
|
|
139
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
140
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
141
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
142
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
143
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
144
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
145
|
+
|
|
146
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
147
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
148
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
149
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
150
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
151
|
+
|
|
152
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
153
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
154
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
155
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
156
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
157
|
+
scheme that sharpened it lost ground there.
|
|
158
|
+
|
|
121
159
|
## Agent-authored descriptions
|
|
122
160
|
|
|
123
161
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -303,6 +341,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
303
341
|
one reply per line:
|
|
304
342
|
|
|
305
343
|
```json
|
|
344
|
+
{"action":"bootstrap"}
|
|
306
345
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
307
346
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
308
347
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -313,6 +352,14 @@ one reply per line:
|
|
|
313
352
|
{"action":"stats"}
|
|
314
353
|
```
|
|
315
354
|
|
|
355
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
356
|
+
path, line range, signature, description, score and matched terms. The code
|
|
357
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
358
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
359
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
360
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
361
|
+
111,843 by serialising the same eight units three times over.
|
|
362
|
+
|
|
316
363
|
**No single request can end the session.** Numeric fields saturate at their
|
|
317
364
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
318
365
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -76,14 +76,23 @@ package itself on first use.
|
|
|
76
76
|
```bash
|
|
77
77
|
pip install rag-your-code
|
|
78
78
|
|
|
79
|
-
rag-your-code index
|
|
79
|
+
rag-your-code bootstrap . # index, and say what is still missing
|
|
80
80
|
rag-your-code search "where are HTTP retries handled" --json
|
|
81
81
|
rag-your-code search "what calls the retry handler" --graph --hops 1 --json
|
|
82
82
|
```
|
|
83
83
|
|
|
84
|
+
`bootstrap` exists because indexing a repository is not the same as making it
|
|
85
|
+
searchable, and nothing used to say so. A fresh index retrieves against the
|
|
86
|
+
sentence the parser generated, which adds no word the source did not already
|
|
87
|
+
have. It reports which rung this repository is on — descriptions still to
|
|
88
|
+
write, a promotion to apply, or nothing left — and hands over that rung's
|
|
89
|
+
work. It reads the state rather than remembering a position, so running it
|
|
90
|
+
again after each round is how you make progress. `index` still exists and does
|
|
91
|
+
only the indexing.
|
|
92
|
+
|
|
84
93
|
The index is written under `.rag-your-code/`; your source files are never
|
|
85
|
-
modified. Later
|
|
86
|
-
|
|
94
|
+
modified. Later runs reuse unchanged files. For a large repository, prefer
|
|
95
|
+
`--compact`.
|
|
87
96
|
|
|
88
97
|
## How it works
|
|
89
98
|
|
|
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
|
|
|
135
144
|
an unrelated pair are indistinguishable, because no shared word is no shared
|
|
136
145
|
word either way.
|
|
137
146
|
|
|
138
|
-
Retrieval works regardless, because **
|
|
139
|
-
natural language** —
|
|
140
|
-
|
|
141
|
-
|
|
147
|
+
Retrieval works regardless, because **the prose people write about code is
|
|
148
|
+
already natural language** — docstrings, comments, descriptions. Identifiers
|
|
149
|
+
are not part of that, and it is worth being exact: `retry_charge` tokenizes to
|
|
150
|
+
one opaque term, not to *retry* and *charge*. Splitting identifiers was
|
|
151
|
+
implemented and measured against all three rulers, with query and stored
|
|
152
|
+
vectors rebuilt together, and it was equal or worse on every one; the pieces it
|
|
153
|
+
makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
|
|
154
|
+
discounts to nothing.
|
|
155
|
+
|
|
156
|
+
So retrieval reaches only concepts somebody wrote down. Two things close the
|
|
157
|
+
rest of the gap, and neither is a model:
|
|
142
158
|
|
|
143
159
|
- **Your agent rewrites the query.** It has the conversation; turning
|
|
144
160
|
"重试扣款" into `retry charge payment gateway` costs it nothing.
|
|
145
161
|
- **Your agent writes the descriptions**, which puts the missing vocabulary
|
|
146
162
|
into the index once instead of into every query.
|
|
147
163
|
|
|
164
|
+
### Seven attempts to make the vector half earn its place
|
|
165
|
+
|
|
166
|
+
Because "just use a better embedding" is the obvious next thought, it was
|
|
167
|
+
measured rather than argued about. Six schemes were implemented — character
|
|
168
|
+
n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
|
|
169
|
+
signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
|
|
170
|
+
lexical postings expansion and embedding only the authored text. **On the
|
|
171
|
+
foreign-repository ruler, not one of them beat using no vector at all.**
|
|
172
|
+
|
|
173
|
+
The reason is architectural, not representational. Retrieval scores only the
|
|
174
|
+
units the lexical half already matched, so a vector can reorder an answer but
|
|
175
|
+
can never make one *retrievable*; pure cosine fires only when nothing matched
|
|
176
|
+
at all, on 1 question of 35. Every scheme was competing for the same one- or
|
|
177
|
+
two-question reshuffle inside a list that had already been chosen.
|
|
178
|
+
|
|
179
|
+
Two things follow, and both are stated here rather than buried. Corpus-learned
|
|
180
|
+
semantics need orders of magnitude more text than a repository has: 65% of the
|
|
181
|
+
foreign corpus's terms appear in four or fewer units, so their co-occurrence
|
|
182
|
+
row is a handful of sightings rather than a distribution. And on a *described*
|
|
183
|
+
repository the shipped hash is useful precisely because it is blunt — every
|
|
184
|
+
scheme that sharpened it lost ground there.
|
|
185
|
+
|
|
148
186
|
## Agent-authored descriptions
|
|
149
187
|
|
|
150
188
|
Every unit carries a description, and that description is indexed. By default
|
|
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
|
|
|
330
368
|
one reply per line:
|
|
331
369
|
|
|
332
370
|
```json
|
|
371
|
+
{"action":"bootstrap"}
|
|
333
372
|
{"action":"search","query":"database transaction rollback","limit":5}
|
|
334
373
|
{"action":"research","query":"trace payment retry behavior","max_steps":2}
|
|
335
374
|
{"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
|
|
@@ -340,6 +379,14 @@ one reply per line:
|
|
|
340
379
|
{"action":"stats"}
|
|
341
380
|
```
|
|
342
381
|
|
|
382
|
+
**A result is navigation, not the file.** `results` carries the identifier,
|
|
383
|
+
path, line range, signature, description, score and matched terms. The code
|
|
384
|
+
arrives once, in the reply's `context`, trimmed to `max_chars`, and
|
|
385
|
+
`omitted_for_budget` says how many results it did not reach. Carrying the
|
|
386
|
+
source per result as well is what let one `search --json` reply reach 65,025
|
|
387
|
+
characters against a stated budget of 12,000, and one `research` reply reach
|
|
388
|
+
111,843 by serialising the same eight units three times over.
|
|
389
|
+
|
|
343
390
|
**No single request can end the session.** Numeric fields saturate at their
|
|
344
391
|
bounds, `open` is bounded in both lines and bytes, and anything unanticipated
|
|
345
392
|
is reported in-band with its exception type. Streams are pinned to UTF-8
|
|
@@ -21,6 +21,7 @@ src/ragyourcode/models.py
|
|
|
21
21
|
src/ragyourcode/parser.py
|
|
22
22
|
src/ragyourcode/py.typed
|
|
23
23
|
src/ragyourcode/search.py
|
|
24
|
+
src/ragyourcode/workflow.py
|
|
24
25
|
tests/test_agent_protocol.py
|
|
25
26
|
tests/test_agentic.py
|
|
26
27
|
tests/test_config.py
|
|
@@ -39,4 +40,5 @@ tests/test_ragyourcode.py
|
|
|
39
40
|
tests/test_ranking.py
|
|
40
41
|
tests/test_repo_queries.py
|
|
41
42
|
tests/test_resilience.py
|
|
42
|
-
tests/test_retrieval_correctness.py
|
|
43
|
+
tests/test_retrieval_correctness.py
|
|
44
|
+
tests/test_workflow.py
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Bounded, observable agentic retrieval (ARAG) orchestration."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .graph import CodeGraph, graph_search
|
|
6
|
+
from .models import CodeUnit, SearchResult
|
|
7
|
+
from .search import DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _result_ids(results: list[SearchResult]) -> set[str]:
|
|
11
|
+
"""Collects the unit identifiers out of a list of search results, so two
|
|
12
|
+
result sets can be compared for overlap or novelty. Used to decide
|
|
13
|
+
whether a second retrieval round actually surfaced anything new.
|
|
14
|
+
"""
|
|
15
|
+
return {result.unit.id for result in results}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
# How far ahead of the runner-up the top result must be before a second
|
|
19
|
+
# retrieval step is skipped, as a fraction of the top score.
|
|
20
|
+
#
|
|
21
|
+
# This replaced an absolute `confidence_threshold` of 0.8, which was a number
|
|
22
|
+
# tied to a scoring scale. When ranking became BM25F the scale moved and the
|
|
23
|
+
# threshold quietly died: measured across 105 ruler questions, only 3% of
|
|
24
|
+
# queries reached 0.8 at all, so the early stop had stopped existing and every
|
|
25
|
+
# research call ran the graph expansion. A margin is a ratio between two scores
|
|
26
|
+
# from the same query, so it cannot drift when the scoring changes again.
|
|
27
|
+
#
|
|
28
|
+
# Measured over those 105 questions: top-1 is correct 42% of the time overall,
|
|
29
|
+
# and 68-73% of the time among the queries this fires on. Anywhere in 0.2-0.5
|
|
30
|
+
# behaves the same at this sample size; 0.30 is the middle of that band, not a
|
|
31
|
+
# measured optimum.
|
|
32
|
+
DEFAULT_DOMINANCE: float = 0.30
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _serialize(results: list[SearchResult]) -> list[dict]:
|
|
36
|
+
"""Converts search results into plain JSON-ready dictionaries for the agent
|
|
37
|
+
protocol reply.
|
|
38
|
+
"""
|
|
39
|
+
return [result.to_dict() for result in results]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _trace(results: list[SearchResult]) -> list[dict]:
|
|
43
|
+
"""One step of the search, as a trace rather than as a payload.
|
|
44
|
+
|
|
45
|
+
A step exists so the caller can see how the answer was reached: which
|
|
46
|
+
units each round reached and how strongly. That needs an identifier, a
|
|
47
|
+
score and the matching words -- not a second and third copy of every
|
|
48
|
+
unit's description, callees and imports. Reporting two steps in full is
|
|
49
|
+
what made a research reply three times the size of the answer inside it.
|
|
50
|
+
"""
|
|
51
|
+
return [
|
|
52
|
+
{"id": result.unit.id, "score": round(result.score, 6), "matched_terms": result.matched_terms}
|
|
53
|
+
for result in results
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def dominance(results: list[SearchResult]) -> float:
|
|
58
|
+
"""How far the top result is ahead of the runner-up, relative to the top.
|
|
59
|
+
|
|
60
|
+
One result alone is unopposed and scores 1.0. Nothing scores 0.0. The
|
|
61
|
+
quantity is scale-free by construction, which is the whole point: it
|
|
62
|
+
compares two numbers produced by the same query under the same scoring
|
|
63
|
+
rule, so no future change to that rule can silently recalibrate it.
|
|
64
|
+
"""
|
|
65
|
+
if not results:
|
|
66
|
+
return 0.0
|
|
67
|
+
top = results[0].score
|
|
68
|
+
if len(results) == 1:
|
|
69
|
+
return 1.0
|
|
70
|
+
return (top - results[1].score) / top if top else 0.0
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def research(
|
|
74
|
+
units: list[CodeUnit],
|
|
75
|
+
query: str,
|
|
76
|
+
limit: int = 8,
|
|
77
|
+
hops: int = 1,
|
|
78
|
+
max_steps: int = 2,
|
|
79
|
+
dominance_threshold: float = DEFAULT_DOMINANCE,
|
|
80
|
+
graph: CodeGraph | None = None,
|
|
81
|
+
search_index: SearchIndex | None = None,
|
|
82
|
+
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
83
|
+
max_chars: int = 12000,
|
|
84
|
+
) -> dict:
|
|
85
|
+
"""Run at most two deterministic retrieval steps and explain the stop.
|
|
86
|
+
|
|
87
|
+
This is deliberately bounded. A future LLM planner can replace the query
|
|
88
|
+
proposal, but the budget, evidence format, and no-progress stop remain
|
|
89
|
+
stable safety contracts.
|
|
90
|
+
|
|
91
|
+
The reply carries the code once, in a context block trimmed to
|
|
92
|
+
``max_chars``. Results are navigation only. Reporting two retrieval steps
|
|
93
|
+
used to mean serialising the same eight units three times with their
|
|
94
|
+
source attached, which is how one research answer reached 111,843
|
|
95
|
+
characters against a stated budget of 12,000.
|
|
96
|
+
"""
|
|
97
|
+
max_steps = min(2, max(1, max_steps))
|
|
98
|
+
steps: list[dict] = []
|
|
99
|
+
initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight)
|
|
100
|
+
steps.append({"action": "search", "query": query, "results": _trace(initial)})
|
|
101
|
+
if not initial:
|
|
102
|
+
return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
|
|
103
|
+
unopposed = dominance(initial) >= dominance_threshold and bool(initial[0].matched_terms)
|
|
104
|
+
if max_steps == 1 or unopposed:
|
|
105
|
+
kept = initial[:limit]
|
|
106
|
+
return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
|
|
107
|
+
|
|
108
|
+
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight)
|
|
109
|
+
steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
|
|
110
|
+
merged = {result.unit.id: result for result in initial}
|
|
111
|
+
for result in expanded:
|
|
112
|
+
current = merged.get(result.unit.id)
|
|
113
|
+
if current is None or result.score > current.score:
|
|
114
|
+
merged[result.unit.id] = result
|
|
115
|
+
final = sorted(merged.values(), key=lambda result: (-result.score, result.unit.id))[:limit]
|
|
116
|
+
new_ids = _result_ids(final) - _result_ids(initial)
|
|
117
|
+
reason = "new_graph_evidence" if new_ids else "no_new_evidence"
|
|
118
|
+
return {"query": query, "results": _serialize(final), "steps": steps, "stop_reason": reason, "context": context(within_budget(final, max_chars), max_chars)}
|