rag-your-code 0.5.0__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.5.0/src/rag_your_code.egg-info → rag_your_code-0.6.0}/PKG-INFO +38 -8
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/README.md +37 -7
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/pyproject.toml +1 -1
- {rag_your_code-0.5.0 → rag_your_code-0.6.0/src/rag_your_code.egg-info}/PKG-INFO +38 -8
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/SOURCES.txt +1 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/cli.py +8 -3
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/models.py +29 -15
- rag_your_code-0.6.0/src/ragyourcode/search.py +278 -0
- rag_your_code-0.6.0/tests/test_ranking.py +202 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_repo_queries.py +40 -5
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_retrieval_correctness.py +4 -22
- rag_your_code-0.5.0/src/ragyourcode/search.py +0 -151
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/LICENSE +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/setup.cfg +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/config.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_agentic.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_config.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_document.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_metadata.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_resilience.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -198,19 +198,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
198
198
|
|
|
199
199
|
| | generated descriptions | agent-written |
|
|
200
200
|
|---|---|---|
|
|
201
|
-
| hit@1 | 0.
|
|
202
|
-
| hit@3 | 0.
|
|
203
|
-
| MRR | 0.
|
|
204
|
-
| answered with no shared word at all |
|
|
201
|
+
| hit@1 | 0.271 | **0.500** |
|
|
202
|
+
| hit@3 | 0.486 | **0.800** |
|
|
203
|
+
| MRR | 0.367 | **0.631** |
|
|
204
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
205
205
|
|
|
206
|
-
Roughly
|
|
207
|
-
|
|
208
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
206
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
207
|
+
is what makes the set usable for measuring the next change;
|
|
208
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
209
|
+
the written column beats the generated one.
|
|
209
210
|
|
|
210
211
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
211
212
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
212
213
|
stemming — exactly the limit documented above.
|
|
213
214
|
|
|
215
|
+
### Measured on a repository nobody here wrote
|
|
216
|
+
|
|
217
|
+
The table above is the warmest case this project supports: its own code, its
|
|
218
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
219
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
220
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
221
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
222
|
+
the words of the docstring that answers it
|
|
223
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
224
|
+
|
|
225
|
+
| | before 0.6.0 | now |
|
|
226
|
+
|---|---|---|
|
|
227
|
+
| hit@1 | 0.086 | **0.257** |
|
|
228
|
+
| hit@3 | 0.229 | **0.400** |
|
|
229
|
+
| MRR | 0.157 | **0.314** |
|
|
230
|
+
|
|
231
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
232
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
233
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
234
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
235
|
+
that repository came back in the top three for four questions out of six. It
|
|
236
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
237
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
238
|
+
buried in a body.
|
|
239
|
+
|
|
240
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
241
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
242
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
243
|
+
|
|
214
244
|
**What this is:** it moves the semantic work from query time to index time.
|
|
215
245
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
216
246
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -171,19 +171,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
171
171
|
|
|
172
172
|
| | generated descriptions | agent-written |
|
|
173
173
|
|---|---|---|
|
|
174
|
-
| hit@1 | 0.
|
|
175
|
-
| hit@3 | 0.
|
|
176
|
-
| MRR | 0.
|
|
177
|
-
| answered with no shared word at all |
|
|
174
|
+
| hit@1 | 0.271 | **0.500** |
|
|
175
|
+
| hit@3 | 0.486 | **0.800** |
|
|
176
|
+
| MRR | 0.367 | **0.631** |
|
|
177
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
178
178
|
|
|
179
|
-
Roughly
|
|
180
|
-
|
|
181
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
179
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
180
|
+
is what makes the set usable for measuring the next change;
|
|
181
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
182
|
+
the written column beats the generated one.
|
|
182
183
|
|
|
183
184
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
184
185
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
185
186
|
stemming — exactly the limit documented above.
|
|
186
187
|
|
|
188
|
+
### Measured on a repository nobody here wrote
|
|
189
|
+
|
|
190
|
+
The table above is the warmest case this project supports: its own code, its
|
|
191
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
192
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
193
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
194
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
195
|
+
the words of the docstring that answers it
|
|
196
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
197
|
+
|
|
198
|
+
| | before 0.6.0 | now |
|
|
199
|
+
|---|---|---|
|
|
200
|
+
| hit@1 | 0.086 | **0.257** |
|
|
201
|
+
| hit@3 | 0.229 | **0.400** |
|
|
202
|
+
| MRR | 0.157 | **0.314** |
|
|
203
|
+
|
|
204
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
205
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
206
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
207
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
208
|
+
that repository came back in the top three for four questions out of six. It
|
|
209
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
210
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
211
|
+
buried in a body.
|
|
212
|
+
|
|
213
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
214
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
215
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
216
|
+
|
|
187
217
|
**What this is:** it moves the semantic work from query time to index time.
|
|
188
218
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
189
219
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -198,19 +198,49 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
198
198
|
|
|
199
199
|
| | generated descriptions | agent-written |
|
|
200
200
|
|---|---|---|
|
|
201
|
-
| hit@1 | 0.
|
|
202
|
-
| hit@3 | 0.
|
|
203
|
-
| MRR | 0.
|
|
204
|
-
| answered with no shared word at all |
|
|
201
|
+
| hit@1 | 0.271 | **0.500** |
|
|
202
|
+
| hit@3 | 0.486 | **0.800** |
|
|
203
|
+
| MRR | 0.367 | **0.631** |
|
|
204
|
+
| answered with no shared word at all | 12.9% | **0%** |
|
|
205
205
|
|
|
206
|
-
Roughly
|
|
207
|
-
|
|
208
|
-
`tests/test_repo_queries.py` asserts that some question always does
|
|
206
|
+
Roughly double the first-place accuracy. Fourteen questions still fail, which
|
|
207
|
+
is what makes the set usable for measuring the next change;
|
|
208
|
+
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
209
|
+
the written column beats the generated one.
|
|
209
210
|
|
|
210
211
|
One failure is worth naming: a query saying `catastrophic backtracking` does
|
|
211
212
|
not reach a description saying `backtracks catastrophically`. There is no
|
|
212
213
|
stemming — exactly the limit documented above.
|
|
213
214
|
|
|
215
|
+
### Measured on a repository nobody here wrote
|
|
216
|
+
|
|
217
|
+
The table above is the warmest case this project supports: its own code, its
|
|
218
|
+
own descriptions, and questions written by the same party. It cannot say what
|
|
219
|
+
a first-time user gets. So there is a second ruler — thirty-five questions
|
|
220
|
+
about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
|
|
221
|
+
descriptions at all, each question phrased in a user's words rather than in
|
|
222
|
+
the words of the docstring that answers it
|
|
223
|
+
([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
|
|
224
|
+
|
|
225
|
+
| | before 0.6.0 | now |
|
|
226
|
+
|---|---|---|
|
|
227
|
+
| hit@1 | 0.086 | **0.257** |
|
|
228
|
+
| hit@3 | 0.229 | **0.400** |
|
|
229
|
+
| MRR | 0.157 | **0.314** |
|
|
230
|
+
|
|
231
|
+
Three times the first-place accuracy, and the same change moved both other
|
|
232
|
+
rulers in the same direction. What it fixed was ranking: scoring used to be
|
|
233
|
+
the fraction of query words a unit contained, so `the` counted for as much as
|
|
234
|
+
`daemon`, and nothing corrected for size — the single largest declaration in
|
|
235
|
+
that repository came back in the top three for four questions out of six. It
|
|
236
|
+
is now BM25 over weighted fields, where a word's worth comes from how rare it
|
|
237
|
+
is in *your* corpus and a word in a declaration's name outweighs the same word
|
|
238
|
+
buried in a body.
|
|
239
|
+
|
|
240
|
+
Twenty-one of the thirty-five still fail, and the largest remaining cause is
|
|
241
|
+
named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
|
|
242
|
+
the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
243
|
+
|
|
214
244
|
**What this is:** it moves the semantic work from query time to index time.
|
|
215
245
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
216
246
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
@@ -18,7 +18,7 @@ from .document import plan as plan_documentation, render_patch, summarise as sum
|
|
|
18
18
|
from .embeddings import embed, embedding_metadata
|
|
19
19
|
from .graph import build_graph, graph_from_dict, graph_search
|
|
20
20
|
from .indexer import StaleMonitor, build_fingerprint, build_units, fingerprint, index_build_fingerprint, read_index, snapshot_repository, write_index
|
|
21
|
-
from .search import build_search_index, context, search
|
|
21
|
+
from .search import build_search_index, context, search, within_budget
|
|
22
22
|
|
|
23
23
|
# Derived from the settings table so the default is written down once.
|
|
24
24
|
# `tests/test_agent_protocol.py` imports these to assert the bound it enforces.
|
|
@@ -157,7 +157,10 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
157
157
|
else search(units, args.query, limit, search_index=search_index, vector_weight=weight)
|
|
158
158
|
)
|
|
159
159
|
if args.json:
|
|
160
|
-
|
|
160
|
+
# One budget decision, applied once: the results an agent reads and the
|
|
161
|
+
# context beside them are the same set, so they cannot disagree.
|
|
162
|
+
shown = within_budget(results, max_chars)
|
|
163
|
+
print(json.dumps({"query": args.query, "mode": "graph" if args.graph else "hybrid", "stale": payload.get("stale", True), "degraded": payload.get("degraded"), "results": [result.to_dict() for result in shown], "omitted_for_budget": len(results) - len(shown), "context": context(shown, max_chars)}, ensure_ascii=False))
|
|
161
164
|
else:
|
|
162
165
|
if payload.get("stale"):
|
|
163
166
|
print("Warning: index is stale; run `rag-your-code index` to refresh.", file=sys.stderr)
|
|
@@ -496,7 +499,9 @@ def _cmd_agent(args: argparse.Namespace) -> int:
|
|
|
496
499
|
if use_graph
|
|
497
500
|
else search(units, query, limit, search_index=search_index, vector_weight=weight)
|
|
498
501
|
)
|
|
499
|
-
|
|
502
|
+
budget = _request_int(request, "max_chars", default_chars, 0, 100000)
|
|
503
|
+
shown = within_budget(results, budget)
|
|
504
|
+
response = {"stale": payload.get("stale", True), "results": [result.to_dict() for result in shown], "omitted_for_budget": len(results) - len(shown), "context": context(shown, budget)}
|
|
500
505
|
elif action == "research":
|
|
501
506
|
response = research(
|
|
502
507
|
units,
|
|
@@ -28,24 +28,38 @@ class CodeUnit:
|
|
|
28
28
|
imports: list[str] = field(default_factory=list)
|
|
29
29
|
vector: Sequence[float] = field(default_factory=list)
|
|
30
30
|
|
|
31
|
+
@property
|
|
32
|
+
def searchable_fields(self) -> dict[str, str]:
|
|
33
|
+
"""Everything retrieval may match, kept apart by where the author
|
|
34
|
+
wrote it. Where a word appears is itself evidence of how much it
|
|
35
|
+
means: a term in a declaration's name is what the author decided to
|
|
36
|
+
call the thing, while the same term two hundred lines into a body is
|
|
37
|
+
a passing mention. Ranking weights those differently, so they cannot
|
|
38
|
+
arrive as one undifferentiated blob -- flattened, a long function
|
|
39
|
+
outranks the function that is actually named after the query, purely
|
|
40
|
+
by owning more words.
|
|
41
|
+
|
|
42
|
+
Because the description is one of these fields, replacing a generated
|
|
43
|
+
description with a better one immediately widens the set of queries
|
|
44
|
+
that can reach this unit.
|
|
45
|
+
"""
|
|
46
|
+
return {
|
|
47
|
+
"name": f"{self.qualified_name} {self.kind}",
|
|
48
|
+
"signature": self.signature,
|
|
49
|
+
"description": self.description,
|
|
50
|
+
"relations": "calls: " + " ".join(self.calls) + "\nimports: " + " ".join(self.imports),
|
|
51
|
+
"body": self.source,
|
|
52
|
+
}
|
|
53
|
+
|
|
31
54
|
@property
|
|
32
55
|
def searchable_text(self) -> str:
|
|
33
|
-
"""
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
reach this unit.
|
|
56
|
+
"""Every searchable field as one block, for callers that want the
|
|
57
|
+
words without caring where they came from -- the unit's own embedding
|
|
58
|
+
is built from this. Derived from ``searchable_fields`` rather than
|
|
59
|
+
rebuilt beside it, so a field added for ranking cannot go missing
|
|
60
|
+
from the text that gets embedded.
|
|
39
61
|
"""
|
|
40
|
-
return "\n".join(
|
|
41
|
-
(
|
|
42
|
-
f"{self.qualified_name} {self.kind} {self.signature}",
|
|
43
|
-
self.description,
|
|
44
|
-
"calls: " + " ".join(self.calls),
|
|
45
|
-
"imports: " + " ".join(self.imports),
|
|
46
|
-
self.source,
|
|
47
|
-
)
|
|
48
|
-
)
|
|
62
|
+
return "\n".join(self.searchable_fields.values())
|
|
49
63
|
|
|
50
64
|
def to_dict(self, include_vector: bool = True) -> dict[str, Any]:
|
|
51
65
|
"""Serialises a unit to a plain dictionary for storage or for an agent
|
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
"""Hybrid lexical/vector retrieval for agent context."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import heapq
|
|
6
|
+
import math
|
|
7
|
+
from bisect import bisect_left
|
|
8
|
+
from collections import Counter, defaultdict
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
|
|
11
|
+
from .config import BY_PATH
|
|
12
|
+
from .embeddings import DEFAULT_DIMENSIONS, embed, tokenize
|
|
13
|
+
from .models import CodeUnit, SearchResult
|
|
14
|
+
|
|
15
|
+
# Named here rather than repeated as a literal so `search.vector_weight` in
|
|
16
|
+
# rag-your-code.toml and the default a direct caller gets cannot drift apart.
|
|
17
|
+
DEFAULT_VECTOR_WEIGHT: float = BY_PATH["search.vector_weight"].default
|
|
18
|
+
|
|
19
|
+
# How much a word counts for, by the field the author wrote it in. A term in
|
|
20
|
+
# the name is what the declaration is called; the same term inside the body is
|
|
21
|
+
# a mention. Every key of `CodeUnit.searchable_fields` must appear here, and a
|
|
22
|
+
# test asserts the two sets agree -- a field with no weight would otherwise
|
|
23
|
+
# vanish from ranking the moment it was added, silently.
|
|
24
|
+
FIELD_WEIGHTS: dict[str, float] = {
|
|
25
|
+
"name": 8.0,
|
|
26
|
+
"signature": 4.0,
|
|
27
|
+
"description": 3.0,
|
|
28
|
+
"relations": 2.0,
|
|
29
|
+
"body": 1.0,
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
# BM25's saturation and length-normalisation constants, at their standard
|
|
33
|
+
# values. `k1` bounds what repeating a word can buy; `b` decides how hard a
|
|
34
|
+
# field is discounted for being longer than its average.
|
|
35
|
+
#
|
|
36
|
+
# Three variations were implemented and measured against all three rulers
|
|
37
|
+
# before settling here, and two were dropped for want of evidence: excluding
|
|
38
|
+
# curated text from the length on the argument that a written description is
|
|
39
|
+
# deliberate rather than incidental, and counting authored words instead of
|
|
40
|
+
# tokeniser output so a run of Chinese expanded into overlapping bigrams would
|
|
41
|
+
# not read as five times the text. Each moved one to four questions in both
|
|
42
|
+
# directions at once, which is this instrument's noise. Lowering `b` to 0.5 or
|
|
43
|
+
# 0.3 under per-field normalisation was measured and was worse than 0.75 on
|
|
44
|
+
# the foreign-repository ruler. See docs/TESTING.md.
|
|
45
|
+
BM25_K1 = 1.2
|
|
46
|
+
BM25_B = 0.75
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass(slots=True)
|
|
50
|
+
class SearchIndex:
|
|
51
|
+
"""In-memory inverted index reused across queries.
|
|
52
|
+
|
|
53
|
+
Building this once avoids re-tokenizing every code unit. Each posting
|
|
54
|
+
carries a term's weight in that unit alongside its id -- already scaled by
|
|
55
|
+
the field the term appeared in and by how long that field is -- because
|
|
56
|
+
ranking needs to know how often a term occurs and where, not merely that
|
|
57
|
+
it occurs somewhere. The weight lives in the posting rather than in a
|
|
58
|
+
per-unit table: an earlier version cached a per-unit frozenset of every
|
|
59
|
+
token, which cost the largest share of resident memory while holding
|
|
60
|
+
nothing the postings did not already have.
|
|
61
|
+
|
|
62
|
+
Lists are kept sorted by unit id so membership can be answered by binary
|
|
63
|
+
search. The same structure can later be backed by SQLite/ANN storage.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
units: dict[str, CodeUnit]
|
|
67
|
+
postings: dict[str, tuple[tuple[str, float], ...]]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def build_search_index(units: list[CodeUnit]) -> SearchIndex:
|
|
71
|
+
"""Builds the inverted lookup table, recording for each term the units that
|
|
72
|
+
contain it and how much it counts for in each.
|
|
73
|
+
|
|
74
|
+
A term's weight is BM25F's: its count in a field, divided by how long that
|
|
75
|
+
field is against the average for that same field, then scaled by what the
|
|
76
|
+
field is worth. Normalising per field is the part that matters. Measured
|
|
77
|
+
against one length for the whole unit, a body repeating a word forty times
|
|
78
|
+
still beat the declaration actually named after it, because a long body's
|
|
79
|
+
advantage in raw count almost exactly cancelled its penalty for being
|
|
80
|
+
long. Comparing each field against its own average removes that cancelling.
|
|
81
|
+
|
|
82
|
+
Two passes are needed because a field's average length is a property of
|
|
83
|
+
the corpus and is not known until every unit has been seen. The second
|
|
84
|
+
pass re-tokenizes rather than holding every unit's tokens in memory at
|
|
85
|
+
once, which costs about half again in time and keeps the peak bounded.
|
|
86
|
+
"""
|
|
87
|
+
field_totals: Counter[str] = Counter()
|
|
88
|
+
for unit in units:
|
|
89
|
+
for field_name, text in unit.searchable_fields.items():
|
|
90
|
+
field_totals[field_name] += len(tokenize(text))
|
|
91
|
+
count = len(units) or 1
|
|
92
|
+
# A field every unit leaves empty would otherwise divide by zero. Its terms
|
|
93
|
+
# cannot reach any unit anyway, so the value only has to be finite.
|
|
94
|
+
averages = {name: (total / count) or 1.0 for name, total in field_totals.items()}
|
|
95
|
+
|
|
96
|
+
postings: dict[str, list[tuple[str, float]]] = defaultdict(list)
|
|
97
|
+
for unit in units:
|
|
98
|
+
weighted: Counter[str] = Counter()
|
|
99
|
+
for field_name, text in unit.searchable_fields.items():
|
|
100
|
+
terms = tokenize(text)
|
|
101
|
+
if not terms:
|
|
102
|
+
continue
|
|
103
|
+
saturation = 1 - BM25_B + BM25_B * len(terms) / averages[field_name]
|
|
104
|
+
weight = FIELD_WEIGHTS[field_name] / saturation
|
|
105
|
+
for term, occurrences in Counter(terms).items():
|
|
106
|
+
weighted[term] += weight * occurrences
|
|
107
|
+
for term, mass in weighted.items():
|
|
108
|
+
postings[term].append((unit.id, mass))
|
|
109
|
+
return SearchIndex(
|
|
110
|
+
{unit.id: unit for unit in units},
|
|
111
|
+
{term: tuple(sorted(entries)) for term, entries in postings.items()},
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _in_posting(posting: tuple[tuple[str, float], ...], unit_id: str) -> bool:
|
|
116
|
+
"""Membership test over a posting list, which build_search_index keeps
|
|
117
|
+
sorted by unit id. A one-element tuple sorts before every pair sharing its
|
|
118
|
+
id, so it locates the entry without needing the mass that follows it.
|
|
119
|
+
"""
|
|
120
|
+
position = bisect_left(posting, (unit_id,))
|
|
121
|
+
return position < len(posting) and posting[position][0] == unit_id
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _inverse_document_frequency(document_count: int, matching: int) -> float:
|
|
125
|
+
"""How much evidence one term carries, from how rare it is in this corpus.
|
|
126
|
+
|
|
127
|
+
A word in nearly every unit says nothing about which unit is wanted, and
|
|
128
|
+
a word in two says a great deal. Deriving that from the corpus rather than
|
|
129
|
+
from a list of stopwords is what makes it work on a repository in any
|
|
130
|
+
language: `the` and `calls` earn their low weight the same way a Chinese
|
|
131
|
+
bigram does, by being everywhere, and no list has to be maintained.
|
|
132
|
+
"""
|
|
133
|
+
return math.log(1 + (document_count - matching + 0.5) / (matching + 0.5))
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def search(
|
|
137
|
+
units: list[CodeUnit],
|
|
138
|
+
query: str,
|
|
139
|
+
limit: int = 8,
|
|
140
|
+
search_index: SearchIndex | None = None,
|
|
141
|
+
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
142
|
+
) -> list[SearchResult]:
|
|
143
|
+
"""Ranks code units against a natural-language query by combining weighted
|
|
144
|
+
word overlap with vector similarity.
|
|
145
|
+
|
|
146
|
+
The lexical half is BM25 over weighted fields: rare words count for more
|
|
147
|
+
than common ones, repeating a word saturates rather than accumulating
|
|
148
|
+
without bound, and a unit is normalised by how much text it owns. An
|
|
149
|
+
earlier version scored the plain fraction of query words present, which
|
|
150
|
+
made `the` worth as much as `daemon` and handed every query to whichever
|
|
151
|
+
unit was longest -- on a foreign repository the single largest declaration
|
|
152
|
+
came back for four questions out of six.
|
|
153
|
+
|
|
154
|
+
Every unit sharing any query word is scored, so nothing that matches is
|
|
155
|
+
left out. Full vector scoring is reserved for units reached by a selective
|
|
156
|
+
term, since computing a dot product for everything a stopword-class term
|
|
157
|
+
touches is pure cost. With no overlap anywhere it falls back to similarity
|
|
158
|
+
alone.
|
|
159
|
+
"""
|
|
160
|
+
query_tokens = set(tokenize(query))
|
|
161
|
+
if limit <= 0 or not query_tokens:
|
|
162
|
+
return []
|
|
163
|
+
query_vector = embed(query, len(units[0].vector) if units and units[0].vector else DEFAULT_DIMENSIONS)
|
|
164
|
+
query_features = [(index, value) for index, value in enumerate(query_vector) if value]
|
|
165
|
+
search_index = search_index or build_search_index(units)
|
|
166
|
+
postings = [(token, search_index.postings.get(token, ())) for token in query_tokens]
|
|
167
|
+
document_count = len(search_index.units) or 1
|
|
168
|
+
|
|
169
|
+
# Accumulate term by term rather than unit by unit: a posting list is
|
|
170
|
+
# exactly the units a term reaches, so this touches no unit the query
|
|
171
|
+
# cannot possibly match. Length normalisation is already inside `mass`, so
|
|
172
|
+
# all that remains is saturation -- the tenth occurrence of a word says
|
|
173
|
+
# much less than the second, and neither should be able to run away with
|
|
174
|
+
# the ranking.
|
|
175
|
+
weights = {token: _inverse_document_frequency(document_count, len(posting)) for token, posting in postings if posting}
|
|
176
|
+
lexical_scores: dict[str, float] = defaultdict(float)
|
|
177
|
+
for token, posting in postings:
|
|
178
|
+
if not posting:
|
|
179
|
+
continue
|
|
180
|
+
weight = weights[token]
|
|
181
|
+
for unit_id, mass in posting:
|
|
182
|
+
lexical_scores[unit_id] += weight * mass * (BM25_K1 + 1) / (mass + BM25_K1)
|
|
183
|
+
|
|
184
|
+
# Divide by what this query could have scored at most, which is a constant
|
|
185
|
+
# for the query and so changes no ranking. It exists to keep the lexical
|
|
186
|
+
# half on the 0..1 scale `search.vector_weight` was documented and bounded
|
|
187
|
+
# against: raw BM25 is unbounded, and against a score of 18 a weight of
|
|
188
|
+
# 0.35 on a cosine would be arithmetic, not a tie-break.
|
|
189
|
+
ceiling = sum(weights.values()) * (BM25_K1 + 1)
|
|
190
|
+
|
|
191
|
+
# A term present in a tenth of the corpus (``function``, ``return``) is not
|
|
192
|
+
# evidence of relevance, and its posting list is effectively the whole index;
|
|
193
|
+
# computing a 384-dimension dot product for everything it reaches is what
|
|
194
|
+
# this threshold exists to avoid. It selects which candidates additionally
|
|
195
|
+
# receive a VECTOR score. It must not decide which candidates are scored at
|
|
196
|
+
# all -- doing that silently dropped units matching MORE query terms and
|
|
197
|
+
# under-filled ``limit`` (116 units, `--limit 8`, one result returned).
|
|
198
|
+
selective_threshold = max(64, min(2048, len(units) // 10))
|
|
199
|
+
vector_ids: set[str] = set()
|
|
200
|
+
for _, posting in postings:
|
|
201
|
+
if 0 < len(posting) <= selective_threshold:
|
|
202
|
+
vector_ids.update(unit_id for unit_id, _ in posting)
|
|
203
|
+
if lexical_scores and not vector_ids and len(lexical_scores) <= selective_threshold:
|
|
204
|
+
vector_ids = set(lexical_scores)
|
|
205
|
+
|
|
206
|
+
# With no lexical overlap anywhere, fall back to pure cosine so a genuine
|
|
207
|
+
# paraphrase still retrieves something.
|
|
208
|
+
candidate_ids = lexical_scores.keys() if lexical_scores else search_index.units.keys()
|
|
209
|
+
scored: list[tuple[float, str]] = []
|
|
210
|
+
for unit_id in candidate_ids:
|
|
211
|
+
unit = search_index.units[unit_id]
|
|
212
|
+
lexical = (lexical_scores.get(unit_id, 0.0) / ceiling) if ceiling else 0.0
|
|
213
|
+
vector_score = (
|
|
214
|
+
sum(value * unit.vector[index] for index, value in query_features)
|
|
215
|
+
if (unit_id in vector_ids or not lexical_scores) and len(unit.vector) == len(query_vector)
|
|
216
|
+
else 0.0
|
|
217
|
+
)
|
|
218
|
+
# Exact symbols and domain terms are high-confidence evidence. Keep
|
|
219
|
+
# lexical overlap dominant so a noisy feature-hash vector cannot push an
|
|
220
|
+
# exact match below an unrelated semantic neighbor; use the vector score
|
|
221
|
+
# to rank paraphrases and break lexical ties.
|
|
222
|
+
score = lexical + vector_weight * max(0.0, vector_score)
|
|
223
|
+
if lexical or score > 0:
|
|
224
|
+
scored.append((score, unit_id))
|
|
225
|
+
# Materialise only the winners. Building a SearchResult for every lexical
|
|
226
|
+
# match and then sorting all of them cost more than the scoring itself once
|
|
227
|
+
# recall became complete: at 10k units that alone was most of a 10x query
|
|
228
|
+
# regression. nsmallest keeps the exact previous ordering -- highest score
|
|
229
|
+
# first, ties broken by ascending unit id -- at O(n log limit).
|
|
230
|
+
winners = heapq.nsmallest(limit, scored, key=lambda item: (-item[0], item[1]))
|
|
231
|
+
# Which terms matched is only needed for the handful actually returned, and
|
|
232
|
+
# postings are stored sorted, so a binary search beats carrying a per-unit
|
|
233
|
+
# term list through the scoring loop for every candidate in the corpus.
|
|
234
|
+
return [
|
|
235
|
+
SearchResult(search_index.units[unit_id], score, sorted(token for token, posting in postings if _in_posting(posting, unit_id)))
|
|
236
|
+
for score, unit_id in winners
|
|
237
|
+
]
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _block(result: SearchResult) -> str:
|
|
241
|
+
"""One result as an agent reads it: identifier, score, why it matched,
|
|
242
|
+
what it is, and the code itself.
|
|
243
|
+
"""
|
|
244
|
+
unit = result.unit
|
|
245
|
+
evidence = "\nEvidence: " + " | ".join(result.evidence) if result.evidence else ""
|
|
246
|
+
return f"[{unit.id}] score={result.score:.3f}{evidence}\n{unit.description}\n```{unit.language}\n{unit.source}\n```"
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def within_budget(results: list[SearchResult], max_chars: int) -> list[SearchResult]:
|
|
250
|
+
"""The leading results whose rendered size fits the caller's budget.
|
|
251
|
+
|
|
252
|
+
The budget has to decide how many results there *are*, not merely how many
|
|
253
|
+
get rendered into one of the two places they are sent. `search --json`
|
|
254
|
+
used to serialise every result in full while capping only the context
|
|
255
|
+
string beside them, so a default query answered with 65,025 characters
|
|
256
|
+
against a stated budget of 12,000 -- and the agent, which reads the
|
|
257
|
+
results, was the side that overran.
|
|
258
|
+
|
|
259
|
+
The first result is always kept. A search that found something and
|
|
260
|
+
returned nothing because the match was large is less useful than one
|
|
261
|
+
oversized answer, and the caller is told how many were dropped.
|
|
262
|
+
"""
|
|
263
|
+
kept: list[SearchResult] = []
|
|
264
|
+
used = 0
|
|
265
|
+
for result in results:
|
|
266
|
+
size = len(_block(result))
|
|
267
|
+
if kept and used + size > max_chars:
|
|
268
|
+
break
|
|
269
|
+
kept.append(result)
|
|
270
|
+
used += size
|
|
271
|
+
return kept
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def context(results: list[SearchResult], max_chars: int = 12000) -> str:
|
|
275
|
+
"""Packs ranked results into one readable block for an agent prompt,
|
|
276
|
+
stopping before the caller's character budget is exceeded.
|
|
277
|
+
"""
|
|
278
|
+
return "\n\n".join(_block(result) for result in within_budget(results, max_chars))
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Guards for the three ranking defects a cold foreign repository exposed.
|
|
2
|
+
|
|
3
|
+
Measured on cc-enforcer -- 1153 units, no written descriptions -- the single
|
|
4
|
+
largest declaration came back in the top three for four questions out of six,
|
|
5
|
+
and the words deciding the ranking were the ones present in nearly every unit.
|
|
6
|
+
Three causes, all in how a match was scored:
|
|
7
|
+
|
|
8
|
+
H: every query word counted the same, so `the` (49% of units) and `calls`
|
|
9
|
+
(97%) outweighed `daemon` (2 units) and `warm` (none).
|
|
10
|
+
I: nothing normalised for size, and the largest unit held 539 distinct terms
|
|
11
|
+
against a median of 52, so it could contain any query by accident.
|
|
12
|
+
J: a term in a declaration's name counted no more than the same term two
|
|
13
|
+
hundred lines into a body, which put test units above the code they test.
|
|
14
|
+
|
|
15
|
+
Each test is built so that the previous rule -- the fraction of query words
|
|
16
|
+
present anywhere in the unit -- ranks the *wrong* answer first.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from ragyourcode.indexer import build_units
|
|
24
|
+
from ragyourcode.models import CodeUnit
|
|
25
|
+
from ragyourcode.search import FIELD_WEIGHTS, _block, build_search_index, context, search, within_budget
|
|
26
|
+
|
|
27
|
+
RARE_QUERY = "handle the request and return the response quiesce"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _with_a_word_nothing_else_uses(root: Path) -> Path:
|
|
31
|
+
"""One more unit, whose only distinguishing word appears nowhere else."""
|
|
32
|
+
(root / "rare.py").write_text(
|
|
33
|
+
"def quiesce(attempt):\n"
|
|
34
|
+
' """Settle down."""\n'
|
|
35
|
+
" return attempt\n",
|
|
36
|
+
encoding="utf-8",
|
|
37
|
+
)
|
|
38
|
+
return root
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_a_word_in_almost_every_unit_cannot_decide_the_ranking(tmp_path: Path, skewed_corpus):
|
|
42
|
+
"""Rarity is evidence; ubiquity is not.
|
|
43
|
+
|
|
44
|
+
Under the previous rule the unit matching six common words out of seven
|
|
45
|
+
scored 0.857 and the only unit that answers the question scored 0.286.
|
|
46
|
+
Nothing here is a stopword list: `the` and `request` earn their low weight
|
|
47
|
+
from this corpus, which is why the mechanism works on a repository written
|
|
48
|
+
in any language, including one where the tokens are Chinese bigrams.
|
|
49
|
+
"""
|
|
50
|
+
units = build_units(_with_a_word_nothing_else_uses(skewed_corpus(tmp_path)))
|
|
51
|
+
results = search(units, RARE_QUERY, limit=3)
|
|
52
|
+
assert results[0].unit.name == "quiesce", [result.unit.id for result in results]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_matched_terms_survive_the_posting_list_lookup(tmp_path: Path, skewed_corpus):
|
|
56
|
+
"""Postings now carry a weight beside each id, and are searched by id.
|
|
57
|
+
|
|
58
|
+
A membership test comparing whole entries would report no matched terms at
|
|
59
|
+
all, turning every result's evidence -- the thing that makes a result
|
|
60
|
+
checkable -- into an empty list.
|
|
61
|
+
"""
|
|
62
|
+
units = build_units(_with_a_word_nothing_else_uses(skewed_corpus(tmp_path)))
|
|
63
|
+
results = search(units, RARE_QUERY, limit=8)
|
|
64
|
+
assert all(result.matched_terms for result in results), [
|
|
65
|
+
(result.unit.id, result.matched_terms) for result in results
|
|
66
|
+
]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_a_long_unit_does_not_win_on_size_alone(tmp_path: Path, skewed_corpus):
|
|
70
|
+
"""Containing a word is not the same as being about it.
|
|
71
|
+
|
|
72
|
+
The corpus has to be this large for the defect to appear at all. Every
|
|
73
|
+
query word here reaches all eighty-one units, which puts its posting list
|
|
74
|
+
past the selective threshold, so no vector score is computed and the
|
|
75
|
+
lexical rule decides alone. In a two-unit fixture the feature-hash vector
|
|
76
|
+
hides the problem, because normalising by a unit's token mass is itself a
|
|
77
|
+
crude length normalisation -- which is why the defect was visible on a
|
|
78
|
+
1153-unit repository and invisible in the small benchmark.
|
|
79
|
+
"""
|
|
80
|
+
filler = "\n".join(f" step_{number} = compute(step_{number - 1})" for number in range(1, 200))
|
|
81
|
+
(skewed_corpus(tmp_path) / "aggregate.py").write_text(
|
|
82
|
+
"def process_everything(request, response):\n"
|
|
83
|
+
' """Handle request and return response."""\n'
|
|
84
|
+
f"{filler}\n"
|
|
85
|
+
" return response\n",
|
|
86
|
+
encoding="utf-8",
|
|
87
|
+
)
|
|
88
|
+
units = build_units(tmp_path)
|
|
89
|
+
results = search(units, "handle the request and return the response", limit=3)
|
|
90
|
+
# Same words as every handler, two hundred lines of unrelated code, and it
|
|
91
|
+
# came back first: matched-word count made them equal and the tie fell to
|
|
92
|
+
# the unit id.
|
|
93
|
+
assert "process_everything" not in [result.unit.name for result in results], [
|
|
94
|
+
result.unit.id for result in results
|
|
95
|
+
]
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def test_the_same_word_counts_for_more_in_a_name_than_in_a_body(tmp_path: Path):
|
|
99
|
+
"""Where the author wrote a word says how much it meant.
|
|
100
|
+
|
|
101
|
+
Both units are the same size and mention the word once, so nothing but the
|
|
102
|
+
field it appears in can separate them. The vector is switched off because
|
|
103
|
+
it would answer this question by a different route -- its cosine already
|
|
104
|
+
favours the unit where the word is a larger share of the text -- and the
|
|
105
|
+
lexical rule is what changed. Without that, both scored exactly 1.0 and
|
|
106
|
+
the tie fell to the unit id.
|
|
107
|
+
|
|
108
|
+
This is the mechanism, not its limit: a body repeating a word forty times
|
|
109
|
+
still outranks the declaration named after it, because forty occurrences
|
|
110
|
+
over an average-length body is genuinely a lot of evidence. See
|
|
111
|
+
docs/TESTING.md for what that still costs.
|
|
112
|
+
"""
|
|
113
|
+
(tmp_path / "named.py").write_text(
|
|
114
|
+
'def checkpoint(job):\n """Do the work and hand it back."""\n return job\n',
|
|
115
|
+
encoding="utf-8",
|
|
116
|
+
)
|
|
117
|
+
(tmp_path / "mentions.py").write_text(
|
|
118
|
+
'def worker(job):\n """Do the work and hand it back."""\n # checkpoint\n return job\n',
|
|
119
|
+
encoding="utf-8",
|
|
120
|
+
)
|
|
121
|
+
units = build_units(tmp_path)
|
|
122
|
+
results = search(units, "checkpoint", limit=2, vector_weight=0.0)
|
|
123
|
+
assert results[0].unit.name == "checkpoint", [result.unit.id for result in results]
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def test_every_searchable_field_carries_a_weight(tmp_path: Path):
|
|
127
|
+
"""A field with no weight would vanish from ranking the moment it appeared.
|
|
128
|
+
|
|
129
|
+
The two live in different modules on purpose -- one says what retrieval may
|
|
130
|
+
match, the other how much each part counts -- so this is the join that
|
|
131
|
+
keeps them from drifting apart.
|
|
132
|
+
"""
|
|
133
|
+
(tmp_path / "one.py").write_text("def f(a):\n return a\n", encoding="utf-8")
|
|
134
|
+
unit = build_units(tmp_path)[0]
|
|
135
|
+
assert set(unit.searchable_fields) == set(FIELD_WEIGHTS)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_searchable_text_is_every_field_and_nothing_else(tmp_path: Path):
|
|
139
|
+
"""The text a unit is embedded from must not lose a field ranking uses."""
|
|
140
|
+
(tmp_path / "one.py").write_text(
|
|
141
|
+
'"""Module."""\n\n\ndef f(a):\n """Add one."""\n return a + 1\n',
|
|
142
|
+
encoding="utf-8",
|
|
143
|
+
)
|
|
144
|
+
unit = build_units(tmp_path)[0]
|
|
145
|
+
assert unit.searchable_text == "\n".join(unit.searchable_fields.values())
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _budgeted(root: Path) -> list:
|
|
149
|
+
for index in range(6):
|
|
150
|
+
(root / f"unit_{index}.py").write_text(
|
|
151
|
+
f"def widget_{index}(value):\n"
|
|
152
|
+
f' """Widget {index} handles the value."""\n'
|
|
153
|
+
+ "".join(f" line_{number} = value\n" for number in range(40))
|
|
154
|
+
+ " return value\n",
|
|
155
|
+
encoding="utf-8",
|
|
156
|
+
)
|
|
157
|
+
return search(build_units(root), "widget handles the value", limit=6)
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def test_the_budget_bounds_the_results_not_only_the_context(tmp_path: Path):
|
|
161
|
+
"""`search --json` served 65,025 characters against a budget of 12,000.
|
|
162
|
+
|
|
163
|
+
The cap applied to the context string while every result was serialised
|
|
164
|
+
beside it in full, so the half an agent reads was the half that overran.
|
|
165
|
+
"""
|
|
166
|
+
results = _budgeted(tmp_path)
|
|
167
|
+
assert len(results) > 1, "the fixture must produce enough results to be trimmed"
|
|
168
|
+
budget = len(_block(results[0])) + 10
|
|
169
|
+
kept = within_budget(results, budget)
|
|
170
|
+
assert len(kept) < len(results)
|
|
171
|
+
assert sum(len(_block(result)) for result in kept) <= budget
|
|
172
|
+
assert context(kept, budget) == context(results, budget)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_a_single_result_larger_than_the_budget_is_still_returned(tmp_path: Path):
|
|
176
|
+
"""Finding something and returning nothing is worse than one oversized hit."""
|
|
177
|
+
results = _budgeted(tmp_path)
|
|
178
|
+
assert [result.unit.id for result in within_budget(results, 0)] == [results[0].unit.id]
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def test_a_field_every_unit_leaves_empty_does_not_divide_by_zero():
|
|
182
|
+
"""Each field is normalised against that field's own average length.
|
|
183
|
+
|
|
184
|
+
A corpus where nobody fills one in -- no signature, no body, no written
|
|
185
|
+
description -- makes that average zero, and the average is a denominator.
|
|
186
|
+
"""
|
|
187
|
+
unit = CodeUnit(
|
|
188
|
+
id="empty.py:1:nothing",
|
|
189
|
+
path="empty.py",
|
|
190
|
+
language="python",
|
|
191
|
+
kind="function",
|
|
192
|
+
name="nothing",
|
|
193
|
+
qualified_name="nothing",
|
|
194
|
+
signature="",
|
|
195
|
+
start_line=1,
|
|
196
|
+
end_line=1,
|
|
197
|
+
source="",
|
|
198
|
+
description="",
|
|
199
|
+
serial=1,
|
|
200
|
+
)
|
|
201
|
+
index = build_search_index([unit])
|
|
202
|
+
assert search([unit], "nothing", limit=1, search_index=index)[0].unit.id == unit.id
|
|
@@ -21,30 +21,42 @@ from ragyourcode import descriptions as descriptions_module
|
|
|
21
21
|
from ragyourcode.indexer import build_units
|
|
22
22
|
|
|
23
23
|
ROOT = Path(__file__).resolve().parents[1]
|
|
24
|
+
COLD_PATH = ROOT / "benchmarks" / "cold_queries.json"
|
|
24
25
|
|
|
25
26
|
|
|
26
27
|
@pytest.fixture(scope="module")
|
|
27
28
|
def units():
|
|
28
29
|
# With the description store, because that is how the CLI builds an index.
|
|
29
|
-
# Without it the same ruler scores 0.171 rather than 0.500 on hit@1, which
|
|
30
|
-
# would be measuring a configuration nobody runs.
|
|
31
30
|
return build_units(ROOT, descriptions=descriptions_module.load(ROOT))
|
|
32
31
|
|
|
33
32
|
|
|
33
|
+
@pytest.fixture(scope="module")
|
|
34
|
+
def cold_units():
|
|
35
|
+
"""The same repository as a first-time user's index sees it: parsed, with
|
|
36
|
+
only the sentence the parser generates and nothing anybody wrote.
|
|
37
|
+
"""
|
|
38
|
+
return build_units(ROOT)
|
|
39
|
+
|
|
40
|
+
|
|
34
41
|
def test_every_acceptable_answer_names_code_that_exists(units):
|
|
35
42
|
problems = check_ruler(load_questions(), units)
|
|
36
43
|
assert not problems, "the ruler has drifted from the code:\n " + "\n ".join(problems)
|
|
37
44
|
|
|
38
45
|
|
|
39
|
-
|
|
40
|
-
|
|
46
|
+
@pytest.mark.parametrize("path", [QUERIES_PATH, COLD_PATH], ids=["repository", "cold"])
|
|
47
|
+
def test_the_ruler_is_well_formed(path: Path):
|
|
48
|
+
questions = load_questions(path)
|
|
41
49
|
entries = questions["queries"]
|
|
42
|
-
assert len(entries) >=
|
|
50
|
+
assert len(entries) >= 30, "too few questions to distinguish a change from noise"
|
|
43
51
|
assert questions["k"] >= 1
|
|
52
|
+
seen: set[str] = set()
|
|
44
53
|
for entry in entries:
|
|
45
54
|
assert entry["query"].strip(), f"{entry['id']}: empty question"
|
|
55
|
+
assert entry["id"] not in seen, f"{entry['id']}: duplicate question id"
|
|
56
|
+
seen.add(entry["id"])
|
|
46
57
|
assert entry["language"] in {"en", "zh"}
|
|
47
58
|
assert entry["kind"] in {"concept", "why", "symbol"}
|
|
59
|
+
assert entry["acceptable"], f"{entry['id']}: no acceptable answer listed"
|
|
48
60
|
assert all(len(pair) == 2 for pair in entry["acceptable"])
|
|
49
61
|
# Both languages have to be represented, because the CJK path through the
|
|
50
62
|
# tokenizer is the one place a query can share no character class with the
|
|
@@ -53,6 +65,29 @@ def test_the_ruler_is_well_formed():
|
|
|
53
65
|
assert languages == {"en", "zh"}
|
|
54
66
|
|
|
55
67
|
|
|
68
|
+
def test_the_cold_ruler_says_which_repository_it_grades():
|
|
69
|
+
"""It grades code that is not in this repository, so it has to name it and
|
|
70
|
+
say why a ruler asked about this project cannot stand in for it.
|
|
71
|
+
"""
|
|
72
|
+
questions = load_questions(COLD_PATH)
|
|
73
|
+
assert questions["repository"], "a ruler over foreign code must name that code"
|
|
74
|
+
assert questions["caveat"].strip()
|
|
75
|
+
assert questions["why"].strip()
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_written_descriptions_beat_generated_ones(units, cold_units):
|
|
79
|
+
"""The reason `describe` exists, asserted rather than stated.
|
|
80
|
+
|
|
81
|
+
This used to be a comment quoting two numbers, which is exactly the kind of
|
|
82
|
+
claim that rots: both had already moved by the time anybody looked.
|
|
83
|
+
"""
|
|
84
|
+
questions = load_questions()
|
|
85
|
+
warm = evaluate(units, questions)["aggregate"]
|
|
86
|
+
cold = evaluate(cold_units, questions)["aggregate"]
|
|
87
|
+
assert warm["hit_at_1"] > cold["hit_at_1"], (warm, cold)
|
|
88
|
+
assert warm["mrr"] > cold["mrr"], (warm, cold)
|
|
89
|
+
|
|
90
|
+
|
|
56
91
|
def test_the_ruler_has_headroom(units):
|
|
57
92
|
"""A ruler everything already passes cannot measure an improvement.
|
|
58
93
|
|
|
@@ -17,31 +17,13 @@ from ragyourcode.indexer import build_units, fingerprint, read_index, snapshot_r
|
|
|
17
17
|
from ragyourcode.search import build_search_index, search
|
|
18
18
|
|
|
19
19
|
|
|
20
|
-
def
|
|
21
|
-
"""More units than the selective threshold's 64 floor, with skewed terms."""
|
|
22
|
-
for index in range(count):
|
|
23
|
-
(tmp_path / f"mod_{index}.py").write_text(
|
|
24
|
-
f"def handler_{index}(request, response):\n"
|
|
25
|
-
f' """Handle request and return response."""\n'
|
|
26
|
-
f" return response\n",
|
|
27
|
-
encoding="utf-8",
|
|
28
|
-
)
|
|
29
|
-
(tmp_path / "special.py").write_text(
|
|
30
|
-
"def retry_request_with_backoff(request, response):\n"
|
|
31
|
-
' """Retry a failed request and return response."""\n'
|
|
32
|
-
" return response\n",
|
|
33
|
-
encoding="utf-8",
|
|
34
|
-
)
|
|
35
|
-
return tmp_path
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path):
|
|
20
|
+
def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path, skewed_corpus):
|
|
39
21
|
"""A rare token must not shrink the candidate set to itself.
|
|
40
22
|
|
|
41
23
|
`backoff` reaches one unit; `request` and `response` reach all 81. Scoring
|
|
42
24
|
only what the rare term reached returned a single result for `--limit 8`.
|
|
43
25
|
"""
|
|
44
|
-
units = build_units(
|
|
26
|
+
units = build_units(skewed_corpus(tmp_path))
|
|
45
27
|
assert len(units) > 64, "the corpus must exceed the selective threshold's floor"
|
|
46
28
|
index = build_search_index(units)
|
|
47
29
|
results = search(units, "backoff request response", limit=8, search_index=index)
|
|
@@ -50,8 +32,8 @@ def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path):
|
|
|
50
32
|
assert all(result.matched_terms for result in results), "every returned unit must carry its evidence"
|
|
51
33
|
|
|
52
34
|
|
|
53
|
-
def test_every_lexically_matching_unit_can_be_returned(tmp_path: Path):
|
|
54
|
-
units = build_units(
|
|
35
|
+
def test_every_lexically_matching_unit_can_be_returned(tmp_path: Path, skewed_corpus):
|
|
36
|
+
units = build_units(skewed_corpus(tmp_path, count=70))
|
|
55
37
|
index = build_search_index(units)
|
|
56
38
|
results = search(units, "handler_7 request", limit=100, search_index=index)
|
|
57
39
|
returned = {result.unit.id for result in results}
|
|
@@ -1,151 +0,0 @@
|
|
|
1
|
-
"""Hybrid lexical/vector retrieval for agent context."""
|
|
2
|
-
|
|
3
|
-
from __future__ import annotations
|
|
4
|
-
|
|
5
|
-
import heapq
|
|
6
|
-
from bisect import bisect_left
|
|
7
|
-
from collections import Counter, defaultdict
|
|
8
|
-
from dataclasses import dataclass
|
|
9
|
-
|
|
10
|
-
from .config import BY_PATH
|
|
11
|
-
from .embeddings import DEFAULT_DIMENSIONS, embed, tokenize
|
|
12
|
-
from .models import CodeUnit, SearchResult
|
|
13
|
-
|
|
14
|
-
# Named here rather than repeated as a literal so `search.vector_weight` in
|
|
15
|
-
# rag-your-code.toml and the default a direct caller gets cannot drift apart.
|
|
16
|
-
DEFAULT_VECTOR_WEIGHT: float = BY_PATH["search.vector_weight"].default
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
@dataclass(slots=True)
|
|
20
|
-
class SearchIndex:
|
|
21
|
-
"""In-memory inverted index reused across queries.
|
|
22
|
-
|
|
23
|
-
Building this once avoids re-tokenizing every code unit. Matched terms are
|
|
24
|
-
read straight out of ``postings``; an earlier version also cached a
|
|
25
|
-
per-unit frozenset of every token, which cost the largest share of the
|
|
26
|
-
index's resident memory while holding nothing ``postings`` did not already
|
|
27
|
-
have. The same structure can later be backed by SQLite/ANN storage.
|
|
28
|
-
"""
|
|
29
|
-
|
|
30
|
-
units: dict[str, CodeUnit]
|
|
31
|
-
postings: dict[str, tuple[str, ...]]
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
def build_search_index(units: list[CodeUnit]) -> SearchIndex:
|
|
35
|
-
"""Builds the inverted lookup table by tokenising every unit once and
|
|
36
|
-
recording, for each term, which units contain it. The lists are kept
|
|
37
|
-
sorted so membership can later be answered by binary search rather than
|
|
38
|
-
by carrying a word set through the scoring loop.
|
|
39
|
-
"""
|
|
40
|
-
postings: dict[str, set[str]] = defaultdict(set)
|
|
41
|
-
for unit in units:
|
|
42
|
-
for term in set(tokenize(unit.searchable_text)):
|
|
43
|
-
postings[term].add(unit.id)
|
|
44
|
-
return SearchIndex({unit.id: unit for unit in units}, {term: tuple(sorted(ids)) for term, ids in postings.items()})
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
def _in_posting(posting: tuple[str, ...], unit_id: str) -> bool:
|
|
48
|
-
"""Membership test over a posting list, which build_search_index keeps sorted."""
|
|
49
|
-
position = bisect_left(posting, unit_id)
|
|
50
|
-
return position < len(posting) and posting[position] == unit_id
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
def search(
|
|
54
|
-
units: list[CodeUnit],
|
|
55
|
-
query: str,
|
|
56
|
-
limit: int = 8,
|
|
57
|
-
search_index: SearchIndex | None = None,
|
|
58
|
-
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
59
|
-
) -> list[SearchResult]:
|
|
60
|
-
"""Ranks code units against a natural-language query by combining word
|
|
61
|
-
overlap with vector similarity. Every unit sharing any query word is
|
|
62
|
-
scored, so nothing that matches is left out: an earlier design let the
|
|
63
|
-
vector shortlist decide who was scored at all, and units matching more
|
|
64
|
-
query words went unranked, returning one result where eight were asked
|
|
65
|
-
for. Full vector scoring is reserved for units reached by a selective
|
|
66
|
-
term, since computing a dot product for everything a stopword-class term
|
|
67
|
-
touches is pure cost. Word overlap stays dominant so an exact symbol
|
|
68
|
-
match cannot be pushed below a noisy neighbour, and with no overlap
|
|
69
|
-
anywhere it falls back to similarity alone.
|
|
70
|
-
"""
|
|
71
|
-
query_tokens = set(tokenize(query))
|
|
72
|
-
if limit <= 0 or not query_tokens:
|
|
73
|
-
return []
|
|
74
|
-
query_vector = embed(query, len(units[0].vector) if units and units[0].vector else DEFAULT_DIMENSIONS)
|
|
75
|
-
query_features = [(index, value) for index, value in enumerate(query_vector) if value]
|
|
76
|
-
search_index = search_index or build_search_index(units)
|
|
77
|
-
postings = [(token, search_index.postings.get(token, ())) for token in query_tokens]
|
|
78
|
-
|
|
79
|
-
# Matched terms come straight from the posting lists. Walking postings costs
|
|
80
|
-
# O(sum of posting lengths) of dict work, where scoring each candidate by
|
|
81
|
-
# intersecting a cached per-unit token set cost a frozenset operation per
|
|
82
|
-
# candidate -- and every lexically matching unit now gets a score.
|
|
83
|
-
matched_counts: Counter[str] = Counter()
|
|
84
|
-
for _, posting in postings:
|
|
85
|
-
matched_counts.update(posting)
|
|
86
|
-
|
|
87
|
-
# A term present in a tenth of the corpus (``function``, ``return``) is not
|
|
88
|
-
# evidence of relevance, and its posting list is effectively the whole index;
|
|
89
|
-
# computing a 384-dimension dot product for everything it reaches is what
|
|
90
|
-
# this threshold exists to avoid. It selects which candidates additionally
|
|
91
|
-
# receive a VECTOR score. It must not decide which candidates are scored at
|
|
92
|
-
# all -- doing that silently dropped units matching MORE query terms and
|
|
93
|
-
# under-filled ``limit`` (116 units, `--limit 8`, one result returned).
|
|
94
|
-
selective_threshold = max(64, min(2048, len(units) // 10))
|
|
95
|
-
vector_ids: set[str] = set()
|
|
96
|
-
for _, posting in postings:
|
|
97
|
-
if 0 < len(posting) <= selective_threshold:
|
|
98
|
-
vector_ids.update(posting)
|
|
99
|
-
if matched_counts and not vector_ids and len(matched_counts) <= selective_threshold:
|
|
100
|
-
vector_ids = set(matched_counts)
|
|
101
|
-
|
|
102
|
-
# With no lexical overlap anywhere, fall back to pure cosine so a genuine
|
|
103
|
-
# paraphrase still retrieves something.
|
|
104
|
-
candidate_ids = matched_counts.keys() if matched_counts else search_index.units.keys()
|
|
105
|
-
scored: list[tuple[float, str]] = []
|
|
106
|
-
for unit_id in candidate_ids:
|
|
107
|
-
unit = search_index.units[unit_id]
|
|
108
|
-
lexical = matched_counts.get(unit_id, 0) / len(query_tokens)
|
|
109
|
-
vector_score = (
|
|
110
|
-
sum(value * unit.vector[index] for index, value in query_features)
|
|
111
|
-
if (unit_id in vector_ids or not matched_counts) and len(unit.vector) == len(query_vector)
|
|
112
|
-
else 0.0
|
|
113
|
-
)
|
|
114
|
-
# Exact symbols and domain terms are high-confidence evidence. Keep
|
|
115
|
-
# lexical overlap dominant so a noisy feature-hash vector cannot push an
|
|
116
|
-
# exact match below an unrelated semantic neighbor; use the vector score
|
|
117
|
-
# to rank paraphrases and break lexical ties.
|
|
118
|
-
score = lexical + vector_weight * max(0.0, vector_score)
|
|
119
|
-
if lexical or score > 0:
|
|
120
|
-
scored.append((score, unit_id))
|
|
121
|
-
# Materialise only the winners. Building a SearchResult for every lexical
|
|
122
|
-
# match and then sorting all of them cost more than the scoring itself once
|
|
123
|
-
# recall became complete: at 10k units that alone was most of a 10x query
|
|
124
|
-
# regression. nsmallest keeps the exact previous ordering -- highest score
|
|
125
|
-
# first, ties broken by ascending unit id -- at O(n log limit).
|
|
126
|
-
winners = heapq.nsmallest(limit, scored, key=lambda item: (-item[0], item[1]))
|
|
127
|
-
# Which terms matched is only needed for the handful actually returned, and
|
|
128
|
-
# postings are stored sorted, so a binary search beats carrying a per-unit
|
|
129
|
-
# term list through the scoring loop for every candidate in the corpus.
|
|
130
|
-
return [
|
|
131
|
-
SearchResult(search_index.units[unit_id], score, sorted(token for token, posting in postings if _in_posting(posting, unit_id)))
|
|
132
|
-
for score, unit_id in winners
|
|
133
|
-
]
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
def context(results: list[SearchResult], max_chars: int = 12000) -> str:
|
|
137
|
-
"""Packs ranked results into one readable block for an agent prompt, each
|
|
138
|
-
carrying identifier, score, matching evidence, description and source,
|
|
139
|
-
and stops before exceeding the caller character budget.
|
|
140
|
-
"""
|
|
141
|
-
blocks: list[str] = []
|
|
142
|
-
used = 0
|
|
143
|
-
for result in results:
|
|
144
|
-
unit = result.unit
|
|
145
|
-
evidence = "\nEvidence: " + " | ".join(result.evidence) if result.evidence else ""
|
|
146
|
-
block = f"[{unit.id}] score={result.score:.3f}{evidence}\n{unit.description}\n```{unit.language}\n{unit.source}\n```"
|
|
147
|
-
if used + len(block) > max_chars:
|
|
148
|
-
break
|
|
149
|
-
blocks.append(block)
|
|
150
|
-
used += len(block)
|
|
151
|
-
return "\n\n".join(blocks)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|