rag-your-code 0.5.0__tar.gz → 0.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {rag_your_code-0.5.0/src/rag_your_code.egg-info → rag_your_code-0.6.0}/PKG-INFO +38 -8
  2. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/README.md +37 -7
  3. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/pyproject.toml +1 -1
  4. {rag_your_code-0.5.0 → rag_your_code-0.6.0/src/rag_your_code.egg-info}/PKG-INFO +38 -8
  5. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/SOURCES.txt +1 -0
  6. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/__init__.py +1 -1
  7. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/cli.py +8 -3
  8. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/models.py +29 -15
  9. rag_your_code-0.6.0/src/ragyourcode/search.py +278 -0
  10. rag_your_code-0.6.0/tests/test_ranking.py +202 -0
  11. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_repo_queries.py +40 -5
  12. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_retrieval_correctness.py +4 -22
  13. rag_your_code-0.5.0/src/ragyourcode/search.py +0 -151
  14. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/LICENSE +0 -0
  15. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/setup.cfg +0 -0
  16. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  17. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  18. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  19. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  20. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/agentic.py +0 -0
  21. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/annotate.py +0 -0
  22. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/config.py +0 -0
  23. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/descriptions.py +0 -0
  24. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/document.py +0 -0
  25. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/embeddings.py +0 -0
  26. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/graph.py +0 -0
  27. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/indexer.py +0 -0
  28. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/parser.py +0 -0
  29. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/src/ragyourcode/py.typed +0 -0
  30. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_agent_protocol.py +0 -0
  31. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_agentic.py +0 -0
  32. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_config.py +0 -0
  33. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_descriptions.py +0 -0
  34. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_doc_comments.py +0 -0
  35. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_document.py +0 -0
  36. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_e2e_cli.py +0 -0
  37. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_golden.py +0 -0
  38. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_graph_incremental.py +0 -0
  39. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_language_fixtures.py +0 -0
  40. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_large_repo.py +0 -0
  41. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_metadata.py +0 -0
  42. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_multilanguage.py +0 -0
  43. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_parser_edges.py +0 -0
  44. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_ragyourcode.py +0 -0
  45. {rag_your_code-0.5.0 → rag_your_code-0.6.0}/tests/test_resilience.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -198,19 +198,49 @@ Chinese, each listing every unit that genuinely answers it
198
198
 
199
199
  | | generated descriptions | agent-written |
200
200
  |---|---|---|
201
- | hit@1 | 0.171 | **0.500** |
202
- | hit@3 | 0.314 | **0.729** |
203
- | MRR | 0.240 | **0.605** |
204
- | answered with no shared word at all | 15.7% | **0%** |
201
+ | hit@1 | 0.271 | **0.500** |
202
+ | hit@3 | 0.486 | **0.800** |
203
+ | MRR | 0.367 | **0.631** |
204
+ | answered with no shared word at all | 12.9% | **0%** |
205
205
 
206
- Roughly a threefold improvement in first-place accuracy. Nineteen questions
207
- still fail, which is what makes the set usable for measuring the next change;
208
- `tests/test_repo_queries.py` asserts that some question always does.
206
+ Roughly double the first-place accuracy. Fourteen questions still fail, which
207
+ is what makes the set usable for measuring the next change;
208
+ `tests/test_repo_queries.py` asserts that some question always does, and that
209
+ the written column beats the generated one.
209
210
 
210
211
  One failure is worth naming: a query saying `catastrophic backtracking` does
211
212
  not reach a description saying `backtracks catastrophically`. There is no
212
213
  stemming — exactly the limit documented above.
213
214
 
215
+ ### Measured on a repository nobody here wrote
216
+
217
+ The table above is the warmest case this project supports: its own code, its
218
+ own descriptions, and questions written by the same party. It cannot say what
219
+ a first-time user gets. So there is a second ruler — thirty-five questions
220
+ about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
221
+ descriptions at all, each question phrased in a user's words rather than in
222
+ the words of the docstring that answers it
223
+ ([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
224
+
225
+ | | before 0.6.0 | now |
226
+ |---|---|---|
227
+ | hit@1 | 0.086 | **0.257** |
228
+ | hit@3 | 0.229 | **0.400** |
229
+ | MRR | 0.157 | **0.314** |
230
+
231
+ Three times the first-place accuracy, and the same change moved both other
232
+ rulers in the same direction. What it fixed was ranking: scoring used to be
233
+ the fraction of query words a unit contained, so `the` counted for as much as
234
+ `daemon`, and nothing corrected for size — the single largest declaration in
235
+ that repository came back in the top three for four questions out of six. It
236
+ is now BM25 over weighted fields, where a word's worth comes from how rare it
237
+ is in *your* corpus and a word in a declaration's name outweighs the same word
238
+ buried in a body.
239
+
240
+ Twenty-one of the thirty-five still fail, and the largest remaining cause is
241
+ named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
242
+ the code it tests, because it repeats that code's vocabulary and adds its own.
243
+
214
244
  **What this is:** it moves the semantic work from query time to index time.
215
245
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
216
246
  bounded by how many ways of saying the thing the agent thought to write down.
@@ -171,19 +171,49 @@ Chinese, each listing every unit that genuinely answers it
171
171
 
172
172
  | | generated descriptions | agent-written |
173
173
  |---|---|---|
174
- | hit@1 | 0.171 | **0.500** |
175
- | hit@3 | 0.314 | **0.729** |
176
- | MRR | 0.240 | **0.605** |
177
- | answered with no shared word at all | 15.7% | **0%** |
174
+ | hit@1 | 0.271 | **0.500** |
175
+ | hit@3 | 0.486 | **0.800** |
176
+ | MRR | 0.367 | **0.631** |
177
+ | answered with no shared word at all | 12.9% | **0%** |
178
178
 
179
- Roughly a threefold improvement in first-place accuracy. Nineteen questions
180
- still fail, which is what makes the set usable for measuring the next change;
181
- `tests/test_repo_queries.py` asserts that some question always does.
179
+ Roughly double the first-place accuracy. Fourteen questions still fail, which
180
+ is what makes the set usable for measuring the next change;
181
+ `tests/test_repo_queries.py` asserts that some question always does, and that
182
+ the written column beats the generated one.
182
183
 
183
184
  One failure is worth naming: a query saying `catastrophic backtracking` does
184
185
  not reach a description saying `backtracks catastrophically`. There is no
185
186
  stemming — exactly the limit documented above.
186
187
 
188
+ ### Measured on a repository nobody here wrote
189
+
190
+ The table above is the warmest case this project supports: its own code, its
191
+ own descriptions, and questions written by the same party. It cannot say what
192
+ a first-time user gets. So there is a second ruler — thirty-five questions
193
+ about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
194
+ descriptions at all, each question phrased in a user's words rather than in
195
+ the words of the docstring that answers it
196
+ ([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
197
+
198
+ | | before 0.6.0 | now |
199
+ |---|---|---|
200
+ | hit@1 | 0.086 | **0.257** |
201
+ | hit@3 | 0.229 | **0.400** |
202
+ | MRR | 0.157 | **0.314** |
203
+
204
+ Three times the first-place accuracy, and the same change moved both other
205
+ rulers in the same direction. What it fixed was ranking: scoring used to be
206
+ the fraction of query words a unit contained, so `the` counted for as much as
207
+ `daemon`, and nothing corrected for size — the single largest declaration in
208
+ that repository came back in the top three for four questions out of six. It
209
+ is now BM25 over weighted fields, where a word's worth comes from how rare it
210
+ is in *your* corpus and a word in a declaration's name outweighs the same word
211
+ buried in a body.
212
+
213
+ Twenty-one of the thirty-five still fail, and the largest remaining cause is
214
+ named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
215
+ the code it tests, because it repeats that code's vocabulary and adds its own.
216
+
187
217
  **What this is:** it moves the semantic work from query time to index time.
188
218
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
189
219
  bounded by how many ways of saying the thing the agent thought to write down.
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "0.5.0"
9
+ version = "0.6.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.5.0
3
+ Version: 0.6.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -198,19 +198,49 @@ Chinese, each listing every unit that genuinely answers it
198
198
 
199
199
  | | generated descriptions | agent-written |
200
200
  |---|---|---|
201
- | hit@1 | 0.171 | **0.500** |
202
- | hit@3 | 0.314 | **0.729** |
203
- | MRR | 0.240 | **0.605** |
204
- | answered with no shared word at all | 15.7% | **0%** |
201
+ | hit@1 | 0.271 | **0.500** |
202
+ | hit@3 | 0.486 | **0.800** |
203
+ | MRR | 0.367 | **0.631** |
204
+ | answered with no shared word at all | 12.9% | **0%** |
205
205
 
206
- Roughly a threefold improvement in first-place accuracy. Nineteen questions
207
- still fail, which is what makes the set usable for measuring the next change;
208
- `tests/test_repo_queries.py` asserts that some question always does.
206
+ Roughly double the first-place accuracy. Fourteen questions still fail, which
207
+ is what makes the set usable for measuring the next change;
208
+ `tests/test_repo_queries.py` asserts that some question always does, and that
209
+ the written column beats the generated one.
209
210
 
210
211
  One failure is worth naming: a query saying `catastrophic backtracking` does
211
212
  not reach a description saying `backtracks catastrophically`. There is no
212
213
  stemming — exactly the limit documented above.
213
214
 
215
+ ### Measured on a repository nobody here wrote
216
+
217
+ The table above is the warmest case this project supports: its own code, its
218
+ own descriptions, and questions written by the same party. It cannot say what
219
+ a first-time user gets. So there is a second ruler — thirty-five questions
220
+ about [cc-enforcer](https://github.com/skymanbp/cc-enforcer), 1153 units, no
221
+ descriptions at all, each question phrased in a user's words rather than in
222
+ the words of the docstring that answers it
223
+ ([`benchmarks/cold_queries.json`](benchmarks/cold_queries.json)):
224
+
225
+ | | before 0.6.0 | now |
226
+ |---|---|---|
227
+ | hit@1 | 0.086 | **0.257** |
228
+ | hit@3 | 0.229 | **0.400** |
229
+ | MRR | 0.157 | **0.314** |
230
+
231
+ Three times the first-place accuracy, and the same change moved both other
232
+ rulers in the same direction. What it fixed was ranking: scoring used to be
233
+ the fraction of query words a unit contained, so `the` counted for as much as
234
+ `daemon`, and nothing corrected for size — the single largest declaration in
235
+ that repository came back in the top three for four questions out of six. It
236
+ is now BM25 over weighted fields, where a word's worth comes from how rare it
237
+ is in *your* corpus and a word in a declaration's name outweighs the same word
238
+ buried in a body.
239
+
240
+ Twenty-one of the thirty-five still fail, and the largest remaining cause is
241
+ named in [docs/TESTING.md](docs/TESTING.md): a test declaration often outranks
242
+ the code it tests, because it repeats that code's vocabulary and adds its own.
243
+
214
244
  **What this is:** it moves the semantic work from query time to index time.
215
245
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
216
246
  bounded by how many ways of saying the thing the agent thought to write down.
@@ -36,6 +36,7 @@ tests/test_metadata.py
36
36
  tests/test_multilanguage.py
37
37
  tests/test_parser_edges.py
38
38
  tests/test_ragyourcode.py
39
+ tests/test_ranking.py
39
40
  tests/test_repo_queries.py
40
41
  tests/test_resilience.py
41
42
  tests/test_retrieval_correctness.py
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "0.5.0"
6
+ __version__ = "0.6.0"
@@ -18,7 +18,7 @@ from .document import plan as plan_documentation, render_patch, summarise as sum
18
18
  from .embeddings import embed, embedding_metadata
19
19
  from .graph import build_graph, graph_from_dict, graph_search
20
20
  from .indexer import StaleMonitor, build_fingerprint, build_units, fingerprint, index_build_fingerprint, read_index, snapshot_repository, write_index
21
- from .search import build_search_index, context, search
21
+ from .search import build_search_index, context, search, within_budget
22
22
 
23
23
  # Derived from the settings table so the default is written down once.
24
24
  # `tests/test_agent_protocol.py` imports these to assert the bound it enforces.
@@ -157,7 +157,10 @@ def _cmd_search(args: argparse.Namespace) -> int:
157
157
  else search(units, args.query, limit, search_index=search_index, vector_weight=weight)
158
158
  )
159
159
  if args.json:
160
- print(json.dumps({"query": args.query, "mode": "graph" if args.graph else "hybrid", "stale": payload.get("stale", True), "degraded": payload.get("degraded"), "results": [result.to_dict() for result in results], "context": context(results, max_chars)}, ensure_ascii=False))
160
+ # One budget decision, applied once: the results an agent reads and the
161
+ # context beside them are the same set, so they cannot disagree.
162
+ shown = within_budget(results, max_chars)
163
+ print(json.dumps({"query": args.query, "mode": "graph" if args.graph else "hybrid", "stale": payload.get("stale", True), "degraded": payload.get("degraded"), "results": [result.to_dict() for result in shown], "omitted_for_budget": len(results) - len(shown), "context": context(shown, max_chars)}, ensure_ascii=False))
161
164
  else:
162
165
  if payload.get("stale"):
163
166
  print("Warning: index is stale; run `rag-your-code index` to refresh.", file=sys.stderr)
@@ -496,7 +499,9 @@ def _cmd_agent(args: argparse.Namespace) -> int:
496
499
  if use_graph
497
500
  else search(units, query, limit, search_index=search_index, vector_weight=weight)
498
501
  )
499
- response = {"stale": payload.get("stale", True), "results": [result.to_dict() for result in results], "context": context(results, _request_int(request, "max_chars", default_chars, 0, 100000))}
502
+ budget = _request_int(request, "max_chars", default_chars, 0, 100000)
503
+ shown = within_budget(results, budget)
504
+ response = {"stale": payload.get("stale", True), "results": [result.to_dict() for result in shown], "omitted_for_budget": len(results) - len(shown), "context": context(shown, budget)}
500
505
  elif action == "research":
501
506
  response = research(
502
507
  units,
@@ -28,24 +28,38 @@ class CodeUnit:
28
28
  imports: list[str] = field(default_factory=list)
29
29
  vector: Sequence[float] = field(default_factory=list)
30
30
 
31
+ @property
32
+ def searchable_fields(self) -> dict[str, str]:
33
+ """Everything retrieval may match, kept apart by where the author
34
+ wrote it. Where a word appears is itself evidence of how much it
35
+ means: a term in a declaration's name is what the author decided to
36
+ call the thing, while the same term two hundred lines into a body is
37
+ a passing mention. Ranking weights those differently, so they cannot
38
+ arrive as one undifferentiated blob -- flattened, a long function
39
+ outranks the function that is actually named after the query, purely
40
+ by owning more words.
41
+
42
+ Because the description is one of these fields, replacing a generated
43
+ description with a better one immediately widens the set of queries
44
+ that can reach this unit.
45
+ """
46
+ return {
47
+ "name": f"{self.qualified_name} {self.kind}",
48
+ "signature": self.signature,
49
+ "description": self.description,
50
+ "relations": "calls: " + " ".join(self.calls) + "\nimports: " + " ".join(self.imports),
51
+ "body": self.source,
52
+ }
53
+
31
54
  @property
32
55
  def searchable_text(self) -> str:
33
- """Assembles everything about a unit that retrieval is allowed to match
34
- against: qualified name, kind and signature, the description, the
35
- names it calls and imports, and the full source. Because the
36
- description is part of this text, replacing a generated description
37
- with a better one immediately widens the set of queries that can
38
- reach this unit.
56
+ """Every searchable field as one block, for callers that want the
57
+ words without caring where they came from -- the unit's own embedding
58
+ is built from this. Derived from ``searchable_fields`` rather than
59
+ rebuilt beside it, so a field added for ranking cannot go missing
60
+ from the text that gets embedded.
39
61
  """
40
- return "\n".join(
41
- (
42
- f"{self.qualified_name} {self.kind} {self.signature}",
43
- self.description,
44
- "calls: " + " ".join(self.calls),
45
- "imports: " + " ".join(self.imports),
46
- self.source,
47
- )
48
- )
62
+ return "\n".join(self.searchable_fields.values())
49
63
 
50
64
  def to_dict(self, include_vector: bool = True) -> dict[str, Any]:
51
65
  """Serialises a unit to a plain dictionary for storage or for an agent
@@ -0,0 +1,278 @@
1
+ """Hybrid lexical/vector retrieval for agent context."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import heapq
6
+ import math
7
+ from bisect import bisect_left
8
+ from collections import Counter, defaultdict
9
+ from dataclasses import dataclass
10
+
11
+ from .config import BY_PATH
12
+ from .embeddings import DEFAULT_DIMENSIONS, embed, tokenize
13
+ from .models import CodeUnit, SearchResult
14
+
15
+ # Named here rather than repeated as a literal so `search.vector_weight` in
16
+ # rag-your-code.toml and the default a direct caller gets cannot drift apart.
17
+ DEFAULT_VECTOR_WEIGHT: float = BY_PATH["search.vector_weight"].default
18
+
19
+ # How much a word counts for, by the field the author wrote it in. A term in
20
+ # the name is what the declaration is called; the same term inside the body is
21
+ # a mention. Every key of `CodeUnit.searchable_fields` must appear here, and a
22
+ # test asserts the two sets agree -- a field with no weight would otherwise
23
+ # vanish from ranking the moment it was added, silently.
24
+ FIELD_WEIGHTS: dict[str, float] = {
25
+ "name": 8.0,
26
+ "signature": 4.0,
27
+ "description": 3.0,
28
+ "relations": 2.0,
29
+ "body": 1.0,
30
+ }
31
+
32
+ # BM25's saturation and length-normalisation constants, at their standard
33
+ # values. `k1` bounds what repeating a word can buy; `b` decides how hard a
34
+ # field is discounted for being longer than its average.
35
+ #
36
+ # Three variations were implemented and measured against all three rulers
37
+ # before settling here, and two were dropped for want of evidence: excluding
38
+ # curated text from the length on the argument that a written description is
39
+ # deliberate rather than incidental, and counting authored words instead of
40
+ # tokeniser output so a run of Chinese expanded into overlapping bigrams would
41
+ # not read as five times the text. Each moved one to four questions in both
42
+ # directions at once, which is this instrument's noise. Lowering `b` to 0.5 or
43
+ # 0.3 under per-field normalisation was measured and was worse than 0.75 on
44
+ # the foreign-repository ruler. See docs/TESTING.md.
45
+ BM25_K1 = 1.2
46
+ BM25_B = 0.75
47
+
48
+
49
+ @dataclass(slots=True)
50
+ class SearchIndex:
51
+ """In-memory inverted index reused across queries.
52
+
53
+ Building this once avoids re-tokenizing every code unit. Each posting
54
+ carries a term's weight in that unit alongside its id -- already scaled by
55
+ the field the term appeared in and by how long that field is -- because
56
+ ranking needs to know how often a term occurs and where, not merely that
57
+ it occurs somewhere. The weight lives in the posting rather than in a
58
+ per-unit table: an earlier version cached a per-unit frozenset of every
59
+ token, which cost the largest share of resident memory while holding
60
+ nothing the postings did not already have.
61
+
62
+ Lists are kept sorted by unit id so membership can be answered by binary
63
+ search. The same structure can later be backed by SQLite/ANN storage.
64
+ """
65
+
66
+ units: dict[str, CodeUnit]
67
+ postings: dict[str, tuple[tuple[str, float], ...]]
68
+
69
+
70
+ def build_search_index(units: list[CodeUnit]) -> SearchIndex:
71
+ """Builds the inverted lookup table, recording for each term the units that
72
+ contain it and how much it counts for in each.
73
+
74
+ A term's weight is BM25F's: its count in a field, divided by how long that
75
+ field is against the average for that same field, then scaled by what the
76
+ field is worth. Normalising per field is the part that matters. Measured
77
+ against one length for the whole unit, a body repeating a word forty times
78
+ still beat the declaration actually named after it, because a long body's
79
+ advantage in raw count almost exactly cancelled its penalty for being
80
+ long. Comparing each field against its own average removes that cancelling.
81
+
82
+ Two passes are needed because a field's average length is a property of
83
+ the corpus and is not known until every unit has been seen. The second
84
+ pass re-tokenizes rather than holding every unit's tokens in memory at
85
+ once, which costs about half again in time and keeps the peak bounded.
86
+ """
87
+ field_totals: Counter[str] = Counter()
88
+ for unit in units:
89
+ for field_name, text in unit.searchable_fields.items():
90
+ field_totals[field_name] += len(tokenize(text))
91
+ count = len(units) or 1
92
+ # A field every unit leaves empty would otherwise divide by zero. Its terms
93
+ # cannot reach any unit anyway, so the value only has to be finite.
94
+ averages = {name: (total / count) or 1.0 for name, total in field_totals.items()}
95
+
96
+ postings: dict[str, list[tuple[str, float]]] = defaultdict(list)
97
+ for unit in units:
98
+ weighted: Counter[str] = Counter()
99
+ for field_name, text in unit.searchable_fields.items():
100
+ terms = tokenize(text)
101
+ if not terms:
102
+ continue
103
+ saturation = 1 - BM25_B + BM25_B * len(terms) / averages[field_name]
104
+ weight = FIELD_WEIGHTS[field_name] / saturation
105
+ for term, occurrences in Counter(terms).items():
106
+ weighted[term] += weight * occurrences
107
+ for term, mass in weighted.items():
108
+ postings[term].append((unit.id, mass))
109
+ return SearchIndex(
110
+ {unit.id: unit for unit in units},
111
+ {term: tuple(sorted(entries)) for term, entries in postings.items()},
112
+ )
113
+
114
+
115
+ def _in_posting(posting: tuple[tuple[str, float], ...], unit_id: str) -> bool:
116
+ """Membership test over a posting list, which build_search_index keeps
117
+ sorted by unit id. A one-element tuple sorts before every pair sharing its
118
+ id, so it locates the entry without needing the mass that follows it.
119
+ """
120
+ position = bisect_left(posting, (unit_id,))
121
+ return position < len(posting) and posting[position][0] == unit_id
122
+
123
+
124
+ def _inverse_document_frequency(document_count: int, matching: int) -> float:
125
+ """How much evidence one term carries, from how rare it is in this corpus.
126
+
127
+ A word in nearly every unit says nothing about which unit is wanted, and
128
+ a word in two says a great deal. Deriving that from the corpus rather than
129
+ from a list of stopwords is what makes it work on a repository in any
130
+ language: `the` and `calls` earn their low weight the same way a Chinese
131
+ bigram does, by being everywhere, and no list has to be maintained.
132
+ """
133
+ return math.log(1 + (document_count - matching + 0.5) / (matching + 0.5))
134
+
135
+
136
+ def search(
137
+ units: list[CodeUnit],
138
+ query: str,
139
+ limit: int = 8,
140
+ search_index: SearchIndex | None = None,
141
+ vector_weight: float = DEFAULT_VECTOR_WEIGHT,
142
+ ) -> list[SearchResult]:
143
+ """Ranks code units against a natural-language query by combining weighted
144
+ word overlap with vector similarity.
145
+
146
+ The lexical half is BM25 over weighted fields: rare words count for more
147
+ than common ones, repeating a word saturates rather than accumulating
148
+ without bound, and a unit is normalised by how much text it owns. An
149
+ earlier version scored the plain fraction of query words present, which
150
+ made `the` worth as much as `daemon` and handed every query to whichever
151
+ unit was longest -- on a foreign repository the single largest declaration
152
+ came back for four questions out of six.
153
+
154
+ Every unit sharing any query word is scored, so nothing that matches is
155
+ left out. Full vector scoring is reserved for units reached by a selective
156
+ term, since computing a dot product for everything a stopword-class term
157
+ touches is pure cost. With no overlap anywhere it falls back to similarity
158
+ alone.
159
+ """
160
+ query_tokens = set(tokenize(query))
161
+ if limit <= 0 or not query_tokens:
162
+ return []
163
+ query_vector = embed(query, len(units[0].vector) if units and units[0].vector else DEFAULT_DIMENSIONS)
164
+ query_features = [(index, value) for index, value in enumerate(query_vector) if value]
165
+ search_index = search_index or build_search_index(units)
166
+ postings = [(token, search_index.postings.get(token, ())) for token in query_tokens]
167
+ document_count = len(search_index.units) or 1
168
+
169
+ # Accumulate term by term rather than unit by unit: a posting list is
170
+ # exactly the units a term reaches, so this touches no unit the query
171
+ # cannot possibly match. Length normalisation is already inside `mass`, so
172
+ # all that remains is saturation -- the tenth occurrence of a word says
173
+ # much less than the second, and neither should be able to run away with
174
+ # the ranking.
175
+ weights = {token: _inverse_document_frequency(document_count, len(posting)) for token, posting in postings if posting}
176
+ lexical_scores: dict[str, float] = defaultdict(float)
177
+ for token, posting in postings:
178
+ if not posting:
179
+ continue
180
+ weight = weights[token]
181
+ for unit_id, mass in posting:
182
+ lexical_scores[unit_id] += weight * mass * (BM25_K1 + 1) / (mass + BM25_K1)
183
+
184
+ # Divide by what this query could have scored at most, which is a constant
185
+ # for the query and so changes no ranking. It exists to keep the lexical
186
+ # half on the 0..1 scale `search.vector_weight` was documented and bounded
187
+ # against: raw BM25 is unbounded, and against a score of 18 a weight of
188
+ # 0.35 on a cosine would be arithmetic, not a tie-break.
189
+ ceiling = sum(weights.values()) * (BM25_K1 + 1)
190
+
191
+ # A term present in a tenth of the corpus (``function``, ``return``) is not
192
+ # evidence of relevance, and its posting list is effectively the whole index;
193
+ # computing a 384-dimension dot product for everything it reaches is what
194
+ # this threshold exists to avoid. It selects which candidates additionally
195
+ # receive a VECTOR score. It must not decide which candidates are scored at
196
+ # all -- doing that silently dropped units matching MORE query terms and
197
+ # under-filled ``limit`` (116 units, `--limit 8`, one result returned).
198
+ selective_threshold = max(64, min(2048, len(units) // 10))
199
+ vector_ids: set[str] = set()
200
+ for _, posting in postings:
201
+ if 0 < len(posting) <= selective_threshold:
202
+ vector_ids.update(unit_id for unit_id, _ in posting)
203
+ if lexical_scores and not vector_ids and len(lexical_scores) <= selective_threshold:
204
+ vector_ids = set(lexical_scores)
205
+
206
+ # With no lexical overlap anywhere, fall back to pure cosine so a genuine
207
+ # paraphrase still retrieves something.
208
+ candidate_ids = lexical_scores.keys() if lexical_scores else search_index.units.keys()
209
+ scored: list[tuple[float, str]] = []
210
+ for unit_id in candidate_ids:
211
+ unit = search_index.units[unit_id]
212
+ lexical = (lexical_scores.get(unit_id, 0.0) / ceiling) if ceiling else 0.0
213
+ vector_score = (
214
+ sum(value * unit.vector[index] for index, value in query_features)
215
+ if (unit_id in vector_ids or not lexical_scores) and len(unit.vector) == len(query_vector)
216
+ else 0.0
217
+ )
218
+ # Exact symbols and domain terms are high-confidence evidence. Keep
219
+ # lexical overlap dominant so a noisy feature-hash vector cannot push an
220
+ # exact match below an unrelated semantic neighbor; use the vector score
221
+ # to rank paraphrases and break lexical ties.
222
+ score = lexical + vector_weight * max(0.0, vector_score)
223
+ if lexical or score > 0:
224
+ scored.append((score, unit_id))
225
+ # Materialise only the winners. Building a SearchResult for every lexical
226
+ # match and then sorting all of them cost more than the scoring itself once
227
+ # recall became complete: at 10k units that alone was most of a 10x query
228
+ # regression. nsmallest keeps the exact previous ordering -- highest score
229
+ # first, ties broken by ascending unit id -- at O(n log limit).
230
+ winners = heapq.nsmallest(limit, scored, key=lambda item: (-item[0], item[1]))
231
+ # Which terms matched is only needed for the handful actually returned, and
232
+ # postings are stored sorted, so a binary search beats carrying a per-unit
233
+ # term list through the scoring loop for every candidate in the corpus.
234
+ return [
235
+ SearchResult(search_index.units[unit_id], score, sorted(token for token, posting in postings if _in_posting(posting, unit_id)))
236
+ for score, unit_id in winners
237
+ ]
238
+
239
+
240
+ def _block(result: SearchResult) -> str:
241
+ """One result as an agent reads it: identifier, score, why it matched,
242
+ what it is, and the code itself.
243
+ """
244
+ unit = result.unit
245
+ evidence = "\nEvidence: " + " | ".join(result.evidence) if result.evidence else ""
246
+ return f"[{unit.id}] score={result.score:.3f}{evidence}\n{unit.description}\n```{unit.language}\n{unit.source}\n```"
247
+
248
+
249
+ def within_budget(results: list[SearchResult], max_chars: int) -> list[SearchResult]:
250
+ """The leading results whose rendered size fits the caller's budget.
251
+
252
+ The budget has to decide how many results there *are*, not merely how many
253
+ get rendered into one of the two places they are sent. `search --json`
254
+ used to serialise every result in full while capping only the context
255
+ string beside them, so a default query answered with 65,025 characters
256
+ against a stated budget of 12,000 -- and the agent, which reads the
257
+ results, was the side that overran.
258
+
259
+ The first result is always kept. A search that found something and
260
+ returned nothing because the match was large is less useful than one
261
+ oversized answer, and the caller is told how many were dropped.
262
+ """
263
+ kept: list[SearchResult] = []
264
+ used = 0
265
+ for result in results:
266
+ size = len(_block(result))
267
+ if kept and used + size > max_chars:
268
+ break
269
+ kept.append(result)
270
+ used += size
271
+ return kept
272
+
273
+
274
+ def context(results: list[SearchResult], max_chars: int = 12000) -> str:
275
+ """Packs ranked results into one readable block for an agent prompt,
276
+ stopping before the caller's character budget is exceeded.
277
+ """
278
+ return "\n\n".join(_block(result) for result in within_budget(results, max_chars))
@@ -0,0 +1,202 @@
1
+ """Guards for the three ranking defects a cold foreign repository exposed.
2
+
3
+ Measured on cc-enforcer -- 1153 units, no written descriptions -- the single
4
+ largest declaration came back in the top three for four questions out of six,
5
+ and the words deciding the ranking were the ones present in nearly every unit.
6
+ Three causes, all in how a match was scored:
7
+
8
+ H: every query word counted the same, so `the` (49% of units) and `calls`
9
+ (97%) outweighed `daemon` (2 units) and `warm` (none).
10
+ I: nothing normalised for size, and the largest unit held 539 distinct terms
11
+ against a median of 52, so it could contain any query by accident.
12
+ J: a term in a declaration's name counted no more than the same term two
13
+ hundred lines into a body, which put test units above the code they test.
14
+
15
+ Each test is built so that the previous rule -- the fraction of query words
16
+ present anywhere in the unit -- ranks the *wrong* answer first.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ from pathlib import Path
22
+
23
+ from ragyourcode.indexer import build_units
24
+ from ragyourcode.models import CodeUnit
25
+ from ragyourcode.search import FIELD_WEIGHTS, _block, build_search_index, context, search, within_budget
26
+
27
+ RARE_QUERY = "handle the request and return the response quiesce"
28
+
29
+
30
+ def _with_a_word_nothing_else_uses(root: Path) -> Path:
31
+ """One more unit, whose only distinguishing word appears nowhere else."""
32
+ (root / "rare.py").write_text(
33
+ "def quiesce(attempt):\n"
34
+ ' """Settle down."""\n'
35
+ " return attempt\n",
36
+ encoding="utf-8",
37
+ )
38
+ return root
39
+
40
+
41
+ def test_a_word_in_almost_every_unit_cannot_decide_the_ranking(tmp_path: Path, skewed_corpus):
42
+ """Rarity is evidence; ubiquity is not.
43
+
44
+ Under the previous rule the unit matching six common words out of seven
45
+ scored 0.857 and the only unit that answers the question scored 0.286.
46
+ Nothing here is a stopword list: `the` and `request` earn their low weight
47
+ from this corpus, which is why the mechanism works on a repository written
48
+ in any language, including one where the tokens are Chinese bigrams.
49
+ """
50
+ units = build_units(_with_a_word_nothing_else_uses(skewed_corpus(tmp_path)))
51
+ results = search(units, RARE_QUERY, limit=3)
52
+ assert results[0].unit.name == "quiesce", [result.unit.id for result in results]
53
+
54
+
55
+ def test_matched_terms_survive_the_posting_list_lookup(tmp_path: Path, skewed_corpus):
56
+ """Postings now carry a weight beside each id, and are searched by id.
57
+
58
+ A membership test comparing whole entries would report no matched terms at
59
+ all, turning every result's evidence -- the thing that makes a result
60
+ checkable -- into an empty list.
61
+ """
62
+ units = build_units(_with_a_word_nothing_else_uses(skewed_corpus(tmp_path)))
63
+ results = search(units, RARE_QUERY, limit=8)
64
+ assert all(result.matched_terms for result in results), [
65
+ (result.unit.id, result.matched_terms) for result in results
66
+ ]
67
+
68
+
69
+ def test_a_long_unit_does_not_win_on_size_alone(tmp_path: Path, skewed_corpus):
70
+ """Containing a word is not the same as being about it.
71
+
72
+ The corpus has to be this large for the defect to appear at all. Every
73
+ query word here reaches all eighty-one units, which puts its posting list
74
+ past the selective threshold, so no vector score is computed and the
75
+ lexical rule decides alone. In a two-unit fixture the feature-hash vector
76
+ hides the problem, because normalising by a unit's token mass is itself a
77
+ crude length normalisation -- which is why the defect was visible on a
78
+ 1153-unit repository and invisible in the small benchmark.
79
+ """
80
+ filler = "\n".join(f" step_{number} = compute(step_{number - 1})" for number in range(1, 200))
81
+ (skewed_corpus(tmp_path) / "aggregate.py").write_text(
82
+ "def process_everything(request, response):\n"
83
+ ' """Handle request and return response."""\n'
84
+ f"{filler}\n"
85
+ " return response\n",
86
+ encoding="utf-8",
87
+ )
88
+ units = build_units(tmp_path)
89
+ results = search(units, "handle the request and return the response", limit=3)
90
+ # Same words as every handler, two hundred lines of unrelated code, and it
91
+ # came back first: matched-word count made them equal and the tie fell to
92
+ # the unit id.
93
+ assert "process_everything" not in [result.unit.name for result in results], [
94
+ result.unit.id for result in results
95
+ ]
96
+
97
+
98
+ def test_the_same_word_counts_for_more_in_a_name_than_in_a_body(tmp_path: Path):
99
+ """Where the author wrote a word says how much it meant.
100
+
101
+ Both units are the same size and mention the word once, so nothing but the
102
+ field it appears in can separate them. The vector is switched off because
103
+ it would answer this question by a different route -- its cosine already
104
+ favours the unit where the word is a larger share of the text -- and the
105
+ lexical rule is what changed. Without that, both scored exactly 1.0 and
106
+ the tie fell to the unit id.
107
+
108
+ This is the mechanism, not its limit: a body repeating a word forty times
109
+ still outranks the declaration named after it, because forty occurrences
110
+ over an average-length body is genuinely a lot of evidence. See
111
+ docs/TESTING.md for what that still costs.
112
+ """
113
+ (tmp_path / "named.py").write_text(
114
+ 'def checkpoint(job):\n """Do the work and hand it back."""\n return job\n',
115
+ encoding="utf-8",
116
+ )
117
+ (tmp_path / "mentions.py").write_text(
118
+ 'def worker(job):\n """Do the work and hand it back."""\n # checkpoint\n return job\n',
119
+ encoding="utf-8",
120
+ )
121
+ units = build_units(tmp_path)
122
+ results = search(units, "checkpoint", limit=2, vector_weight=0.0)
123
+ assert results[0].unit.name == "checkpoint", [result.unit.id for result in results]
124
+
125
+
126
+ def test_every_searchable_field_carries_a_weight(tmp_path: Path):
127
+ """A field with no weight would vanish from ranking the moment it appeared.
128
+
129
+ The two live in different modules on purpose -- one says what retrieval may
130
+ match, the other how much each part counts -- so this is the join that
131
+ keeps them from drifting apart.
132
+ """
133
+ (tmp_path / "one.py").write_text("def f(a):\n return a\n", encoding="utf-8")
134
+ unit = build_units(tmp_path)[0]
135
+ assert set(unit.searchable_fields) == set(FIELD_WEIGHTS)
136
+
137
+
138
+ def test_searchable_text_is_every_field_and_nothing_else(tmp_path: Path):
139
+ """The text a unit is embedded from must not lose a field ranking uses."""
140
+ (tmp_path / "one.py").write_text(
141
+ '"""Module."""\n\n\ndef f(a):\n """Add one."""\n return a + 1\n',
142
+ encoding="utf-8",
143
+ )
144
+ unit = build_units(tmp_path)[0]
145
+ assert unit.searchable_text == "\n".join(unit.searchable_fields.values())
146
+
147
+
148
+ def _budgeted(root: Path) -> list:
149
+ for index in range(6):
150
+ (root / f"unit_{index}.py").write_text(
151
+ f"def widget_{index}(value):\n"
152
+ f' """Widget {index} handles the value."""\n'
153
+ + "".join(f" line_{number} = value\n" for number in range(40))
154
+ + " return value\n",
155
+ encoding="utf-8",
156
+ )
157
+ return search(build_units(root), "widget handles the value", limit=6)
158
+
159
+
160
+ def test_the_budget_bounds_the_results_not_only_the_context(tmp_path: Path):
161
+ """`search --json` served 65,025 characters against a budget of 12,000.
162
+
163
+ The cap applied to the context string while every result was serialised
164
+ beside it in full, so the half an agent reads was the half that overran.
165
+ """
166
+ results = _budgeted(tmp_path)
167
+ assert len(results) > 1, "the fixture must produce enough results to be trimmed"
168
+ budget = len(_block(results[0])) + 10
169
+ kept = within_budget(results, budget)
170
+ assert len(kept) < len(results)
171
+ assert sum(len(_block(result)) for result in kept) <= budget
172
+ assert context(kept, budget) == context(results, budget)
173
+
174
+
175
+ def test_a_single_result_larger_than_the_budget_is_still_returned(tmp_path: Path):
176
+ """Finding something and returning nothing is worse than one oversized hit."""
177
+ results = _budgeted(tmp_path)
178
+ assert [result.unit.id for result in within_budget(results, 0)] == [results[0].unit.id]
179
+
180
+
181
+ def test_a_field_every_unit_leaves_empty_does_not_divide_by_zero():
182
+ """Each field is normalised against that field's own average length.
183
+
184
+ A corpus where nobody fills one in -- no signature, no body, no written
185
+ description -- makes that average zero, and the average is a denominator.
186
+ """
187
+ unit = CodeUnit(
188
+ id="empty.py:1:nothing",
189
+ path="empty.py",
190
+ language="python",
191
+ kind="function",
192
+ name="nothing",
193
+ qualified_name="nothing",
194
+ signature="",
195
+ start_line=1,
196
+ end_line=1,
197
+ source="",
198
+ description="",
199
+ serial=1,
200
+ )
201
+ index = build_search_index([unit])
202
+ assert search([unit], "nothing", limit=1, search_index=index)[0].unit.id == unit.id
@@ -21,30 +21,42 @@ from ragyourcode import descriptions as descriptions_module
21
21
  from ragyourcode.indexer import build_units
22
22
 
23
23
  ROOT = Path(__file__).resolve().parents[1]
24
+ COLD_PATH = ROOT / "benchmarks" / "cold_queries.json"
24
25
 
25
26
 
26
27
  @pytest.fixture(scope="module")
27
28
  def units():
28
29
  # With the description store, because that is how the CLI builds an index.
29
- # Without it the same ruler scores 0.171 rather than 0.500 on hit@1, which
30
- # would be measuring a configuration nobody runs.
31
30
  return build_units(ROOT, descriptions=descriptions_module.load(ROOT))
32
31
 
33
32
 
33
+ @pytest.fixture(scope="module")
34
+ def cold_units():
35
+ """The same repository as a first-time user's index sees it: parsed, with
36
+ only the sentence the parser generates and nothing anybody wrote.
37
+ """
38
+ return build_units(ROOT)
39
+
40
+
34
41
  def test_every_acceptable_answer_names_code_that_exists(units):
35
42
  problems = check_ruler(load_questions(), units)
36
43
  assert not problems, "the ruler has drifted from the code:\n " + "\n ".join(problems)
37
44
 
38
45
 
39
- def test_the_ruler_is_well_formed():
40
- questions = load_questions()
46
+ @pytest.mark.parametrize("path", [QUERIES_PATH, COLD_PATH], ids=["repository", "cold"])
47
+ def test_the_ruler_is_well_formed(path: Path):
48
+ questions = load_questions(path)
41
49
  entries = questions["queries"]
42
- assert len(entries) >= 50, "too few questions to distinguish a change from noise"
50
+ assert len(entries) >= 30, "too few questions to distinguish a change from noise"
43
51
  assert questions["k"] >= 1
52
+ seen: set[str] = set()
44
53
  for entry in entries:
45
54
  assert entry["query"].strip(), f"{entry['id']}: empty question"
55
+ assert entry["id"] not in seen, f"{entry['id']}: duplicate question id"
56
+ seen.add(entry["id"])
46
57
  assert entry["language"] in {"en", "zh"}
47
58
  assert entry["kind"] in {"concept", "why", "symbol"}
59
+ assert entry["acceptable"], f"{entry['id']}: no acceptable answer listed"
48
60
  assert all(len(pair) == 2 for pair in entry["acceptable"])
49
61
  # Both languages have to be represented, because the CJK path through the
50
62
  # tokenizer is the one place a query can share no character class with the
@@ -53,6 +65,29 @@ def test_the_ruler_is_well_formed():
53
65
  assert languages == {"en", "zh"}
54
66
 
55
67
 
68
+ def test_the_cold_ruler_says_which_repository_it_grades():
69
+ """It grades code that is not in this repository, so it has to name it and
70
+ say why a ruler asked about this project cannot stand in for it.
71
+ """
72
+ questions = load_questions(COLD_PATH)
73
+ assert questions["repository"], "a ruler over foreign code must name that code"
74
+ assert questions["caveat"].strip()
75
+ assert questions["why"].strip()
76
+
77
+
78
+ def test_written_descriptions_beat_generated_ones(units, cold_units):
79
+ """The reason `describe` exists, asserted rather than stated.
80
+
81
+ This used to be a comment quoting two numbers, which is exactly the kind of
82
+ claim that rots: both had already moved by the time anybody looked.
83
+ """
84
+ questions = load_questions()
85
+ warm = evaluate(units, questions)["aggregate"]
86
+ cold = evaluate(cold_units, questions)["aggregate"]
87
+ assert warm["hit_at_1"] > cold["hit_at_1"], (warm, cold)
88
+ assert warm["mrr"] > cold["mrr"], (warm, cold)
89
+
90
+
56
91
  def test_the_ruler_has_headroom(units):
57
92
  """A ruler everything already passes cannot measure an improvement.
58
93
 
@@ -17,31 +17,13 @@ from ragyourcode.indexer import build_units, fingerprint, read_index, snapshot_r
17
17
  from ragyourcode.search import build_search_index, search
18
18
 
19
19
 
20
- def _corpus(tmp_path: Path, count: int = 80) -> Path:
21
- """More units than the selective threshold's 64 floor, with skewed terms."""
22
- for index in range(count):
23
- (tmp_path / f"mod_{index}.py").write_text(
24
- f"def handler_{index}(request, response):\n"
25
- f' """Handle request and return response."""\n'
26
- f" return response\n",
27
- encoding="utf-8",
28
- )
29
- (tmp_path / "special.py").write_text(
30
- "def retry_request_with_backoff(request, response):\n"
31
- ' """Retry a failed request and return response."""\n'
32
- " return response\n",
33
- encoding="utf-8",
34
- )
35
- return tmp_path
36
-
37
-
38
- def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path):
20
+ def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path, skewed_corpus):
39
21
  """A rare token must not shrink the candidate set to itself.
40
22
 
41
23
  `backoff` reaches one unit; `request` and `response` reach all 81. Scoring
42
24
  only what the rare term reached returned a single result for `--limit 8`.
43
25
  """
44
- units = build_units(_corpus(tmp_path))
26
+ units = build_units(skewed_corpus(tmp_path))
45
27
  assert len(units) > 64, "the corpus must exceed the selective threshold's floor"
46
28
  index = build_search_index(units)
47
29
  results = search(units, "backoff request response", limit=8, search_index=index)
@@ -50,8 +32,8 @@ def test_limit_is_filled_when_a_rare_term_joins_common_ones(tmp_path: Path):
50
32
  assert all(result.matched_terms for result in results), "every returned unit must carry its evidence"
51
33
 
52
34
 
53
- def test_every_lexically_matching_unit_can_be_returned(tmp_path: Path):
54
- units = build_units(_corpus(tmp_path, count=70))
35
+ def test_every_lexically_matching_unit_can_be_returned(tmp_path: Path, skewed_corpus):
36
+ units = build_units(skewed_corpus(tmp_path, count=70))
55
37
  index = build_search_index(units)
56
38
  results = search(units, "handler_7 request", limit=100, search_index=index)
57
39
  returned = {result.unit.id for result in results}
@@ -1,151 +0,0 @@
1
- """Hybrid lexical/vector retrieval for agent context."""
2
-
3
- from __future__ import annotations
4
-
5
- import heapq
6
- from bisect import bisect_left
7
- from collections import Counter, defaultdict
8
- from dataclasses import dataclass
9
-
10
- from .config import BY_PATH
11
- from .embeddings import DEFAULT_DIMENSIONS, embed, tokenize
12
- from .models import CodeUnit, SearchResult
13
-
14
- # Named here rather than repeated as a literal so `search.vector_weight` in
15
- # rag-your-code.toml and the default a direct caller gets cannot drift apart.
16
- DEFAULT_VECTOR_WEIGHT: float = BY_PATH["search.vector_weight"].default
17
-
18
-
19
- @dataclass(slots=True)
20
- class SearchIndex:
21
- """In-memory inverted index reused across queries.
22
-
23
- Building this once avoids re-tokenizing every code unit. Matched terms are
24
- read straight out of ``postings``; an earlier version also cached a
25
- per-unit frozenset of every token, which cost the largest share of the
26
- index's resident memory while holding nothing ``postings`` did not already
27
- have. The same structure can later be backed by SQLite/ANN storage.
28
- """
29
-
30
- units: dict[str, CodeUnit]
31
- postings: dict[str, tuple[str, ...]]
32
-
33
-
34
- def build_search_index(units: list[CodeUnit]) -> SearchIndex:
35
- """Builds the inverted lookup table by tokenising every unit once and
36
- recording, for each term, which units contain it. The lists are kept
37
- sorted so membership can later be answered by binary search rather than
38
- by carrying a word set through the scoring loop.
39
- """
40
- postings: dict[str, set[str]] = defaultdict(set)
41
- for unit in units:
42
- for term in set(tokenize(unit.searchable_text)):
43
- postings[term].add(unit.id)
44
- return SearchIndex({unit.id: unit for unit in units}, {term: tuple(sorted(ids)) for term, ids in postings.items()})
45
-
46
-
47
- def _in_posting(posting: tuple[str, ...], unit_id: str) -> bool:
48
- """Membership test over a posting list, which build_search_index keeps sorted."""
49
- position = bisect_left(posting, unit_id)
50
- return position < len(posting) and posting[position] == unit_id
51
-
52
-
53
- def search(
54
- units: list[CodeUnit],
55
- query: str,
56
- limit: int = 8,
57
- search_index: SearchIndex | None = None,
58
- vector_weight: float = DEFAULT_VECTOR_WEIGHT,
59
- ) -> list[SearchResult]:
60
- """Ranks code units against a natural-language query by combining word
61
- overlap with vector similarity. Every unit sharing any query word is
62
- scored, so nothing that matches is left out: an earlier design let the
63
- vector shortlist decide who was scored at all, and units matching more
64
- query words went unranked, returning one result where eight were asked
65
- for. Full vector scoring is reserved for units reached by a selective
66
- term, since computing a dot product for everything a stopword-class term
67
- touches is pure cost. Word overlap stays dominant so an exact symbol
68
- match cannot be pushed below a noisy neighbour, and with no overlap
69
- anywhere it falls back to similarity alone.
70
- """
71
- query_tokens = set(tokenize(query))
72
- if limit <= 0 or not query_tokens:
73
- return []
74
- query_vector = embed(query, len(units[0].vector) if units and units[0].vector else DEFAULT_DIMENSIONS)
75
- query_features = [(index, value) for index, value in enumerate(query_vector) if value]
76
- search_index = search_index or build_search_index(units)
77
- postings = [(token, search_index.postings.get(token, ())) for token in query_tokens]
78
-
79
- # Matched terms come straight from the posting lists. Walking postings costs
80
- # O(sum of posting lengths) of dict work, where scoring each candidate by
81
- # intersecting a cached per-unit token set cost a frozenset operation per
82
- # candidate -- and every lexically matching unit now gets a score.
83
- matched_counts: Counter[str] = Counter()
84
- for _, posting in postings:
85
- matched_counts.update(posting)
86
-
87
- # A term present in a tenth of the corpus (``function``, ``return``) is not
88
- # evidence of relevance, and its posting list is effectively the whole index;
89
- # computing a 384-dimension dot product for everything it reaches is what
90
- # this threshold exists to avoid. It selects which candidates additionally
91
- # receive a VECTOR score. It must not decide which candidates are scored at
92
- # all -- doing that silently dropped units matching MORE query terms and
93
- # under-filled ``limit`` (116 units, `--limit 8`, one result returned).
94
- selective_threshold = max(64, min(2048, len(units) // 10))
95
- vector_ids: set[str] = set()
96
- for _, posting in postings:
97
- if 0 < len(posting) <= selective_threshold:
98
- vector_ids.update(posting)
99
- if matched_counts and not vector_ids and len(matched_counts) <= selective_threshold:
100
- vector_ids = set(matched_counts)
101
-
102
- # With no lexical overlap anywhere, fall back to pure cosine so a genuine
103
- # paraphrase still retrieves something.
104
- candidate_ids = matched_counts.keys() if matched_counts else search_index.units.keys()
105
- scored: list[tuple[float, str]] = []
106
- for unit_id in candidate_ids:
107
- unit = search_index.units[unit_id]
108
- lexical = matched_counts.get(unit_id, 0) / len(query_tokens)
109
- vector_score = (
110
- sum(value * unit.vector[index] for index, value in query_features)
111
- if (unit_id in vector_ids or not matched_counts) and len(unit.vector) == len(query_vector)
112
- else 0.0
113
- )
114
- # Exact symbols and domain terms are high-confidence evidence. Keep
115
- # lexical overlap dominant so a noisy feature-hash vector cannot push an
116
- # exact match below an unrelated semantic neighbor; use the vector score
117
- # to rank paraphrases and break lexical ties.
118
- score = lexical + vector_weight * max(0.0, vector_score)
119
- if lexical or score > 0:
120
- scored.append((score, unit_id))
121
- # Materialise only the winners. Building a SearchResult for every lexical
122
- # match and then sorting all of them cost more than the scoring itself once
123
- # recall became complete: at 10k units that alone was most of a 10x query
124
- # regression. nsmallest keeps the exact previous ordering -- highest score
125
- # first, ties broken by ascending unit id -- at O(n log limit).
126
- winners = heapq.nsmallest(limit, scored, key=lambda item: (-item[0], item[1]))
127
- # Which terms matched is only needed for the handful actually returned, and
128
- # postings are stored sorted, so a binary search beats carrying a per-unit
129
- # term list through the scoring loop for every candidate in the corpus.
130
- return [
131
- SearchResult(search_index.units[unit_id], score, sorted(token for token, posting in postings if _in_posting(posting, unit_id)))
132
- for score, unit_id in winners
133
- ]
134
-
135
-
136
- def context(results: list[SearchResult], max_chars: int = 12000) -> str:
137
- """Packs ranked results into one readable block for an agent prompt, each
138
- carrying identifier, score, matching evidence, description and source,
139
- and stops before exceeding the caller character budget.
140
- """
141
- blocks: list[str] = []
142
- used = 0
143
- for result in results:
144
- unit = result.unit
145
- evidence = "\nEvidence: " + " | ".join(result.evidence) if result.evidence else ""
146
- block = f"[{unit.id}] score={result.score:.3f}{evidence}\n{unit.description}\n```{unit.language}\n{unit.source}\n```"
147
- if used + len(block) > max_chars:
148
- break
149
- blocks.append(block)
150
- used += len(block)
151
- return "\n\n".join(blocks)
File without changes
File without changes