rag-your-code 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {rag_your_code-0.6.0/src/rag_your_code.egg-info → rag_your_code-0.7.0}/PKG-INFO +55 -8
  2. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/README.md +54 -7
  3. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/pyproject.toml +1 -1
  4. {rag_your_code-0.6.0 → rag_your_code-0.7.0/src/rag_your_code.egg-info}/PKG-INFO +55 -8
  5. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/SOURCES.txt +3 -1
  6. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/__init__.py +1 -1
  7. rag_your_code-0.7.0/src/ragyourcode/agentic.py +118 -0
  8. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/cli.py +70 -118
  9. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/models.py +23 -11
  10. rag_your_code-0.7.0/src/ragyourcode/workflow.py +184 -0
  11. rag_your_code-0.7.0/tests/test_agentic.py +146 -0
  12. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_metadata.py +26 -2
  13. rag_your_code-0.7.0/tests/test_workflow.py +127 -0
  14. rag_your_code-0.6.0/src/ragyourcode/agentic.py +0 -62
  15. rag_your_code-0.6.0/tests/test_agentic.py +0 -29
  16. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/LICENSE +0 -0
  17. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/setup.cfg +0 -0
  18. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  19. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  20. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  21. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  22. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/annotate.py +0 -0
  23. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/config.py +0 -0
  24. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/descriptions.py +0 -0
  25. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/document.py +0 -0
  26. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/embeddings.py +0 -0
  27. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/graph.py +0 -0
  28. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/indexer.py +0 -0
  29. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/parser.py +0 -0
  30. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/py.typed +0 -0
  31. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/src/ragyourcode/search.py +0 -0
  32. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_agent_protocol.py +0 -0
  33. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_config.py +0 -0
  34. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_descriptions.py +0 -0
  35. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_doc_comments.py +0 -0
  36. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_document.py +0 -0
  37. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_e2e_cli.py +0 -0
  38. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_golden.py +0 -0
  39. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_graph_incremental.py +0 -0
  40. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_language_fixtures.py +0 -0
  41. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_large_repo.py +0 -0
  42. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_multilanguage.py +0 -0
  43. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_parser_edges.py +0 -0
  44. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_ragyourcode.py +0 -0
  45. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_ranking.py +0 -0
  46. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_repo_queries.py +0 -0
  47. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_resilience.py +0 -0
  48. {rag_your_code-0.6.0 → rag_your_code-0.7.0}/tests/test_retrieval_correctness.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -76,14 +76,23 @@ package itself on first use.
76
76
  ```bash
77
77
  pip install rag-your-code
78
78
 
79
- rag-your-code index .
79
+ rag-your-code bootstrap . # index, and say what is still missing
80
80
  rag-your-code search "where are HTTP retries handled" --json
81
81
  rag-your-code search "what calls the retry handler" --graph --hops 1 --json
82
82
  ```
83
83
 
84
+ `bootstrap` exists because indexing a repository is not the same as making it
85
+ searchable, and nothing used to say so. A fresh index retrieves against the
86
+ sentence the parser generated, which adds no word the source did not already
87
+ have. It reports which rung this repository is on — descriptions still to
88
+ write, a promotion to apply, or nothing left — and hands over that rung's
89
+ work. It reads the state rather than remembering a position, so running it
90
+ again after each round is how you make progress. `index` still exists and does
91
+ only the indexing.
92
+
84
93
  The index is written under `.rag-your-code/`; your source files are never
85
- modified. Later `index` runs reuse unchanged files. For a large repository,
86
- prefer `rag-your-code index . --compact`.
94
+ modified. Later runs reuse unchanged files. For a large repository, prefer
95
+ `--compact`.
87
96
 
88
97
  ## How it works
89
98
 
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
135
144
  an unrelated pair are indistinguishable, because no shared word is no shared
136
145
  word either way.
137
146
 
138
- Retrieval works regardless, because **identifiers and docstrings are already
139
- natural language** — `retry_charge` contains *retry* and *charge*. But it
140
- reaches only concepts somebody wrote down. Two things close the rest of the
141
- gap, and neither is a model:
147
+ Retrieval works regardless, because **the prose people write about code is
148
+ already natural language** — docstrings, comments, descriptions. Identifiers
149
+ are not part of that, and it is worth being exact: `retry_charge` tokenizes to
150
+ one opaque term, not to *retry* and *charge*. Splitting identifiers was
151
+ implemented and measured against all three rulers, with query and stored
152
+ vectors rebuilt together, and it was equal or worse on every one; the pieces it
153
+ makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
154
+ discounts to nothing.
155
+
156
+ So retrieval reaches only concepts somebody wrote down. Two things close the
157
+ rest of the gap, and neither is a model:
142
158
 
143
159
  - **Your agent rewrites the query.** It has the conversation; turning
144
160
  "重试扣款" into `retry charge payment gateway` costs it nothing.
145
161
  - **Your agent writes the descriptions**, which puts the missing vocabulary
146
162
  into the index once instead of into every query.
147
163
 
164
+ ### Seven attempts to make the vector half earn its place
165
+
166
+ Because "just use a better embedding" is the obvious next thought, it was
167
+ measured rather than argued about. Six schemes were implemented — character
168
+ n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
169
+ signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
170
+ lexical postings expansion and embedding only the authored text. **On the
171
+ foreign-repository ruler, not one of them beat using no vector at all.**
172
+
173
+ The reason is architectural, not representational. Retrieval scores only the
174
+ units the lexical half already matched, so a vector can reorder an answer but
175
+ can never make one *retrievable*; pure cosine fires only when nothing matched
176
+ at all, on 1 question of 35. Every scheme was competing for the same one- or
177
+ two-question reshuffle inside a list that had already been chosen.
178
+
179
+ Two things follow, and both are stated here rather than buried. Corpus-learned
180
+ semantics need orders of magnitude more text than a repository has: 65% of the
181
+ foreign corpus's terms appear in four or fewer units, so their co-occurrence
182
+ row is a handful of sightings rather than a distribution. And on a *described*
183
+ repository the shipped hash is useful precisely because it is blunt — every
184
+ scheme that sharpened it lost ground there.
185
+
148
186
  ## Agent-authored descriptions
149
187
 
150
188
  Every unit carries a description, and that description is indexed. By default
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
330
368
  one reply per line:
331
369
 
332
370
  ```json
371
+ {"action":"bootstrap"}
333
372
  {"action":"search","query":"database transaction rollback","limit":5}
334
373
  {"action":"research","query":"trace payment retry behavior","max_steps":2}
335
374
  {"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
@@ -340,6 +379,14 @@ one reply per line:
340
379
  {"action":"stats"}
341
380
  ```
342
381
 
382
+ **A result is navigation, not the file.** `results` carries the identifier,
383
+ path, line range, signature, description, score and matched terms. The code
384
+ arrives once, in the reply's `context`, trimmed to `max_chars`, and
385
+ `omitted_for_budget` says how many results it did not reach. Carrying the
386
+ source per result as well is what let one `search --json` reply reach 65,025
387
+ characters against a stated budget of 12,000, and one `research` reply reach
388
+ 111,843 by serialising the same eight units three times over.
389
+
343
390
  **No single request can end the session.** Numeric fields saturate at their
344
391
  bounds, `open` is bounded in both lines and bytes, and anything unanticipated
345
392
  is reported in-band with its exception type. Streams are pinned to UTF-8
@@ -49,14 +49,23 @@ package itself on first use.
49
49
  ```bash
50
50
  pip install rag-your-code
51
51
 
52
- rag-your-code index .
52
+ rag-your-code bootstrap . # index, and say what is still missing
53
53
  rag-your-code search "where are HTTP retries handled" --json
54
54
  rag-your-code search "what calls the retry handler" --graph --hops 1 --json
55
55
  ```
56
56
 
57
+ `bootstrap` exists because indexing a repository is not the same as making it
58
+ searchable, and nothing used to say so. A fresh index retrieves against the
59
+ sentence the parser generated, which adds no word the source did not already
60
+ have. It reports which rung this repository is on — descriptions still to
61
+ write, a promotion to apply, or nothing left — and hands over that rung's
62
+ work. It reads the state rather than remembering a position, so running it
63
+ again after each round is how you make progress. `index` still exists and does
64
+ only the indexing.
65
+
57
66
  The index is written under `.rag-your-code/`; your source files are never
58
- modified. Later `index` runs reuse unchanged files. For a large repository,
59
- prefer `rag-your-code index . --compact`.
67
+ modified. Later runs reuse unchanged files. For a large repository, prefer
68
+ `--compact`.
60
69
 
61
70
  ## How it works
62
71
 
@@ -108,16 +117,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
108
117
  an unrelated pair are indistinguishable, because no shared word is no shared
109
118
  word either way.
110
119
 
111
- Retrieval works regardless, because **identifiers and docstrings are already
112
- natural language** — `retry_charge` contains *retry* and *charge*. But it
113
- reaches only concepts somebody wrote down. Two things close the rest of the
114
- gap, and neither is a model:
120
+ Retrieval works regardless, because **the prose people write about code is
121
+ already natural language** — docstrings, comments, descriptions. Identifiers
122
+ are not part of that, and it is worth being exact: `retry_charge` tokenizes to
123
+ one opaque term, not to *retry* and *charge*. Splitting identifiers was
124
+ implemented and measured against all three rulers, with query and stored
125
+ vectors rebuilt together, and it was equal or worse on every one; the pieces it
126
+ makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
127
+ discounts to nothing.
128
+
129
+ So retrieval reaches only concepts somebody wrote down. Two things close the
130
+ rest of the gap, and neither is a model:
115
131
 
116
132
  - **Your agent rewrites the query.** It has the conversation; turning
117
133
  "重试扣款" into `retry charge payment gateway` costs it nothing.
118
134
  - **Your agent writes the descriptions**, which puts the missing vocabulary
119
135
  into the index once instead of into every query.
120
136
 
137
+ ### Seven attempts to make the vector half earn its place
138
+
139
+ Because "just use a better embedding" is the obvious next thought, it was
140
+ measured rather than argued about. Six schemes were implemented — character
141
+ n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
142
+ signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
143
+ lexical postings expansion and embedding only the authored text. **On the
144
+ foreign-repository ruler, not one of them beat using no vector at all.**
145
+
146
+ The reason is architectural, not representational. Retrieval scores only the
147
+ units the lexical half already matched, so a vector can reorder an answer but
148
+ can never make one *retrievable*; pure cosine fires only when nothing matched
149
+ at all, on 1 question of 35. Every scheme was competing for the same one- or
150
+ two-question reshuffle inside a list that had already been chosen.
151
+
152
+ Two things follow, and both are stated here rather than buried. Corpus-learned
153
+ semantics need orders of magnitude more text than a repository has: 65% of the
154
+ foreign corpus's terms appear in four or fewer units, so their co-occurrence
155
+ row is a handful of sightings rather than a distribution. And on a *described*
156
+ repository the shipped hash is useful precisely because it is blunt — every
157
+ scheme that sharpened it lost ground there.
158
+
121
159
  ## Agent-authored descriptions
122
160
 
123
161
  Every unit carries a description, and that description is indexed. By default
@@ -303,6 +341,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
303
341
  one reply per line:
304
342
 
305
343
  ```json
344
+ {"action":"bootstrap"}
306
345
  {"action":"search","query":"database transaction rollback","limit":5}
307
346
  {"action":"research","query":"trace payment retry behavior","max_steps":2}
308
347
  {"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
@@ -313,6 +352,14 @@ one reply per line:
313
352
  {"action":"stats"}
314
353
  ```
315
354
 
355
+ **A result is navigation, not the file.** `results` carries the identifier,
356
+ path, line range, signature, description, score and matched terms. The code
357
+ arrives once, in the reply's `context`, trimmed to `max_chars`, and
358
+ `omitted_for_budget` says how many results it did not reach. Carrying the
359
+ source per result as well is what let one `search --json` reply reach 65,025
360
+ characters against a stated budget of 12,000, and one `research` reply reach
361
+ 111,843 by serialising the same eight units three times over.
362
+
316
363
  **No single request can end the session.** Numeric fields saturate at their
317
364
  bounds, `open` is bounded in both lines and bytes, and anything unanticipated
318
365
  is reported in-band with its exception type. Streams are pinned to UTF-8
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "0.6.0"
9
+ version = "0.7.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -76,14 +76,23 @@ package itself on first use.
76
76
  ```bash
77
77
  pip install rag-your-code
78
78
 
79
- rag-your-code index .
79
+ rag-your-code bootstrap . # index, and say what is still missing
80
80
  rag-your-code search "where are HTTP retries handled" --json
81
81
  rag-your-code search "what calls the retry handler" --graph --hops 1 --json
82
82
  ```
83
83
 
84
+ `bootstrap` exists because indexing a repository is not the same as making it
85
+ searchable, and nothing used to say so. A fresh index retrieves against the
86
+ sentence the parser generated, which adds no word the source did not already
87
+ have. It reports which rung this repository is on — descriptions still to
88
+ write, a promotion to apply, or nothing left — and hands over that rung's
89
+ work. It reads the state rather than remembering a position, so running it
90
+ again after each round is how you make progress. `index` still exists and does
91
+ only the indexing.
92
+
84
93
  The index is written under `.rag-your-code/`; your source files are never
85
- modified. Later `index` runs reuse unchanged files. For a large repository,
86
- prefer `rag-your-code index . --compact`.
94
+ modified. Later runs reuse unchanged files. For a large repository, prefer
95
+ `--compact`.
87
96
 
88
97
  ## How it works
89
98
 
@@ -135,16 +144,45 @@ A trained embedding model scores row 2 at around 0.8. Here a synonym pair and
135
144
  an unrelated pair are indistinguishable, because no shared word is no shared
136
145
  word either way.
137
146
 
138
- Retrieval works regardless, because **identifiers and docstrings are already
139
- natural language** — `retry_charge` contains *retry* and *charge*. But it
140
- reaches only concepts somebody wrote down. Two things close the rest of the
141
- gap, and neither is a model:
147
+ Retrieval works regardless, because **the prose people write about code is
148
+ already natural language** — docstrings, comments, descriptions. Identifiers
149
+ are not part of that, and it is worth being exact: `retry_charge` tokenizes to
150
+ one opaque term, not to *retry* and *charge*. Splitting identifiers was
151
+ implemented and measured against all three rulers, with query and stored
152
+ vectors rebuilt together, and it was equal or worse on every one; the pieces it
153
+ makes are `get`, `find`, `check`, `test`, which rarity weighting immediately
154
+ discounts to nothing.
155
+
156
+ So retrieval reaches only concepts somebody wrote down. Two things close the
157
+ rest of the gap, and neither is a model:
142
158
 
143
159
  - **Your agent rewrites the query.** It has the conversation; turning
144
160
  "重试扣款" into `retry charge payment gateway` costs it nothing.
145
161
  - **Your agent writes the descriptions**, which puts the missing vocabulary
146
162
  into the index once instead of into every query.
147
163
 
164
+ ### Seven attempts to make the vector half earn its place
165
+
166
+ Because "just use a better embedding" is the obvious next thought, it was
167
+ measured rather than argued about. Six schemes were implemented — character
168
+ n-grams, corpus co-occurrence via random indexing, truncated SVD, posting-list
169
+ signatures, a rarity- and field-weighted hash, and call-graph diffusion — plus
170
+ lexical postings expansion and embedding only the authored text. **On the
171
+ foreign-repository ruler, not one of them beat using no vector at all.**
172
+
173
+ The reason is architectural, not representational. Retrieval scores only the
174
+ units the lexical half already matched, so a vector can reorder an answer but
175
+ can never make one *retrievable*; pure cosine fires only when nothing matched
176
+ at all, on 1 question of 35. Every scheme was competing for the same one- or
177
+ two-question reshuffle inside a list that had already been chosen.
178
+
179
+ Two things follow, and both are stated here rather than buried. Corpus-learned
180
+ semantics need orders of magnitude more text than a repository has: 65% of the
181
+ foreign corpus's terms appear in four or fewer units, so their co-occurrence
182
+ row is a handful of sightings rather than a distribution. And on a *described*
183
+ repository the shipped hash is useful precisely because it is blunt — every
184
+ scheme that sharpened it lost ground there.
185
+
148
186
  ## Agent-authored descriptions
149
187
 
150
188
  Every unit carries a description, and that description is indexed. By default
@@ -330,6 +368,7 @@ full rebuild. The rest take effect immediately and invalidate nothing.
330
368
  one reply per line:
331
369
 
332
370
  ```json
371
+ {"action":"bootstrap"}
333
372
  {"action":"search","query":"database transaction rollback","limit":5}
334
373
  {"action":"research","query":"trace payment retry behavior","max_steps":2}
335
374
  {"action":"neighbors","id":"payments.py:4:retry_charge","hops":1}
@@ -340,6 +379,14 @@ one reply per line:
340
379
  {"action":"stats"}
341
380
  ```
342
381
 
382
+ **A result is navigation, not the file.** `results` carries the identifier,
383
+ path, line range, signature, description, score and matched terms. The code
384
+ arrives once, in the reply's `context`, trimmed to `max_chars`, and
385
+ `omitted_for_budget` says how many results it did not reach. Carrying the
386
+ source per result as well is what let one `search --json` reply reach 65,025
387
+ characters against a stated budget of 12,000, and one `research` reply reach
388
+ 111,843 by serialising the same eight units three times over.
389
+
343
390
  **No single request can end the session.** Numeric fields saturate at their
344
391
  bounds, `open` is bounded in both lines and bytes, and anything unanticipated
345
392
  is reported in-band with its exception type. Streams are pinned to UTF-8
@@ -21,6 +21,7 @@ src/ragyourcode/models.py
21
21
  src/ragyourcode/parser.py
22
22
  src/ragyourcode/py.typed
23
23
  src/ragyourcode/search.py
24
+ src/ragyourcode/workflow.py
24
25
  tests/test_agent_protocol.py
25
26
  tests/test_agentic.py
26
27
  tests/test_config.py
@@ -39,4 +40,5 @@ tests/test_ragyourcode.py
39
40
  tests/test_ranking.py
40
41
  tests/test_repo_queries.py
41
42
  tests/test_resilience.py
42
- tests/test_retrieval_correctness.py
43
+ tests/test_retrieval_correctness.py
44
+ tests/test_workflow.py
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "0.6.0"
6
+ __version__ = "0.7.0"
@@ -0,0 +1,118 @@
1
+ """Bounded, observable agentic retrieval (ARAG) orchestration."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from .graph import CodeGraph, graph_search
6
+ from .models import CodeUnit, SearchResult
7
+ from .search import DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
8
+
9
+
10
+ def _result_ids(results: list[SearchResult]) -> set[str]:
11
+ """Collects the unit identifiers out of a list of search results, so two
12
+ result sets can be compared for overlap or novelty. Used to decide
13
+ whether a second retrieval round actually surfaced anything new.
14
+ """
15
+ return {result.unit.id for result in results}
16
+
17
+
18
+ # How far ahead of the runner-up the top result must be before a second
19
+ # retrieval step is skipped, as a fraction of the top score.
20
+ #
21
+ # This replaced an absolute `confidence_threshold` of 0.8, which was a number
22
+ # tied to a scoring scale. When ranking became BM25F the scale moved and the
23
+ # threshold quietly died: measured across 105 ruler questions, only 3% of
24
+ # queries reached 0.8 at all, so the early stop had stopped existing and every
25
+ # research call ran the graph expansion. A margin is a ratio between two scores
26
+ # from the same query, so it cannot drift when the scoring changes again.
27
+ #
28
+ # Measured over those 105 questions: top-1 is correct 42% of the time overall,
29
+ # and 68-73% of the time among the queries this fires on. Anywhere in 0.2-0.5
30
+ # behaves the same at this sample size; 0.30 is the middle of that band, not a
31
+ # measured optimum.
32
+ DEFAULT_DOMINANCE: float = 0.30
33
+
34
+
35
+ def _serialize(results: list[SearchResult]) -> list[dict]:
36
+ """Converts search results into plain JSON-ready dictionaries for the agent
37
+ protocol reply.
38
+ """
39
+ return [result.to_dict() for result in results]
40
+
41
+
42
+ def _trace(results: list[SearchResult]) -> list[dict]:
43
+ """One step of the search, as a trace rather than as a payload.
44
+
45
+ A step exists so the caller can see how the answer was reached: which
46
+ units each round reached and how strongly. That needs an identifier, a
47
+ score and the matching words -- not a second and third copy of every
48
+ unit's description, callees and imports. Reporting two steps in full is
49
+ what made a research reply three times the size of the answer inside it.
50
+ """
51
+ return [
52
+ {"id": result.unit.id, "score": round(result.score, 6), "matched_terms": result.matched_terms}
53
+ for result in results
54
+ ]
55
+
56
+
57
+ def dominance(results: list[SearchResult]) -> float:
58
+ """How far the top result is ahead of the runner-up, relative to the top.
59
+
60
+ One result alone is unopposed and scores 1.0. Nothing scores 0.0. The
61
+ quantity is scale-free by construction, which is the whole point: it
62
+ compares two numbers produced by the same query under the same scoring
63
+ rule, so no future change to that rule can silently recalibrate it.
64
+ """
65
+ if not results:
66
+ return 0.0
67
+ top = results[0].score
68
+ if len(results) == 1:
69
+ return 1.0
70
+ return (top - results[1].score) / top if top else 0.0
71
+
72
+
73
+ def research(
74
+ units: list[CodeUnit],
75
+ query: str,
76
+ limit: int = 8,
77
+ hops: int = 1,
78
+ max_steps: int = 2,
79
+ dominance_threshold: float = DEFAULT_DOMINANCE,
80
+ graph: CodeGraph | None = None,
81
+ search_index: SearchIndex | None = None,
82
+ vector_weight: float = DEFAULT_VECTOR_WEIGHT,
83
+ max_chars: int = 12000,
84
+ ) -> dict:
85
+ """Run at most two deterministic retrieval steps and explain the stop.
86
+
87
+ This is deliberately bounded. A future LLM planner can replace the query
88
+ proposal, but the budget, evidence format, and no-progress stop remain
89
+ stable safety contracts.
90
+
91
+ The reply carries the code once, in a context block trimmed to
92
+ ``max_chars``. Results are navigation only. Reporting two retrieval steps
93
+ used to mean serialising the same eight units three times with their
94
+ source attached, which is how one research answer reached 111,843
95
+ characters against a stated budget of 12,000.
96
+ """
97
+ max_steps = min(2, max(1, max_steps))
98
+ steps: list[dict] = []
99
+ initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight)
100
+ steps.append({"action": "search", "query": query, "results": _trace(initial)})
101
+ if not initial:
102
+ return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
103
+ unopposed = dominance(initial) >= dominance_threshold and bool(initial[0].matched_terms)
104
+ if max_steps == 1 or unopposed:
105
+ kept = initial[:limit]
106
+ return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
107
+
108
+ expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight)
109
+ steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
110
+ merged = {result.unit.id: result for result in initial}
111
+ for result in expanded:
112
+ current = merged.get(result.unit.id)
113
+ if current is None or result.score > current.score:
114
+ merged[result.unit.id] = result
115
+ final = sorted(merged.values(), key=lambda result: (-result.score, result.unit.id))[:limit]
116
+ new_ids = _result_ids(final) - _result_ids(initial)
117
+ reason = "new_graph_evidence" if new_ids else "no_new_evidence"
118
+ return {"query": query, "results": _serialize(final), "steps": steps, "stop_reason": reason, "context": context(within_budget(final, max_chars), max_chars)}