rag-your-code 0.8.0__tar.gz → 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.8.0/src/rag_your_code.egg-info → rag_your_code-1.0.0}/PKG-INFO +94 -17
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/README.md +93 -16
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/pyproject.toml +1 -1
- {rag_your_code-0.8.0 → rag_your_code-1.0.0/src/rag_your_code.egg-info}/PKG-INFO +94 -17
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/SOURCES.txt +2 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/agentic.py +26 -4
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/cli.py +27 -8
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/config.py +16 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/graph.py +9 -3
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/indexer.py +32 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/search.py +168 -1
- rag_your_code-1.0.0/tests/test_absent_queries.py +114 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_descriptions.py +65 -0
- rag_your_code-1.0.0/tests/test_evidence.py +201 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_metadata.py +24 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/LICENSE +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/setup.cfg +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/workflow.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_agentic.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_config.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_document.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_providers.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_ranking.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_resilience.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 1.0.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -226,9 +226,9 @@ and left Chinese retrieval unchanged.
|
|
|
226
226
|
|
|
227
227
|
### Measured on this repository
|
|
228
228
|
|
|
229
|
-
This project describes its own implementation:
|
|
230
|
-
an agent-written bilingual description, committed to the repo, and
|
|
231
|
-
|
|
229
|
+
This project describes its own implementation: all 163 units under `src/` carry
|
|
230
|
+
an agent-written bilingual description, committed to the repo, and 155 of those
|
|
231
|
+
163 declarations also carry the author's own documentation in the source.
|
|
232
232
|
|
|
233
233
|
Seventy natural-language questions about this codebase, in English and
|
|
234
234
|
Chinese, each listing every unit that genuinely answers it
|
|
@@ -236,13 +236,13 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
236
236
|
|
|
237
237
|
| | generated descriptions | agent-written |
|
|
238
238
|
|---|---|---|
|
|
239
|
-
| hit@1 | 0.
|
|
240
|
-
| hit@3 | 0.
|
|
241
|
-
| MRR | 0.
|
|
242
|
-
|
|
|
239
|
+
| hit@1 | 0.314 | **0.500** |
|
|
240
|
+
| hit@3 | 0.457 | **0.729** |
|
|
241
|
+
| MRR | 0.379 | **0.583** |
|
|
242
|
+
| questions declined for want of evidence | 13.0% | **4.3%** |
|
|
243
243
|
|
|
244
|
-
|
|
245
|
-
|
|
244
|
+
Half again the first-place accuracy. Nineteen questions still fail, which is
|
|
245
|
+
what makes the set usable for measuring the next change;
|
|
246
246
|
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
247
247
|
the written column beats the generated one.
|
|
248
248
|
|
|
@@ -283,6 +283,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
|
283
283
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
284
284
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
285
285
|
|
|
286
|
+
### The question a ranking cannot answer
|
|
287
|
+
|
|
288
|
+
Both rulers above ask questions that *have* an answer, so both can only score
|
|
289
|
+
whether it was found. Neither can see the opposite failure. A ranking always
|
|
290
|
+
produces a least-bad unit and hands it back with a score and a rank that read
|
|
291
|
+
exactly like an answer — and it does that whether or not the repository
|
|
292
|
+
contains anything relevant at all.
|
|
293
|
+
|
|
294
|
+
So there is a third ruler: thirty questions about subjects neither repository
|
|
295
|
+
implements, where the only correct reply is nothing
|
|
296
|
+
([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
|
|
297
|
+
1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
|
|
298
|
+
languages:
|
|
299
|
+
|
|
300
|
+
| asked of a repository with no such code | answered with | on the evidence of |
|
|
301
|
+
|---|---|---|
|
|
302
|
+
| `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
|
|
303
|
+
| `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
|
|
304
|
+
| `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
|
|
305
|
+
|
|
306
|
+
Not a Chinese problem and not a ranking problem — a missing question. Nothing
|
|
307
|
+
in the pipeline asked *is any of this evidence*; it only asked which ranks
|
|
308
|
+
highest. Retrieval now asks both, and returns nothing when the answer to the
|
|
309
|
+
first is no:
|
|
310
|
+
|
|
311
|
+
| | before | now |
|
|
312
|
+
|---|---|---|
|
|
313
|
+
| unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
|
|
314
|
+
| the same, on the foreign repository | 0.000 | **0.800** |
|
|
315
|
+
| results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
|
|
316
|
+
| hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
|
|
317
|
+
| hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
|
|
318
|
+
|
|
319
|
+
The bar is the share of a question's **discriminating** words that occur in
|
|
320
|
+
the index — words the repository uses everywhere are dropped from both sides
|
|
321
|
+
of that fraction, which is the part that does the work. Half of `where are
|
|
322
|
+
CUDA kernels dispatched to the device` matches, and it looks like evidence
|
|
323
|
+
until you notice which half. Counting only words that distinguish silenced 18
|
|
324
|
+
of 30 unanswerable English questions that no plain coverage threshold reached
|
|
325
|
+
at all, at identical cost in real answers — 97 of 98 either way.
|
|
326
|
+
|
|
327
|
+
It is a ratio inside the query rather than a threshold on a score, because a
|
|
328
|
+
score threshold is tied to whatever scale the ranking currently produces —
|
|
329
|
+
this project has already had one of those stop meaning anything the moment
|
|
330
|
+
BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
|
|
331
|
+
behaviour exactly.
|
|
332
|
+
|
|
333
|
+
**What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
|
|
334
|
+
会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
|
|
335
|
+
repository contains are `utf` and `8`, both of which it uses everywhere. The
|
|
336
|
+
gate says too little of that question is distinctive, which is defensible, and
|
|
337
|
+
it was getting the right answer on a coincidence.
|
|
338
|
+
|
|
339
|
+
**What it does not fix:** an English question whose words genuinely occur here
|
|
340
|
+
in another sense. `how is the OAuth refresh token rotated` matches `refresh`
|
|
341
|
+
because this repository refreshes *indexes*, and no threshold separates those.
|
|
342
|
+
Six of fifteen English absent questions still get answered for that reason.
|
|
343
|
+
That is the case a real embedding model exists for, and it is measurable now
|
|
344
|
+
that the ruler exists.
|
|
345
|
+
|
|
346
|
+
An empty answer says which kind of empty it is, because each is recovered by a
|
|
347
|
+
different move:
|
|
348
|
+
|
|
349
|
+
```json
|
|
350
|
+
{"results": [],
|
|
351
|
+
"diagnosis": {"reason": "only_ubiquitous_terms_matched",
|
|
352
|
+
"matched_terms": [], "ubiquitous_terms": ["the", "to"],
|
|
353
|
+
"coverage": 0.0, "min_coverage": 0.4,
|
|
354
|
+
"hint": "The only words that matched are ones this repository uses throughout ..."}}
|
|
355
|
+
```
|
|
356
|
+
|
|
286
357
|
Descriptions live in `rag-your-code.descriptions.json` at the repository root
|
|
287
358
|
and are meant to be committed, so one person's pass benefits everyone who
|
|
288
359
|
clones. Each is keyed by unit id **and a digest of the unit's source**: when
|
|
@@ -332,23 +403,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
|
|
|
332
403
|
|
|
333
404
|
## Configuration
|
|
334
405
|
|
|
335
|
-
|
|
406
|
+
21 settings in `rag-your-code.toml` at the repository root:
|
|
336
407
|
|
|
337
408
|
```bash
|
|
338
409
|
rag-your-code config init # a commented file, all defaults
|
|
339
410
|
rag-your-code config list # effective values and their source
|
|
340
411
|
rag-your-code config set index.ignore '["vendor", "generated"]'
|
|
341
|
-
rag-your-code config set search.
|
|
412
|
+
rag-your-code config set search.min_coverage 0.25
|
|
342
413
|
```
|
|
343
414
|
|
|
344
415
|
| section | settings |
|
|
345
416
|
|---|---|
|
|
346
417
|
| `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
|
|
347
|
-
| `[embedding]` | `dimensions` |
|
|
348
|
-
| `[search]` | `vector_weight`, `limit`, `max_chars` |
|
|
418
|
+
| `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
|
|
419
|
+
| `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
|
|
349
420
|
| `[agent]` | `max_open_bytes`, `max_open_chars` |
|
|
350
421
|
| `[describe]` | `languages`, `batch`, `max_chars` |
|
|
351
422
|
|
|
423
|
+
This table is asserted against the settings table in `config.py`, in both
|
|
424
|
+
directions, by `tests/test_metadata.py` — it had already fallen nine settings
|
|
425
|
+
behind by 1.0.0, and a section listing three quarters of what exists is worse
|
|
426
|
+
than none, because it reads as complete.
|
|
427
|
+
|
|
352
428
|
Resolution is CLI flag > file > built-in default. There is no environment
|
|
353
429
|
layer: an index is an artifact of a repository, not of a shell.
|
|
354
430
|
|
|
@@ -358,9 +434,10 @@ silently dropped is indistinguishable from one that had no effect.
|
|
|
358
434
|
suffix it cannot read is walked, parsed to nothing, and reported as a clean
|
|
359
435
|
index of zero units.
|
|
360
436
|
|
|
361
|
-
The
|
|
362
|
-
*contains
|
|
363
|
-
|
|
437
|
+
The settings under `[index]` and `[embedding]` that decide what an index
|
|
438
|
+
*contains* — including which provider and model computed its vectors — have a
|
|
439
|
+
digest stored in the index, and changing one forces a full rebuild. The rest
|
|
440
|
+
take effect immediately and invalidate nothing.
|
|
364
441
|
|
|
365
442
|
## Agent protocol
|
|
366
443
|
|
|
@@ -199,9 +199,9 @@ and left Chinese retrieval unchanged.
|
|
|
199
199
|
|
|
200
200
|
### Measured on this repository
|
|
201
201
|
|
|
202
|
-
This project describes its own implementation:
|
|
203
|
-
an agent-written bilingual description, committed to the repo, and
|
|
204
|
-
|
|
202
|
+
This project describes its own implementation: all 163 units under `src/` carry
|
|
203
|
+
an agent-written bilingual description, committed to the repo, and 155 of those
|
|
204
|
+
163 declarations also carry the author's own documentation in the source.
|
|
205
205
|
|
|
206
206
|
Seventy natural-language questions about this codebase, in English and
|
|
207
207
|
Chinese, each listing every unit that genuinely answers it
|
|
@@ -209,13 +209,13 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
209
209
|
|
|
210
210
|
| | generated descriptions | agent-written |
|
|
211
211
|
|---|---|---|
|
|
212
|
-
| hit@1 | 0.
|
|
213
|
-
| hit@3 | 0.
|
|
214
|
-
| MRR | 0.
|
|
215
|
-
|
|
|
212
|
+
| hit@1 | 0.314 | **0.500** |
|
|
213
|
+
| hit@3 | 0.457 | **0.729** |
|
|
214
|
+
| MRR | 0.379 | **0.583** |
|
|
215
|
+
| questions declined for want of evidence | 13.0% | **4.3%** |
|
|
216
216
|
|
|
217
|
-
|
|
218
|
-
|
|
217
|
+
Half again the first-place accuracy. Nineteen questions still fail, which is
|
|
218
|
+
what makes the set usable for measuring the next change;
|
|
219
219
|
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
220
220
|
the written column beats the generated one.
|
|
221
221
|
|
|
@@ -256,6 +256,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
|
256
256
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
257
257
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
258
258
|
|
|
259
|
+
### The question a ranking cannot answer
|
|
260
|
+
|
|
261
|
+
Both rulers above ask questions that *have* an answer, so both can only score
|
|
262
|
+
whether it was found. Neither can see the opposite failure. A ranking always
|
|
263
|
+
produces a least-bad unit and hands it back with a score and a rank that read
|
|
264
|
+
exactly like an answer — and it does that whether or not the repository
|
|
265
|
+
contains anything relevant at all.
|
|
266
|
+
|
|
267
|
+
So there is a third ruler: thirty questions about subjects neither repository
|
|
268
|
+
implements, where the only correct reply is nothing
|
|
269
|
+
([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
|
|
270
|
+
1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
|
|
271
|
+
languages:
|
|
272
|
+
|
|
273
|
+
| asked of a repository with no such code | answered with | on the evidence of |
|
|
274
|
+
|---|---|---|
|
|
275
|
+
| `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
|
|
276
|
+
| `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
|
|
277
|
+
| `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
|
|
278
|
+
|
|
279
|
+
Not a Chinese problem and not a ranking problem — a missing question. Nothing
|
|
280
|
+
in the pipeline asked *is any of this evidence*; it only asked which ranks
|
|
281
|
+
highest. Retrieval now asks both, and returns nothing when the answer to the
|
|
282
|
+
first is no:
|
|
283
|
+
|
|
284
|
+
| | before | now |
|
|
285
|
+
|---|---|---|
|
|
286
|
+
| unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
|
|
287
|
+
| the same, on the foreign repository | 0.000 | **0.800** |
|
|
288
|
+
| results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
|
|
289
|
+
| hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
|
|
290
|
+
| hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
|
|
291
|
+
|
|
292
|
+
The bar is the share of a question's **discriminating** words that occur in
|
|
293
|
+
the index — words the repository uses everywhere are dropped from both sides
|
|
294
|
+
of that fraction, which is the part that does the work. Half of `where are
|
|
295
|
+
CUDA kernels dispatched to the device` matches, and it looks like evidence
|
|
296
|
+
until you notice which half. Counting only words that distinguish silenced 18
|
|
297
|
+
of 30 unanswerable English questions that no plain coverage threshold reached
|
|
298
|
+
at all, at identical cost in real answers — 97 of 98 either way.
|
|
299
|
+
|
|
300
|
+
It is a ratio inside the query rather than a threshold on a score, because a
|
|
301
|
+
score threshold is tied to whatever scale the ranking currently produces —
|
|
302
|
+
this project has already had one of those stop meaning anything the moment
|
|
303
|
+
BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
|
|
304
|
+
behaviour exactly.
|
|
305
|
+
|
|
306
|
+
**What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
|
|
307
|
+
会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
|
|
308
|
+
repository contains are `utf` and `8`, both of which it uses everywhere. The
|
|
309
|
+
gate says too little of that question is distinctive, which is defensible, and
|
|
310
|
+
it was getting the right answer on a coincidence.
|
|
311
|
+
|
|
312
|
+
**What it does not fix:** an English question whose words genuinely occur here
|
|
313
|
+
in another sense. `how is the OAuth refresh token rotated` matches `refresh`
|
|
314
|
+
because this repository refreshes *indexes*, and no threshold separates those.
|
|
315
|
+
Six of fifteen English absent questions still get answered for that reason.
|
|
316
|
+
That is the case a real embedding model exists for, and it is measurable now
|
|
317
|
+
that the ruler exists.
|
|
318
|
+
|
|
319
|
+
An empty answer says which kind of empty it is, because each is recovered by a
|
|
320
|
+
different move:
|
|
321
|
+
|
|
322
|
+
```json
|
|
323
|
+
{"results": [],
|
|
324
|
+
"diagnosis": {"reason": "only_ubiquitous_terms_matched",
|
|
325
|
+
"matched_terms": [], "ubiquitous_terms": ["the", "to"],
|
|
326
|
+
"coverage": 0.0, "min_coverage": 0.4,
|
|
327
|
+
"hint": "The only words that matched are ones this repository uses throughout ..."}}
|
|
328
|
+
```
|
|
329
|
+
|
|
259
330
|
Descriptions live in `rag-your-code.descriptions.json` at the repository root
|
|
260
331
|
and are meant to be committed, so one person's pass benefits everyone who
|
|
261
332
|
clones. Each is keyed by unit id **and a digest of the unit's source**: when
|
|
@@ -305,23 +376,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
|
|
|
305
376
|
|
|
306
377
|
## Configuration
|
|
307
378
|
|
|
308
|
-
|
|
379
|
+
21 settings in `rag-your-code.toml` at the repository root:
|
|
309
380
|
|
|
310
381
|
```bash
|
|
311
382
|
rag-your-code config init # a commented file, all defaults
|
|
312
383
|
rag-your-code config list # effective values and their source
|
|
313
384
|
rag-your-code config set index.ignore '["vendor", "generated"]'
|
|
314
|
-
rag-your-code config set search.
|
|
385
|
+
rag-your-code config set search.min_coverage 0.25
|
|
315
386
|
```
|
|
316
387
|
|
|
317
388
|
| section | settings |
|
|
318
389
|
|---|---|
|
|
319
390
|
| `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
|
|
320
|
-
| `[embedding]` | `dimensions` |
|
|
321
|
-
| `[search]` | `vector_weight`, `limit`, `max_chars` |
|
|
391
|
+
| `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
|
|
392
|
+
| `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
|
|
322
393
|
| `[agent]` | `max_open_bytes`, `max_open_chars` |
|
|
323
394
|
| `[describe]` | `languages`, `batch`, `max_chars` |
|
|
324
395
|
|
|
396
|
+
This table is asserted against the settings table in `config.py`, in both
|
|
397
|
+
directions, by `tests/test_metadata.py` — it had already fallen nine settings
|
|
398
|
+
behind by 1.0.0, and a section listing three quarters of what exists is worse
|
|
399
|
+
than none, because it reads as complete.
|
|
400
|
+
|
|
325
401
|
Resolution is CLI flag > file > built-in default. There is no environment
|
|
326
402
|
layer: an index is an artifact of a repository, not of a shell.
|
|
327
403
|
|
|
@@ -331,9 +407,10 @@ silently dropped is indistinguishable from one that had no effect.
|
|
|
331
407
|
suffix it cannot read is walked, parsed to nothing, and reported as a clean
|
|
332
408
|
index of zero units.
|
|
333
409
|
|
|
334
|
-
The
|
|
335
|
-
*contains
|
|
336
|
-
|
|
410
|
+
The settings under `[index]` and `[embedding]` that decide what an index
|
|
411
|
+
*contains* — including which provider and model computed its vectors — have a
|
|
412
|
+
digest stored in the index, and changing one forces a full rebuild. The rest
|
|
413
|
+
take effect immediately and invalidate nothing.
|
|
337
414
|
|
|
338
415
|
## Agent protocol
|
|
339
416
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 1.0.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -226,9 +226,9 @@ and left Chinese retrieval unchanged.
|
|
|
226
226
|
|
|
227
227
|
### Measured on this repository
|
|
228
228
|
|
|
229
|
-
This project describes its own implementation:
|
|
230
|
-
an agent-written bilingual description, committed to the repo, and
|
|
231
|
-
|
|
229
|
+
This project describes its own implementation: all 163 units under `src/` carry
|
|
230
|
+
an agent-written bilingual description, committed to the repo, and 155 of those
|
|
231
|
+
163 declarations also carry the author's own documentation in the source.
|
|
232
232
|
|
|
233
233
|
Seventy natural-language questions about this codebase, in English and
|
|
234
234
|
Chinese, each listing every unit that genuinely answers it
|
|
@@ -236,13 +236,13 @@ Chinese, each listing every unit that genuinely answers it
|
|
|
236
236
|
|
|
237
237
|
| | generated descriptions | agent-written |
|
|
238
238
|
|---|---|---|
|
|
239
|
-
| hit@1 | 0.
|
|
240
|
-
| hit@3 | 0.
|
|
241
|
-
| MRR | 0.
|
|
242
|
-
|
|
|
239
|
+
| hit@1 | 0.314 | **0.500** |
|
|
240
|
+
| hit@3 | 0.457 | **0.729** |
|
|
241
|
+
| MRR | 0.379 | **0.583** |
|
|
242
|
+
| questions declined for want of evidence | 13.0% | **4.3%** |
|
|
243
243
|
|
|
244
|
-
|
|
245
|
-
|
|
244
|
+
Half again the first-place accuracy. Nineteen questions still fail, which is
|
|
245
|
+
what makes the set usable for measuring the next change;
|
|
246
246
|
`tests/test_repo_queries.py` asserts that some question always does, and that
|
|
247
247
|
the written column beats the generated one.
|
|
248
248
|
|
|
@@ -283,6 +283,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
|
|
|
283
283
|
Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
|
|
284
284
|
bounded by how many ways of saying the thing the agent thought to write down.
|
|
285
285
|
|
|
286
|
+
### The question a ranking cannot answer
|
|
287
|
+
|
|
288
|
+
Both rulers above ask questions that *have* an answer, so both can only score
|
|
289
|
+
whether it was found. Neither can see the opposite failure. A ranking always
|
|
290
|
+
produces a least-bad unit and hands it back with a score and a rank that read
|
|
291
|
+
exactly like an answer — and it does that whether or not the repository
|
|
292
|
+
contains anything relevant at all.
|
|
293
|
+
|
|
294
|
+
So there is a third ruler: thirty questions about subjects neither repository
|
|
295
|
+
implements, where the only correct reply is nothing
|
|
296
|
+
([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
|
|
297
|
+
1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
|
|
298
|
+
languages:
|
|
299
|
+
|
|
300
|
+
| asked of a repository with no such code | answered with | on the evidence of |
|
|
301
|
+
|---|---|---|
|
|
302
|
+
| `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
|
|
303
|
+
| `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
|
|
304
|
+
| `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
|
|
305
|
+
|
|
306
|
+
Not a Chinese problem and not a ranking problem — a missing question. Nothing
|
|
307
|
+
in the pipeline asked *is any of this evidence*; it only asked which ranks
|
|
308
|
+
highest. Retrieval now asks both, and returns nothing when the answer to the
|
|
309
|
+
first is no:
|
|
310
|
+
|
|
311
|
+
| | before | now |
|
|
312
|
+
|---|---|---|
|
|
313
|
+
| unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
|
|
314
|
+
| the same, on the foreign repository | 0.000 | **0.800** |
|
|
315
|
+
| results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
|
|
316
|
+
| hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
|
|
317
|
+
| hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
|
|
318
|
+
|
|
319
|
+
The bar is the share of a question's **discriminating** words that occur in
|
|
320
|
+
the index — words the repository uses everywhere are dropped from both sides
|
|
321
|
+
of that fraction, which is the part that does the work. Half of `where are
|
|
322
|
+
CUDA kernels dispatched to the device` matches, and it looks like evidence
|
|
323
|
+
until you notice which half. Counting only words that distinguish silenced 18
|
|
324
|
+
of 30 unanswerable English questions that no plain coverage threshold reached
|
|
325
|
+
at all, at identical cost in real answers — 97 of 98 either way.
|
|
326
|
+
|
|
327
|
+
It is a ratio inside the query rather than a threshold on a score, because a
|
|
328
|
+
score threshold is tied to whatever scale the ranking currently produces —
|
|
329
|
+
this project has already had one of those stop meaning anything the moment
|
|
330
|
+
BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
|
|
331
|
+
behaviour exactly.
|
|
332
|
+
|
|
333
|
+
**What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
|
|
334
|
+
会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
|
|
335
|
+
repository contains are `utf` and `8`, both of which it uses everywhere. The
|
|
336
|
+
gate says too little of that question is distinctive, which is defensible, and
|
|
337
|
+
it was getting the right answer on a coincidence.
|
|
338
|
+
|
|
339
|
+
**What it does not fix:** an English question whose words genuinely occur here
|
|
340
|
+
in another sense. `how is the OAuth refresh token rotated` matches `refresh`
|
|
341
|
+
because this repository refreshes *indexes*, and no threshold separates those.
|
|
342
|
+
Six of fifteen English absent questions still get answered for that reason.
|
|
343
|
+
That is the case a real embedding model exists for, and it is measurable now
|
|
344
|
+
that the ruler exists.
|
|
345
|
+
|
|
346
|
+
An empty answer says which kind of empty it is, because each is recovered by a
|
|
347
|
+
different move:
|
|
348
|
+
|
|
349
|
+
```json
|
|
350
|
+
{"results": [],
|
|
351
|
+
"diagnosis": {"reason": "only_ubiquitous_terms_matched",
|
|
352
|
+
"matched_terms": [], "ubiquitous_terms": ["the", "to"],
|
|
353
|
+
"coverage": 0.0, "min_coverage": 0.4,
|
|
354
|
+
"hint": "The only words that matched are ones this repository uses throughout ..."}}
|
|
355
|
+
```
|
|
356
|
+
|
|
286
357
|
Descriptions live in `rag-your-code.descriptions.json` at the repository root
|
|
287
358
|
and are meant to be committed, so one person's pass benefits everyone who
|
|
288
359
|
clones. Each is keyed by unit id **and a digest of the unit's source**: when
|
|
@@ -332,23 +403,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
|
|
|
332
403
|
|
|
333
404
|
## Configuration
|
|
334
405
|
|
|
335
|
-
|
|
406
|
+
21 settings in `rag-your-code.toml` at the repository root:
|
|
336
407
|
|
|
337
408
|
```bash
|
|
338
409
|
rag-your-code config init # a commented file, all defaults
|
|
339
410
|
rag-your-code config list # effective values and their source
|
|
340
411
|
rag-your-code config set index.ignore '["vendor", "generated"]'
|
|
341
|
-
rag-your-code config set search.
|
|
412
|
+
rag-your-code config set search.min_coverage 0.25
|
|
342
413
|
```
|
|
343
414
|
|
|
344
415
|
| section | settings |
|
|
345
416
|
|---|---|
|
|
346
417
|
| `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
|
|
347
|
-
| `[embedding]` | `dimensions` |
|
|
348
|
-
| `[search]` | `vector_weight`, `limit`, `max_chars` |
|
|
418
|
+
| `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
|
|
419
|
+
| `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
|
|
349
420
|
| `[agent]` | `max_open_bytes`, `max_open_chars` |
|
|
350
421
|
| `[describe]` | `languages`, `batch`, `max_chars` |
|
|
351
422
|
|
|
423
|
+
This table is asserted against the settings table in `config.py`, in both
|
|
424
|
+
directions, by `tests/test_metadata.py` — it had already fallen nine settings
|
|
425
|
+
behind by 1.0.0, and a section listing three quarters of what exists is worse
|
|
426
|
+
than none, because it reads as complete.
|
|
427
|
+
|
|
352
428
|
Resolution is CLI flag > file > built-in default. There is no environment
|
|
353
429
|
layer: an index is an artifact of a repository, not of a shell.
|
|
354
430
|
|
|
@@ -358,9 +434,10 @@ silently dropped is indistinguishable from one that had no effect.
|
|
|
358
434
|
suffix it cannot read is walked, parsed to nothing, and reported as a clean
|
|
359
435
|
index of zero units.
|
|
360
436
|
|
|
361
|
-
The
|
|
362
|
-
*contains
|
|
363
|
-
|
|
437
|
+
The settings under `[index]` and `[embedding]` that decide what an index
|
|
438
|
+
*contains* — including which provider and model computed its vectors — have a
|
|
439
|
+
digest stored in the index, and changing one forces a full rebuild. The rest
|
|
440
|
+
take effect immediately and invalidate nothing.
|
|
364
441
|
|
|
365
442
|
## Agent protocol
|
|
366
443
|
|
|
@@ -23,6 +23,7 @@ src/ragyourcode/providers.py
|
|
|
23
23
|
src/ragyourcode/py.typed
|
|
24
24
|
src/ragyourcode/search.py
|
|
25
25
|
src/ragyourcode/workflow.py
|
|
26
|
+
tests/test_absent_queries.py
|
|
26
27
|
tests/test_agent_protocol.py
|
|
27
28
|
tests/test_agentic.py
|
|
28
29
|
tests/test_config.py
|
|
@@ -30,6 +31,7 @@ tests/test_descriptions.py
|
|
|
30
31
|
tests/test_doc_comments.py
|
|
31
32
|
tests/test_document.py
|
|
32
33
|
tests/test_e2e_cli.py
|
|
34
|
+
tests/test_evidence.py
|
|
33
35
|
tests/test_golden.py
|
|
34
36
|
tests/test_graph_incremental.py
|
|
35
37
|
tests/test_language_fixtures.py
|
|
@@ -4,7 +4,18 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
from .graph import CodeGraph, graph_search
|
|
6
6
|
from .models import CodeUnit, SearchResult
|
|
7
|
-
from .search import
|
|
7
|
+
from .search import (
|
|
8
|
+
DEFAULT_MIN_COVERAGE,
|
|
9
|
+
DEFAULT_VECTOR_RECALL,
|
|
10
|
+
DEFAULT_VECTOR_WEIGHT,
|
|
11
|
+
SearchIndex,
|
|
12
|
+
assess,
|
|
13
|
+
build_search_index,
|
|
14
|
+
context,
|
|
15
|
+
diagnose,
|
|
16
|
+
search,
|
|
17
|
+
within_budget,
|
|
18
|
+
)
|
|
8
19
|
|
|
9
20
|
|
|
10
21
|
def _result_ids(results: list[SearchResult]) -> set[str]:
|
|
@@ -81,6 +92,7 @@ def research(
|
|
|
81
92
|
search_index: SearchIndex | None = None,
|
|
82
93
|
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
83
94
|
vector_recall: int = DEFAULT_VECTOR_RECALL,
|
|
95
|
+
min_coverage: float = DEFAULT_MIN_COVERAGE,
|
|
84
96
|
max_chars: int = 12000,
|
|
85
97
|
) -> dict:
|
|
86
98
|
"""Run at most two deterministic retrieval steps and explain the stop.
|
|
@@ -97,16 +109,26 @@ def research(
|
|
|
97
109
|
"""
|
|
98
110
|
max_steps = min(2, max(1, max_steps))
|
|
99
111
|
steps: list[dict] = []
|
|
100
|
-
|
|
112
|
+
search_index = search_index or build_search_index(units)
|
|
113
|
+
initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall, min_coverage=min_coverage)
|
|
101
114
|
steps.append({"action": "search", "query": query, "results": _trace(initial)})
|
|
102
115
|
if not initial:
|
|
103
|
-
|
|
116
|
+
# Two observable steps are worth nothing if the first one stopping is
|
|
117
|
+
# reported as a bare emptiness. A second step cannot recover a query
|
|
118
|
+
# whose words are not in this index, so the reply says so instead of
|
|
119
|
+
# spending the budget to arrive at the same silence.
|
|
120
|
+
# `stop_reason` keeps its published set of values -- the specific reason
|
|
121
|
+
# goes in the new `diagnosis` field beside it. Widening an enumeration
|
|
122
|
+
# callers already branch on is a breaking change wearing the clothes of
|
|
123
|
+
# an improvement; adding a field next to it is not.
|
|
124
|
+
report = diagnose(assess(search_index, query, min_coverage), min_coverage)
|
|
125
|
+
return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": "", "diagnosis": report}
|
|
104
126
|
unopposed = dominance(initial) >= dominance_threshold and bool(initial[0].matched_terms)
|
|
105
127
|
if max_steps == 1 or unopposed:
|
|
106
128
|
kept = initial[:limit]
|
|
107
129
|
return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
|
|
108
130
|
|
|
109
|
-
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
|
|
131
|
+
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall, min_coverage=min_coverage)
|
|
110
132
|
steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
|
|
111
133
|
merged = {result.unit.id: result for result in initial}
|
|
112
134
|
for result in expanded:
|