rag-your-code 0.8.0__tar.gz → 1.0.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {rag_your_code-0.8.0/src/rag_your_code.egg-info → rag_your_code-1.0.0}/PKG-INFO +94 -17
  2. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/README.md +93 -16
  3. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/pyproject.toml +1 -1
  4. {rag_your_code-0.8.0 → rag_your_code-1.0.0/src/rag_your_code.egg-info}/PKG-INFO +94 -17
  5. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/SOURCES.txt +2 -0
  6. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/__init__.py +1 -1
  7. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/agentic.py +26 -4
  8. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/cli.py +27 -8
  9. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/config.py +16 -0
  10. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/graph.py +9 -3
  11. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/indexer.py +32 -0
  12. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/search.py +168 -1
  13. rag_your_code-1.0.0/tests/test_absent_queries.py +114 -0
  14. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_descriptions.py +65 -0
  15. rag_your_code-1.0.0/tests/test_evidence.py +201 -0
  16. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_metadata.py +24 -0
  17. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/LICENSE +0 -0
  18. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/setup.cfg +0 -0
  19. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  20. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  21. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  22. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  23. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/annotate.py +0 -0
  24. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/descriptions.py +0 -0
  25. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/document.py +0 -0
  26. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/embeddings.py +0 -0
  27. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/models.py +0 -0
  28. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/parser.py +0 -0
  29. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/providers.py +0 -0
  30. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/py.typed +0 -0
  31. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/src/ragyourcode/workflow.py +0 -0
  32. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_agent_protocol.py +0 -0
  33. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_agentic.py +0 -0
  34. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_config.py +0 -0
  35. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_doc_comments.py +0 -0
  36. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_document.py +0 -0
  37. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_e2e_cli.py +0 -0
  38. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_golden.py +0 -0
  39. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_graph_incremental.py +0 -0
  40. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_language_fixtures.py +0 -0
  41. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_large_repo.py +0 -0
  42. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_multilanguage.py +0 -0
  43. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_parser_edges.py +0 -0
  44. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_providers.py +0 -0
  45. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_ragyourcode.py +0 -0
  46. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_ranking.py +0 -0
  47. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_repo_queries.py +0 -0
  48. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_resilience.py +0 -0
  49. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_retrieval_correctness.py +0 -0
  50. {rag_your_code-0.8.0 → rag_your_code-1.0.0}/tests/test_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.8.0
3
+ Version: 1.0.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -226,9 +226,9 @@ and left Chinese retrieval unchanged.
226
226
 
227
227
  ### Measured on this repository
228
228
 
229
- This project describes its own implementation: every unit under `src/` carries
230
- an agent-written bilingual description, committed to the repo, and 68 of them
231
- have been promoted into the source as doc comments.
229
+ This project describes its own implementation: all 163 units under `src/` carry
230
+ an agent-written bilingual description, committed to the repo, and 155 of those
231
+ 163 declarations also carry the author's own documentation in the source.
232
232
 
233
233
  Seventy natural-language questions about this codebase, in English and
234
234
  Chinese, each listing every unit that genuinely answers it
@@ -236,13 +236,13 @@ Chinese, each listing every unit that genuinely answers it
236
236
 
237
237
  | | generated descriptions | agent-written |
238
238
  |---|---|---|
239
- | hit@1 | 0.271 | **0.500** |
240
- | hit@3 | 0.486 | **0.800** |
241
- | MRR | 0.367 | **0.631** |
242
- | answered with no shared word at all | 12.9% | **0%** |
239
+ | hit@1 | 0.314 | **0.500** |
240
+ | hit@3 | 0.457 | **0.729** |
241
+ | MRR | 0.379 | **0.583** |
242
+ | questions declined for want of evidence | 13.0% | **4.3%** |
243
243
 
244
- Roughly double the first-place accuracy. Fourteen questions still fail, which
245
- is what makes the set usable for measuring the next change;
244
+ Half again the first-place accuracy. Nineteen questions still fail, which is
245
+ what makes the set usable for measuring the next change;
246
246
  `tests/test_repo_queries.py` asserts that some question always does, and that
247
247
  the written column beats the generated one.
248
248
 
@@ -283,6 +283,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
283
283
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
284
284
  bounded by how many ways of saying the thing the agent thought to write down.
285
285
 
286
+ ### The question a ranking cannot answer
287
+
288
+ Both rulers above ask questions that *have* an answer, so both can only score
289
+ whether it was found. Neither can see the opposite failure. A ranking always
290
+ produces a least-bad unit and hands it back with a score and a rank that read
291
+ exactly like an answer — and it does that whether or not the repository
292
+ contains anything relevant at all.
293
+
294
+ So there is a third ruler: thirty questions about subjects neither repository
295
+ implements, where the only correct reply is nothing
296
+ ([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
297
+ 1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
298
+ languages:
299
+
300
+ | asked of a repository with no such code | answered with | on the evidence of |
301
+ |---|---|---|
302
+ | `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
303
+ | `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
304
+ | `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
305
+
306
+ Not a Chinese problem and not a ranking problem — a missing question. Nothing
307
+ in the pipeline asked *is any of this evidence*; it only asked which ranks
308
+ highest. Retrieval now asks both, and returns nothing when the answer to the
309
+ first is no:
310
+
311
+ | | before | now |
312
+ |---|---|---|
313
+ | unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
314
+ | the same, on the foreign repository | 0.000 | **0.800** |
315
+ | results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
316
+ | hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
317
+ | hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
318
+
319
+ The bar is the share of a question's **discriminating** words that occur in
320
+ the index — words the repository uses everywhere are dropped from both sides
321
+ of that fraction, which is the part that does the work. Half of `where are
322
+ CUDA kernels dispatched to the device` matches, and it looks like evidence
323
+ until you notice which half. Counting only words that distinguish silenced 18
324
+ of 30 unanswerable English questions that no plain coverage threshold reached
325
+ at all, at identical cost in real answers — 97 of 98 either way.
326
+
327
+ It is a ratio inside the query rather than a threshold on a score, because a
328
+ score threshold is tied to whatever scale the ranking currently produces —
329
+ this project has already had one of those stop meaning anything the moment
330
+ BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
331
+ behaviour exactly.
332
+
333
+ **What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
334
+ 会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
335
+ repository contains are `utf` and `8`, both of which it uses everywhere. The
336
+ gate says too little of that question is distinctive, which is defensible, and
337
+ it was getting the right answer on a coincidence.
338
+
339
+ **What it does not fix:** an English question whose words genuinely occur here
340
+ in another sense. `how is the OAuth refresh token rotated` matches `refresh`
341
+ because this repository refreshes *indexes*, and no threshold separates those.
342
+ Six of fifteen English absent questions still get answered for that reason.
343
+ That is the case a real embedding model exists for, and it is measurable now
344
+ that the ruler exists.
345
+
346
+ An empty answer says which kind of empty it is, because each is recovered by a
347
+ different move:
348
+
349
+ ```json
350
+ {"results": [],
351
+ "diagnosis": {"reason": "only_ubiquitous_terms_matched",
352
+ "matched_terms": [], "ubiquitous_terms": ["the", "to"],
353
+ "coverage": 0.0, "min_coverage": 0.4,
354
+ "hint": "The only words that matched are ones this repository uses throughout ..."}}
355
+ ```
356
+
286
357
  Descriptions live in `rag-your-code.descriptions.json` at the repository root
287
358
  and are meant to be committed, so one person's pass benefits everyone who
288
359
  clones. Each is keyed by unit id **and a digest of the unit's source**: when
@@ -332,23 +403,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
332
403
 
333
404
  ## Configuration
334
405
 
335
- Twelve settings in `rag-your-code.toml` at the repository root:
406
+ 21 settings in `rag-your-code.toml` at the repository root:
336
407
 
337
408
  ```bash
338
409
  rag-your-code config init # a commented file, all defaults
339
410
  rag-your-code config list # effective values and their source
340
411
  rag-your-code config set index.ignore '["vendor", "generated"]'
341
- rag-your-code config set search.vector_weight 0.25
412
+ rag-your-code config set search.min_coverage 0.25
342
413
  ```
343
414
 
344
415
  | section | settings |
345
416
  |---|---|
346
417
  | `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
347
- | `[embedding]` | `dimensions` |
348
- | `[search]` | `vector_weight`, `limit`, `max_chars` |
418
+ | `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
419
+ | `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
349
420
  | `[agent]` | `max_open_bytes`, `max_open_chars` |
350
421
  | `[describe]` | `languages`, `batch`, `max_chars` |
351
422
 
423
+ This table is asserted against the settings table in `config.py`, in both
424
+ directions, by `tests/test_metadata.py` — it had already fallen nine settings
425
+ behind by 1.0.0, and a section listing three quarters of what exists is worse
426
+ than none, because it reads as complete.
427
+
352
428
  Resolution is CLI flag > file > built-in default. There is no environment
353
429
  layer: an index is an artifact of a repository, not of a shell.
354
430
 
@@ -358,9 +434,10 @@ silently dropped is indistinguishable from one that had no effect.
358
434
  suffix it cannot read is walked, parsed to nothing, and reported as a clean
359
435
  index of zero units.
360
436
 
361
- The four settings under `[index]` and `[embedding]` decide what an index
362
- *contains*, so a digest of them is stored in the index and a change forces a
363
- full rebuild. The rest take effect immediately and invalidate nothing.
437
+ The settings under `[index]` and `[embedding]` that decide what an index
438
+ *contains* — including which provider and model computed its vectors — have a
439
+ digest stored in the index, and changing one forces a full rebuild. The rest
440
+ take effect immediately and invalidate nothing.
364
441
 
365
442
  ## Agent protocol
366
443
 
@@ -199,9 +199,9 @@ and left Chinese retrieval unchanged.
199
199
 
200
200
  ### Measured on this repository
201
201
 
202
- This project describes its own implementation: every unit under `src/` carries
203
- an agent-written bilingual description, committed to the repo, and 68 of them
204
- have been promoted into the source as doc comments.
202
+ This project describes its own implementation: all 163 units under `src/` carry
203
+ an agent-written bilingual description, committed to the repo, and 155 of those
204
+ 163 declarations also carry the author's own documentation in the source.
205
205
 
206
206
  Seventy natural-language questions about this codebase, in English and
207
207
  Chinese, each listing every unit that genuinely answers it
@@ -209,13 +209,13 @@ Chinese, each listing every unit that genuinely answers it
209
209
 
210
210
  | | generated descriptions | agent-written |
211
211
  |---|---|---|
212
- | hit@1 | 0.271 | **0.500** |
213
- | hit@3 | 0.486 | **0.800** |
214
- | MRR | 0.367 | **0.631** |
215
- | answered with no shared word at all | 12.9% | **0%** |
212
+ | hit@1 | 0.314 | **0.500** |
213
+ | hit@3 | 0.457 | **0.729** |
214
+ | MRR | 0.379 | **0.583** |
215
+ | questions declined for want of evidence | 13.0% | **4.3%** |
216
216
 
217
- Roughly double the first-place accuracy. Fourteen questions still fail, which
218
- is what makes the set usable for measuring the next change;
217
+ Half again the first-place accuracy. Nineteen questions still fail, which is
218
+ what makes the set usable for measuring the next change;
219
219
  `tests/test_repo_queries.py` asserts that some question always does, and that
220
220
  the written column beats the generated one.
221
221
 
@@ -256,6 +256,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
256
256
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
257
257
  bounded by how many ways of saying the thing the agent thought to write down.
258
258
 
259
+ ### The question a ranking cannot answer
260
+
261
+ Both rulers above ask questions that *have* an answer, so both can only score
262
+ whether it was found. Neither can see the opposite failure. A ranking always
263
+ produces a least-bad unit and hands it back with a score and a rank that read
264
+ exactly like an answer — and it does that whether or not the repository
265
+ contains anything relevant at all.
266
+
267
+ So there is a third ruler: thirty questions about subjects neither repository
268
+ implements, where the only correct reply is nothing
269
+ ([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
270
+ 1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
271
+ languages:
272
+
273
+ | asked of a repository with no such code | answered with | on the evidence of |
274
+ |---|---|---|
275
+ | `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
276
+ | `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
277
+ | `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
278
+
279
+ Not a Chinese problem and not a ranking problem — a missing question. Nothing
280
+ in the pipeline asked *is any of this evidence*; it only asked which ranks
281
+ highest. Retrieval now asks both, and returns nothing when the answer to the
282
+ first is no:
283
+
284
+ | | before | now |
285
+ |---|---|---|
286
+ | unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
287
+ | the same, on the foreign repository | 0.000 | **0.800** |
288
+ | results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
289
+ | hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
290
+ | hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
291
+
292
+ The bar is the share of a question's **discriminating** words that occur in
293
+ the index — words the repository uses everywhere are dropped from both sides
294
+ of that fraction, which is the part that does the work. Half of `where are
295
+ CUDA kernels dispatched to the device` matches, and it looks like evidence
296
+ until you notice which half. Counting only words that distinguish silenced 18
297
+ of 30 unanswerable English questions that no plain coverage threshold reached
298
+ at all, at identical cost in real answers — 97 of 98 either way.
299
+
300
+ It is a ratio inside the query rather than a threshold on a score, because a
301
+ score threshold is tied to whatever scale the ranking currently produces —
302
+ this project has already had one of those stop meaning anything the moment
303
+ BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
304
+ behaviour exactly.
305
+
306
+ **What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
307
+ 会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
308
+ repository contains are `utf` and `8`, both of which it uses everywhere. The
309
+ gate says too little of that question is distinctive, which is defensible, and
310
+ it was getting the right answer on a coincidence.
311
+
312
+ **What it does not fix:** an English question whose words genuinely occur here
313
+ in another sense. `how is the OAuth refresh token rotated` matches `refresh`
314
+ because this repository refreshes *indexes*, and no threshold separates those.
315
+ Six of fifteen English absent questions still get answered for that reason.
316
+ That is the case a real embedding model exists for, and it is measurable now
317
+ that the ruler exists.
318
+
319
+ An empty answer says which kind of empty it is, because each is recovered by a
320
+ different move:
321
+
322
+ ```json
323
+ {"results": [],
324
+ "diagnosis": {"reason": "only_ubiquitous_terms_matched",
325
+ "matched_terms": [], "ubiquitous_terms": ["the", "to"],
326
+ "coverage": 0.0, "min_coverage": 0.4,
327
+ "hint": "The only words that matched are ones this repository uses throughout ..."}}
328
+ ```
329
+
259
330
  Descriptions live in `rag-your-code.descriptions.json` at the repository root
260
331
  and are meant to be committed, so one person's pass benefits everyone who
261
332
  clones. Each is keyed by unit id **and a digest of the unit's source**: when
@@ -305,23 +376,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
305
376
 
306
377
  ## Configuration
307
378
 
308
- Twelve settings in `rag-your-code.toml` at the repository root:
379
+ 21 settings in `rag-your-code.toml` at the repository root:
309
380
 
310
381
  ```bash
311
382
  rag-your-code config init # a commented file, all defaults
312
383
  rag-your-code config list # effective values and their source
313
384
  rag-your-code config set index.ignore '["vendor", "generated"]'
314
- rag-your-code config set search.vector_weight 0.25
385
+ rag-your-code config set search.min_coverage 0.25
315
386
  ```
316
387
 
317
388
  | section | settings |
318
389
  |---|---|
319
390
  | `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
320
- | `[embedding]` | `dimensions` |
321
- | `[search]` | `vector_weight`, `limit`, `max_chars` |
391
+ | `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
392
+ | `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
322
393
  | `[agent]` | `max_open_bytes`, `max_open_chars` |
323
394
  | `[describe]` | `languages`, `batch`, `max_chars` |
324
395
 
396
+ This table is asserted against the settings table in `config.py`, in both
397
+ directions, by `tests/test_metadata.py` — it had already fallen nine settings
398
+ behind by 1.0.0, and a section listing three quarters of what exists is worse
399
+ than none, because it reads as complete.
400
+
325
401
  Resolution is CLI flag > file > built-in default. There is no environment
326
402
  layer: an index is an artifact of a repository, not of a shell.
327
403
 
@@ -331,9 +407,10 @@ silently dropped is indistinguishable from one that had no effect.
331
407
  suffix it cannot read is walked, parsed to nothing, and reported as a clean
332
408
  index of zero units.
333
409
 
334
- The four settings under `[index]` and `[embedding]` decide what an index
335
- *contains*, so a digest of them is stored in the index and a change forces a
336
- full rebuild. The rest take effect immediately and invalidate nothing.
410
+ The settings under `[index]` and `[embedding]` that decide what an index
411
+ *contains* — including which provider and model computed its vectors — have a
412
+ digest stored in the index, and changing one forces a full rebuild. The rest
413
+ take effect immediately and invalidate nothing.
337
414
 
338
415
  ## Agent protocol
339
416
 
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "0.8.0"
9
+ version = "1.0.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.8.0
3
+ Version: 1.0.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -226,9 +226,9 @@ and left Chinese retrieval unchanged.
226
226
 
227
227
  ### Measured on this repository
228
228
 
229
- This project describes its own implementation: every unit under `src/` carries
230
- an agent-written bilingual description, committed to the repo, and 68 of them
231
- have been promoted into the source as doc comments.
229
+ This project describes its own implementation: all 163 units under `src/` carry
230
+ an agent-written bilingual description, committed to the repo, and 155 of those
231
+ 163 declarations also carry the author's own documentation in the source.
232
232
 
233
233
  Seventy natural-language questions about this codebase, in English and
234
234
  Chinese, each listing every unit that genuinely answers it
@@ -236,13 +236,13 @@ Chinese, each listing every unit that genuinely answers it
236
236
 
237
237
  | | generated descriptions | agent-written |
238
238
  |---|---|---|
239
- | hit@1 | 0.271 | **0.500** |
240
- | hit@3 | 0.486 | **0.800** |
241
- | MRR | 0.367 | **0.631** |
242
- | answered with no shared word at all | 12.9% | **0%** |
239
+ | hit@1 | 0.314 | **0.500** |
240
+ | hit@3 | 0.457 | **0.729** |
241
+ | MRR | 0.379 | **0.583** |
242
+ | questions declined for want of evidence | 13.0% | **4.3%** |
243
243
 
244
- Roughly double the first-place accuracy. Fourteen questions still fail, which
245
- is what makes the set usable for measuring the next change;
244
+ Half again the first-place accuracy. Nineteen questions still fail, which is
245
+ what makes the set usable for measuring the next change;
246
246
  `tests/test_repo_queries.py` asserts that some question always does, and that
247
247
  the written column beats the generated one.
248
248
 
@@ -283,6 +283,77 @@ the code it tests, because it repeats that code's vocabulary and adds its own.
283
283
  Matching stays lexical. It is LLM-authored keyword expansion, and its reach is
284
284
  bounded by how many ways of saying the thing the agent thought to write down.
285
285
 
286
+ ### The question a ranking cannot answer
287
+
288
+ Both rulers above ask questions that *have* an answer, so both can only score
289
+ whether it was found. Neither can see the opposite failure. A ranking always
290
+ produces a least-bad unit and hands it back with a score and a rank that read
291
+ exactly like an answer — and it does that whether or not the repository
292
+ contains anything relevant at all.
293
+
294
+ So there is a third ruler: thirty questions about subjects neither repository
295
+ implements, where the only correct reply is nothing
296
+ ([`benchmarks/absent_queries.json`](benchmarks/absent_queries.json)). Before
297
+ 1.0.0 it scored **zero**. All thirty answered, on both repositories, in both
298
+ languages:
299
+
300
+ | asked of a repository with no such code | answered with | on the evidence of |
301
+ |---|---|---|
302
+ | `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
303
+ | `准入控制为什么会拒绝没有资源限额的容器组` | the UTF-8 console setup | `拒绝` `控制` `没有` |
304
+ | `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
305
+
306
+ Not a Chinese problem and not a ranking problem — a missing question. Nothing
307
+ in the pipeline asked *is any of this evidence*; it only asked which ranks
308
+ highest. Retrieval now asks both, and returns nothing when the answer to the
309
+ first is no:
310
+
311
+ | | before | now |
312
+ |---|---|---|
313
+ | unanswerable questions correctly met with silence, this repository | 0.000 | **0.733** |
314
+ | the same, on the foreign repository | 0.000 | **0.800** |
315
+ | results resting on no lexical evidence at all, all three rulers | 0.029 – 0.129 | **0.000** |
316
+ | hit@1 / hit@3 / MRR on the foreign ruler | 0.257 / 0.400 / 0.314 | **unchanged** |
317
+ | hit@1 / hit@3 / MRR on this repository | 0.471 / 0.686 / 0.557 | **unchanged** |
318
+
319
+ The bar is the share of a question's **discriminating** words that occur in
320
+ the index — words the repository uses everywhere are dropped from both sides
321
+ of that fraction, which is the part that does the work. Half of `where are
322
+ CUDA kernels dispatched to the device` matches, and it looks like evidence
323
+ until you notice which half. Counting only words that distinguish silenced 18
324
+ of 30 unanswerable English questions that no plain coverage threshold reached
325
+ at all, at identical cost in real answers — 97 of 98 either way.
326
+
327
+ It is a ratio inside the query rather than a threshold on a score, because a
328
+ score threshold is tied to whatever scale the ranking currently produces —
329
+ this project has already had one of those stop meaning anything the moment
330
+ BM25F changed the scale. `search.min_coverage` sets it; `0` restores the old
331
+ behaviour exactly.
332
+
333
+ **What it costs:** one question of the 158 measured. `控制台编码不是 UTF-8
334
+ 会怎么样` was reaching `_use_utf8_streams`, and the only words in it this
335
+ repository contains are `utf` and `8`, both of which it uses everywhere. The
336
+ gate says too little of that question is distinctive, which is defensible, and
337
+ it was getting the right answer on a coincidence.
338
+
339
+ **What it does not fix:** an English question whose words genuinely occur here
340
+ in another sense. `how is the OAuth refresh token rotated` matches `refresh`
341
+ because this repository refreshes *indexes*, and no threshold separates those.
342
+ Six of fifteen English absent questions still get answered for that reason.
343
+ That is the case a real embedding model exists for, and it is measurable now
344
+ that the ruler exists.
345
+
346
+ An empty answer says which kind of empty it is, because each is recovered by a
347
+ different move:
348
+
349
+ ```json
350
+ {"results": [],
351
+ "diagnosis": {"reason": "only_ubiquitous_terms_matched",
352
+ "matched_terms": [], "ubiquitous_terms": ["the", "to"],
353
+ "coverage": 0.0, "min_coverage": 0.4,
354
+ "hint": "The only words that matched are ones this repository uses throughout ..."}}
355
+ ```
356
+
286
357
  Descriptions live in `rag-your-code.descriptions.json` at the repository root
287
358
  and are meant to be committed, so one person's pass benefits everyone who
288
359
  clones. Each is keyed by unit id **and a digest of the unit's source**: when
@@ -332,23 +403,28 @@ per-release counts are in [CHANGELOG.md](CHANGELOG.md).
332
403
 
333
404
  ## Configuration
334
405
 
335
- Twelve settings in `rag-your-code.toml` at the repository root:
406
+ 21 settings in `rag-your-code.toml` at the repository root:
336
407
 
337
408
  ```bash
338
409
  rag-your-code config init # a commented file, all defaults
339
410
  rag-your-code config list # effective values and their source
340
411
  rag-your-code config set index.ignore '["vendor", "generated"]'
341
- rag-your-code config set search.vector_weight 0.25
412
+ rag-your-code config set search.min_coverage 0.25
342
413
  ```
343
414
 
344
415
  | section | settings |
345
416
  |---|---|
346
417
  | `[index]` | `ignore`, `suffixes`, `max_file_bytes` |
347
- | `[embedding]` | `dimensions` |
348
- | `[search]` | `vector_weight`, `limit`, `max_chars` |
418
+ | `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
419
+ | `[search]` | `min_coverage`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
349
420
  | `[agent]` | `max_open_bytes`, `max_open_chars` |
350
421
  | `[describe]` | `languages`, `batch`, `max_chars` |
351
422
 
423
+ This table is asserted against the settings table in `config.py`, in both
424
+ directions, by `tests/test_metadata.py` — it had already fallen nine settings
425
+ behind by 1.0.0, and a section listing three quarters of what exists is worse
426
+ than none, because it reads as complete.
427
+
352
428
  Resolution is CLI flag > file > built-in default. There is no environment
353
429
  layer: an index is an artifact of a repository, not of a shell.
354
430
 
@@ -358,9 +434,10 @@ silently dropped is indistinguishable from one that had no effect.
358
434
  suffix it cannot read is walked, parsed to nothing, and reported as a clean
359
435
  index of zero units.
360
436
 
361
- The four settings under `[index]` and `[embedding]` decide what an index
362
- *contains*, so a digest of them is stored in the index and a change forces a
363
- full rebuild. The rest take effect immediately and invalidate nothing.
437
+ The settings under `[index]` and `[embedding]` that decide what an index
438
+ *contains* — including which provider and model computed its vectors — have a
439
+ digest stored in the index, and changing one forces a full rebuild. The rest
440
+ take effect immediately and invalidate nothing.
364
441
 
365
442
  ## Agent protocol
366
443
 
@@ -23,6 +23,7 @@ src/ragyourcode/providers.py
23
23
  src/ragyourcode/py.typed
24
24
  src/ragyourcode/search.py
25
25
  src/ragyourcode/workflow.py
26
+ tests/test_absent_queries.py
26
27
  tests/test_agent_protocol.py
27
28
  tests/test_agentic.py
28
29
  tests/test_config.py
@@ -30,6 +31,7 @@ tests/test_descriptions.py
30
31
  tests/test_doc_comments.py
31
32
  tests/test_document.py
32
33
  tests/test_e2e_cli.py
34
+ tests/test_evidence.py
33
35
  tests/test_golden.py
34
36
  tests/test_graph_incremental.py
35
37
  tests/test_language_fixtures.py
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "0.8.0"
6
+ __version__ = "1.0.0"
@@ -4,7 +4,18 @@ from __future__ import annotations
4
4
 
5
5
  from .graph import CodeGraph, graph_search
6
6
  from .models import CodeUnit, SearchResult
7
- from .search import DEFAULT_VECTOR_RECALL, DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
7
+ from .search import (
8
+ DEFAULT_MIN_COVERAGE,
9
+ DEFAULT_VECTOR_RECALL,
10
+ DEFAULT_VECTOR_WEIGHT,
11
+ SearchIndex,
12
+ assess,
13
+ build_search_index,
14
+ context,
15
+ diagnose,
16
+ search,
17
+ within_budget,
18
+ )
8
19
 
9
20
 
10
21
  def _result_ids(results: list[SearchResult]) -> set[str]:
@@ -81,6 +92,7 @@ def research(
81
92
  search_index: SearchIndex | None = None,
82
93
  vector_weight: float = DEFAULT_VECTOR_WEIGHT,
83
94
  vector_recall: int = DEFAULT_VECTOR_RECALL,
95
+ min_coverage: float = DEFAULT_MIN_COVERAGE,
84
96
  max_chars: int = 12000,
85
97
  ) -> dict:
86
98
  """Run at most two deterministic retrieval steps and explain the stop.
@@ -97,16 +109,26 @@ def research(
97
109
  """
98
110
  max_steps = min(2, max(1, max_steps))
99
111
  steps: list[dict] = []
100
- initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
112
+ search_index = search_index or build_search_index(units)
113
+ initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall, min_coverage=min_coverage)
101
114
  steps.append({"action": "search", "query": query, "results": _trace(initial)})
102
115
  if not initial:
103
- return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
116
+ # Two observable steps are worth nothing if the first one stopping is
117
+ # reported as a bare emptiness. A second step cannot recover a query
118
+ # whose words are not in this index, so the reply says so instead of
119
+ # spending the budget to arrive at the same silence.
120
+ # `stop_reason` keeps its published set of values -- the specific reason
121
+ # goes in the new `diagnosis` field beside it. Widening an enumeration
122
+ # callers already branch on is a breaking change wearing the clothes of
123
+ # an improvement; adding a field next to it is not.
124
+ report = diagnose(assess(search_index, query, min_coverage), min_coverage)
125
+ return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": "", "diagnosis": report}
104
126
  unopposed = dominance(initial) >= dominance_threshold and bool(initial[0].matched_terms)
105
127
  if max_steps == 1 or unopposed:
106
128
  kept = initial[:limit]
107
129
  return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
108
130
 
109
- expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
131
+ expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall, min_coverage=min_coverage)
110
132
  steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
111
133
  merged = {result.unit.id: result for result in initial}
112
134
  for result in expanded: