rag-your-code 1.2.1__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {rag_your_code-1.2.1/src/rag_your_code.egg-info → rag_your_code-1.4.0}/PKG-INFO +148 -63
  2. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/README.md +147 -62
  3. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/pyproject.toml +1 -1
  4. {rag_your_code-1.2.1 → rag_your_code-1.4.0/src/rag_your_code.egg-info}/PKG-INFO +148 -63
  5. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/__init__.py +1 -1
  6. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/config.py +14 -6
  7. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/search.py +1 -1
  8. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_absent_queries.py +42 -9
  9. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_evidence.py +4 -3
  10. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_metadata.py +32 -0
  11. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/LICENSE +0 -0
  12. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/setup.cfg +0 -0
  13. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
  14. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  15. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  16. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  17. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  18. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/agentic.py +0 -0
  19. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/annotate.py +0 -0
  20. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/cli.py +0 -0
  21. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/descriptions.py +0 -0
  22. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/document.py +0 -0
  23. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/embeddings.py +0 -0
  24. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/graph.py +0 -0
  25. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/indexer.py +0 -0
  26. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/models.py +0 -0
  27. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/parser.py +0 -0
  28. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/providers.py +0 -0
  29. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/py.typed +0 -0
  30. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/workflow.py +0 -0
  31. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_agent_protocol.py +0 -0
  32. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_agentic.py +0 -0
  33. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_config.py +0 -0
  34. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_descriptions.py +0 -0
  35. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_doc_comments.py +0 -0
  36. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_document.py +0 -0
  37. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_e2e_cli.py +0 -0
  38. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_golden.py +0 -0
  39. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_graph_incremental.py +0 -0
  40. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_language_fixtures.py +0 -0
  41. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_large_repo.py +0 -0
  42. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_local_model.py +0 -0
  43. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_multilanguage.py +0 -0
  44. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_parser_edges.py +0 -0
  45. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_providers.py +0 -0
  46. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_ragyourcode.py +0 -0
  47. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_ranking.py +0 -0
  48. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_repo_queries.py +0 -0
  49. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_resilience.py +0 -0
  50. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_retrieval_correctness.py +0 -0
  51. {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.2.1
3
+ Version: 1.4.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -293,42 +293,87 @@ repository had grown by ninety units.
293
293
 
294
294
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
295
295
  |---|---|---|---|---|---|
296
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.400 | 0.300 |
296
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
297
297
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
298
298
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
299
299
 
300
+ The corpora, without which none of the above is reproducible — **A** 1,572
301
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
302
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
303
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
304
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
305
+ that cost three things: two questions pointed at a declaration the subject had
306
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
307
+ subject had grown, and the model comparison below was taken against two
308
+ different states of it. All three are now a `git clone` away from being
309
+ checked, and CI runs this ruler as an ordinary job.
310
+
300
311
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
301
312
 
302
- | | this repo | foreign repo |
313
+ | | this repo | Flask |
303
314
  |---|---|---|
304
- | correctly met with silence | **0.967** | **0.933** |
305
- | English only | **0.933** | **0.867** |
315
+ | correctly met with silence | **0.967** | **0.833** |
316
+ | English only | **0.933** | 0.667 |
306
317
  | Chinese only | **1.000** | **1.000** |
307
318
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
308
319
 
320
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
321
+ cause is a limit of the design rather than a defect. A word counts as evidence
322
+ unless it occurs in more than 5% of units — a stopword list derived from the
323
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
324
+ `does` and `are` are everywhere, because 304 units carry written English prose.
325
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
326
+ them and start counting as evidence. Five English questions about subjects
327
+ Flask does not implement get through on exactly that.
328
+
309
329
  **What each bar costs and buys** — one corpus, gate varied alone:
310
330
 
311
331
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
312
332
  |---|---|---|---|---|
313
- | neither (pre-1.0.0) | 0.229/0.400/0.300 | 0.314/0.486/0.391 | 0.471/0.686/0.552 | 0.000 / 0.000 |
314
- | coverage only (1.0.0) | 0.229/0.400/0.300 | 0.314/0.471/0.383 | 0.471/0.671/0.548 | 0.700 / 0.767 |
315
- | **both (1.1.0)** | **0.229/0.400/0.300** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
316
-
317
- Rulers A and B are **identical to three decimals**. The entire cost is four
318
- questions of seventy on the warmest ruler. On these four rulers concentration
319
- subsumes coverage — stated plainly because it is true; coverage is kept because
320
- it answers a different question and names a different diagnosis.
321
-
322
- **Latency** — warm corpus, 557 units, 420 samples after warm-up:
323
-
324
- | | |
325
- |---|---|
326
- | query, median | **0.83 ms** |
327
- | query, p95 | 1.68 ms |
328
- | refusing an unanswerable query | **0.03 ms** |
329
-
330
- Refusal is cheaper than answering by a factor of forty: an unanswerable query
331
- touches only the posting lists of its own distinctive words, never the corpus.
333
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
334
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
335
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
336
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
337
+
338
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
339
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
340
+
341
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
342
+ this project did not choose, it does not.** Both bars together silence 0.833 of
343
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
344
+ for coverage alone. One question — and the first time in four releases that
345
+ keeping both has been worth a measurable amount rather than worth a different
346
+ diagnosis.
347
+
348
+ Raising the concentration bar buys the remaining silence, and is refused,
349
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
350
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
351
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
352
+ existed and survived meeting it**, which is the only kind of evidence a default
353
+ can have.
354
+
355
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
356
+ invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
357
+ each):
358
+
359
+ | | | across the five |
360
+ |---|---|---|
361
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
362
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
363
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
364
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
365
+
366
+ Two significant figures and a spread, because that is the precision the
367
+ measurement has. Across fifteen invocations over two releases on the same idle
368
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
369
+ to 7.34 ms — a band wider than any change the code has ever made to this
370
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
371
+ figures from a script that was never committed; both values sit inside that
372
+ band, which is the point: they were unfalsifiable rather than wrong.
373
+
374
+ Refusal is cheap for a structural reason, not a tuned one: an unanswerable
375
+ query touches only the posting lists of its own distinctive words, and never
376
+ reaches ranking at all.
332
377
 
333
378
  **Scale**, synthetic 10,000-unit repository (500 files):
334
379
 
@@ -350,8 +395,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
350
395
  | with a usable signature | **91 / 91** |
351
396
  | units invented that do not exist | **0** |
352
397
 
353
- Directional local measurements, not service levels. Reproduce with
354
- `python benchmarks/repo_queries.py` and `python benchmarks/large_repo.py`.
398
+ Directional local measurements, not service levels — but every one of them is
399
+ now a command rather than a memory, which two of them were not before. Each
400
+ prints the corpus fingerprint beside its score; quote both or neither.
401
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
402
+ each is for, and the corpus one of them grades is now carried here too.
355
403
 
356
404
  ## 7 · `rag-your-code search` vs a Grep loop
357
405
 
@@ -361,45 +409,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
361
409
  ruler, scored at **file** granularity so Grep is not penalised for lacking
362
410
  declaration spans.
363
411
 
364
- **On a repository nobody has described, Grep wins.** That is the measured
365
- result and it is not softened here.
412
+ **Which side wins on an undescribed repository depends on the repository.**
413
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
414
+ undescribed repository ever measured was a hook-heavy tool whose questions were
415
+ answerable by matching identifiers. Swapping the subject for a public web
416
+ framework reversed it. The honest claim is narrower than either table alone:
366
417
 
367
- | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
418
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
368
419
  |---|---|---|
369
- | right file first | **34.3%** | 31.4% |
370
- | right file in top 3 | **60.0%** | 48.6% |
371
- | lines matched across the repo, all questions | 33,213 | — |
372
- | characters returned, all questions | — | **163,294** |
373
- | questions it answers | 35 | 28 |
420
+ | right file first | 22.9% | **37.1%** |
421
+ | right file in top 3 | 45.7% | **57.1%** |
422
+ | lines it hands back, all questions | 17,641 | — |
423
+ | characters returned, all questions | 1,415,656 | **249,720** |
424
+ | questions it answers | 30 | 30 |
374
425
 
375
426
  **Once the vocabulary exists, it is not close.**
376
427
 
377
- | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
428
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
378
429
  |---|---|---|
379
- | right file first | 25.7% | **58.6%** |
380
- | right file in top 3 | 64.3% | **75.7%** |
381
- | lines matched across the repo, all questions | 40,150 | — |
382
- | characters returned, all questions | — | **277,327** |
383
- | questions it answers | 70 | 60 |
430
+ | right file first | 22.9% | **58.6%** |
431
+ | right file in top 3 | 54.3% | **75.7%** |
432
+ | lines it hands back, all questions | 11,833 | — |
433
+ | characters returned, all questions | 1,122,902 | **626,022** |
434
+ | questions it answers | **61** | 60 |
384
435
 
385
436
  Those two tables are the whole argument of section 3.3, measured against a real
386
437
  baseline instead of asserted. A cold index retrieves against a sentence the
387
- parser generated from identifiers the author already chose — so it is competing
388
- with Grep using Grep's own information, and losing, because Grep does not have
389
- to guess which of the matching files is the definition. Descriptions put words
390
- in the index that the source never contained, and first-place accuracy goes
391
- from below Grep's to **more than double** it.
392
-
393
- Three qualifications, because the table would otherwise flatter both sides:
438
+ parser generated from identifiers the author already chose, plus whatever
439
+ docstrings the author wrote — so how it fares against Grep is decided by how
440
+ much prose the repository already contains. Flask has a written docstring on
441
+ most public methods, and the cold index beats Grep there without a single
442
+ description being added. On the previous subject, a tool with terse comments
443
+ and long identifiers, the same cold index lost to Grep by the same margin.
444
+
445
+ What does not depend on the subject is what descriptions buy: on this
446
+ repository first-place accuracy goes to **more than double** Grep's, and the
447
+ payload comes back ranked, spanned, and roughly half the size.
448
+
449
+ **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
450
+ changed in 1.3.0. Until then this section — the strongest claim the project
451
+ makes — was published from a script that had never been committed, so nothing
452
+ here could be checked and the word "Grep loop" had no precise meaning. The
453
+ committed version defines it: take the query's words, drop the ones the corpus
454
+ itself shows are everywhere, run one substring search per remaining word over
455
+ exactly the files the index was built from, rank each file by how many distinct
456
+ words hit it, break ties on path. Reconstructing it reproduced this side's
457
+ figures exactly and moved Grep's, which is the expected shape — the ranked arm
458
+ was always a call into shipped code, and the baseline never was.
459
+
460
+ Four qualifications, because the table would otherwise flatter both sides:
394
461
 
395
462
  - **Scored at file granularity**, which understates this side. A Grep hit is a
396
463
  file; a hit here is a declaration with an exact span, a score, and the words
397
464
  it matched on. The agent that reads the result opens 40 lines, not a file.
398
- - **Grep answers everything.** It never declines, which is why it hands back
399
- 33,213 matching lines for 35 questions — about 950 lines per question, no
400
- ranking, no spans, no indication which match is the definition. This returns
401
- roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
402
- questions come back empty here instead, with a reason.
465
+ - **Dropping the corpus-common words is generous to Grep**, and it is what
466
+ makes the baseline a fair one rather than a straw man: an agent that greps
467
+ `the` gets every file back in no order. It is also why Grep declines nine of
468
+ the seventy questions here — those had no word left that this corpus does not
469
+ use everywhere.
470
+ - **Payload is counted in characters on both sides.** On this repository Grep
471
+ hands back 18,400 characters per question it answers, unranked and without
472
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
473
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
474
+ 8,300 — because a framework repeats its own vocabulary across many files and
475
+ Grep has no way to rank what it finds. Both sides decline the same five of
476
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
477
+ not a substring of English source and it is not a token in an index built
478
+ from English source, so on a repository written in one language the cold
479
+ cross-language case is not this tool's failure but the corpus's.
403
480
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
404
481
  exact, instant and complete, and nothing here replaces it.
405
482
 
@@ -464,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
464
541
  install gets, the import happens inside the constructor, and a test asserts the
465
542
  default provider imports none of it.
466
543
 
467
- **Measured, on the same four rulers.** This is the first release with a real
468
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
469
- benefit was unmeasured.
544
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
545
+ published this comparison and read it as a win. Its largest gain was on the
546
+ foreign ruler, whose two arms turned out to have been taken against two
547
+ different states of a repository being edited while the script ran. Repeated
548
+ against a pinned corpus:
470
549
 
471
- | ruler | signed hash (default) | MiniLM, local |
472
- |---|---|---|
473
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
474
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
475
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
476
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
550
+ | ruler | corpus | signed hash (default) | MiniLM, local |
551
+ |---|---|---|---|
552
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
553
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
554
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
555
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
477
556
 
478
- Better on every positive ruler, with refusal unchanged. The pairs the hash
479
- scores exactly zero:
557
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
558
+ has to install, because it does one thing the hash cannot do at all and these
559
+ rulers cannot see: reach a unit that shares no word with the question. The
560
+ pairs the hash scores exactly zero:
480
561
 
481
562
  | pair | signed hash | MiniLM |
482
563
  |---|---|---|
@@ -494,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
494
575
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
495
576
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
496
577
 
578
+ If you install it expecting the hit rates above to move, they will not. Install
579
+ it for the cross-language and paraphrase cases in the table above, which is
580
+ where the difference between the two columns actually lives.
581
+
497
582
  **A hosted endpoint** is the third option, and the only one that sends your
498
583
  source anywhere:
499
584
 
@@ -264,42 +264,87 @@ repository had grown by ninety units.
264
264
 
265
265
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
266
266
  |---|---|---|---|---|---|
267
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.400 | 0.300 |
267
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
268
268
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
269
269
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
270
270
 
271
+ The corpora, without which none of the above is reproducible — **A** 1,572
272
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
273
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
274
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
275
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
276
+ that cost three things: two questions pointed at a declaration the subject had
277
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
278
+ subject had grown, and the model comparison below was taken against two
279
+ different states of it. All three are now a `git clone` away from being
280
+ checked, and CI runs this ruler as an ordinary job.
281
+
271
282
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
272
283
 
273
- | | this repo | foreign repo |
284
+ | | this repo | Flask |
274
285
  |---|---|---|
275
- | correctly met with silence | **0.967** | **0.933** |
276
- | English only | **0.933** | **0.867** |
286
+ | correctly met with silence | **0.967** | **0.833** |
287
+ | English only | **0.933** | 0.667 |
277
288
  | Chinese only | **1.000** | **1.000** |
278
289
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
279
290
 
291
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
292
+ cause is a limit of the design rather than a defect. A word counts as evidence
293
+ unless it occurs in more than 5% of units — a stopword list derived from the
294
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
295
+ `does` and `are` are everywhere, because 304 units carry written English prose.
296
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
297
+ them and start counting as evidence. Five English questions about subjects
298
+ Flask does not implement get through on exactly that.
299
+
280
300
  **What each bar costs and buys** — one corpus, gate varied alone:
281
301
 
282
302
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
283
303
  |---|---|---|---|---|
284
- | neither (pre-1.0.0) | 0.229/0.400/0.300 | 0.314/0.486/0.391 | 0.471/0.686/0.552 | 0.000 / 0.000 |
285
- | coverage only (1.0.0) | 0.229/0.400/0.300 | 0.314/0.471/0.383 | 0.471/0.671/0.548 | 0.700 / 0.767 |
286
- | **both (1.1.0)** | **0.229/0.400/0.300** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
287
-
288
- Rulers A and B are **identical to three decimals**. The entire cost is four
289
- questions of seventy on the warmest ruler. On these four rulers concentration
290
- subsumes coverage — stated plainly because it is true; coverage is kept because
291
- it answers a different question and names a different diagnosis.
292
-
293
- **Latency** — warm corpus, 557 units, 420 samples after warm-up:
294
-
295
- | | |
296
- |---|---|
297
- | query, median | **0.83 ms** |
298
- | query, p95 | 1.68 ms |
299
- | refusing an unanswerable query | **0.03 ms** |
300
-
301
- Refusal is cheaper than answering by a factor of forty: an unanswerable query
302
- touches only the posting lists of its own distinctive words, never the corpus.
304
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
305
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
306
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
307
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
308
+
309
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
310
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
311
+
312
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
313
+ this project did not choose, it does not.** Both bars together silence 0.833 of
314
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
315
+ for coverage alone. One question — and the first time in four releases that
316
+ keeping both has been worth a measurable amount rather than worth a different
317
+ diagnosis.
318
+
319
+ Raising the concentration bar buys the remaining silence, and is refused,
320
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
321
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
322
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
323
+ existed and survived meeting it**, which is the only kind of evidence a default
324
+ can have.
325
+
326
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
327
+ invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
328
+ each):
329
+
330
+ | | | across the five |
331
+ |---|---|---|
332
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
333
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
334
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
335
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
336
+
337
+ Two significant figures and a spread, because that is the precision the
338
+ measurement has. Across fifteen invocations over two releases on the same idle
339
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
340
+ to 7.34 ms — a band wider than any change the code has ever made to this
341
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
342
+ figures from a script that was never committed; both values sit inside that
343
+ band, which is the point: they were unfalsifiable rather than wrong.
344
+
345
+ Refusal is cheap for a structural reason, not a tuned one: an unanswerable
346
+ query touches only the posting lists of its own distinctive words, and never
347
+ reaches ranking at all.
303
348
 
304
349
  **Scale**, synthetic 10,000-unit repository (500 files):
305
350
 
@@ -321,8 +366,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
321
366
  | with a usable signature | **91 / 91** |
322
367
  | units invented that do not exist | **0** |
323
368
 
324
- Directional local measurements, not service levels. Reproduce with
325
- `python benchmarks/repo_queries.py` and `python benchmarks/large_repo.py`.
369
+ Directional local measurements, not service levels — but every one of them is
370
+ now a command rather than a memory, which two of them were not before. Each
371
+ prints the corpus fingerprint beside its score; quote both or neither.
372
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
373
+ each is for, and the corpus one of them grades is now carried here too.
326
374
 
327
375
  ## 7 · `rag-your-code search` vs a Grep loop
328
376
 
@@ -332,45 +380,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
332
380
  ruler, scored at **file** granularity so Grep is not penalised for lacking
333
381
  declaration spans.
334
382
 
335
- **On a repository nobody has described, Grep wins.** That is the measured
336
- result and it is not softened here.
383
+ **Which side wins on an undescribed repository depends on the repository.**
384
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
385
+ undescribed repository ever measured was a hook-heavy tool whose questions were
386
+ answerable by matching identifiers. Swapping the subject for a public web
387
+ framework reversed it. The honest claim is narrower than either table alone:
337
388
 
338
- | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
389
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
339
390
  |---|---|---|
340
- | right file first | **34.3%** | 31.4% |
341
- | right file in top 3 | **60.0%** | 48.6% |
342
- | lines matched across the repo, all questions | 33,213 | — |
343
- | characters returned, all questions | — | **163,294** |
344
- | questions it answers | 35 | 28 |
391
+ | right file first | 22.9% | **37.1%** |
392
+ | right file in top 3 | 45.7% | **57.1%** |
393
+ | lines it hands back, all questions | 17,641 | — |
394
+ | characters returned, all questions | 1,415,656 | **249,720** |
395
+ | questions it answers | 30 | 30 |
345
396
 
346
397
  **Once the vocabulary exists, it is not close.**
347
398
 
348
- | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
399
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
349
400
  |---|---|---|
350
- | right file first | 25.7% | **58.6%** |
351
- | right file in top 3 | 64.3% | **75.7%** |
352
- | lines matched across the repo, all questions | 40,150 | — |
353
- | characters returned, all questions | — | **277,327** |
354
- | questions it answers | 70 | 60 |
401
+ | right file first | 22.9% | **58.6%** |
402
+ | right file in top 3 | 54.3% | **75.7%** |
403
+ | lines it hands back, all questions | 11,833 | — |
404
+ | characters returned, all questions | 1,122,902 | **626,022** |
405
+ | questions it answers | **61** | 60 |
355
406
 
356
407
  Those two tables are the whole argument of section 3.3, measured against a real
357
408
  baseline instead of asserted. A cold index retrieves against a sentence the
358
- parser generated from identifiers the author already chose — so it is competing
359
- with Grep using Grep's own information, and losing, because Grep does not have
360
- to guess which of the matching files is the definition. Descriptions put words
361
- in the index that the source never contained, and first-place accuracy goes
362
- from below Grep's to **more than double** it.
363
-
364
- Three qualifications, because the table would otherwise flatter both sides:
409
+ parser generated from identifiers the author already chose, plus whatever
410
+ docstrings the author wrote — so how it fares against Grep is decided by how
411
+ much prose the repository already contains. Flask has a written docstring on
412
+ most public methods, and the cold index beats Grep there without a single
413
+ description being added. On the previous subject, a tool with terse comments
414
+ and long identifiers, the same cold index lost to Grep by the same margin.
415
+
416
+ What does not depend on the subject is what descriptions buy: on this
417
+ repository first-place accuracy goes to **more than double** Grep's, and the
418
+ payload comes back ranked, spanned, and roughly half the size.
419
+
420
+ **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
421
+ changed in 1.3.0. Until then this section — the strongest claim the project
422
+ makes — was published from a script that had never been committed, so nothing
423
+ here could be checked and the word "Grep loop" had no precise meaning. The
424
+ committed version defines it: take the query's words, drop the ones the corpus
425
+ itself shows are everywhere, run one substring search per remaining word over
426
+ exactly the files the index was built from, rank each file by how many distinct
427
+ words hit it, break ties on path. Reconstructing it reproduced this side's
428
+ figures exactly and moved Grep's, which is the expected shape — the ranked arm
429
+ was always a call into shipped code, and the baseline never was.
430
+
431
+ Four qualifications, because the table would otherwise flatter both sides:
365
432
 
366
433
  - **Scored at file granularity**, which understates this side. A Grep hit is a
367
434
  file; a hit here is a declaration with an exact span, a score, and the words
368
435
  it matched on. The agent that reads the result opens 40 lines, not a file.
369
- - **Grep answers everything.** It never declines, which is why it hands back
370
- 33,213 matching lines for 35 questions — about 950 lines per question, no
371
- ranking, no spans, no indication which match is the definition. This returns
372
- roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
373
- questions come back empty here instead, with a reason.
436
+ - **Dropping the corpus-common words is generous to Grep**, and it is what
437
+ makes the baseline a fair one rather than a straw man: an agent that greps
438
+ `the` gets every file back in no order. It is also why Grep declines nine of
439
+ the seventy questions here — those had no word left that this corpus does not
440
+ use everywhere.
441
+ - **Payload is counted in characters on both sides.** On this repository Grep
442
+ hands back 18,400 characters per question it answers, unranked and without
443
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
444
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
445
+ 8,300 — because a framework repeats its own vocabulary across many files and
446
+ Grep has no way to rank what it finds. Both sides decline the same five of
447
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
448
+ not a substring of English source and it is not a token in an index built
449
+ from English source, so on a repository written in one language the cold
450
+ cross-language case is not this tool's failure but the corpus's.
374
451
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
375
452
  exact, instant and complete, and nothing here replaces it.
376
453
 
@@ -435,19 +512,23 @@ The extra is optional by construction: `dependencies = []` is what a default
435
512
  install gets, the import happens inside the constructor, and a test asserts the
436
513
  default provider imports none of it.
437
514
 
438
- **Measured, on the same four rulers.** This is the first release with a real
439
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
440
- benefit was unmeasured.
515
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
516
+ published this comparison and read it as a win. Its largest gain was on the
517
+ foreign ruler, whose two arms turned out to have been taken against two
518
+ different states of a repository being edited while the script ran. Repeated
519
+ against a pinned corpus:
441
520
 
442
- | ruler | signed hash (default) | MiniLM, local |
443
- |---|---|---|
444
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
445
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
446
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
447
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
521
+ | ruler | corpus | signed hash (default) | MiniLM, local |
522
+ |---|---|---|---|
523
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
524
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
525
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
526
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
448
527
 
449
- Better on every positive ruler, with refusal unchanged. The pairs the hash
450
- scores exactly zero:
528
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
529
+ has to install, because it does one thing the hash cannot do at all and these
530
+ rulers cannot see: reach a unit that shares no word with the question. The
531
+ pairs the hash scores exactly zero:
451
532
 
452
533
  | pair | signed hash | MiniLM |
453
534
  |---|---|---|
@@ -465,6 +546,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
465
546
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
466
547
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
467
548
 
549
+ If you install it expecting the hit rates above to move, they will not. Install
550
+ it for the cross-language and paraphrase cases in the table above, which is
551
+ where the difference between the two columns actually lives.
552
+
468
553
  **A hosted endpoint** is the third option, and the only one that sends your
469
554
  source anywhere:
470
555
 
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "1.2.1"
9
+ version = "1.4.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.2.1
3
+ Version: 1.4.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -293,42 +293,87 @@ repository had grown by ninety units.
293
293
 
294
294
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
295
295
  |---|---|---|---|---|---|
296
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.400 | 0.300 |
296
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
297
297
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
298
298
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
299
299
 
300
+ The corpora, without which none of the above is reproducible — **A** 1,572
301
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
302
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
303
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
304
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
305
+ that cost three things: two questions pointed at a declaration the subject had
306
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
307
+ subject had grown, and the model comparison below was taken against two
308
+ different states of it. All three are now a `git clone` away from being
309
+ checked, and CI runs this ruler as an ordinary job.
310
+
300
311
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
301
312
 
302
- | | this repo | foreign repo |
313
+ | | this repo | Flask |
303
314
  |---|---|---|
304
- | correctly met with silence | **0.967** | **0.933** |
305
- | English only | **0.933** | **0.867** |
315
+ | correctly met with silence | **0.967** | **0.833** |
316
+ | English only | **0.933** | 0.667 |
306
317
  | Chinese only | **1.000** | **1.000** |
307
318
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
308
319
 
320
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
321
+ cause is a limit of the design rather than a defect. A word counts as evidence
322
+ unless it occurs in more than 5% of units — a stopword list derived from the
323
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
324
+ `does` and `are` are everywhere, because 304 units carry written English prose.
325
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
326
+ them and start counting as evidence. Five English questions about subjects
327
+ Flask does not implement get through on exactly that.
328
+
309
329
  **What each bar costs and buys** — one corpus, gate varied alone:
310
330
 
311
331
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
312
332
  |---|---|---|---|---|
313
- | neither (pre-1.0.0) | 0.229/0.400/0.300 | 0.314/0.486/0.391 | 0.471/0.686/0.552 | 0.000 / 0.000 |
314
- | coverage only (1.0.0) | 0.229/0.400/0.300 | 0.314/0.471/0.383 | 0.471/0.671/0.548 | 0.700 / 0.767 |
315
- | **both (1.1.0)** | **0.229/0.400/0.300** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
316
-
317
- Rulers A and B are **identical to three decimals**. The entire cost is four
318
- questions of seventy on the warmest ruler. On these four rulers concentration
319
- subsumes coverage — stated plainly because it is true; coverage is kept because
320
- it answers a different question and names a different diagnosis.
321
-
322
- **Latency** — warm corpus, 557 units, 420 samples after warm-up:
323
-
324
- | | |
325
- |---|---|
326
- | query, median | **0.83 ms** |
327
- | query, p95 | 1.68 ms |
328
- | refusing an unanswerable query | **0.03 ms** |
329
-
330
- Refusal is cheaper than answering by a factor of forty: an unanswerable query
331
- touches only the posting lists of its own distinctive words, never the corpus.
333
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
334
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
335
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
336
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
337
+
338
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
339
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
340
+
341
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
342
+ this project did not choose, it does not.** Both bars together silence 0.833 of
343
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
344
+ for coverage alone. One question — and the first time in four releases that
345
+ keeping both has been worth a measurable amount rather than worth a different
346
+ diagnosis.
347
+
348
+ Raising the concentration bar buys the remaining silence, and is refused,
349
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
350
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
351
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
352
+ existed and survived meeting it**, which is the only kind of evidence a default
353
+ can have.
354
+
355
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
356
+ invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
357
+ each):
358
+
359
+ | | | across the five |
360
+ |---|---|---|
361
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
362
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
363
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
364
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
365
+
366
+ Two significant figures and a spread, because that is the precision the
367
+ measurement has. Across fifteen invocations over two releases on the same idle
368
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
369
+ to 7.34 ms — a band wider than any change the code has ever made to this
370
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
371
+ figures from a script that was never committed; both values sit inside that
372
+ band, which is the point: they were unfalsifiable rather than wrong.
373
+
374
+ Refusal is cheap for a structural reason, not a tuned one: an unanswerable
375
+ query touches only the posting lists of its own distinctive words, and never
376
+ reaches ranking at all.
332
377
 
333
378
  **Scale**, synthetic 10,000-unit repository (500 files):
334
379
 
@@ -350,8 +395,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
350
395
  | with a usable signature | **91 / 91** |
351
396
  | units invented that do not exist | **0** |
352
397
 
353
- Directional local measurements, not service levels. Reproduce with
354
- `python benchmarks/repo_queries.py` and `python benchmarks/large_repo.py`.
398
+ Directional local measurements, not service levels — but every one of them is
399
+ now a command rather than a memory, which two of them were not before. Each
400
+ prints the corpus fingerprint beside its score; quote both or neither.
401
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
402
+ each is for, and the corpus one of them grades is now carried here too.
355
403
 
356
404
  ## 7 · `rag-your-code search` vs a Grep loop
357
405
 
@@ -361,45 +409,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
361
409
  ruler, scored at **file** granularity so Grep is not penalised for lacking
362
410
  declaration spans.
363
411
 
364
- **On a repository nobody has described, Grep wins.** That is the measured
365
- result and it is not softened here.
412
+ **Which side wins on an undescribed repository depends on the repository.**
413
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
414
+ undescribed repository ever measured was a hook-heavy tool whose questions were
415
+ answerable by matching identifiers. Swapping the subject for a public web
416
+ framework reversed it. The honest claim is narrower than either table alone:
366
417
 
367
- | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
418
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
368
419
  |---|---|---|
369
- | right file first | **34.3%** | 31.4% |
370
- | right file in top 3 | **60.0%** | 48.6% |
371
- | lines matched across the repo, all questions | 33,213 | — |
372
- | characters returned, all questions | — | **163,294** |
373
- | questions it answers | 35 | 28 |
420
+ | right file first | 22.9% | **37.1%** |
421
+ | right file in top 3 | 45.7% | **57.1%** |
422
+ | lines it hands back, all questions | 17,641 | — |
423
+ | characters returned, all questions | 1,415,656 | **249,720** |
424
+ | questions it answers | 30 | 30 |
374
425
 
375
426
  **Once the vocabulary exists, it is not close.**
376
427
 
377
- | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
428
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
378
429
  |---|---|---|
379
- | right file first | 25.7% | **58.6%** |
380
- | right file in top 3 | 64.3% | **75.7%** |
381
- | lines matched across the repo, all questions | 40,150 | — |
382
- | characters returned, all questions | — | **277,327** |
383
- | questions it answers | 70 | 60 |
430
+ | right file first | 22.9% | **58.6%** |
431
+ | right file in top 3 | 54.3% | **75.7%** |
432
+ | lines it hands back, all questions | 11,833 | — |
433
+ | characters returned, all questions | 1,122,902 | **626,022** |
434
+ | questions it answers | **61** | 60 |
384
435
 
385
436
  Those two tables are the whole argument of section 3.3, measured against a real
386
437
  baseline instead of asserted. A cold index retrieves against a sentence the
387
- parser generated from identifiers the author already chose — so it is competing
388
- with Grep using Grep's own information, and losing, because Grep does not have
389
- to guess which of the matching files is the definition. Descriptions put words
390
- in the index that the source never contained, and first-place accuracy goes
391
- from below Grep's to **more than double** it.
392
-
393
- Three qualifications, because the table would otherwise flatter both sides:
438
+ parser generated from identifiers the author already chose, plus whatever
439
+ docstrings the author wrote — so how it fares against Grep is decided by how
440
+ much prose the repository already contains. Flask has a written docstring on
441
+ most public methods, and the cold index beats Grep there without a single
442
+ description being added. On the previous subject, a tool with terse comments
443
+ and long identifiers, the same cold index lost to Grep by the same margin.
444
+
445
+ What does not depend on the subject is what descriptions buy: on this
446
+ repository first-place accuracy goes to **more than double** Grep's, and the
447
+ payload comes back ranked, spanned, and roughly half the size.
448
+
449
+ **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
450
+ changed in 1.3.0. Until then this section — the strongest claim the project
451
+ makes — was published from a script that had never been committed, so nothing
452
+ here could be checked and the word "Grep loop" had no precise meaning. The
453
+ committed version defines it: take the query's words, drop the ones the corpus
454
+ itself shows are everywhere, run one substring search per remaining word over
455
+ exactly the files the index was built from, rank each file by how many distinct
456
+ words hit it, break ties on path. Reconstructing it reproduced this side's
457
+ figures exactly and moved Grep's, which is the expected shape — the ranked arm
458
+ was always a call into shipped code, and the baseline never was.
459
+
460
+ Four qualifications, because the table would otherwise flatter both sides:
394
461
 
395
462
  - **Scored at file granularity**, which understates this side. A Grep hit is a
396
463
  file; a hit here is a declaration with an exact span, a score, and the words
397
464
  it matched on. The agent that reads the result opens 40 lines, not a file.
398
- - **Grep answers everything.** It never declines, which is why it hands back
399
- 33,213 matching lines for 35 questions — about 950 lines per question, no
400
- ranking, no spans, no indication which match is the definition. This returns
401
- roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
402
- questions come back empty here instead, with a reason.
465
+ - **Dropping the corpus-common words is generous to Grep**, and it is what
466
+ makes the baseline a fair one rather than a straw man: an agent that greps
467
+ `the` gets every file back in no order. It is also why Grep declines nine of
468
+ the seventy questions here — those had no word left that this corpus does not
469
+ use everywhere.
470
+ - **Payload is counted in characters on both sides.** On this repository Grep
471
+ hands back 18,400 characters per question it answers, unranked and without
472
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
473
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
474
+ 8,300 — because a framework repeats its own vocabulary across many files and
475
+ Grep has no way to rank what it finds. Both sides decline the same five of
476
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
477
+ not a substring of English source and it is not a token in an index built
478
+ from English source, so on a repository written in one language the cold
479
+ cross-language case is not this tool's failure but the corpus's.
403
480
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
404
481
  exact, instant and complete, and nothing here replaces it.
405
482
 
@@ -464,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
464
541
  install gets, the import happens inside the constructor, and a test asserts the
465
542
  default provider imports none of it.
466
543
 
467
- **Measured, on the same four rulers.** This is the first release with a real
468
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
469
- benefit was unmeasured.
544
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
545
+ published this comparison and read it as a win. Its largest gain was on the
546
+ foreign ruler, whose two arms turned out to have been taken against two
547
+ different states of a repository being edited while the script ran. Repeated
548
+ against a pinned corpus:
470
549
 
471
- | ruler | signed hash (default) | MiniLM, local |
472
- |---|---|---|
473
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
474
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
475
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
476
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
550
+ | ruler | corpus | signed hash (default) | MiniLM, local |
551
+ |---|---|---|---|
552
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
553
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
554
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
555
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
477
556
 
478
- Better on every positive ruler, with refusal unchanged. The pairs the hash
479
- scores exactly zero:
557
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
558
+ has to install, because it does one thing the hash cannot do at all and these
559
+ rulers cannot see: reach a unit that shares no word with the question. The
560
+ pairs the hash scores exactly zero:
480
561
 
481
562
  | pair | signed hash | MiniLM |
482
563
  |---|---|---|
@@ -494,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
494
575
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
495
576
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
496
577
 
578
+ If you install it expecting the hit rates above to move, they will not. Install
579
+ it for the cross-language and paraphrase cases in the table above, which is
580
+ where the difference between the two columns actually lives.
581
+
497
582
  **A hosted endpoint** is the third option, and the only one that sends your
498
583
  source anywhere:
499
584
 
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "1.2.1"
6
+ __version__ = "1.4.0"
@@ -201,10 +201,15 @@ SETTINGS: tuple[Setting, ...] = (
201
201
  ),
202
202
  # How much of a question has to reach the index before an answer counts as
203
203
  # evidence rather than as a guess. Measured, not chosen: across two
204
- # repositories, two languages and 126 questions, 0.40 is the largest value
205
- # at which every question that was being answered correctly still is, and
206
- # it silences 47% of questions whose answer is not in the repository at
207
- # all. It is a ratio inside the query, so unlike a score threshold it does
204
+ # repositories, two languages and every question set under `benchmarks/`,
205
+ # 0.40 is the largest value at which every question that was being answered
206
+ # correctly still is. On its own it silences three fifths of the questions
207
+ # whose answer is not in the repository at all; the rest is what
208
+ # `search.min_concentration` below adds. The command is the claim --
209
+ # `repo_queries --questions benchmarks/absent_queries.json
210
+ # --min-concentration 0` -- because a count typed into a comment is a figure
211
+ # nothing checks, and both numbers this sentence used to carry had rotted.
212
+ # It is a ratio inside the query, so unlike a score threshold it does
208
213
  # not move when the corpus or the scale of the ranking does -- the defect
209
214
  # that made `confidence_threshold = 0.8` stop meaning anything.
210
215
  Setting(
@@ -221,8 +226,11 @@ SETTINGS: tuple[Setting, ...] = (
221
226
  # unrelated declarations -- four of six words found in four places with
222
227
  # nothing to do with one another or with what was asked. Measured across
223
228
  # four rulers, requiring a quarter of a query's rarity to land inside one
224
- # unit leaves the two rulers over undescribed code unchanged and roughly
225
- # halves the questions that get an answer they should not have. Rarity-
229
+ # unit leaves the two rulers over undescribed code unchanged and removes
230
+ # most of what the coverage bar alone still answers; the ablation table
231
+ # carries the numbers, in docs/ROADMAP.md, rather than this line, which
232
+ # said "roughly halves" while that table said an order of magnitude.
233
+ # Rarity-
226
234
  # weighted rather than counted, because a unit holding two ordinary words is
227
235
  # not better evidence than one holding the rare word the question is about.
228
236
  Setting(
@@ -119,7 +119,7 @@ class Evidence:
119
119
 
120
120
  Ranking answers "which of these is best". It cannot answer "is any of this
121
121
  an answer", and reading the first as the second is what let a repository
122
- reply to all 32 questions in `benchmarks/absent_queries.json` -- every one
122
+ reply to every question in `benchmarks/absent_queries.json` -- each one
123
123
  about a subject neither repository implements. `where are CUDA kernels
124
124
  dispatched to the device` came back with a test about word counting, on the
125
125
  evidence of `are`, `the`, `to` and `where`. That is not a Chinese problem
@@ -53,23 +53,56 @@ def test_the_absent_ruler_is_well_formed():
53
53
  assert questions["why"].strip() and questions["caveat"].strip()
54
54
 
55
55
 
56
- def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
57
- """The absence claim, re-derived rather than trusted.
58
-
59
- This is the assertion the ruler's own caveat promises, and it is the one
60
- that fails first when the repository grows into a subject the ruler
61
- assumed it would never contain.
56
+ def _intruders(units) -> list[tuple[str, str, int]]:
57
+ """Which absent question's own vocabulary reaches a corpus that must not
58
+ contain it.
62
59
  """
63
60
  index = build_search_index(units)
64
- intruders = [
61
+ return [
65
62
  (entry["id"], term, len(index.postings[term]))
66
63
  for entry in load_questions(ABSENT_PATH)["queries"]
67
64
  for term in entry["subject"]
68
65
  if index.postings.get(term)
69
66
  ]
70
- assert not intruders, (
67
+
68
+
69
+ def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
70
+ """The absence claim, re-derived rather than trusted.
71
+
72
+ This is the assertion the ruler's own caveat promises, and it is the one
73
+ that fails first when the repository grows into a subject the ruler
74
+ assumed it would never contain.
75
+ """
76
+ assert not _intruders(units), (
71
77
  "this repository now contains the vocabulary of a question the ruler calls unanswerable; "
72
- f"retire or rewrite those questions rather than letting them score as misses: {intruders}"
78
+ f"retire or rewrite those questions rather than letting them score as misses: {_intruders(units)}"
79
+ )
80
+
81
+
82
+ def test_no_subject_of_an_absent_question_exists_in_the_vendored_corpus():
83
+ """The other half of the same claim, which used to be unenforceable.
84
+
85
+ This ruler asserts its questions are unanswerable in *both* graded
86
+ repositories. Until the foreign one was vendored that half could only be
87
+ asserted, because the subject was somebody else's checkout and a re-run
88
+ here could not see what it had become. It is now a directory in this
89
+ repository at a pinned tag, so the claim is a test.
90
+
91
+ It has already earned its place. One question's subject list carried a
92
+ generic English verb where a specific term belonged, and the corpus below
93
+ uses that verb in a comment about signing keys -- so the question was one
94
+ ordinary word away from being answerable by accident. The verb was
95
+ replaced with a term that names the thing.
96
+
97
+ Naming the subject here would put it in the index and break the very
98
+ claim this asserts, which is why the paragraph above is written around it.
99
+ That has now happened five times; CONTRIBUTING.md keeps the tally.
100
+ """
101
+ corpus = ROOT / "benchmarks" / "corpus" / "flask"
102
+ assert corpus.is_dir(), f"the vendored corpus is missing from {corpus}"
103
+ assert not _intruders(build_units(corpus)), (
104
+ "the vendored corpus contains the vocabulary of a question the ruler calls unanswerable; "
105
+ f"rewrite the question rather than letting it score as a miss: {_intruders(build_units(corpus))}"
73
106
  )
74
107
 
75
108
 
@@ -1,9 +1,10 @@
1
1
  """Retrieval must be able to say it has no answer.
2
2
 
3
3
  Ranking always produces a least-bad unit and returns it with a score and a
4
- rank, which read exactly like an answer. Graded against 32 questions about
5
- subjects neither this repository nor `benchmarks/cold_queries.json`'s
6
- repository implements, every single one came back answered -- `where are CUDA
4
+ rank, which read exactly like an answer. Graded against every question in
5
+ `benchmarks/absent_queries.json` -- subjects neither this repository nor
6
+ `benchmarks/cold_queries.json`'s repository implements -- every single one came
7
+ back answered, on both repositories. `where are CUDA
7
8
  kernels dispatched to the device` on the evidence of `are`, `the`, `to` and
8
9
  `where`. These assert the second question retrieval now asks: not which unit
9
10
  ranks highest, but whether any of this is evidence at all.
@@ -321,3 +321,35 @@ def test_every_documented_provider_block_actually_configures_that_provider(tmp_p
321
321
  # arrangement exists to prevent.
322
322
  assert "sk-" not in settings
323
323
  assert {"sentence-transformers", "openai-compatible"} <= seen, f"undocumented providers; README shows {sorted(seen)}"
324
+
325
+
326
+ # --- a measurement nobody can re-run is a claim, not a measurement ----------
327
+
328
+ BENCHMARKS = tuple(
329
+ sorted(
330
+ path.name
331
+ for path in (ROOT / "benchmarks").glob("*.py")
332
+ if path.name != "__init__.py"
333
+ )
334
+ )
335
+
336
+
337
+ def test_every_benchmark_script_is_listed_in_its_own_index():
338
+ """The index of the rulers is discovered, not maintained by hand.
339
+
340
+ 1.3.0 found two figures in the README -- query latency, and the entire Grep
341
+ head-to-head -- produced by scripts that had never been committed. Nobody
342
+ could re-derive either, and the second one meant the phrase "a Grep loop"
343
+ had no definition a reader could argue with.
344
+
345
+ Both directions matter. A script absent from the index is one nobody knows
346
+ to run; a command in the index naming a script that does not exist is the
347
+ install-line defect this repository shipped twice. Discovery by glob is
348
+ what keeps the seventh script from being the one nothing checks.
349
+ """
350
+ index = (ROOT / "benchmarks" / "README.md").read_text(encoding="utf-8")
351
+ assert BENCHMARKS, "benchmarks/ holds no scripts; this guard would pass vacuously"
352
+ for name in BENCHMARKS:
353
+ assert f"benchmarks.{Path(name).stem}" in index, f"{name} is in benchmarks/ and not in its README"
354
+ for module in re.findall(r"benchmarks\.([a-z_]+)", index):
355
+ assert (ROOT / "benchmarks" / f"{module}.py").is_file(), f"the README runs benchmarks.{module}, which does not exist"
File without changes
File without changes