rag-your-code 1.3.0__tar.gz → 1.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {rag_your_code-1.3.0/src/rag_your_code.egg-info → rag_your_code-1.4.0}/PKG-INFO +113 -70
  2. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/README.md +112 -69
  3. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/pyproject.toml +1 -1
  4. {rag_your_code-1.3.0 → rag_your_code-1.4.0/src/rag_your_code.egg-info}/PKG-INFO +113 -70
  5. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/__init__.py +1 -1
  6. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_absent_queries.py +42 -9
  7. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/LICENSE +0 -0
  8. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/setup.cfg +0 -0
  9. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
  10. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  11. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  12. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  13. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  14. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/agentic.py +0 -0
  15. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/annotate.py +0 -0
  16. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/cli.py +0 -0
  17. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/config.py +0 -0
  18. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/descriptions.py +0 -0
  19. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/document.py +0 -0
  20. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/embeddings.py +0 -0
  21. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/graph.py +0 -0
  22. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/indexer.py +0 -0
  23. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/models.py +0 -0
  24. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/parser.py +0 -0
  25. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/providers.py +0 -0
  26. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/py.typed +0 -0
  27. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/search.py +0 -0
  28. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/workflow.py +0 -0
  29. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_agent_protocol.py +0 -0
  30. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_agentic.py +0 -0
  31. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_config.py +0 -0
  32. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_descriptions.py +0 -0
  33. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_doc_comments.py +0 -0
  34. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_document.py +0 -0
  35. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_e2e_cli.py +0 -0
  36. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_evidence.py +0 -0
  37. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_golden.py +0 -0
  38. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_graph_incremental.py +0 -0
  39. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_language_fixtures.py +0 -0
  40. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_large_repo.py +0 -0
  41. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_local_model.py +0 -0
  42. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_metadata.py +0 -0
  43. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_multilanguage.py +0 -0
  44. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_parser_edges.py +0 -0
  45. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_providers.py +0 -0
  46. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_ragyourcode.py +0 -0
  47. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_ranking.py +0 -0
  48. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_repo_queries.py +0 -0
  49. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_resilience.py +0 -0
  50. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_retrieval_correctness.py +0 -0
  51. {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.3.0
3
+ Version: 1.4.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -293,60 +293,83 @@ repository had grown by ninety units.
293
293
 
294
294
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
295
295
  |---|---|---|---|---|---|
296
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.371 | 0.286 |
296
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
297
297
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
298
298
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
299
299
 
300
- The corpora, without which none of the above is reproducible — **A**
301
- cc-enforcer, 1,345 units, `471b78f9f806`; **B** 579 units, `dca2a4656659`;
302
- **C** 579 units, `0a2f050dbafd`. Ruler A's repository is one this project does
303
- not control. Between the previous release and this table it grew by 78 units
304
- and renamed a declaration two questions pointed at, which stopped the ruler
305
- running at all until they were repointed — hit@3 0.400 → 0.371 and MRR 0.300 →
306
- 0.286 are that, not a change in retrieval.
300
+ The corpora, without which none of the above is reproducible — **A** 1,572
301
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
302
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
303
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
304
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
305
+ that cost three things: two questions pointed at a declaration the subject had
306
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
307
+ subject had grown, and the model comparison below was taken against two
308
+ different states of it. All three are now a `git clone` away from being
309
+ checked, and CI runs this ruler as an ordinary job.
307
310
 
308
311
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
309
312
 
310
- | | this repo | foreign repo |
313
+ | | this repo | Flask |
311
314
  |---|---|---|
312
- | correctly met with silence | **0.967** | **0.933** |
313
- | English only | **0.933** | **0.867** |
315
+ | correctly met with silence | **0.967** | **0.833** |
316
+ | English only | **0.933** | 0.667 |
314
317
  | Chinese only | **1.000** | **1.000** |
315
318
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
316
319
 
320
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
321
+ cause is a limit of the design rather than a defect. A word counts as evidence
322
+ unless it occurs in more than 5% of units — a stopword list derived from the
323
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
324
+ `does` and `are` are everywhere, because 304 units carry written English prose.
325
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
326
+ them and start counting as evidence. Five English questions about subjects
327
+ Flask does not implement get through on exactly that.
328
+
317
329
  **What each bar costs and buys** — one corpus, gate varied alone:
318
330
 
319
331
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
320
332
  |---|---|---|---|---|
321
- | neither (pre-1.0.0) | 0.229/0.371/0.286 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
322
- | coverage only (1.0.0) | 0.229/0.371/0.286 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.767 |
323
- | **both (1.1.0)** | **0.229/0.371/0.286** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
324
-
325
- Rulers A and B are **identical to three decimals**. The entire cost is three
326
- questions of seventy at hit@1 on the warmest ruler, and six at hit@3. On these
327
- four rulers concentration
328
- subsumes coverage — stated plainly because it is true; coverage is kept because
329
- it answers a different question and names a different diagnosis.
330
-
331
- **Latency** — warm corpus, 579 units `0a2f050dbafd`, five consecutive
333
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
334
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
335
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
336
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
337
+
338
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
339
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
340
+
341
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
342
+ this project did not choose, it does not.** Both bars together silence 0.833 of
343
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
344
+ for coverage alone. One question — and the first time in four releases that
345
+ keeping both has been worth a measurable amount rather than worth a different
346
+ diagnosis.
347
+
348
+ Raising the concentration bar buys the remaining silence, and is refused,
349
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
350
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
351
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
352
+ existed and survived meeting it**, which is the only kind of evidence a default
353
+ can have.
354
+
355
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
332
356
  invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
333
357
  each):
334
358
 
335
359
  | | | across the five |
336
360
  |---|---|---|
337
- | query, median | **0.65 ms** | 0.62 – 0.92 |
338
- | query, p95 | 1.13 ms | 1.09 – 1.48 |
339
- | refusing an unanswerable query | **0.022 ms** | 0.019 – 0.024 |
340
- | refusal cheaper than answering by | **~35×** | 31 – 39 |
361
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
362
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
363
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
364
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
341
365
 
342
366
  Two significant figures and a spread, because that is the precision the
343
- measurement has. A second set of five taken earlier the same hour, while the
344
- machine was busy, put the median at 1.0 ms and p95 at 3.0 ms; the gap between
345
- the two sets is the machine, and it is wider than any change the code has ever
346
- made to this number. Earlier releases published `0.83 ms / p95 1.68 ms` to
347
- three figures from a script that was never committed, on a corpus that no
348
- longer exists — both values sit inside today's range, which is the point: they
349
- were unfalsifiable rather than wrong.
367
+ measurement has. Across fifteen invocations over two releases on the same idle
368
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
369
+ to 7.34 ms — a band wider than any change the code has ever made to this
370
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
371
+ figures from a script that was never committed; both values sit inside that
372
+ band, which is the point: they were unfalsifiable rather than wrong.
350
373
 
351
374
  Refusal is cheap for a structural reason, not a tuned one: an unanswerable
352
375
  query touches only the posting lists of its own distinctive words, and never
@@ -375,8 +398,8 @@ reaches ranking at all.
375
398
  Directional local measurements, not service levels — but every one of them is
376
399
  now a command rather than a memory, which two of them were not before. Each
377
400
  prints the corpus fingerprint beside its score; quote both or neither.
378
- [`benchmarks/README.md`](benchmarks/README.md) lists the five scripts and what
379
- each is for.
401
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
402
+ each is for, and the corpus one of them grades is now carried here too.
380
403
 
381
404
  ## 7 · `rag-your-code search` vs a Grep loop
382
405
 
@@ -386,34 +409,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
386
409
  ruler, scored at **file** granularity so Grep is not penalised for lacking
387
410
  declaration spans.
388
411
 
389
- **On a repository nobody has described, Grep wins.** That is the measured
390
- result and it is not softened here.
412
+ **Which side wins on an undescribed repository depends on the repository.**
413
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
414
+ undescribed repository ever measured was a hook-heavy tool whose questions were
415
+ answerable by matching identifiers. Swapping the subject for a public web
416
+ framework reversed it. The honest claim is narrower than either table alone:
391
417
 
392
- | foreign repository · 35 questions · 1,345 units `471b78f9f806` · no descriptions | Grep loop | rag-your-code |
418
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
393
419
  |---|---|---|
394
- | right file first | **40.0%** | 28.6% |
395
- | right file in top 3 | **62.9%** | 48.6% |
396
- | lines it hands back, all questions | 8,937 | — |
397
- | characters returned, all questions | 841,484 | **282,102** |
398
- | questions it answers | **35** | 28 |
420
+ | right file first | 22.9% | **37.1%** |
421
+ | right file in top 3 | 45.7% | **57.1%** |
422
+ | lines it hands back, all questions | 17,641 | — |
423
+ | characters returned, all questions | 1,415,656 | **249,720** |
424
+ | questions it answers | 30 | 30 |
399
425
 
400
426
  **Once the vocabulary exists, it is not close.**
401
427
 
402
- | this repository · 70 questions · 579 units `0a2f050dbafd` · 304 described | Grep loop | rag-your-code |
428
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
403
429
  |---|---|---|
404
430
  | right file first | 22.9% | **58.6%** |
405
- | right file in top 3 | 55.7% | **75.7%** |
406
- | lines it hands back, all questions | 11,636 | — |
407
- | characters returned, all questions | 1,101,618 | **621,837** |
431
+ | right file in top 3 | 54.3% | **75.7%** |
432
+ | lines it hands back, all questions | 11,833 | — |
433
+ | characters returned, all questions | 1,122,902 | **626,022** |
408
434
  | questions it answers | **61** | 60 |
409
435
 
410
436
  Those two tables are the whole argument of section 3.3, measured against a real
411
437
  baseline instead of asserted. A cold index retrieves against a sentence the
412
- parser generated from identifiers the author already chose — so it is competing
413
- with Grep using Grep's own information, and losing, because Grep does not have
414
- to guess which of the matching files is the definition. Descriptions put words
415
- in the index that the source never contained, and first-place accuracy goes
416
- from below Grep's to **more than double** it.
438
+ parser generated from identifiers the author already chose, plus whatever
439
+ docstrings the author wrote — so how it fares against Grep is decided by how
440
+ much prose the repository already contains. Flask has a written docstring on
441
+ most public methods, and the cold index beats Grep there without a single
442
+ description being added. On the previous subject, a tool with terse comments
443
+ and long identifiers, the same cold index lost to Grep by the same margin.
444
+
445
+ What does not depend on the subject is what descriptions buy: on this
446
+ repository first-place accuracy goes to **more than double** Grep's, and the
447
+ payload comes back ranked, spanned, and roughly half the size.
417
448
 
418
449
  **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
419
450
  changed in 1.3.0. Until then this section — the strongest claim the project
@@ -436,12 +467,16 @@ Four qualifications, because the table would otherwise flatter both sides:
436
467
  `the` gets every file back in no order. It is also why Grep declines nine of
437
468
  the seventy questions here — those had no word left that this corpus does not
438
469
  use everywhere.
439
- - **Payload is counted in characters on both sides.** Grep hands back 18,000
440
- characters per question it answers, unranked and without spans; this returns
441
- 10,400, ranked, capped by `search.max_chars`. That is a factor of 1.7, not
442
- the order of magnitude an earlier version of this table implied by counting
443
- one side in lines and the other in characters. Seven of 35 and ten of 70
444
- questions come back empty here instead, with a reason.
470
+ - **Payload is counted in characters on both sides.** On this repository Grep
471
+ hands back 18,400 characters per question it answers, unranked and without
472
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
473
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
474
+ 8,300 — because a framework repeats its own vocabulary across many files and
475
+ Grep has no way to rank what it finds. Both sides decline the same five of
476
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
477
+ not a substring of English source and it is not a token in an index built
478
+ from English source, so on a repository written in one language the cold
479
+ cross-language case is not this tool's failure but the corpus's.
445
480
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
446
481
  exact, instant and complete, and nothing here replaces it.
447
482
 
@@ -506,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
506
541
  install gets, the import happens inside the constructor, and a test asserts the
507
542
  default provider imports none of it.
508
543
 
509
- **Measured, on the same four rulers.** This is the first release with a real
510
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
511
- benefit was unmeasured.
544
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
545
+ published this comparison and read it as a win. Its largest gain was on the
546
+ foreign ruler, whose two arms turned out to have been taken against two
547
+ different states of a repository being edited while the script ran. Repeated
548
+ against a pinned corpus:
512
549
 
513
- | ruler | signed hash (default) | MiniLM, local |
514
- |---|---|---|
515
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
516
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
517
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
518
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
550
+ | ruler | corpus | signed hash (default) | MiniLM, local |
551
+ |---|---|---|---|
552
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
553
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
554
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
555
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
519
556
 
520
- Better on every positive ruler, with refusal unchanged. The pairs the hash
521
- scores exactly zero:
557
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
558
+ has to install, because it does one thing the hash cannot do at all and these
559
+ rulers cannot see: reach a unit that shares no word with the question. The
560
+ pairs the hash scores exactly zero:
522
561
 
523
562
  | pair | signed hash | MiniLM |
524
563
  |---|---|---|
@@ -536,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
536
575
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
537
576
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
538
577
 
578
+ If you install it expecting the hit rates above to move, they will not. Install
579
+ it for the cross-language and paraphrase cases in the table above, which is
580
+ where the difference between the two columns actually lives.
581
+
539
582
  **A hosted endpoint** is the third option, and the only one that sends your
540
583
  source anywhere:
541
584
 
@@ -264,60 +264,83 @@ repository had grown by ninety units.
264
264
 
265
265
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
266
266
  |---|---|---|---|---|---|
267
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.371 | 0.286 |
267
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
268
268
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
269
269
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
270
270
 
271
- The corpora, without which none of the above is reproducible — **A**
272
- cc-enforcer, 1,345 units, `471b78f9f806`; **B** 579 units, `dca2a4656659`;
273
- **C** 579 units, `0a2f050dbafd`. Ruler A's repository is one this project does
274
- not control. Between the previous release and this table it grew by 78 units
275
- and renamed a declaration two questions pointed at, which stopped the ruler
276
- running at all until they were repointed — hit@3 0.400 → 0.371 and MRR 0.300 →
277
- 0.286 are that, not a change in retrieval.
271
+ The corpora, without which none of the above is reproducible — **A** 1,572
272
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
273
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
274
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
275
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
276
+ that cost three things: two questions pointed at a declaration the subject had
277
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
278
+ subject had grown, and the model comparison below was taken against two
279
+ different states of it. All three are now a `git clone` away from being
280
+ checked, and CI runs this ruler as an ordinary job.
278
281
 
279
282
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
280
283
 
281
- | | this repo | foreign repo |
284
+ | | this repo | Flask |
282
285
  |---|---|---|
283
- | correctly met with silence | **0.967** | **0.933** |
284
- | English only | **0.933** | **0.867** |
286
+ | correctly met with silence | **0.967** | **0.833** |
287
+ | English only | **0.933** | 0.667 |
285
288
  | Chinese only | **1.000** | **1.000** |
286
289
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
287
290
 
291
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
292
+ cause is a limit of the design rather than a defect. A word counts as evidence
293
+ unless it occurs in more than 5% of units — a stopword list derived from the
294
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
295
+ `does` and `are` are everywhere, because 304 units carry written English prose.
296
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
297
+ them and start counting as evidence. Five English questions about subjects
298
+ Flask does not implement get through on exactly that.
299
+
288
300
  **What each bar costs and buys** — one corpus, gate varied alone:
289
301
 
290
302
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
291
303
  |---|---|---|---|---|
292
- | neither (pre-1.0.0) | 0.229/0.371/0.286 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
293
- | coverage only (1.0.0) | 0.229/0.371/0.286 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.767 |
294
- | **both (1.1.0)** | **0.229/0.371/0.286** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
295
-
296
- Rulers A and B are **identical to three decimals**. The entire cost is three
297
- questions of seventy at hit@1 on the warmest ruler, and six at hit@3. On these
298
- four rulers concentration
299
- subsumes coverage — stated plainly because it is true; coverage is kept because
300
- it answers a different question and names a different diagnosis.
301
-
302
- **Latency** — warm corpus, 579 units `0a2f050dbafd`, five consecutive
304
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
305
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
306
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
307
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
308
+
309
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
310
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
311
+
312
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
313
+ this project did not choose, it does not.** Both bars together silence 0.833 of
314
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
315
+ for coverage alone. One question — and the first time in four releases that
316
+ keeping both has been worth a measurable amount rather than worth a different
317
+ diagnosis.
318
+
319
+ Raising the concentration bar buys the remaining silence, and is refused,
320
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
321
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
322
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
323
+ existed and survived meeting it**, which is the only kind of evidence a default
324
+ can have.
325
+
326
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
303
327
  invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
304
328
  each):
305
329
 
306
330
  | | | across the five |
307
331
  |---|---|---|
308
- | query, median | **0.65 ms** | 0.62 – 0.92 |
309
- | query, p95 | 1.13 ms | 1.09 – 1.48 |
310
- | refusing an unanswerable query | **0.022 ms** | 0.019 – 0.024 |
311
- | refusal cheaper than answering by | **~35×** | 31 – 39 |
332
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
333
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
334
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
335
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
312
336
 
313
337
  Two significant figures and a spread, because that is the precision the
314
- measurement has. A second set of five taken earlier the same hour, while the
315
- machine was busy, put the median at 1.0 ms and p95 at 3.0 ms; the gap between
316
- the two sets is the machine, and it is wider than any change the code has ever
317
- made to this number. Earlier releases published `0.83 ms / p95 1.68 ms` to
318
- three figures from a script that was never committed, on a corpus that no
319
- longer exists — both values sit inside today's range, which is the point: they
320
- were unfalsifiable rather than wrong.
338
+ measurement has. Across fifteen invocations over two releases on the same idle
339
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
340
+ to 7.34 ms — a band wider than any change the code has ever made to this
341
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
342
+ figures from a script that was never committed; both values sit inside that
343
+ band, which is the point: they were unfalsifiable rather than wrong.
321
344
 
322
345
  Refusal is cheap for a structural reason, not a tuned one: an unanswerable
323
346
  query touches only the posting lists of its own distinctive words, and never
@@ -346,8 +369,8 @@ reaches ranking at all.
346
369
  Directional local measurements, not service levels — but every one of them is
347
370
  now a command rather than a memory, which two of them were not before. Each
348
371
  prints the corpus fingerprint beside its score; quote both or neither.
349
- [`benchmarks/README.md`](benchmarks/README.md) lists the five scripts and what
350
- each is for.
372
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
373
+ each is for, and the corpus one of them grades is now carried here too.
351
374
 
352
375
  ## 7 · `rag-your-code search` vs a Grep loop
353
376
 
@@ -357,34 +380,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
357
380
  ruler, scored at **file** granularity so Grep is not penalised for lacking
358
381
  declaration spans.
359
382
 
360
- **On a repository nobody has described, Grep wins.** That is the measured
361
- result and it is not softened here.
383
+ **Which side wins on an undescribed repository depends on the repository.**
384
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
385
+ undescribed repository ever measured was a hook-heavy tool whose questions were
386
+ answerable by matching identifiers. Swapping the subject for a public web
387
+ framework reversed it. The honest claim is narrower than either table alone:
362
388
 
363
- | foreign repository · 35 questions · 1,345 units `471b78f9f806` · no descriptions | Grep loop | rag-your-code |
389
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
364
390
  |---|---|---|
365
- | right file first | **40.0%** | 28.6% |
366
- | right file in top 3 | **62.9%** | 48.6% |
367
- | lines it hands back, all questions | 8,937 | — |
368
- | characters returned, all questions | 841,484 | **282,102** |
369
- | questions it answers | **35** | 28 |
391
+ | right file first | 22.9% | **37.1%** |
392
+ | right file in top 3 | 45.7% | **57.1%** |
393
+ | lines it hands back, all questions | 17,641 | — |
394
+ | characters returned, all questions | 1,415,656 | **249,720** |
395
+ | questions it answers | 30 | 30 |
370
396
 
371
397
  **Once the vocabulary exists, it is not close.**
372
398
 
373
- | this repository · 70 questions · 579 units `0a2f050dbafd` · 304 described | Grep loop | rag-your-code |
399
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
374
400
  |---|---|---|
375
401
  | right file first | 22.9% | **58.6%** |
376
- | right file in top 3 | 55.7% | **75.7%** |
377
- | lines it hands back, all questions | 11,636 | — |
378
- | characters returned, all questions | 1,101,618 | **621,837** |
402
+ | right file in top 3 | 54.3% | **75.7%** |
403
+ | lines it hands back, all questions | 11,833 | — |
404
+ | characters returned, all questions | 1,122,902 | **626,022** |
379
405
  | questions it answers | **61** | 60 |
380
406
 
381
407
  Those two tables are the whole argument of section 3.3, measured against a real
382
408
  baseline instead of asserted. A cold index retrieves against a sentence the
383
- parser generated from identifiers the author already chose — so it is competing
384
- with Grep using Grep's own information, and losing, because Grep does not have
385
- to guess which of the matching files is the definition. Descriptions put words
386
- in the index that the source never contained, and first-place accuracy goes
387
- from below Grep's to **more than double** it.
409
+ parser generated from identifiers the author already chose, plus whatever
410
+ docstrings the author wrote — so how it fares against Grep is decided by how
411
+ much prose the repository already contains. Flask has a written docstring on
412
+ most public methods, and the cold index beats Grep there without a single
413
+ description being added. On the previous subject, a tool with terse comments
414
+ and long identifiers, the same cold index lost to Grep by the same margin.
415
+
416
+ What does not depend on the subject is what descriptions buy: on this
417
+ repository first-place accuracy goes to **more than double** Grep's, and the
418
+ payload comes back ranked, spanned, and roughly half the size.
388
419
 
389
420
  **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
390
421
  changed in 1.3.0. Until then this section — the strongest claim the project
@@ -407,12 +438,16 @@ Four qualifications, because the table would otherwise flatter both sides:
407
438
  `the` gets every file back in no order. It is also why Grep declines nine of
408
439
  the seventy questions here — those had no word left that this corpus does not
409
440
  use everywhere.
410
- - **Payload is counted in characters on both sides.** Grep hands back 18,000
411
- characters per question it answers, unranked and without spans; this returns
412
- 10,400, ranked, capped by `search.max_chars`. That is a factor of 1.7, not
413
- the order of magnitude an earlier version of this table implied by counting
414
- one side in lines and the other in characters. Seven of 35 and ten of 70
415
- questions come back empty here instead, with a reason.
441
+ - **Payload is counted in characters on both sides.** On this repository Grep
442
+ hands back 18,400 characters per question it answers, unranked and without
443
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
444
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
445
+ 8,300 — because a framework repeats its own vocabulary across many files and
446
+ Grep has no way to rank what it finds. Both sides decline the same five of
447
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
448
+ not a substring of English source and it is not a token in an index built
449
+ from English source, so on a repository written in one language the cold
450
+ cross-language case is not this tool's failure but the corpus's.
416
451
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
417
452
  exact, instant and complete, and nothing here replaces it.
418
453
 
@@ -477,19 +512,23 @@ The extra is optional by construction: `dependencies = []` is what a default
477
512
  install gets, the import happens inside the constructor, and a test asserts the
478
513
  default provider imports none of it.
479
514
 
480
- **Measured, on the same four rulers.** This is the first release with a real
481
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
482
- benefit was unmeasured.
515
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
516
+ published this comparison and read it as a win. Its largest gain was on the
517
+ foreign ruler, whose two arms turned out to have been taken against two
518
+ different states of a repository being edited while the script ran. Repeated
519
+ against a pinned corpus:
483
520
 
484
- | ruler | signed hash (default) | MiniLM, local |
485
- |---|---|---|
486
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
487
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
488
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
489
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
521
+ | ruler | corpus | signed hash (default) | MiniLM, local |
522
+ |---|---|---|---|
523
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
524
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
525
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
526
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
490
527
 
491
- Better on every positive ruler, with refusal unchanged. The pairs the hash
492
- scores exactly zero:
528
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
529
+ has to install, because it does one thing the hash cannot do at all and these
530
+ rulers cannot see: reach a unit that shares no word with the question. The
531
+ pairs the hash scores exactly zero:
493
532
 
494
533
  | pair | signed hash | MiniLM |
495
534
  |---|---|---|
@@ -507,6 +546,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
507
546
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
508
547
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
509
548
 
549
+ If you install it expecting the hit rates above to move, they will not. Install
550
+ it for the cross-language and paraphrase cases in the table above, which is
551
+ where the difference between the two columns actually lives.
552
+
510
553
  **A hosted endpoint** is the third option, and the only one that sends your
511
554
  source anywhere:
512
555
 
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "1.3.0"
9
+ version = "1.4.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.3.0
3
+ Version: 1.4.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -293,60 +293,83 @@ repository had grown by ninety units.
293
293
 
294
294
  | ruler | what it represents | n | hit@1 | hit@3 | MRR |
295
295
  |---|---|---|---|---|---|
296
- | **A** foreign repo, no descriptions | what a first-time user gets | 35 | 0.229 | 0.371 | 0.286 |
296
+ | **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
297
297
  | **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
298
298
  | **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
299
299
 
300
- The corpora, without which none of the above is reproducible — **A**
301
- cc-enforcer, 1,345 units, `471b78f9f806`; **B** 579 units, `dca2a4656659`;
302
- **C** 579 units, `0a2f050dbafd`. Ruler A's repository is one this project does
303
- not control. Between the previous release and this table it grew by 78 units
304
- and renamed a declaration two questions pointed at, which stopped the ruler
305
- running at all until they were repointed — hit@3 0.400 → 0.371 and MRR 0.300 →
306
- 0.286 are that, not a change in retrieval.
300
+ The corpora, without which none of the above is reproducible — **A** 1,572
301
+ units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
302
+ `978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
303
+ repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
304
+ to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
305
+ that cost three things: two questions pointed at a declaration the subject had
306
+ renamed, a published score moved 0.257 → 0.229 with no code change because the
307
+ subject had grown, and the model comparison below was taken against two
308
+ different states of it. All three are now a `git clone` away from being
309
+ checked, and CI runs this ruler as an ordinary job.
307
310
 
308
311
  **Refusal — the fourth ruler, 30 questions with no answer anywhere**
309
312
 
310
- | | this repo | foreign repo |
313
+ | | this repo | Flask |
311
314
  |---|---|---|
312
- | correctly met with silence | **0.967** | **0.933** |
313
- | English only | **0.933** | **0.867** |
315
+ | correctly met with silence | **0.967** | **0.833** |
316
+ | English only | **0.933** | 0.667 |
314
317
  | Chinese only | **1.000** | **1.000** |
315
318
  | results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
316
319
 
320
+ Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
321
+ cause is a limit of the design rather than a defect. A word counts as evidence
322
+ unless it occurs in more than 5% of units — a stopword list derived from the
323
+ corpus, so that it needs no list and works in any language. Here `how`, `when`,
324
+ `does` and `are` are everywhere, because 304 units carry written English prose.
325
+ Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
326
+ them and start counting as evidence. Five English questions about subjects
327
+ Flask does not implement get through on exactly that.
328
+
317
329
  **What each bar costs and buys** — one corpus, gate varied alone:
318
330
 
319
331
  | gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
320
332
  |---|---|---|---|---|
321
- | neither (pre-1.0.0) | 0.229/0.371/0.286 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
322
- | coverage only (1.0.0) | 0.229/0.371/0.286 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.767 |
323
- | **both (1.1.0)** | **0.229/0.371/0.286** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.933** |
324
-
325
- Rulers A and B are **identical to three decimals**. The entire cost is three
326
- questions of seventy at hit@1 on the warmest ruler, and six at hit@3. On these
327
- four rulers concentration
328
- subsumes coverage — stated plainly because it is true; coverage is kept because
329
- it answers a different question and names a different diagnosis.
330
-
331
- **Latency** — warm corpus, 579 units `0a2f050dbafd`, five consecutive
333
+ | neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
334
+ | coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
335
+ | concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
336
+ | **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
337
+
338
+ Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
339
+ three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
340
+
341
+ Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
342
+ this project did not choose, it does not.** Both bars together silence 0.833 of
343
+ the foreign absent questions, against 0.800 for concentration alone and 0.733
344
+ for coverage alone. One question — and the first time in four releases that
345
+ keeping both has been worth a measurable amount rather than worth a different
346
+ diagnosis.
347
+
348
+ Raising the concentration bar buys the remaining silence, and is refused,
349
+ because it is bought out of the answers: at 0.50 the foreign absent ruler is
350
+ silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
351
+ from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
352
+ existed and survived meeting it**, which is the only kind of evidence a default
353
+ can have.
354
+
355
+ **Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
332
356
  invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
333
357
  each):
334
358
 
335
359
  | | | across the five |
336
360
  |---|---|---|
337
- | query, median | **0.65 ms** | 0.62 – 0.92 |
338
- | query, p95 | 1.13 ms | 1.09 – 1.48 |
339
- | refusing an unanswerable query | **0.022 ms** | 0.019 – 0.024 |
340
- | refusal cheaper than answering by | **~35×** | 31 – 39 |
361
+ | query, median | **1.0 ms** | 0.71 – 1.26 |
362
+ | query, p95 | 1.7 ms | 1.2 – 3.7 |
363
+ | refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
364
+ | refusal cheaper than answering by | **~36×** | 31 – 44 |
341
365
 
342
366
  Two significant figures and a spread, because that is the precision the
343
- measurement has. A second set of five taken earlier the same hour, while the
344
- machine was busy, put the median at 1.0 ms and p95 at 3.0 ms; the gap between
345
- the two sets is the machine, and it is wider than any change the code has ever
346
- made to this number. Earlier releases published `0.83 ms / p95 1.68 ms` to
347
- three figures from a script that was never committed, on a corpus that no
348
- longer exists — both values sit inside today's range, which is the point: they
349
- were unfalsifiable rather than wrong.
367
+ measurement has. Across fifteen invocations over two releases on the same idle
368
+ machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
369
+ to 7.34 ms — a band wider than any change the code has ever made to this
370
+ number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
371
+ figures from a script that was never committed; both values sit inside that
372
+ band, which is the point: they were unfalsifiable rather than wrong.
350
373
 
351
374
  Refusal is cheap for a structural reason, not a tuned one: an unanswerable
352
375
  query touches only the posting lists of its own distinctive words, and never
@@ -375,8 +398,8 @@ reaches ranking at all.
375
398
  Directional local measurements, not service levels — but every one of them is
376
399
  now a command rather than a memory, which two of them were not before. Each
377
400
  prints the corpus fingerprint beside its score; quote both or neither.
378
- [`benchmarks/README.md`](benchmarks/README.md) lists the five scripts and what
379
- each is for.
401
+ [`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
402
+ each is for, and the corpus one of them grades is now carried here too.
380
403
 
381
404
  ## 7 · `rag-your-code search` vs a Grep loop
382
405
 
@@ -386,34 +409,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
386
409
  ruler, scored at **file** granularity so Grep is not penalised for lacking
387
410
  declaration spans.
388
411
 
389
- **On a repository nobody has described, Grep wins.** That is the measured
390
- result and it is not softened here.
412
+ **Which side wins on an undescribed repository depends on the repository.**
413
+ Through 1.3.0 this section said flatly that Grep wins there, because the one
414
+ undescribed repository ever measured was a hook-heavy tool whose questions were
415
+ answerable by matching identifiers. Swapping the subject for a public web
416
+ framework reversed it. The honest claim is narrower than either table alone:
391
417
 
392
- | foreign repository · 35 questions · 1,345 units `471b78f9f806` · no descriptions | Grep loop | rag-your-code |
418
+ | Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
393
419
  |---|---|---|
394
- | right file first | **40.0%** | 28.6% |
395
- | right file in top 3 | **62.9%** | 48.6% |
396
- | lines it hands back, all questions | 8,937 | — |
397
- | characters returned, all questions | 841,484 | **282,102** |
398
- | questions it answers | **35** | 28 |
420
+ | right file first | 22.9% | **37.1%** |
421
+ | right file in top 3 | 45.7% | **57.1%** |
422
+ | lines it hands back, all questions | 17,641 | — |
423
+ | characters returned, all questions | 1,415,656 | **249,720** |
424
+ | questions it answers | 30 | 30 |
399
425
 
400
426
  **Once the vocabulary exists, it is not close.**
401
427
 
402
- | this repository · 70 questions · 579 units `0a2f050dbafd` · 304 described | Grep loop | rag-your-code |
428
+ | this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
403
429
  |---|---|---|
404
430
  | right file first | 22.9% | **58.6%** |
405
- | right file in top 3 | 55.7% | **75.7%** |
406
- | lines it hands back, all questions | 11,636 | — |
407
- | characters returned, all questions | 1,101,618 | **621,837** |
431
+ | right file in top 3 | 54.3% | **75.7%** |
432
+ | lines it hands back, all questions | 11,833 | — |
433
+ | characters returned, all questions | 1,122,902 | **626,022** |
408
434
  | questions it answers | **61** | 60 |
409
435
 
410
436
  Those two tables are the whole argument of section 3.3, measured against a real
411
437
  baseline instead of asserted. A cold index retrieves against a sentence the
412
- parser generated from identifiers the author already chose — so it is competing
413
- with Grep using Grep's own information, and losing, because Grep does not have
414
- to guess which of the matching files is the definition. Descriptions put words
415
- in the index that the source never contained, and first-place accuracy goes
416
- from below Grep's to **more than double** it.
438
+ parser generated from identifiers the author already chose, plus whatever
439
+ docstrings the author wrote — so how it fares against Grep is decided by how
440
+ much prose the repository already contains. Flask has a written docstring on
441
+ most public methods, and the cold index beats Grep there without a single
442
+ description being added. On the previous subject, a tool with terse comments
443
+ and long identifiers, the same cold index lost to Grep by the same margin.
444
+
445
+ What does not depend on the subject is what descriptions buy: on this
446
+ repository first-place accuracy goes to **more than double** Grep's, and the
447
+ payload comes back ranked, spanned, and roughly half the size.
417
448
 
418
449
  **Both tables come from `python -m benchmarks.grep_baseline`**, which is what
419
450
  changed in 1.3.0. Until then this section — the strongest claim the project
@@ -436,12 +467,16 @@ Four qualifications, because the table would otherwise flatter both sides:
436
467
  `the` gets every file back in no order. It is also why Grep declines nine of
437
468
  the seventy questions here — those had no word left that this corpus does not
438
469
  use everywhere.
439
- - **Payload is counted in characters on both sides.** Grep hands back 18,000
440
- characters per question it answers, unranked and without spans; this returns
441
- 10,400, ranked, capped by `search.max_chars`. That is a factor of 1.7, not
442
- the order of magnitude an earlier version of this table implied by counting
443
- one side in lines and the other in characters. Seven of 35 and ten of 70
444
- questions come back empty here instead, with a reason.
470
+ - **Payload is counted in characters on both sides.** On this repository Grep
471
+ hands back 18,400 characters per question it answers, unranked and without
472
+ spans, against 10,400 here, ranked and capped by `search.max_chars` — a
473
+ factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
474
+ 8,300 — because a framework repeats its own vocabulary across many files and
475
+ Grep has no way to rank what it finds. Both sides decline the same five of
476
+ those 35 — and they are the five Chinese ones, all of them. A Chinese word is
477
+ not a substring of English source and it is not a token in an index built
478
+ from English source, so on a repository written in one language the cold
479
+ cross-language case is not this tool's failure but the corpus's.
445
480
  - **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
446
481
  exact, instant and complete, and nothing here replaces it.
447
482
 
@@ -506,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
506
541
  install gets, the import happens inside the constructor, and a test asserts the
507
542
  default provider imports none of it.
508
543
 
509
- **Measured, on the same four rulers.** This is the first release with a real
510
- model behind these numbers — 0.8.0 shipped the seam and said plainly that its
511
- benefit was unmeasured.
544
+ **Measured on the same four rulers, both arms against one corpus.** 1.1.0
545
+ published this comparison and read it as a win. Its largest gain was on the
546
+ foreign ruler, whose two arms turned out to have been taken against two
547
+ different states of a repository being edited while the script ran. Repeated
548
+ against a pinned corpus:
512
549
 
513
- | ruler | signed hash (default) | MiniLM, local |
514
- |---|---|---|
515
- | **A** foreign, cold | 0.229 / 0.400 / 0.300 | **0.286 / 0.457 / 0.357** |
516
- | **B** own, cold | 0.314 / 0.471 / 0.383 | **0.329 / 0.486 / 0.400** |
517
- | **C** own, described | 0.443 / 0.614 / 0.507 | 0.443 / **0.671 / 0.540** |
518
- | **D** silence, own / foreign | 0.967 / 0.933 | 0.967 / 0.933 |
550
+ | ruler | corpus | signed hash (default) | MiniLM, local |
551
+ |---|---|---|---|
552
+ | **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
553
+ | **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
554
+ | **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
555
+ | **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
519
556
 
520
- Better on every positive ruler, with refusal unchanged. The pairs the hash
521
- scores exactly zero:
557
+ **Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
558
+ has to install, because it does one thing the hash cannot do at all and these
559
+ rulers cannot see: reach a unit that shares no word with the question. The
560
+ pairs the hash scores exactly zero:
522
561
 
523
562
  | pair | signed hash | MiniLM |
524
563
  |---|---|---|
@@ -536,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
536
575
  and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
537
576
  thirds of the silence. Applying the lexical bars costs ruler A nothing.
538
577
 
578
+ If you install it expecting the hit rates above to move, they will not. Install
579
+ it for the cross-language and paraphrase cases in the table above, which is
580
+ where the difference between the two columns actually lives.
581
+
539
582
  **A hosted endpoint** is the third option, and the only one that sends your
540
583
  source anywhere:
541
584
 
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "1.3.0"
6
+ __version__ = "1.4.0"
@@ -53,23 +53,56 @@ def test_the_absent_ruler_is_well_formed():
53
53
  assert questions["why"].strip() and questions["caveat"].strip()
54
54
 
55
55
 
56
- def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
57
- """The absence claim, re-derived rather than trusted.
58
-
59
- This is the assertion the ruler's own caveat promises, and it is the one
60
- that fails first when the repository grows into a subject the ruler
61
- assumed it would never contain.
56
+ def _intruders(units) -> list[tuple[str, str, int]]:
57
+ """Which absent question's own vocabulary reaches a corpus that must not
58
+ contain it.
62
59
  """
63
60
  index = build_search_index(units)
64
- intruders = [
61
+ return [
65
62
  (entry["id"], term, len(index.postings[term]))
66
63
  for entry in load_questions(ABSENT_PATH)["queries"]
67
64
  for term in entry["subject"]
68
65
  if index.postings.get(term)
69
66
  ]
70
- assert not intruders, (
67
+
68
+
69
+ def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
70
+ """The absence claim, re-derived rather than trusted.
71
+
72
+ This is the assertion the ruler's own caveat promises, and it is the one
73
+ that fails first when the repository grows into a subject the ruler
74
+ assumed it would never contain.
75
+ """
76
+ assert not _intruders(units), (
71
77
  "this repository now contains the vocabulary of a question the ruler calls unanswerable; "
72
- f"retire or rewrite those questions rather than letting them score as misses: {intruders}"
78
+ f"retire or rewrite those questions rather than letting them score as misses: {_intruders(units)}"
79
+ )
80
+
81
+
82
+ def test_no_subject_of_an_absent_question_exists_in_the_vendored_corpus():
83
+ """The other half of the same claim, which used to be unenforceable.
84
+
85
+ This ruler asserts its questions are unanswerable in *both* graded
86
+ repositories. Until the foreign one was vendored that half could only be
87
+ asserted, because the subject was somebody else's checkout and a re-run
88
+ here could not see what it had become. It is now a directory in this
89
+ repository at a pinned tag, so the claim is a test.
90
+
91
+ It has already earned its place. One question's subject list carried a
92
+ generic English verb where a specific term belonged, and the corpus below
93
+ uses that verb in a comment about signing keys -- so the question was one
94
+ ordinary word away from being answerable by accident. The verb was
95
+ replaced with a term that names the thing.
96
+
97
+ Naming the subject here would put it in the index and break the very
98
+ claim this asserts, which is why the paragraph above is written around it.
99
+ That has now happened five times; CONTRIBUTING.md keeps the tally.
100
+ """
101
+ corpus = ROOT / "benchmarks" / "corpus" / "flask"
102
+ assert corpus.is_dir(), f"the vendored corpus is missing from {corpus}"
103
+ assert not _intruders(build_units(corpus)), (
104
+ "the vendored corpus contains the vocabulary of a question the ruler calls unanswerable; "
105
+ f"rewrite the question rather than letting it score as a miss: {_intruders(build_units(corpus))}"
73
106
  )
74
107
 
75
108
 
File without changes
File without changes