rag-your-code 1.3.0__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.3.0/src/rag_your_code.egg-info → rag_your_code-1.4.0}/PKG-INFO +113 -70
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/README.md +112 -69
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/pyproject.toml +1 -1
- {rag_your_code-1.3.0 → rag_your_code-1.4.0/src/rag_your_code.egg-info}/PKG-INFO +113 -70
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_absent_queries.py +42 -9
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/LICENSE +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/setup.cfg +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/cli.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/config.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/search.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/src/ragyourcode/workflow.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_agentic.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_config.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_document.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_evidence.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_golden.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_local_model.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_metadata.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_providers.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_ranking.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_resilience.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.3.0 → rag_your_code-1.4.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -293,60 +293,83 @@ repository had grown by ninety units.
|
|
|
293
293
|
|
|
294
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
295
295
|
|---|---|---|---|---|---|
|
|
296
|
-
| **A**
|
|
296
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
297
297
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
298
298
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
299
299
|
|
|
300
|
-
The corpora, without which none of the above is reproducible — **A**
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
0.
|
|
300
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
301
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
302
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
303
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
304
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
305
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
306
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
307
|
+
subject had grown, and the model comparison below was taken against two
|
|
308
|
+
different states of it. All three are now a `git clone` away from being
|
|
309
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
307
310
|
|
|
308
311
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
309
312
|
|
|
310
|
-
| | this repo |
|
|
313
|
+
| | this repo | Flask |
|
|
311
314
|
|---|---|---|
|
|
312
|
-
| correctly met with silence | **0.967** | **0.
|
|
313
|
-
| English only | **0.933** |
|
|
315
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
316
|
+
| English only | **0.933** | 0.667 |
|
|
314
317
|
| Chinese only | **1.000** | **1.000** |
|
|
315
318
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
316
319
|
|
|
320
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
321
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
322
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
323
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
324
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
325
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
326
|
+
them and start counting as evidence. Five English questions about subjects
|
|
327
|
+
Flask does not implement get through on exactly that.
|
|
328
|
+
|
|
317
329
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
318
330
|
|
|
319
331
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
320
332
|
|---|---|---|---|---|
|
|
321
|
-
| neither (pre-1.0.0) | 0.
|
|
322
|
-
| coverage only (1.0.0) | 0.
|
|
323
|
-
|
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
333
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
334
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
335
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
336
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
337
|
+
|
|
338
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
339
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
340
|
+
|
|
341
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
342
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
343
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
344
|
+
for coverage alone. One question — and the first time in four releases that
|
|
345
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
346
|
+
diagnosis.
|
|
347
|
+
|
|
348
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
349
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
350
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
351
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
352
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
353
|
+
can have.
|
|
354
|
+
|
|
355
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
332
356
|
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
333
357
|
each):
|
|
334
358
|
|
|
335
359
|
| | | across the five |
|
|
336
360
|
|---|---|---|
|
|
337
|
-
| query, median | **0
|
|
338
|
-
| query, p95 | 1.
|
|
339
|
-
| refusing an unanswerable query | **0.
|
|
340
|
-
| refusal cheaper than answering by | **~
|
|
361
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
362
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
363
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
364
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
341
365
|
|
|
342
366
|
Two significant figures and a spread, because that is the precision the
|
|
343
|
-
measurement has.
|
|
344
|
-
machine
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
were unfalsifiable rather than wrong.
|
|
367
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
368
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
369
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
370
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
371
|
+
figures from a script that was never committed; both values sit inside that
|
|
372
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
350
373
|
|
|
351
374
|
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
352
375
|
query touches only the posting lists of its own distinctive words, and never
|
|
@@ -375,8 +398,8 @@ reaches ranking at all.
|
|
|
375
398
|
Directional local measurements, not service levels — but every one of them is
|
|
376
399
|
now a command rather than a memory, which two of them were not before. Each
|
|
377
400
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
378
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
379
|
-
each is for.
|
|
401
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
402
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
380
403
|
|
|
381
404
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
382
405
|
|
|
@@ -386,34 +409,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
386
409
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
387
410
|
declaration spans.
|
|
388
411
|
|
|
389
|
-
**
|
|
390
|
-
|
|
412
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
413
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
414
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
415
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
416
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
391
417
|
|
|
392
|
-
|
|
|
418
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
393
419
|
|---|---|---|
|
|
394
|
-
| right file first |
|
|
395
|
-
| right file in top 3 |
|
|
396
|
-
| lines it hands back, all questions |
|
|
397
|
-
| characters returned, all questions |
|
|
398
|
-
| questions it answers |
|
|
420
|
+
| right file first | 22.9% | **37.1%** |
|
|
421
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
422
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
423
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
424
|
+
| questions it answers | 30 | 30 |
|
|
399
425
|
|
|
400
426
|
**Once the vocabulary exists, it is not close.**
|
|
401
427
|
|
|
402
|
-
| this repository · 70 questions ·
|
|
428
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
403
429
|
|---|---|---|
|
|
404
430
|
| right file first | 22.9% | **58.6%** |
|
|
405
|
-
| right file in top 3 |
|
|
406
|
-
| lines it hands back, all questions | 11,
|
|
407
|
-
| characters returned, all questions | 1,
|
|
431
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
432
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
433
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
408
434
|
| questions it answers | **61** | 60 |
|
|
409
435
|
|
|
410
436
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
411
437
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
412
|
-
parser generated from identifiers the author already chose
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
438
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
439
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
440
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
441
|
+
most public methods, and the cold index beats Grep there without a single
|
|
442
|
+
description being added. On the previous subject, a tool with terse comments
|
|
443
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
444
|
+
|
|
445
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
446
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
447
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
417
448
|
|
|
418
449
|
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
419
450
|
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
@@ -436,12 +467,16 @@ Four qualifications, because the table would otherwise flatter both sides:
|
|
|
436
467
|
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
437
468
|
the seventy questions here — those had no word left that this corpus does not
|
|
438
469
|
use everywhere.
|
|
439
|
-
- **Payload is counted in characters on both sides.**
|
|
440
|
-
characters per question it answers, unranked and without
|
|
441
|
-
10,400, ranked
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
470
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
471
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
472
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
473
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
474
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
475
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
476
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
477
|
+
not a substring of English source and it is not a token in an index built
|
|
478
|
+
from English source, so on a repository written in one language the cold
|
|
479
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
445
480
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
446
481
|
exact, instant and complete, and nothing here replaces it.
|
|
447
482
|
|
|
@@ -506,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
506
541
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
507
542
|
default provider imports none of it.
|
|
508
543
|
|
|
509
|
-
**Measured
|
|
510
|
-
|
|
511
|
-
|
|
544
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
545
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
546
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
547
|
+
different states of a repository being edited while the script ran. Repeated
|
|
548
|
+
against a pinned corpus:
|
|
512
549
|
|
|
513
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
514
|
-
|
|
515
|
-
| **A** foreign, cold | 0.
|
|
516
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
517
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
518
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
550
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
551
|
+
|---|---|---|---|
|
|
552
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
553
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
554
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
555
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
519
556
|
|
|
520
|
-
|
|
521
|
-
|
|
557
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
558
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
559
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
560
|
+
pairs the hash scores exactly zero:
|
|
522
561
|
|
|
523
562
|
| pair | signed hash | MiniLM |
|
|
524
563
|
|---|---|---|
|
|
@@ -536,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
536
575
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
537
576
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
538
577
|
|
|
578
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
579
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
580
|
+
where the difference between the two columns actually lives.
|
|
581
|
+
|
|
539
582
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
540
583
|
source anywhere:
|
|
541
584
|
|
|
@@ -264,60 +264,83 @@ repository had grown by ninety units.
|
|
|
264
264
|
|
|
265
265
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
266
266
|
|---|---|---|---|---|---|
|
|
267
|
-
| **A**
|
|
267
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
268
268
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
269
269
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
270
270
|
|
|
271
|
-
The corpora, without which none of the above is reproducible — **A**
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
0.
|
|
271
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
272
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
273
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
274
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
275
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
276
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
277
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
278
|
+
subject had grown, and the model comparison below was taken against two
|
|
279
|
+
different states of it. All three are now a `git clone` away from being
|
|
280
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
278
281
|
|
|
279
282
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
280
283
|
|
|
281
|
-
| | this repo |
|
|
284
|
+
| | this repo | Flask |
|
|
282
285
|
|---|---|---|
|
|
283
|
-
| correctly met with silence | **0.967** | **0.
|
|
284
|
-
| English only | **0.933** |
|
|
286
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
287
|
+
| English only | **0.933** | 0.667 |
|
|
285
288
|
| Chinese only | **1.000** | **1.000** |
|
|
286
289
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
287
290
|
|
|
291
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
292
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
293
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
294
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
295
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
296
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
297
|
+
them and start counting as evidence. Five English questions about subjects
|
|
298
|
+
Flask does not implement get through on exactly that.
|
|
299
|
+
|
|
288
300
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
289
301
|
|
|
290
302
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
291
303
|
|---|---|---|---|---|
|
|
292
|
-
| neither (pre-1.0.0) | 0.
|
|
293
|
-
| coverage only (1.0.0) | 0.
|
|
294
|
-
|
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
304
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
305
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
306
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
307
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
308
|
+
|
|
309
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
310
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
311
|
+
|
|
312
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
313
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
314
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
315
|
+
for coverage alone. One question — and the first time in four releases that
|
|
316
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
317
|
+
diagnosis.
|
|
318
|
+
|
|
319
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
320
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
321
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
322
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
323
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
324
|
+
can have.
|
|
325
|
+
|
|
326
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
303
327
|
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
304
328
|
each):
|
|
305
329
|
|
|
306
330
|
| | | across the five |
|
|
307
331
|
|---|---|---|
|
|
308
|
-
| query, median | **0
|
|
309
|
-
| query, p95 | 1.
|
|
310
|
-
| refusing an unanswerable query | **0.
|
|
311
|
-
| refusal cheaper than answering by | **~
|
|
332
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
333
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
334
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
335
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
312
336
|
|
|
313
337
|
Two significant figures and a spread, because that is the precision the
|
|
314
|
-
measurement has.
|
|
315
|
-
machine
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
were unfalsifiable rather than wrong.
|
|
338
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
339
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
340
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
341
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
342
|
+
figures from a script that was never committed; both values sit inside that
|
|
343
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
321
344
|
|
|
322
345
|
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
323
346
|
query touches only the posting lists of its own distinctive words, and never
|
|
@@ -346,8 +369,8 @@ reaches ranking at all.
|
|
|
346
369
|
Directional local measurements, not service levels — but every one of them is
|
|
347
370
|
now a command rather than a memory, which two of them were not before. Each
|
|
348
371
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
349
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
350
|
-
each is for.
|
|
372
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
373
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
351
374
|
|
|
352
375
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
353
376
|
|
|
@@ -357,34 +380,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
357
380
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
358
381
|
declaration spans.
|
|
359
382
|
|
|
360
|
-
**
|
|
361
|
-
|
|
383
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
384
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
385
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
386
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
387
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
362
388
|
|
|
363
|
-
|
|
|
389
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
364
390
|
|---|---|---|
|
|
365
|
-
| right file first |
|
|
366
|
-
| right file in top 3 |
|
|
367
|
-
| lines it hands back, all questions |
|
|
368
|
-
| characters returned, all questions |
|
|
369
|
-
| questions it answers |
|
|
391
|
+
| right file first | 22.9% | **37.1%** |
|
|
392
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
393
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
394
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
395
|
+
| questions it answers | 30 | 30 |
|
|
370
396
|
|
|
371
397
|
**Once the vocabulary exists, it is not close.**
|
|
372
398
|
|
|
373
|
-
| this repository · 70 questions ·
|
|
399
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
374
400
|
|---|---|---|
|
|
375
401
|
| right file first | 22.9% | **58.6%** |
|
|
376
|
-
| right file in top 3 |
|
|
377
|
-
| lines it hands back, all questions | 11,
|
|
378
|
-
| characters returned, all questions | 1,
|
|
402
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
403
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
404
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
379
405
|
| questions it answers | **61** | 60 |
|
|
380
406
|
|
|
381
407
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
382
408
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
383
|
-
parser generated from identifiers the author already chose
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
409
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
410
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
411
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
412
|
+
most public methods, and the cold index beats Grep there without a single
|
|
413
|
+
description being added. On the previous subject, a tool with terse comments
|
|
414
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
415
|
+
|
|
416
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
417
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
418
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
388
419
|
|
|
389
420
|
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
390
421
|
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
@@ -407,12 +438,16 @@ Four qualifications, because the table would otherwise flatter both sides:
|
|
|
407
438
|
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
408
439
|
the seventy questions here — those had no word left that this corpus does not
|
|
409
440
|
use everywhere.
|
|
410
|
-
- **Payload is counted in characters on both sides.**
|
|
411
|
-
characters per question it answers, unranked and without
|
|
412
|
-
10,400, ranked
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
441
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
442
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
443
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
444
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
445
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
446
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
447
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
448
|
+
not a substring of English source and it is not a token in an index built
|
|
449
|
+
from English source, so on a repository written in one language the cold
|
|
450
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
416
451
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
417
452
|
exact, instant and complete, and nothing here replaces it.
|
|
418
453
|
|
|
@@ -477,19 +512,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
477
512
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
478
513
|
default provider imports none of it.
|
|
479
514
|
|
|
480
|
-
**Measured
|
|
481
|
-
|
|
482
|
-
|
|
515
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
516
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
517
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
518
|
+
different states of a repository being edited while the script ran. Repeated
|
|
519
|
+
against a pinned corpus:
|
|
483
520
|
|
|
484
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
485
|
-
|
|
486
|
-
| **A** foreign, cold | 0.
|
|
487
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
488
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
489
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
521
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
522
|
+
|---|---|---|---|
|
|
523
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
524
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
525
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
526
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
490
527
|
|
|
491
|
-
|
|
492
|
-
|
|
528
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
529
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
530
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
531
|
+
pairs the hash scores exactly zero:
|
|
493
532
|
|
|
494
533
|
| pair | signed hash | MiniLM |
|
|
495
534
|
|---|---|---|
|
|
@@ -507,6 +546,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
507
546
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
508
547
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
509
548
|
|
|
549
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
550
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
551
|
+
where the difference between the two columns actually lives.
|
|
552
|
+
|
|
510
553
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
511
554
|
source anywhere:
|
|
512
555
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -293,60 +293,83 @@ repository had grown by ninety units.
|
|
|
293
293
|
|
|
294
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
295
295
|
|---|---|---|---|---|---|
|
|
296
|
-
| **A**
|
|
296
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
297
297
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
298
298
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
299
299
|
|
|
300
|
-
The corpora, without which none of the above is reproducible — **A**
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
0.
|
|
300
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
301
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
302
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
303
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
304
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
305
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
306
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
307
|
+
subject had grown, and the model comparison below was taken against two
|
|
308
|
+
different states of it. All three are now a `git clone` away from being
|
|
309
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
307
310
|
|
|
308
311
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
309
312
|
|
|
310
|
-
| | this repo |
|
|
313
|
+
| | this repo | Flask |
|
|
311
314
|
|---|---|---|
|
|
312
|
-
| correctly met with silence | **0.967** | **0.
|
|
313
|
-
| English only | **0.933** |
|
|
315
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
316
|
+
| English only | **0.933** | 0.667 |
|
|
314
317
|
| Chinese only | **1.000** | **1.000** |
|
|
315
318
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
316
319
|
|
|
320
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
321
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
322
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
323
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
324
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
325
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
326
|
+
them and start counting as evidence. Five English questions about subjects
|
|
327
|
+
Flask does not implement get through on exactly that.
|
|
328
|
+
|
|
317
329
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
318
330
|
|
|
319
331
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
320
332
|
|---|---|---|---|---|
|
|
321
|
-
| neither (pre-1.0.0) | 0.
|
|
322
|
-
| coverage only (1.0.0) | 0.
|
|
323
|
-
|
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
333
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
334
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
335
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
336
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
337
|
+
|
|
338
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
339
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
340
|
+
|
|
341
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
342
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
343
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
344
|
+
for coverage alone. One question — and the first time in four releases that
|
|
345
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
346
|
+
diagnosis.
|
|
347
|
+
|
|
348
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
349
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
350
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
351
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
352
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
353
|
+
can have.
|
|
354
|
+
|
|
355
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
332
356
|
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
333
357
|
each):
|
|
334
358
|
|
|
335
359
|
| | | across the five |
|
|
336
360
|
|---|---|---|
|
|
337
|
-
| query, median | **0
|
|
338
|
-
| query, p95 | 1.
|
|
339
|
-
| refusing an unanswerable query | **0.
|
|
340
|
-
| refusal cheaper than answering by | **~
|
|
361
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
362
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
363
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
364
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
341
365
|
|
|
342
366
|
Two significant figures and a spread, because that is the precision the
|
|
343
|
-
measurement has.
|
|
344
|
-
machine
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
were unfalsifiable rather than wrong.
|
|
367
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
368
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
369
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
370
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
371
|
+
figures from a script that was never committed; both values sit inside that
|
|
372
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
350
373
|
|
|
351
374
|
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
352
375
|
query touches only the posting lists of its own distinctive words, and never
|
|
@@ -375,8 +398,8 @@ reaches ranking at all.
|
|
|
375
398
|
Directional local measurements, not service levels — but every one of them is
|
|
376
399
|
now a command rather than a memory, which two of them were not before. Each
|
|
377
400
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
378
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
379
|
-
each is for.
|
|
401
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
402
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
380
403
|
|
|
381
404
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
382
405
|
|
|
@@ -386,34 +409,42 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
386
409
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
387
410
|
declaration spans.
|
|
388
411
|
|
|
389
|
-
**
|
|
390
|
-
|
|
412
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
413
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
414
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
415
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
416
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
391
417
|
|
|
392
|
-
|
|
|
418
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
393
419
|
|---|---|---|
|
|
394
|
-
| right file first |
|
|
395
|
-
| right file in top 3 |
|
|
396
|
-
| lines it hands back, all questions |
|
|
397
|
-
| characters returned, all questions |
|
|
398
|
-
| questions it answers |
|
|
420
|
+
| right file first | 22.9% | **37.1%** |
|
|
421
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
422
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
423
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
424
|
+
| questions it answers | 30 | 30 |
|
|
399
425
|
|
|
400
426
|
**Once the vocabulary exists, it is not close.**
|
|
401
427
|
|
|
402
|
-
| this repository · 70 questions ·
|
|
428
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
403
429
|
|---|---|---|
|
|
404
430
|
| right file first | 22.9% | **58.6%** |
|
|
405
|
-
| right file in top 3 |
|
|
406
|
-
| lines it hands back, all questions | 11,
|
|
407
|
-
| characters returned, all questions | 1,
|
|
431
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
432
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
433
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
408
434
|
| questions it answers | **61** | 60 |
|
|
409
435
|
|
|
410
436
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
411
437
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
412
|
-
parser generated from identifiers the author already chose
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
438
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
439
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
440
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
441
|
+
most public methods, and the cold index beats Grep there without a single
|
|
442
|
+
description being added. On the previous subject, a tool with terse comments
|
|
443
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
444
|
+
|
|
445
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
446
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
447
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
417
448
|
|
|
418
449
|
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
419
450
|
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
@@ -436,12 +467,16 @@ Four qualifications, because the table would otherwise flatter both sides:
|
|
|
436
467
|
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
437
468
|
the seventy questions here — those had no word left that this corpus does not
|
|
438
469
|
use everywhere.
|
|
439
|
-
- **Payload is counted in characters on both sides.**
|
|
440
|
-
characters per question it answers, unranked and without
|
|
441
|
-
10,400, ranked
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
470
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
471
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
472
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
473
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
474
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
475
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
476
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
477
|
+
not a substring of English source and it is not a token in an index built
|
|
478
|
+
from English source, so on a repository written in one language the cold
|
|
479
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
445
480
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
446
481
|
exact, instant and complete, and nothing here replaces it.
|
|
447
482
|
|
|
@@ -506,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
506
541
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
507
542
|
default provider imports none of it.
|
|
508
543
|
|
|
509
|
-
**Measured
|
|
510
|
-
|
|
511
|
-
|
|
544
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
545
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
546
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
547
|
+
different states of a repository being edited while the script ran. Repeated
|
|
548
|
+
against a pinned corpus:
|
|
512
549
|
|
|
513
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
514
|
-
|
|
515
|
-
| **A** foreign, cold | 0.
|
|
516
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
517
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
518
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
550
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
551
|
+
|---|---|---|---|
|
|
552
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
553
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
554
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
555
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
519
556
|
|
|
520
|
-
|
|
521
|
-
|
|
557
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
558
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
559
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
560
|
+
pairs the hash scores exactly zero:
|
|
522
561
|
|
|
523
562
|
| pair | signed hash | MiniLM |
|
|
524
563
|
|---|---|---|
|
|
@@ -536,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
536
575
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
537
576
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
538
577
|
|
|
578
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
579
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
580
|
+
where the difference between the two columns actually lives.
|
|
581
|
+
|
|
539
582
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
540
583
|
source anywhere:
|
|
541
584
|
|
|
@@ -53,23 +53,56 @@ def test_the_absent_ruler_is_well_formed():
|
|
|
53
53
|
assert questions["why"].strip() and questions["caveat"].strip()
|
|
54
54
|
|
|
55
55
|
|
|
56
|
-
def
|
|
57
|
-
"""
|
|
58
|
-
|
|
59
|
-
This is the assertion the ruler's own caveat promises, and it is the one
|
|
60
|
-
that fails first when the repository grows into a subject the ruler
|
|
61
|
-
assumed it would never contain.
|
|
56
|
+
def _intruders(units) -> list[tuple[str, str, int]]:
|
|
57
|
+
"""Which absent question's own vocabulary reaches a corpus that must not
|
|
58
|
+
contain it.
|
|
62
59
|
"""
|
|
63
60
|
index = build_search_index(units)
|
|
64
|
-
|
|
61
|
+
return [
|
|
65
62
|
(entry["id"], term, len(index.postings[term]))
|
|
66
63
|
for entry in load_questions(ABSENT_PATH)["queries"]
|
|
67
64
|
for term in entry["subject"]
|
|
68
65
|
if index.postings.get(term)
|
|
69
66
|
]
|
|
70
|
-
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
|
|
70
|
+
"""The absence claim, re-derived rather than trusted.
|
|
71
|
+
|
|
72
|
+
This is the assertion the ruler's own caveat promises, and it is the one
|
|
73
|
+
that fails first when the repository grows into a subject the ruler
|
|
74
|
+
assumed it would never contain.
|
|
75
|
+
"""
|
|
76
|
+
assert not _intruders(units), (
|
|
71
77
|
"this repository now contains the vocabulary of a question the ruler calls unanswerable; "
|
|
72
|
-
f"retire or rewrite those questions rather than letting them score as misses: {
|
|
78
|
+
f"retire or rewrite those questions rather than letting them score as misses: {_intruders(units)}"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_no_subject_of_an_absent_question_exists_in_the_vendored_corpus():
|
|
83
|
+
"""The other half of the same claim, which used to be unenforceable.
|
|
84
|
+
|
|
85
|
+
This ruler asserts its questions are unanswerable in *both* graded
|
|
86
|
+
repositories. Until the foreign one was vendored that half could only be
|
|
87
|
+
asserted, because the subject was somebody else's checkout and a re-run
|
|
88
|
+
here could not see what it had become. It is now a directory in this
|
|
89
|
+
repository at a pinned tag, so the claim is a test.
|
|
90
|
+
|
|
91
|
+
It has already earned its place. One question's subject list carried a
|
|
92
|
+
generic English verb where a specific term belonged, and the corpus below
|
|
93
|
+
uses that verb in a comment about signing keys -- so the question was one
|
|
94
|
+
ordinary word away from being answerable by accident. The verb was
|
|
95
|
+
replaced with a term that names the thing.
|
|
96
|
+
|
|
97
|
+
Naming the subject here would put it in the index and break the very
|
|
98
|
+
claim this asserts, which is why the paragraph above is written around it.
|
|
99
|
+
That has now happened five times; CONTRIBUTING.md keeps the tally.
|
|
100
|
+
"""
|
|
101
|
+
corpus = ROOT / "benchmarks" / "corpus" / "flask"
|
|
102
|
+
assert corpus.is_dir(), f"the vendored corpus is missing from {corpus}"
|
|
103
|
+
assert not _intruders(build_units(corpus)), (
|
|
104
|
+
"the vendored corpus contains the vocabulary of a question the ruler calls unanswerable; "
|
|
105
|
+
f"rewrite the question rather than letting it score as a miss: {_intruders(build_units(corpus))}"
|
|
73
106
|
)
|
|
74
107
|
|
|
75
108
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|