rag-your-code 1.2.1__tar.gz → 1.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.2.1/src/rag_your_code.egg-info → rag_your_code-1.4.0}/PKG-INFO +148 -63
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/README.md +147 -62
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/pyproject.toml +1 -1
- {rag_your_code-1.2.1 → rag_your_code-1.4.0/src/rag_your_code.egg-info}/PKG-INFO +148 -63
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/config.py +14 -6
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/search.py +1 -1
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_absent_queries.py +42 -9
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_evidence.py +4 -3
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_metadata.py +32 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/LICENSE +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/setup.cfg +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/cli.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/src/ragyourcode/workflow.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_agentic.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_config.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_document.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_golden.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_local_model.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_providers.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_ranking.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_resilience.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.2.1 → rag_your_code-1.4.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -293,42 +293,87 @@ repository had grown by ninety units.
|
|
|
293
293
|
|
|
294
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
295
295
|
|---|---|---|---|---|---|
|
|
296
|
-
| **A**
|
|
296
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
297
297
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
298
298
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
299
299
|
|
|
300
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
301
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
302
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
303
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
304
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
305
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
306
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
307
|
+
subject had grown, and the model comparison below was taken against two
|
|
308
|
+
different states of it. All three are now a `git clone` away from being
|
|
309
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
310
|
+
|
|
300
311
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
301
312
|
|
|
302
|
-
| | this repo |
|
|
313
|
+
| | this repo | Flask |
|
|
303
314
|
|---|---|---|
|
|
304
|
-
| correctly met with silence | **0.967** | **0.
|
|
305
|
-
| English only | **0.933** |
|
|
315
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
316
|
+
| English only | **0.933** | 0.667 |
|
|
306
317
|
| Chinese only | **1.000** | **1.000** |
|
|
307
318
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
308
319
|
|
|
320
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
321
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
322
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
323
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
324
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
325
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
326
|
+
them and start counting as evidence. Five English questions about subjects
|
|
327
|
+
Flask does not implement get through on exactly that.
|
|
328
|
+
|
|
309
329
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
310
330
|
|
|
311
331
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
312
332
|
|---|---|---|---|---|
|
|
313
|
-
| neither (pre-1.0.0) | 0.
|
|
314
|
-
| coverage only (1.0.0) | 0.
|
|
315
|
-
|
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
333
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
334
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
335
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
336
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
337
|
+
|
|
338
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
339
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
340
|
+
|
|
341
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
342
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
343
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
344
|
+
for coverage alone. One question — and the first time in four releases that
|
|
345
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
346
|
+
diagnosis.
|
|
347
|
+
|
|
348
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
349
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
350
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
351
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
352
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
353
|
+
can have.
|
|
354
|
+
|
|
355
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
356
|
+
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
357
|
+
each):
|
|
358
|
+
|
|
359
|
+
| | | across the five |
|
|
360
|
+
|---|---|---|
|
|
361
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
362
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
363
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
364
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
365
|
+
|
|
366
|
+
Two significant figures and a spread, because that is the precision the
|
|
367
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
368
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
369
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
370
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
371
|
+
figures from a script that was never committed; both values sit inside that
|
|
372
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
373
|
+
|
|
374
|
+
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
375
|
+
query touches only the posting lists of its own distinctive words, and never
|
|
376
|
+
reaches ranking at all.
|
|
332
377
|
|
|
333
378
|
**Scale**, synthetic 10,000-unit repository (500 files):
|
|
334
379
|
|
|
@@ -350,8 +395,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
|
|
|
350
395
|
| with a usable signature | **91 / 91** |
|
|
351
396
|
| units invented that do not exist | **0** |
|
|
352
397
|
|
|
353
|
-
Directional local measurements, not service levels
|
|
354
|
-
|
|
398
|
+
Directional local measurements, not service levels — but every one of them is
|
|
399
|
+
now a command rather than a memory, which two of them were not before. Each
|
|
400
|
+
prints the corpus fingerprint beside its score; quote both or neither.
|
|
401
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
402
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
355
403
|
|
|
356
404
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
357
405
|
|
|
@@ -361,45 +409,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
361
409
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
362
410
|
declaration spans.
|
|
363
411
|
|
|
364
|
-
**
|
|
365
|
-
|
|
412
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
413
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
414
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
415
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
416
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
366
417
|
|
|
367
|
-
|
|
|
418
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
368
419
|
|---|---|---|
|
|
369
|
-
| right file first |
|
|
370
|
-
| right file in top 3 |
|
|
371
|
-
| lines
|
|
372
|
-
| characters returned, all questions |
|
|
373
|
-
| questions it answers |
|
|
420
|
+
| right file first | 22.9% | **37.1%** |
|
|
421
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
422
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
423
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
424
|
+
| questions it answers | 30 | 30 |
|
|
374
425
|
|
|
375
426
|
**Once the vocabulary exists, it is not close.**
|
|
376
427
|
|
|
377
|
-
| this repository · 70 questions ·
|
|
428
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
378
429
|
|---|---|---|
|
|
379
|
-
| right file first |
|
|
380
|
-
| right file in top 3 |
|
|
381
|
-
| lines
|
|
382
|
-
| characters returned, all questions |
|
|
383
|
-
| questions it answers |
|
|
430
|
+
| right file first | 22.9% | **58.6%** |
|
|
431
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
432
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
433
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
434
|
+
| questions it answers | **61** | 60 |
|
|
384
435
|
|
|
385
436
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
386
437
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
387
|
-
parser generated from identifiers the author already chose
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
438
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
439
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
440
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
441
|
+
most public methods, and the cold index beats Grep there without a single
|
|
442
|
+
description being added. On the previous subject, a tool with terse comments
|
|
443
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
444
|
+
|
|
445
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
446
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
447
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
448
|
+
|
|
449
|
+
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
450
|
+
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
451
|
+
makes — was published from a script that had never been committed, so nothing
|
|
452
|
+
here could be checked and the word "Grep loop" had no precise meaning. The
|
|
453
|
+
committed version defines it: take the query's words, drop the ones the corpus
|
|
454
|
+
itself shows are everywhere, run one substring search per remaining word over
|
|
455
|
+
exactly the files the index was built from, rank each file by how many distinct
|
|
456
|
+
words hit it, break ties on path. Reconstructing it reproduced this side's
|
|
457
|
+
figures exactly and moved Grep's, which is the expected shape — the ranked arm
|
|
458
|
+
was always a call into shipped code, and the baseline never was.
|
|
459
|
+
|
|
460
|
+
Four qualifications, because the table would otherwise flatter both sides:
|
|
394
461
|
|
|
395
462
|
- **Scored at file granularity**, which understates this side. A Grep hit is a
|
|
396
463
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
397
464
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
398
|
-
- **
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
465
|
+
- **Dropping the corpus-common words is generous to Grep**, and it is what
|
|
466
|
+
makes the baseline a fair one rather than a straw man: an agent that greps
|
|
467
|
+
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
468
|
+
the seventy questions here — those had no word left that this corpus does not
|
|
469
|
+
use everywhere.
|
|
470
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
471
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
472
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
473
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
474
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
475
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
476
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
477
|
+
not a substring of English source and it is not a token in an index built
|
|
478
|
+
from English source, so on a repository written in one language the cold
|
|
479
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
403
480
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
404
481
|
exact, instant and complete, and nothing here replaces it.
|
|
405
482
|
|
|
@@ -464,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
464
541
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
465
542
|
default provider imports none of it.
|
|
466
543
|
|
|
467
|
-
**Measured
|
|
468
|
-
|
|
469
|
-
|
|
544
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
545
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
546
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
547
|
+
different states of a repository being edited while the script ran. Repeated
|
|
548
|
+
against a pinned corpus:
|
|
470
549
|
|
|
471
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
472
|
-
|
|
473
|
-
| **A** foreign, cold | 0.
|
|
474
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
475
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
476
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
550
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
551
|
+
|---|---|---|---|
|
|
552
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
553
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
554
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
555
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
477
556
|
|
|
478
|
-
|
|
479
|
-
|
|
557
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
558
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
559
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
560
|
+
pairs the hash scores exactly zero:
|
|
480
561
|
|
|
481
562
|
| pair | signed hash | MiniLM |
|
|
482
563
|
|---|---|---|
|
|
@@ -494,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
494
575
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
495
576
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
496
577
|
|
|
578
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
579
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
580
|
+
where the difference between the two columns actually lives.
|
|
581
|
+
|
|
497
582
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
498
583
|
source anywhere:
|
|
499
584
|
|
|
@@ -264,42 +264,87 @@ repository had grown by ninety units.
|
|
|
264
264
|
|
|
265
265
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
266
266
|
|---|---|---|---|---|---|
|
|
267
|
-
| **A**
|
|
267
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
268
268
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
269
269
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
270
270
|
|
|
271
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
272
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
273
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
274
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
275
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
276
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
277
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
278
|
+
subject had grown, and the model comparison below was taken against two
|
|
279
|
+
different states of it. All three are now a `git clone` away from being
|
|
280
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
281
|
+
|
|
271
282
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
272
283
|
|
|
273
|
-
| | this repo |
|
|
284
|
+
| | this repo | Flask |
|
|
274
285
|
|---|---|---|
|
|
275
|
-
| correctly met with silence | **0.967** | **0.
|
|
276
|
-
| English only | **0.933** |
|
|
286
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
287
|
+
| English only | **0.933** | 0.667 |
|
|
277
288
|
| Chinese only | **1.000** | **1.000** |
|
|
278
289
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
279
290
|
|
|
291
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
292
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
293
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
294
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
295
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
296
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
297
|
+
them and start counting as evidence. Five English questions about subjects
|
|
298
|
+
Flask does not implement get through on exactly that.
|
|
299
|
+
|
|
280
300
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
281
301
|
|
|
282
302
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
283
303
|
|---|---|---|---|---|
|
|
284
|
-
| neither (pre-1.0.0) | 0.
|
|
285
|
-
| coverage only (1.0.0) | 0.
|
|
286
|
-
|
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
304
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
305
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
306
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
307
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
308
|
+
|
|
309
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
310
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
311
|
+
|
|
312
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
313
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
314
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
315
|
+
for coverage alone. One question — and the first time in four releases that
|
|
316
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
317
|
+
diagnosis.
|
|
318
|
+
|
|
319
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
320
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
321
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
322
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
323
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
324
|
+
can have.
|
|
325
|
+
|
|
326
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
327
|
+
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
328
|
+
each):
|
|
329
|
+
|
|
330
|
+
| | | across the five |
|
|
331
|
+
|---|---|---|
|
|
332
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
333
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
334
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
335
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
336
|
+
|
|
337
|
+
Two significant figures and a spread, because that is the precision the
|
|
338
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
339
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
340
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
341
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
342
|
+
figures from a script that was never committed; both values sit inside that
|
|
343
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
344
|
+
|
|
345
|
+
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
346
|
+
query touches only the posting lists of its own distinctive words, and never
|
|
347
|
+
reaches ranking at all.
|
|
303
348
|
|
|
304
349
|
**Scale**, synthetic 10,000-unit repository (500 files):
|
|
305
350
|
|
|
@@ -321,8 +366,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
|
|
|
321
366
|
| with a usable signature | **91 / 91** |
|
|
322
367
|
| units invented that do not exist | **0** |
|
|
323
368
|
|
|
324
|
-
Directional local measurements, not service levels
|
|
325
|
-
|
|
369
|
+
Directional local measurements, not service levels — but every one of them is
|
|
370
|
+
now a command rather than a memory, which two of them were not before. Each
|
|
371
|
+
prints the corpus fingerprint beside its score; quote both or neither.
|
|
372
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
373
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
326
374
|
|
|
327
375
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
328
376
|
|
|
@@ -332,45 +380,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
332
380
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
333
381
|
declaration spans.
|
|
334
382
|
|
|
335
|
-
**
|
|
336
|
-
|
|
383
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
384
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
385
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
386
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
387
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
337
388
|
|
|
338
|
-
|
|
|
389
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
339
390
|
|---|---|---|
|
|
340
|
-
| right file first |
|
|
341
|
-
| right file in top 3 |
|
|
342
|
-
| lines
|
|
343
|
-
| characters returned, all questions |
|
|
344
|
-
| questions it answers |
|
|
391
|
+
| right file first | 22.9% | **37.1%** |
|
|
392
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
393
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
394
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
395
|
+
| questions it answers | 30 | 30 |
|
|
345
396
|
|
|
346
397
|
**Once the vocabulary exists, it is not close.**
|
|
347
398
|
|
|
348
|
-
| this repository · 70 questions ·
|
|
399
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
349
400
|
|---|---|---|
|
|
350
|
-
| right file first |
|
|
351
|
-
| right file in top 3 |
|
|
352
|
-
| lines
|
|
353
|
-
| characters returned, all questions |
|
|
354
|
-
| questions it answers |
|
|
401
|
+
| right file first | 22.9% | **58.6%** |
|
|
402
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
403
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
404
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
405
|
+
| questions it answers | **61** | 60 |
|
|
355
406
|
|
|
356
407
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
357
408
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
358
|
-
parser generated from identifiers the author already chose
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
409
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
410
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
411
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
412
|
+
most public methods, and the cold index beats Grep there without a single
|
|
413
|
+
description being added. On the previous subject, a tool with terse comments
|
|
414
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
415
|
+
|
|
416
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
417
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
418
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
419
|
+
|
|
420
|
+
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
421
|
+
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
422
|
+
makes — was published from a script that had never been committed, so nothing
|
|
423
|
+
here could be checked and the word "Grep loop" had no precise meaning. The
|
|
424
|
+
committed version defines it: take the query's words, drop the ones the corpus
|
|
425
|
+
itself shows are everywhere, run one substring search per remaining word over
|
|
426
|
+
exactly the files the index was built from, rank each file by how many distinct
|
|
427
|
+
words hit it, break ties on path. Reconstructing it reproduced this side's
|
|
428
|
+
figures exactly and moved Grep's, which is the expected shape — the ranked arm
|
|
429
|
+
was always a call into shipped code, and the baseline never was.
|
|
430
|
+
|
|
431
|
+
Four qualifications, because the table would otherwise flatter both sides:
|
|
365
432
|
|
|
366
433
|
- **Scored at file granularity**, which understates this side. A Grep hit is a
|
|
367
434
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
368
435
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
369
|
-
- **
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
436
|
+
- **Dropping the corpus-common words is generous to Grep**, and it is what
|
|
437
|
+
makes the baseline a fair one rather than a straw man: an agent that greps
|
|
438
|
+
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
439
|
+
the seventy questions here — those had no word left that this corpus does not
|
|
440
|
+
use everywhere.
|
|
441
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
442
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
443
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
444
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
445
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
446
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
447
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
448
|
+
not a substring of English source and it is not a token in an index built
|
|
449
|
+
from English source, so on a repository written in one language the cold
|
|
450
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
374
451
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
375
452
|
exact, instant and complete, and nothing here replaces it.
|
|
376
453
|
|
|
@@ -435,19 +512,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
435
512
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
436
513
|
default provider imports none of it.
|
|
437
514
|
|
|
438
|
-
**Measured
|
|
439
|
-
|
|
440
|
-
|
|
515
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
516
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
517
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
518
|
+
different states of a repository being edited while the script ran. Repeated
|
|
519
|
+
against a pinned corpus:
|
|
441
520
|
|
|
442
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
443
|
-
|
|
444
|
-
| **A** foreign, cold | 0.
|
|
445
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
446
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
447
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
521
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
522
|
+
|---|---|---|---|
|
|
523
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
524
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
525
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
526
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
448
527
|
|
|
449
|
-
|
|
450
|
-
|
|
528
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
529
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
530
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
531
|
+
pairs the hash scores exactly zero:
|
|
451
532
|
|
|
452
533
|
| pair | signed hash | MiniLM |
|
|
453
534
|
|---|---|---|
|
|
@@ -465,6 +546,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
465
546
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
466
547
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
467
548
|
|
|
549
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
550
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
551
|
+
where the difference between the two columns actually lives.
|
|
552
|
+
|
|
468
553
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
469
554
|
source anywhere:
|
|
470
555
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.4.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -293,42 +293,87 @@ repository had grown by ninety units.
|
|
|
293
293
|
|
|
294
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
295
295
|
|---|---|---|---|---|---|
|
|
296
|
-
| **A**
|
|
296
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
297
297
|
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.383 |
|
|
298
298
|
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.443 | 0.614 | 0.507 |
|
|
299
299
|
|
|
300
|
+
The corpora, without which none of the above is reproducible — **A** 1,572
|
|
301
|
+
units, `5fd51169eacc`; **B** 581 units, `8e1e71942c1c`; **C** 581 units,
|
|
302
|
+
`978a1d48a82a`. Ruler A grades **a copy of Flask 3.1.3 carried in this
|
|
303
|
+
repository**, at [`benchmarks/corpus/flask`](benchmarks/corpus/flask), pinned
|
|
304
|
+
to commit `22d9247`. Through 1.3.0 it graded a checkout on one machine, and
|
|
305
|
+
that cost three things: two questions pointed at a declaration the subject had
|
|
306
|
+
renamed, a published score moved 0.257 → 0.229 with no code change because the
|
|
307
|
+
subject had grown, and the model comparison below was taken against two
|
|
308
|
+
different states of it. All three are now a `git clone` away from being
|
|
309
|
+
checked, and CI runs this ruler as an ordinary job.
|
|
310
|
+
|
|
300
311
|
**Refusal — the fourth ruler, 30 questions with no answer anywhere**
|
|
301
312
|
|
|
302
|
-
| | this repo |
|
|
313
|
+
| | this repo | Flask |
|
|
303
314
|
|---|---|---|
|
|
304
|
-
| correctly met with silence | **0.967** | **0.
|
|
305
|
-
| English only | **0.933** |
|
|
315
|
+
| correctly met with silence | **0.967** | **0.833** |
|
|
316
|
+
| English only | **0.933** | 0.667 |
|
|
306
317
|
| Chinese only | **1.000** | **1.000** |
|
|
307
318
|
| results resting on no lexical evidence, rulers A–C | **0.000** | **0.000** |
|
|
308
319
|
|
|
320
|
+
Foreign silence fell from 0.933 to 0.833 when the subject changed, and the
|
|
321
|
+
cause is a limit of the design rather than a defect. A word counts as evidence
|
|
322
|
+
unless it occurs in more than 5% of units — a stopword list derived from the
|
|
323
|
+
corpus, so that it needs no list and works in any language. Here `how`, `when`,
|
|
324
|
+
`does` and `are` are everywhere, because 304 units carry written English prose.
|
|
325
|
+
Across 1,572 units of mostly short, undocumented methods they occur in 1–5% of
|
|
326
|
+
them and start counting as evidence. Five English questions about subjects
|
|
327
|
+
Flask does not implement get through on exactly that.
|
|
328
|
+
|
|
309
329
|
**What each bar costs and buys** — one corpus, gate varied alone:
|
|
310
330
|
|
|
311
331
|
| gate | A hit@1/3/MRR | B hit@1/3/MRR | C hit@1/3/MRR | silence own / foreign |
|
|
312
332
|
|---|---|---|---|---|
|
|
313
|
-
| neither (pre-1.0.0) | 0.
|
|
314
|
-
| coverage only (1.0.0) | 0.
|
|
315
|
-
|
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
333
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.391 | 0.486/0.700/0.567 | 0.000 / 0.000 |
|
|
334
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.486/0.686/0.562 | 0.600 / 0.733 |
|
|
335
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.383 | 0.443/0.614/0.507 | 0.967 / 0.800 |
|
|
336
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.383** | 0.443/0.614/0.507 | **0.967 / 0.833** |
|
|
337
|
+
|
|
338
|
+
Ruler A is **unmoved by either bar**, and B by concentration. The whole cost is
|
|
339
|
+
three questions of seventy at hit@1 on the warmest ruler, and six at hit@3.
|
|
340
|
+
|
|
341
|
+
Through 1.3.0 this section said concentration subsumes coverage. **On a corpus
|
|
342
|
+
this project did not choose, it does not.** Both bars together silence 0.833 of
|
|
343
|
+
the foreign absent questions, against 0.800 for concentration alone and 0.733
|
|
344
|
+
for coverage alone. One question — and the first time in four releases that
|
|
345
|
+
keeping both has been worth a measurable amount rather than worth a different
|
|
346
|
+
diagnosis.
|
|
347
|
+
|
|
348
|
+
Raising the concentration bar buys the remaining silence, and is refused,
|
|
349
|
+
because it is bought out of the answers: at 0.50 the foreign absent ruler is
|
|
350
|
+
silent on all thirty while ruler A falls to 0.086 hit@1 from 0.200, B to 0.214
|
|
351
|
+
from 0.314 and C to 0.329 from 0.443. **0.28 was chosen before this corpus
|
|
352
|
+
existed and survived meeting it**, which is the only kind of evidence a default
|
|
353
|
+
can have.
|
|
354
|
+
|
|
355
|
+
**Latency** — warm corpus, 581 units `978a1d48a82a`, five consecutive
|
|
356
|
+
invocations of `python -m benchmarks.query_latency --repeats 10` (420 samples
|
|
357
|
+
each):
|
|
358
|
+
|
|
359
|
+
| | | across the five |
|
|
360
|
+
|---|---|---|
|
|
361
|
+
| query, median | **1.0 ms** | 0.71 – 1.26 |
|
|
362
|
+
| query, p95 | 1.7 ms | 1.2 – 3.7 |
|
|
363
|
+
| refusing an unanswerable query | **0.029 ms** | 0.023 – 0.031 |
|
|
364
|
+
| refusal cheaper than answering by | **~36×** | 31 – 44 |
|
|
365
|
+
|
|
366
|
+
Two significant figures and a spread, because that is the precision the
|
|
367
|
+
measurement has. Across fifteen invocations over two releases on the same idle
|
|
368
|
+
machine the median has landed anywhere from 0.62 to 1.44 ms and p95 from 1.09
|
|
369
|
+
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
370
|
+
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
371
|
+
figures from a script that was never committed; both values sit inside that
|
|
372
|
+
band, which is the point: they were unfalsifiable rather than wrong.
|
|
373
|
+
|
|
374
|
+
Refusal is cheap for a structural reason, not a tuned one: an unanswerable
|
|
375
|
+
query touches only the posting lists of its own distinctive words, and never
|
|
376
|
+
reaches ranking at all.
|
|
332
377
|
|
|
333
378
|
**Scale**, synthetic 10,000-unit repository (500 files):
|
|
334
379
|
|
|
@@ -350,8 +395,11 @@ touches only the posting lists of its own distinctive words, never the corpus.
|
|
|
350
395
|
| with a usable signature | **91 / 91** |
|
|
351
396
|
| units invented that do not exist | **0** |
|
|
352
397
|
|
|
353
|
-
Directional local measurements, not service levels
|
|
354
|
-
|
|
398
|
+
Directional local measurements, not service levels — but every one of them is
|
|
399
|
+
now a command rather than a memory, which two of them were not before. Each
|
|
400
|
+
prints the corpus fingerprint beside its score; quote both or neither.
|
|
401
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
402
|
+
each is for, and the corpus one of them grades is now carried here too.
|
|
355
403
|
|
|
356
404
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
357
405
|
|
|
@@ -361,45 +409,74 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
361
409
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
362
410
|
declaration spans.
|
|
363
411
|
|
|
364
|
-
**
|
|
365
|
-
|
|
412
|
+
**Which side wins on an undescribed repository depends on the repository.**
|
|
413
|
+
Through 1.3.0 this section said flatly that Grep wins there, because the one
|
|
414
|
+
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
415
|
+
answerable by matching identifiers. Swapping the subject for a public web
|
|
416
|
+
framework reversed it. The honest claim is narrower than either table alone:
|
|
366
417
|
|
|
367
|
-
|
|
|
418
|
+
| Flask 3.1.3 · 35 questions · 1,572 units `5fd51169eacc` · no descriptions | Grep loop | rag-your-code |
|
|
368
419
|
|---|---|---|
|
|
369
|
-
| right file first |
|
|
370
|
-
| right file in top 3 |
|
|
371
|
-
| lines
|
|
372
|
-
| characters returned, all questions |
|
|
373
|
-
| questions it answers |
|
|
420
|
+
| right file first | 22.9% | **37.1%** |
|
|
421
|
+
| right file in top 3 | 45.7% | **57.1%** |
|
|
422
|
+
| lines it hands back, all questions | 17,641 | — |
|
|
423
|
+
| characters returned, all questions | 1,415,656 | **249,720** |
|
|
424
|
+
| questions it answers | 30 | 30 |
|
|
374
425
|
|
|
375
426
|
**Once the vocabulary exists, it is not close.**
|
|
376
427
|
|
|
377
|
-
| this repository · 70 questions ·
|
|
428
|
+
| this repository · 70 questions · 581 units `978a1d48a82a` · 304 described | Grep loop | rag-your-code |
|
|
378
429
|
|---|---|---|
|
|
379
|
-
| right file first |
|
|
380
|
-
| right file in top 3 |
|
|
381
|
-
| lines
|
|
382
|
-
| characters returned, all questions |
|
|
383
|
-
| questions it answers |
|
|
430
|
+
| right file first | 22.9% | **58.6%** |
|
|
431
|
+
| right file in top 3 | 54.3% | **75.7%** |
|
|
432
|
+
| lines it hands back, all questions | 11,833 | — |
|
|
433
|
+
| characters returned, all questions | 1,122,902 | **626,022** |
|
|
434
|
+
| questions it answers | **61** | 60 |
|
|
384
435
|
|
|
385
436
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
386
437
|
baseline instead of asserted. A cold index retrieves against a sentence the
|
|
387
|
-
parser generated from identifiers the author already chose
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
438
|
+
parser generated from identifiers the author already chose, plus whatever
|
|
439
|
+
docstrings the author wrote — so how it fares against Grep is decided by how
|
|
440
|
+
much prose the repository already contains. Flask has a written docstring on
|
|
441
|
+
most public methods, and the cold index beats Grep there without a single
|
|
442
|
+
description being added. On the previous subject, a tool with terse comments
|
|
443
|
+
and long identifiers, the same cold index lost to Grep by the same margin.
|
|
444
|
+
|
|
445
|
+
What does not depend on the subject is what descriptions buy: on this
|
|
446
|
+
repository first-place accuracy goes to **more than double** Grep's, and the
|
|
447
|
+
payload comes back ranked, spanned, and roughly half the size.
|
|
448
|
+
|
|
449
|
+
**Both tables come from `python -m benchmarks.grep_baseline`**, which is what
|
|
450
|
+
changed in 1.3.0. Until then this section — the strongest claim the project
|
|
451
|
+
makes — was published from a script that had never been committed, so nothing
|
|
452
|
+
here could be checked and the word "Grep loop" had no precise meaning. The
|
|
453
|
+
committed version defines it: take the query's words, drop the ones the corpus
|
|
454
|
+
itself shows are everywhere, run one substring search per remaining word over
|
|
455
|
+
exactly the files the index was built from, rank each file by how many distinct
|
|
456
|
+
words hit it, break ties on path. Reconstructing it reproduced this side's
|
|
457
|
+
figures exactly and moved Grep's, which is the expected shape — the ranked arm
|
|
458
|
+
was always a call into shipped code, and the baseline never was.
|
|
459
|
+
|
|
460
|
+
Four qualifications, because the table would otherwise flatter both sides:
|
|
394
461
|
|
|
395
462
|
- **Scored at file granularity**, which understates this side. A Grep hit is a
|
|
396
463
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
397
464
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
398
|
-
- **
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
465
|
+
- **Dropping the corpus-common words is generous to Grep**, and it is what
|
|
466
|
+
makes the baseline a fair one rather than a straw man: an agent that greps
|
|
467
|
+
`the` gets every file back in no order. It is also why Grep declines nine of
|
|
468
|
+
the seventy questions here — those had no word left that this corpus does not
|
|
469
|
+
use everywhere.
|
|
470
|
+
- **Payload is counted in characters on both sides.** On this repository Grep
|
|
471
|
+
hands back 18,400 characters per question it answers, unranked and without
|
|
472
|
+
spans, against 10,400 here, ranked and capped by `search.max_chars` — a
|
|
473
|
+
factor of 1.8. On Flask it is a factor of 5.7 — 47,200 characters against
|
|
474
|
+
8,300 — because a framework repeats its own vocabulary across many files and
|
|
475
|
+
Grep has no way to rank what it finds. Both sides decline the same five of
|
|
476
|
+
those 35 — and they are the five Chinese ones, all of them. A Chinese word is
|
|
477
|
+
not a substring of English source and it is not a token in an index built
|
|
478
|
+
from English source, so on a repository written in one language the cold
|
|
479
|
+
cross-language case is not this tool's failure but the corpus's.
|
|
403
480
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
404
481
|
exact, instant and complete, and nothing here replaces it.
|
|
405
482
|
|
|
@@ -464,19 +541,23 @@ The extra is optional by construction: `dependencies = []` is what a default
|
|
|
464
541
|
install gets, the import happens inside the constructor, and a test asserts the
|
|
465
542
|
default provider imports none of it.
|
|
466
543
|
|
|
467
|
-
**Measured
|
|
468
|
-
|
|
469
|
-
|
|
544
|
+
**Measured on the same four rulers, both arms against one corpus.** 1.1.0
|
|
545
|
+
published this comparison and read it as a win. Its largest gain was on the
|
|
546
|
+
foreign ruler, whose two arms turned out to have been taken against two
|
|
547
|
+
different states of a repository being edited while the script ran. Repeated
|
|
548
|
+
against a pinned corpus:
|
|
470
549
|
|
|
471
|
-
| ruler | signed hash (default) | MiniLM, local |
|
|
472
|
-
|
|
473
|
-
| **A** foreign, cold | 0.
|
|
474
|
-
| **B** own, cold | 0.314 / 0.471 / 0.383 |
|
|
475
|
-
| **C** own, described | 0.443 / 0.614 / 0.507 | 0.
|
|
476
|
-
| **D** silence, own / foreign | 0.967 / 0.
|
|
550
|
+
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
551
|
+
|---|---|---|---|
|
|
552
|
+
| **A** foreign, cold | 1,572 `5fd51169eacc` | **0.200 / 0.286 / 0.238** | 0.171 / 0.257 / 0.214 |
|
|
553
|
+
| **B** own, cold | 581 `8e1e71942c1c` | 0.314 / 0.471 / 0.383 | 0.314 / 0.471 / 0.383 |
|
|
554
|
+
| **C** own, described | 581 `978a1d48a82a` | **0.443 / 0.614 / 0.507** | 0.429 / 0.600 / 0.500 |
|
|
555
|
+
| **D** silence, own / foreign | as above | 0.967 / 0.833 | 0.967 / 0.833 |
|
|
477
556
|
|
|
478
|
-
|
|
479
|
-
|
|
557
|
+
**Worse or identical on every ruler.** It is shipped anyway, as an extra nobody
|
|
558
|
+
has to install, because it does one thing the hash cannot do at all and these
|
|
559
|
+
rulers cannot see: reach a unit that shares no word with the question. The
|
|
560
|
+
pairs the hash scores exactly zero:
|
|
480
561
|
|
|
481
562
|
| pair | signed hash | MiniLM |
|
|
482
563
|
|---|---|---|
|
|
@@ -494,6 +575,10 @@ a threshold on a score and the distributions overlap (0.469 vs 0.418 median),
|
|
|
494
575
|
and a scale-free standout metric took ruler B from 0.329 to 0.186 for two
|
|
495
576
|
thirds of the silence. Applying the lexical bars costs ruler A nothing.
|
|
496
577
|
|
|
578
|
+
If you install it expecting the hit rates above to move, they will not. Install
|
|
579
|
+
it for the cross-language and paraphrase cases in the table above, which is
|
|
580
|
+
where the difference between the two columns actually lives.
|
|
581
|
+
|
|
497
582
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
498
583
|
source anywhere:
|
|
499
584
|
|
|
@@ -201,10 +201,15 @@ SETTINGS: tuple[Setting, ...] = (
|
|
|
201
201
|
),
|
|
202
202
|
# How much of a question has to reach the index before an answer counts as
|
|
203
203
|
# evidence rather than as a guess. Measured, not chosen: across two
|
|
204
|
-
# repositories, two languages and
|
|
205
|
-
# at which every question that was being answered
|
|
206
|
-
#
|
|
207
|
-
#
|
|
204
|
+
# repositories, two languages and every question set under `benchmarks/`,
|
|
205
|
+
# 0.40 is the largest value at which every question that was being answered
|
|
206
|
+
# correctly still is. On its own it silences three fifths of the questions
|
|
207
|
+
# whose answer is not in the repository at all; the rest is what
|
|
208
|
+
# `search.min_concentration` below adds. The command is the claim --
|
|
209
|
+
# `repo_queries --questions benchmarks/absent_queries.json
|
|
210
|
+
# --min-concentration 0` -- because a count typed into a comment is a figure
|
|
211
|
+
# nothing checks, and both numbers this sentence used to carry had rotted.
|
|
212
|
+
# It is a ratio inside the query, so unlike a score threshold it does
|
|
208
213
|
# not move when the corpus or the scale of the ranking does -- the defect
|
|
209
214
|
# that made `confidence_threshold = 0.8` stop meaning anything.
|
|
210
215
|
Setting(
|
|
@@ -221,8 +226,11 @@ SETTINGS: tuple[Setting, ...] = (
|
|
|
221
226
|
# unrelated declarations -- four of six words found in four places with
|
|
222
227
|
# nothing to do with one another or with what was asked. Measured across
|
|
223
228
|
# four rulers, requiring a quarter of a query's rarity to land inside one
|
|
224
|
-
# unit leaves the two rulers over undescribed code unchanged and
|
|
225
|
-
#
|
|
229
|
+
# unit leaves the two rulers over undescribed code unchanged and removes
|
|
230
|
+
# most of what the coverage bar alone still answers; the ablation table
|
|
231
|
+
# carries the numbers, in docs/ROADMAP.md, rather than this line, which
|
|
232
|
+
# said "roughly halves" while that table said an order of magnitude.
|
|
233
|
+
# Rarity-
|
|
226
234
|
# weighted rather than counted, because a unit holding two ordinary words is
|
|
227
235
|
# not better evidence than one holding the rare word the question is about.
|
|
228
236
|
Setting(
|
|
@@ -119,7 +119,7 @@ class Evidence:
|
|
|
119
119
|
|
|
120
120
|
Ranking answers "which of these is best". It cannot answer "is any of this
|
|
121
121
|
an answer", and reading the first as the second is what let a repository
|
|
122
|
-
reply to
|
|
122
|
+
reply to every question in `benchmarks/absent_queries.json` -- each one
|
|
123
123
|
about a subject neither repository implements. `where are CUDA kernels
|
|
124
124
|
dispatched to the device` came back with a test about word counting, on the
|
|
125
125
|
evidence of `are`, `the`, `to` and `where`. That is not a Chinese problem
|
|
@@ -53,23 +53,56 @@ def test_the_absent_ruler_is_well_formed():
|
|
|
53
53
|
assert questions["why"].strip() and questions["caveat"].strip()
|
|
54
54
|
|
|
55
55
|
|
|
56
|
-
def
|
|
57
|
-
"""
|
|
58
|
-
|
|
59
|
-
This is the assertion the ruler's own caveat promises, and it is the one
|
|
60
|
-
that fails first when the repository grows into a subject the ruler
|
|
61
|
-
assumed it would never contain.
|
|
56
|
+
def _intruders(units) -> list[tuple[str, str, int]]:
|
|
57
|
+
"""Which absent question's own vocabulary reaches a corpus that must not
|
|
58
|
+
contain it.
|
|
62
59
|
"""
|
|
63
60
|
index = build_search_index(units)
|
|
64
|
-
|
|
61
|
+
return [
|
|
65
62
|
(entry["id"], term, len(index.postings[term]))
|
|
66
63
|
for entry in load_questions(ABSENT_PATH)["queries"]
|
|
67
64
|
for term in entry["subject"]
|
|
68
65
|
if index.postings.get(term)
|
|
69
66
|
]
|
|
70
|
-
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def test_no_subject_of_an_absent_question_exists_in_this_repository(units):
|
|
70
|
+
"""The absence claim, re-derived rather than trusted.
|
|
71
|
+
|
|
72
|
+
This is the assertion the ruler's own caveat promises, and it is the one
|
|
73
|
+
that fails first when the repository grows into a subject the ruler
|
|
74
|
+
assumed it would never contain.
|
|
75
|
+
"""
|
|
76
|
+
assert not _intruders(units), (
|
|
71
77
|
"this repository now contains the vocabulary of a question the ruler calls unanswerable; "
|
|
72
|
-
f"retire or rewrite those questions rather than letting them score as misses: {
|
|
78
|
+
f"retire or rewrite those questions rather than letting them score as misses: {_intruders(units)}"
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def test_no_subject_of_an_absent_question_exists_in_the_vendored_corpus():
|
|
83
|
+
"""The other half of the same claim, which used to be unenforceable.
|
|
84
|
+
|
|
85
|
+
This ruler asserts its questions are unanswerable in *both* graded
|
|
86
|
+
repositories. Until the foreign one was vendored that half could only be
|
|
87
|
+
asserted, because the subject was somebody else's checkout and a re-run
|
|
88
|
+
here could not see what it had become. It is now a directory in this
|
|
89
|
+
repository at a pinned tag, so the claim is a test.
|
|
90
|
+
|
|
91
|
+
It has already earned its place. One question's subject list carried a
|
|
92
|
+
generic English verb where a specific term belonged, and the corpus below
|
|
93
|
+
uses that verb in a comment about signing keys -- so the question was one
|
|
94
|
+
ordinary word away from being answerable by accident. The verb was
|
|
95
|
+
replaced with a term that names the thing.
|
|
96
|
+
|
|
97
|
+
Naming the subject here would put it in the index and break the very
|
|
98
|
+
claim this asserts, which is why the paragraph above is written around it.
|
|
99
|
+
That has now happened five times; CONTRIBUTING.md keeps the tally.
|
|
100
|
+
"""
|
|
101
|
+
corpus = ROOT / "benchmarks" / "corpus" / "flask"
|
|
102
|
+
assert corpus.is_dir(), f"the vendored corpus is missing from {corpus}"
|
|
103
|
+
assert not _intruders(build_units(corpus)), (
|
|
104
|
+
"the vendored corpus contains the vocabulary of a question the ruler calls unanswerable; "
|
|
105
|
+
f"rewrite the question rather than letting it score as a miss: {_intruders(build_units(corpus))}"
|
|
73
106
|
)
|
|
74
107
|
|
|
75
108
|
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
"""Retrieval must be able to say it has no answer.
|
|
2
2
|
|
|
3
3
|
Ranking always produces a least-bad unit and returns it with a score and a
|
|
4
|
-
rank, which read exactly like an answer. Graded against
|
|
5
|
-
subjects neither this repository nor
|
|
6
|
-
repository implements
|
|
4
|
+
rank, which read exactly like an answer. Graded against every question in
|
|
5
|
+
`benchmarks/absent_queries.json` -- subjects neither this repository nor
|
|
6
|
+
`benchmarks/cold_queries.json`'s repository implements -- every single one came
|
|
7
|
+
back answered, on both repositories. `where are CUDA
|
|
7
8
|
kernels dispatched to the device` on the evidence of `are`, `the`, `to` and
|
|
8
9
|
`where`. These assert the second question retrieval now asks: not which unit
|
|
9
10
|
ranks highest, but whether any of this is evidence at all.
|
|
@@ -321,3 +321,35 @@ def test_every_documented_provider_block_actually_configures_that_provider(tmp_p
|
|
|
321
321
|
# arrangement exists to prevent.
|
|
322
322
|
assert "sk-" not in settings
|
|
323
323
|
assert {"sentence-transformers", "openai-compatible"} <= seen, f"undocumented providers; README shows {sorted(seen)}"
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
# --- a measurement nobody can re-run is a claim, not a measurement ----------
|
|
327
|
+
|
|
328
|
+
BENCHMARKS = tuple(
|
|
329
|
+
sorted(
|
|
330
|
+
path.name
|
|
331
|
+
for path in (ROOT / "benchmarks").glob("*.py")
|
|
332
|
+
if path.name != "__init__.py"
|
|
333
|
+
)
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def test_every_benchmark_script_is_listed_in_its_own_index():
|
|
338
|
+
"""The index of the rulers is discovered, not maintained by hand.
|
|
339
|
+
|
|
340
|
+
1.3.0 found two figures in the README -- query latency, and the entire Grep
|
|
341
|
+
head-to-head -- produced by scripts that had never been committed. Nobody
|
|
342
|
+
could re-derive either, and the second one meant the phrase "a Grep loop"
|
|
343
|
+
had no definition a reader could argue with.
|
|
344
|
+
|
|
345
|
+
Both directions matter. A script absent from the index is one nobody knows
|
|
346
|
+
to run; a command in the index naming a script that does not exist is the
|
|
347
|
+
install-line defect this repository shipped twice. Discovery by glob is
|
|
348
|
+
what keeps the seventh script from being the one nothing checks.
|
|
349
|
+
"""
|
|
350
|
+
index = (ROOT / "benchmarks" / "README.md").read_text(encoding="utf-8")
|
|
351
|
+
assert BENCHMARKS, "benchmarks/ holds no scripts; this guard would pass vacuously"
|
|
352
|
+
for name in BENCHMARKS:
|
|
353
|
+
assert f"benchmarks.{Path(name).stem}" in index, f"{name} is in benchmarks/ and not in its README"
|
|
354
|
+
for module in re.findall(r"benchmarks\.([a-z_]+)", index):
|
|
355
|
+
assert (ROOT / "benchmarks" / f"{module}.py").is_file(), f"the README runs benchmarks.{module}, which does not exist"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|