rag-your-code 1.5.1__tar.gz → 1.5.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.5.1/src/rag_your_code.egg-info → rag_your_code-1.5.3}/PKG-INFO +22 -25
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/README.md +21 -24
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/pyproject.toml +1 -1
- {rag_your_code-1.5.1 → rag_your_code-1.5.3/src/rag_your_code.egg-info}/PKG-INFO +22 -25
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_metadata.py +1 -1
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/LICENSE +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/setup.cfg +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/cli.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/config.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/search.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/src/ragyourcode/workflow.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_absent_queries.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_agentic.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_config.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_descriptions.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_diagrams.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_document.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_evidence.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_golden.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_local_model.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_providers.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_ranking.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_repo_queries.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_resilience.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.5.1 → rag_your_code-1.5.3}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.5.
|
|
3
|
+
Version: 1.5.3
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -224,7 +224,7 @@ A question with no lexical shortcut, asked of this repository:
|
|
|
224
224
|
|
|
225
225
|
````console
|
|
226
226
|
$ rag-your-code search "where does it decide whether to answer at all" --limit 1
|
|
227
|
-
[src/ragyourcode/search.py:117:Evidence] score=0.
|
|
227
|
+
[src/ragyourcode/search.py:117:Evidence] score=0.446
|
|
228
228
|
The verdict on whether a question reached this index at all, kept separate from
|
|
229
229
|
how results rank. ... 中文:判定一个提问究竟有没有够到索引的结论。...
|
|
230
230
|
```python
|
|
@@ -237,7 +237,7 @@ There is no string here to grep for: *decide* occurs nowhere in that
|
|
|
237
237
|
declaration and matched nothing. What ranked it first is ordinary words —
|
|
238
238
|
*answer*, *whether*, *where* — rare enough in this corpus to tell declarations
|
|
239
239
|
apart. What the agent-written description adds is the other language:
|
|
240
|
-
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.
|
|
240
|
+
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.396,
|
|
241
241
|
sharing not one character with its source.
|
|
242
242
|
|
|
243
243
|
Now the case that motivated 1.0.0 — a question with no answer here at all:
|
|
@@ -261,12 +261,12 @@ it happens to use elsewhere.
|
|
|
261
261
|
"matched_terms": ["job","leave","print"],
|
|
262
262
|
"ubiquitous_terms": ["a","does","the","why"],
|
|
263
263
|
"coverage": 0.5, "min_coverage": 0.4,
|
|
264
|
-
"concentration": 0.
|
|
264
|
+
"concentration": 0.1693, "min_concentration": 0.28,
|
|
265
265
|
"applied_min_coverage": 0.4, "applied_min_concentration": 0.28,
|
|
266
266
|
"hint": "..."}}
|
|
267
267
|
```
|
|
268
268
|
|
|
269
|
-
Read `coverage: 0.5` against `concentration: 0.
|
|
269
|
+
Read `coverage: 0.5` against `concentration: 0.1693`. Half the distinctive
|
|
270
270
|
words are here — `job`, `leave`, `print` — and spread thin enough that no
|
|
271
271
|
declaration holds a fifth of what was asked, against a bar of 0.28. Before
|
|
272
272
|
1.1.0 it came back with a confident-looking result.
|
|
@@ -366,18 +366,14 @@ a default can have.
|
|
|
366
366
|
| refusing an unanswerable query | **0.016 ms** | 0.015 – 0.018 |
|
|
367
367
|
| refusal cheaper than answering by | **~44×** | 40 – 47 |
|
|
368
368
|
|
|
369
|
-
*Idle* is load-bearing
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
has made to it. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to
|
|
378
|
-
three figures from an uncommitted script; both sit inside that band, which is
|
|
379
|
-
the point — unfalsifiable rather than wrong. Refusal is cheap structurally: an
|
|
380
|
-
unanswerable query touches only its distinctive words' posting lists, never ranking.
|
|
369
|
+
*Idle* is load-bearing, and nothing in the report can see whether it was: the
|
|
370
|
+
same corpus at the same commit measures about twice this median while another
|
|
371
|
+
job is running. Two significant figures and a spread, because across releases
|
|
372
|
+
that spread has been wider than any change the code has made to this number —
|
|
373
|
+
which is why the row ships with its range and its corpus stamp rather than to
|
|
374
|
+
three figures. Refusal is cheap structurally, not by tuning: an unanswerable
|
|
375
|
+
query touches only its distinctive words' posting lists and never reaches
|
|
376
|
+
ranking.
|
|
381
377
|
|
|
382
378
|
**Scale**, synthetic 10,000-unit repository (500 files), re-measured in 1.5.0
|
|
383
379
|
— the previous row of figures was optimistic by more than noise:
|
|
@@ -407,7 +403,7 @@ it again.
|
|
|
407
403
|
Directional local measurements, not service levels — but each is a command
|
|
408
404
|
rather than a memory, which two of them were not before. Each
|
|
409
405
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
410
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
406
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the eight scripts and what
|
|
411
407
|
each is for, and the corpus one of them grades is now carried here too.
|
|
412
408
|
|
|
413
409
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
@@ -466,8 +462,8 @@ Qualifications, because the tables would otherwise flatter both sides:
|
|
|
466
462
|
every file back in no order. It is also why Grep declines nine of the seventy
|
|
467
463
|
questions here — no word was left that this corpus does not use everywhere.
|
|
468
464
|
- **Payload is counted in characters on both sides.** Grep hands back roughly
|
|
469
|
-
19,
|
|
470
|
-
against 10,
|
|
465
|
+
19,500 characters per question it answers here, unranked and without spans,
|
|
466
|
+
against 10,200 ranked and capped by `search.max_chars` — a factor of 1.9,
|
|
471
467
|
5.5 on Flask and 5.1 on cobra, where a framework repeats its vocabulary
|
|
472
468
|
across files and Grep cannot rank what it finds. 1.4.1 changed what fits in
|
|
473
469
|
that cap: the block had been reprinting the docstring the code below already
|
|
@@ -731,11 +727,11 @@ by which the author's docstring reaches the weight-3 description field — so
|
|
|
731
727
|
writing one demotes it to the weight-1 body. On `parser.py::_generic_units` a
|
|
732
728
|
long description cost one graded question and a short one cost three;
|
|
733
729
|
appending the docstring to every description instead cost the 1.5.0 corpus
|
|
734
|
-
0.
|
|
730
|
+
0.429 → 0.414 hit@1. `describe.skip` records the decision.
|
|
735
731
|
|
|
736
|
-
**The vectors are 72.1% of the index and earn at most two questions**
|
|
737
|
-
default embedder — 74.8% Flask, 79.7%
|
|
738
|
-
|
|
732
|
+
**The vectors are 72.1% of the index and earn at most two questions** on any
|
|
733
|
+
ruler, in either direction, under the default embedder — 74.8% Flask, 79.7%
|
|
734
|
+
cobra. Kept: that same storage is what an optional model needs.
|
|
739
735
|
|
|
740
736
|
**`search.vector_recall` scans every vector per query** — under a semantic
|
|
741
737
|
embedder. The default hash never widens at all. Affordable at the measured
|
|
@@ -767,7 +763,8 @@ that rots. `pytest --cov=ragyourcode` is the command behind the coverage one.
|
|
|
767
763
|
CI runs Python 3.10–3.13 on Linux and Windows, installs the built wheel into a
|
|
768
764
|
clean environment and runs the documented CLI end to end — `bootstrap` through
|
|
769
765
|
`describe promote` — plus the skill's own install line verbatim, and grades
|
|
770
|
-
|
|
766
|
+
rulers A, C, D and E across this repository and both vendored corpora. Ruler B,
|
|
767
|
+
the cold parse of this repository, is a local command.
|
|
771
768
|
|
|
772
769
|
- [docs/FLOW.md](docs/FLOW.md) — the whole thing in four diagrams
|
|
773
770
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) — how each stage works and why
|
|
@@ -194,7 +194,7 @@ A question with no lexical shortcut, asked of this repository:
|
|
|
194
194
|
|
|
195
195
|
````console
|
|
196
196
|
$ rag-your-code search "where does it decide whether to answer at all" --limit 1
|
|
197
|
-
[src/ragyourcode/search.py:117:Evidence] score=0.
|
|
197
|
+
[src/ragyourcode/search.py:117:Evidence] score=0.446
|
|
198
198
|
The verdict on whether a question reached this index at all, kept separate from
|
|
199
199
|
how results rank. ... 中文:判定一个提问究竟有没有够到索引的结论。...
|
|
200
200
|
```python
|
|
@@ -207,7 +207,7 @@ There is no string here to grep for: *decide* occurs nowhere in that
|
|
|
207
207
|
declaration and matched nothing. What ranked it first is ordinary words —
|
|
208
208
|
*answer*, *whether*, *where* — rare enough in this corpus to tell declarations
|
|
209
209
|
apart. What the agent-written description adds is the other language:
|
|
210
|
-
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.
|
|
210
|
+
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.396,
|
|
211
211
|
sharing not one character with its source.
|
|
212
212
|
|
|
213
213
|
Now the case that motivated 1.0.0 — a question with no answer here at all:
|
|
@@ -231,12 +231,12 @@ it happens to use elsewhere.
|
|
|
231
231
|
"matched_terms": ["job","leave","print"],
|
|
232
232
|
"ubiquitous_terms": ["a","does","the","why"],
|
|
233
233
|
"coverage": 0.5, "min_coverage": 0.4,
|
|
234
|
-
"concentration": 0.
|
|
234
|
+
"concentration": 0.1693, "min_concentration": 0.28,
|
|
235
235
|
"applied_min_coverage": 0.4, "applied_min_concentration": 0.28,
|
|
236
236
|
"hint": "..."}}
|
|
237
237
|
```
|
|
238
238
|
|
|
239
|
-
Read `coverage: 0.5` against `concentration: 0.
|
|
239
|
+
Read `coverage: 0.5` against `concentration: 0.1693`. Half the distinctive
|
|
240
240
|
words are here — `job`, `leave`, `print` — and spread thin enough that no
|
|
241
241
|
declaration holds a fifth of what was asked, against a bar of 0.28. Before
|
|
242
242
|
1.1.0 it came back with a confident-looking result.
|
|
@@ -336,18 +336,14 @@ a default can have.
|
|
|
336
336
|
| refusing an unanswerable query | **0.016 ms** | 0.015 – 0.018 |
|
|
337
337
|
| refusal cheaper than answering by | **~44×** | 40 – 47 |
|
|
338
338
|
|
|
339
|
-
*Idle* is load-bearing
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
has made to it. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to
|
|
348
|
-
three figures from an uncommitted script; both sit inside that band, which is
|
|
349
|
-
the point — unfalsifiable rather than wrong. Refusal is cheap structurally: an
|
|
350
|
-
unanswerable query touches only its distinctive words' posting lists, never ranking.
|
|
339
|
+
*Idle* is load-bearing, and nothing in the report can see whether it was: the
|
|
340
|
+
same corpus at the same commit measures about twice this median while another
|
|
341
|
+
job is running. Two significant figures and a spread, because across releases
|
|
342
|
+
that spread has been wider than any change the code has made to this number —
|
|
343
|
+
which is why the row ships with its range and its corpus stamp rather than to
|
|
344
|
+
three figures. Refusal is cheap structurally, not by tuning: an unanswerable
|
|
345
|
+
query touches only its distinctive words' posting lists and never reaches
|
|
346
|
+
ranking.
|
|
351
347
|
|
|
352
348
|
**Scale**, synthetic 10,000-unit repository (500 files), re-measured in 1.5.0
|
|
353
349
|
— the previous row of figures was optimistic by more than noise:
|
|
@@ -377,7 +373,7 @@ it again.
|
|
|
377
373
|
Directional local measurements, not service levels — but each is a command
|
|
378
374
|
rather than a memory, which two of them were not before. Each
|
|
379
375
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
380
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
376
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the eight scripts and what
|
|
381
377
|
each is for, and the corpus one of them grades is now carried here too.
|
|
382
378
|
|
|
383
379
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
@@ -436,8 +432,8 @@ Qualifications, because the tables would otherwise flatter both sides:
|
|
|
436
432
|
every file back in no order. It is also why Grep declines nine of the seventy
|
|
437
433
|
questions here — no word was left that this corpus does not use everywhere.
|
|
438
434
|
- **Payload is counted in characters on both sides.** Grep hands back roughly
|
|
439
|
-
19,
|
|
440
|
-
against 10,
|
|
435
|
+
19,500 characters per question it answers here, unranked and without spans,
|
|
436
|
+
against 10,200 ranked and capped by `search.max_chars` — a factor of 1.9,
|
|
441
437
|
5.5 on Flask and 5.1 on cobra, where a framework repeats its vocabulary
|
|
442
438
|
across files and Grep cannot rank what it finds. 1.4.1 changed what fits in
|
|
443
439
|
that cap: the block had been reprinting the docstring the code below already
|
|
@@ -701,11 +697,11 @@ by which the author's docstring reaches the weight-3 description field — so
|
|
|
701
697
|
writing one demotes it to the weight-1 body. On `parser.py::_generic_units` a
|
|
702
698
|
long description cost one graded question and a short one cost three;
|
|
703
699
|
appending the docstring to every description instead cost the 1.5.0 corpus
|
|
704
|
-
0.
|
|
700
|
+
0.429 → 0.414 hit@1. `describe.skip` records the decision.
|
|
705
701
|
|
|
706
|
-
**The vectors are 72.1% of the index and earn at most two questions**
|
|
707
|
-
default embedder — 74.8% Flask, 79.7%
|
|
708
|
-
|
|
702
|
+
**The vectors are 72.1% of the index and earn at most two questions** on any
|
|
703
|
+
ruler, in either direction, under the default embedder — 74.8% Flask, 79.7%
|
|
704
|
+
cobra. Kept: that same storage is what an optional model needs.
|
|
709
705
|
|
|
710
706
|
**`search.vector_recall` scans every vector per query** — under a semantic
|
|
711
707
|
embedder. The default hash never widens at all. Affordable at the measured
|
|
@@ -737,7 +733,8 @@ that rots. `pytest --cov=ragyourcode` is the command behind the coverage one.
|
|
|
737
733
|
CI runs Python 3.10–3.13 on Linux and Windows, installs the built wheel into a
|
|
738
734
|
clean environment and runs the documented CLI end to end — `bootstrap` through
|
|
739
735
|
`describe promote` — plus the skill's own install line verbatim, and grades
|
|
740
|
-
|
|
736
|
+
rulers A, C, D and E across this repository and both vendored corpora. Ruler B,
|
|
737
|
+
the cold parse of this repository, is a local command.
|
|
741
738
|
|
|
742
739
|
- [docs/FLOW.md](docs/FLOW.md) — the whole thing in four diagrams
|
|
743
740
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) — how each stage works and why
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.5.
|
|
3
|
+
Version: 1.5.3
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -224,7 +224,7 @@ A question with no lexical shortcut, asked of this repository:
|
|
|
224
224
|
|
|
225
225
|
````console
|
|
226
226
|
$ rag-your-code search "where does it decide whether to answer at all" --limit 1
|
|
227
|
-
[src/ragyourcode/search.py:117:Evidence] score=0.
|
|
227
|
+
[src/ragyourcode/search.py:117:Evidence] score=0.446
|
|
228
228
|
The verdict on whether a question reached this index at all, kept separate from
|
|
229
229
|
how results rank. ... 中文:判定一个提问究竟有没有够到索引的结论。...
|
|
230
230
|
```python
|
|
@@ -237,7 +237,7 @@ There is no string here to grep for: *decide* occurs nowhere in that
|
|
|
237
237
|
declaration and matched nothing. What ranked it first is ordinary words —
|
|
238
238
|
*answer*, *whether*, *where* — rare enough in this corpus to tell declarations
|
|
239
239
|
apart. What the agent-written description adds is the other language:
|
|
240
|
-
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.
|
|
240
|
+
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.396,
|
|
241
241
|
sharing not one character with its source.
|
|
242
242
|
|
|
243
243
|
Now the case that motivated 1.0.0 — a question with no answer here at all:
|
|
@@ -261,12 +261,12 @@ it happens to use elsewhere.
|
|
|
261
261
|
"matched_terms": ["job","leave","print"],
|
|
262
262
|
"ubiquitous_terms": ["a","does","the","why"],
|
|
263
263
|
"coverage": 0.5, "min_coverage": 0.4,
|
|
264
|
-
"concentration": 0.
|
|
264
|
+
"concentration": 0.1693, "min_concentration": 0.28,
|
|
265
265
|
"applied_min_coverage": 0.4, "applied_min_concentration": 0.28,
|
|
266
266
|
"hint": "..."}}
|
|
267
267
|
```
|
|
268
268
|
|
|
269
|
-
Read `coverage: 0.5` against `concentration: 0.
|
|
269
|
+
Read `coverage: 0.5` against `concentration: 0.1693`. Half the distinctive
|
|
270
270
|
words are here — `job`, `leave`, `print` — and spread thin enough that no
|
|
271
271
|
declaration holds a fifth of what was asked, against a bar of 0.28. Before
|
|
272
272
|
1.1.0 it came back with a confident-looking result.
|
|
@@ -366,18 +366,14 @@ a default can have.
|
|
|
366
366
|
| refusing an unanswerable query | **0.016 ms** | 0.015 – 0.018 |
|
|
367
367
|
| refusal cheaper than answering by | **~44×** | 40 – 47 |
|
|
368
368
|
|
|
369
|
-
*Idle* is load-bearing
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
has made to it. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to
|
|
378
|
-
three figures from an uncommitted script; both sit inside that band, which is
|
|
379
|
-
the point — unfalsifiable rather than wrong. Refusal is cheap structurally: an
|
|
380
|
-
unanswerable query touches only its distinctive words' posting lists, never ranking.
|
|
369
|
+
*Idle* is load-bearing, and nothing in the report can see whether it was: the
|
|
370
|
+
same corpus at the same commit measures about twice this median while another
|
|
371
|
+
job is running. Two significant figures and a spread, because across releases
|
|
372
|
+
that spread has been wider than any change the code has made to this number —
|
|
373
|
+
which is why the row ships with its range and its corpus stamp rather than to
|
|
374
|
+
three figures. Refusal is cheap structurally, not by tuning: an unanswerable
|
|
375
|
+
query touches only its distinctive words' posting lists and never reaches
|
|
376
|
+
ranking.
|
|
381
377
|
|
|
382
378
|
**Scale**, synthetic 10,000-unit repository (500 files), re-measured in 1.5.0
|
|
383
379
|
— the previous row of figures was optimistic by more than noise:
|
|
@@ -407,7 +403,7 @@ it again.
|
|
|
407
403
|
Directional local measurements, not service levels — but each is a command
|
|
408
404
|
rather than a memory, which two of them were not before. Each
|
|
409
405
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
410
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
406
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the eight scripts and what
|
|
411
407
|
each is for, and the corpus one of them grades is now carried here too.
|
|
412
408
|
|
|
413
409
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
@@ -466,8 +462,8 @@ Qualifications, because the tables would otherwise flatter both sides:
|
|
|
466
462
|
every file back in no order. It is also why Grep declines nine of the seventy
|
|
467
463
|
questions here — no word was left that this corpus does not use everywhere.
|
|
468
464
|
- **Payload is counted in characters on both sides.** Grep hands back roughly
|
|
469
|
-
19,
|
|
470
|
-
against 10,
|
|
465
|
+
19,500 characters per question it answers here, unranked and without spans,
|
|
466
|
+
against 10,200 ranked and capped by `search.max_chars` — a factor of 1.9,
|
|
471
467
|
5.5 on Flask and 5.1 on cobra, where a framework repeats its vocabulary
|
|
472
468
|
across files and Grep cannot rank what it finds. 1.4.1 changed what fits in
|
|
473
469
|
that cap: the block had been reprinting the docstring the code below already
|
|
@@ -731,11 +727,11 @@ by which the author's docstring reaches the weight-3 description field — so
|
|
|
731
727
|
writing one demotes it to the weight-1 body. On `parser.py::_generic_units` a
|
|
732
728
|
long description cost one graded question and a short one cost three;
|
|
733
729
|
appending the docstring to every description instead cost the 1.5.0 corpus
|
|
734
|
-
0.
|
|
730
|
+
0.429 → 0.414 hit@1. `describe.skip` records the decision.
|
|
735
731
|
|
|
736
|
-
**The vectors are 72.1% of the index and earn at most two questions**
|
|
737
|
-
default embedder — 74.8% Flask, 79.7%
|
|
738
|
-
|
|
732
|
+
**The vectors are 72.1% of the index and earn at most two questions** on any
|
|
733
|
+
ruler, in either direction, under the default embedder — 74.8% Flask, 79.7%
|
|
734
|
+
cobra. Kept: that same storage is what an optional model needs.
|
|
739
735
|
|
|
740
736
|
**`search.vector_recall` scans every vector per query** — under a semantic
|
|
741
737
|
embedder. The default hash never widens at all. Affordable at the measured
|
|
@@ -767,7 +763,8 @@ that rots. `pytest --cov=ragyourcode` is the command behind the coverage one.
|
|
|
767
763
|
CI runs Python 3.10–3.13 on Linux and Windows, installs the built wheel into a
|
|
768
764
|
clean environment and runs the documented CLI end to end — `bootstrap` through
|
|
769
765
|
`describe promote` — plus the skill's own install line verbatim, and grades
|
|
770
|
-
|
|
766
|
+
rulers A, C, D and E across this repository and both vendored corpora. Ruler B,
|
|
767
|
+
the cold parse of this repository, is a local command.
|
|
771
768
|
|
|
772
769
|
- [docs/FLOW.md](docs/FLOW.md) — the whole thing in four diagrams
|
|
773
770
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) — how each stage works and why
|
|
@@ -345,7 +345,7 @@ def test_every_benchmark_script_is_listed_in_its_own_index():
|
|
|
345
345
|
Both directions matter. A script absent from the index is one nobody knows
|
|
346
346
|
to run; a command in the index naming a script that does not exist is the
|
|
347
347
|
install-line defect this repository shipped twice. Discovery by glob is
|
|
348
|
-
what keeps the
|
|
348
|
+
what keeps the newest script from being the one nothing checks.
|
|
349
349
|
"""
|
|
350
350
|
index = (ROOT / "benchmarks" / "README.md").read_text(encoding="utf-8")
|
|
351
351
|
assert BENCHMARKS, "benchmarks/ holds no scripts; this guard would pass vacuously"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|