rag-your-code 1.4.3__tar.gz → 1.5.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.4.3/src/rag_your_code.egg-info → rag_your_code-1.5.1}/PKG-INFO +288 -287
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/README.md +286 -286
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/pyproject.toml +5 -1
- {rag_your_code-1.4.3 → rag_your_code-1.5.1/src/rag_your_code.egg-info}/PKG-INFO +288 -287
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/rag_your_code.egg-info/SOURCES.txt +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/rag_your_code.egg-info/requires.txt +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/cli.py +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/config.py +14 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/descriptions.py +41 -3
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/workflow.py +13 -4
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_absent_queries.py +10 -2
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_descriptions.py +83 -0
- rag_your_code-1.5.1/tests/test_diagrams.py +151 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_repo_queries.py +20 -3
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/LICENSE +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/setup.cfg +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/src/ragyourcode/search.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_agentic.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_config.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_document.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_evidence.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_golden.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_local_model.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_metadata.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_providers.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_ranking.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_resilience.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.1}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.1
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -22,6 +22,7 @@ Description-Content-Type: text/markdown
|
|
|
22
22
|
License-File: LICENSE
|
|
23
23
|
Provides-Extra: dev
|
|
24
24
|
Requires-Dist: pytest>=7; extra == "dev"
|
|
25
|
+
Requires-Dist: pytest-cov>=4; extra == "dev"
|
|
25
26
|
Requires-Dist: tomli>=2.0; python_version < "3.11" and extra == "dev"
|
|
26
27
|
Provides-Extra: sentence-transformers
|
|
27
28
|
Requires-Dist: sentence-transformers>=2.2; extra == "sentence-transformers"
|
|
@@ -57,14 +58,12 @@ An agent looking for something in an unfamiliar codebase has two bad options.
|
|
|
57
58
|
|
|
58
59
|
**Grep** is fast and exact, and it only finds the string you already guessed.
|
|
59
60
|
Ask "where does it decide whether to answer at all" and there is no string to
|
|
60
|
-
grep for. **Reading whole files** is thorough and blows the context budget
|
|
61
|
-
five files of a real repository is tens of thousands of tokens, most of them
|
|
62
|
-
irrelevant.
|
|
61
|
+
grep for. **Reading whole files** is thorough and blows the context budget.
|
|
63
62
|
|
|
64
|
-
Retrieval sits in between
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
63
|
+
Retrieval sits in between and brings a third problem neither has: Grep can say
|
|
64
|
+
it found nothing, and **a ranking cannot.** It always produces a least-bad
|
|
65
|
+
candidate and returns it with a score and rank that read exactly like an answer,
|
|
66
|
+
whether or not the repository holds anything relevant.
|
|
68
67
|
|
|
69
68
|
## 2 · What it does
|
|
70
69
|
|
|
@@ -72,15 +71,14 @@ exactly like an answer, whether or not the repository holds anything relevant.
|
|
|
72
71
|
|---|---|
|
|
73
72
|
| **Index** | Every function, method and class in 15 languages becomes one `CodeUnit`: id, signature, exact line range, source, calls, imports, description. |
|
|
74
73
|
| **Retrieve** | BM25F over five weighted fields, blended with vector similarity. Results carry the terms they matched on. |
|
|
75
|
-
| **Refuse** | Two evidence tests decide whether *any* result is an answer
|
|
76
|
-
| **Expand** | Optional bounded walk over `calls` / `imports` / `contains`,
|
|
74
|
+
| **Refuse** | Two evidence tests decide whether *any* result is an answer; when neither is met, retrieval returns nothing plus a machine-readable diagnosis. |
|
|
75
|
+
| **Expand** | Optional bounded walk over `calls` / `imports` / `contains`, each hop carrying its edge path as evidence. |
|
|
77
76
|
| **Describe** | Your agent writes the vocabulary the source never contained, stored in a committed sidecar or promoted into the code as a reviewable diff. |
|
|
78
|
-
| **Serve** | A CLI,
|
|
77
|
+
| **Serve** | A CLI, plus a JSON-lines protocol for a long-lived agent subprocess. |
|
|
79
78
|
|
|
80
|
-
**Scope.** Retrieval over source declarations
|
|
79
|
+
**Scope.** Retrieval over source declarations — not a code-understanding model,
|
|
81
80
|
not a generation step, not an IDE index. Questions are answered in vocabulary
|
|
82
|
-
somebody wrote down: in the code, its documentation, or
|
|
83
|
-
added.
|
|
81
|
+
somebody wrote down: in the code, its documentation, or an agent's description.
|
|
84
82
|
|
|
85
83
|
## 3 · What is actually hard here
|
|
86
84
|
|
|
@@ -90,14 +88,14 @@ Three things, and all three are measured rather than argued.
|
|
|
90
88
|
|
|
91
89
|
Eight releases measured how well retrieval *finds* the answer. None could see
|
|
92
90
|
what it does when there is none, because every question graded had one. A
|
|
93
|
-
|
|
91
|
+
ruler of its own — thirty questions about subjects no graded repository
|
|
94
92
|
implements — settled it in one run: **all thirty answered**, both languages,
|
|
95
|
-
|
|
93
|
+
every corpus.
|
|
96
94
|
|
|
97
95
|
| asked of a repository containing no such code | answered with | on the evidence of |
|
|
98
96
|
|---|---|---|
|
|
99
97
|
| `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
|
|
100
|
-
|
|
|
98
|
+
| `准入钩子为什么会拒绝没有资源限额的工作负载` | the UTF-8 console setup | `拒绝` `没有` `为什么` |
|
|
101
99
|
| `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
|
|
102
100
|
|
|
103
101
|
Not a Chinese problem and not a ranking problem — a **missing question**: nothing
|
|
@@ -113,29 +111,29 @@ until you notice which half.
|
|
|
113
111
|
|
|
114
112
|
**Concentration** — what share of the query's *rarity* lands inside a single
|
|
115
113
|
declaration. Coverage alone asks whether each word occurs somewhere, which a
|
|
116
|
-
question about
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
114
|
+
question about an unimplemented subject can satisfy entirely out of unrelated
|
|
115
|
+
units: four of six words in four declarations with nothing to do with the
|
|
116
|
+
question or with one another. Rarity-weighted rather than counted, because two
|
|
117
|
+
ordinary words are not better evidence than the rare word asked about.
|
|
120
118
|
|
|
121
119
|
Both are **ratios inside the query**, never thresholds on a score: a score
|
|
122
|
-
threshold is tied to whatever scale the ranking produces, and
|
|
123
|
-
|
|
120
|
+
threshold is tied to whatever scale the ranking produces, and one here silently
|
|
121
|
+
stopped existing the moment BM25F changed that scale.
|
|
124
122
|
|
|
125
123
|
### 3.2 · The vector was carrying nothing, and here is why
|
|
126
124
|
|
|
127
125
|
The default embedder is a signed feature hash. Ablating it entirely moves the
|
|
128
|
-
|
|
129
|
-
occupy **72.1%** of the index.
|
|
126
|
+
four positive rulers by **at most two questions, and in both directions**, while
|
|
127
|
+
the vectors occupy **72.1%** of the index. Known since 0.6.0 and unexplained.
|
|
130
128
|
The explanation, measured here:
|
|
131
129
|
|
|
132
130
|
- **Not saturation.** Median 56 distinct tokens per unit into 384 buckets, 0.4%
|
|
133
|
-
|
|
134
|
-
|
|
131
|
+
over the width; widening to 16,384 raises fidelity from r=0.40 to r=0.56 and
|
|
132
|
+
buys no ranking.
|
|
135
133
|
- **Not redundancy.** Its cosine correlates only **+0.45** with BM25F over
|
|
136
134
|
26,490 scored candidates, so it does carry variance of its own.
|
|
137
135
|
- **The variance is the wrong variance.** A signed hash counts every token
|
|
138
|
-
equally
|
|
136
|
+
equally, so the independent part of what it measures is precisely the
|
|
139
137
|
contribution of words that are everywhere — the part rarity weighting exists
|
|
140
138
|
to discard. Independent *noise*, not independent signal.
|
|
141
139
|
- **And it can only reorder.** Candidates come from the lexical half, so a
|
|
@@ -143,18 +141,18 @@ The explanation, measured here:
|
|
|
143
141
|
questions have an accepted answer sharing **no token at all** with the query.
|
|
144
142
|
|
|
145
143
|
Eight replacement schemes were measured across releases — character n-grams,
|
|
146
|
-
random indexing, truncated SVD, posting-list signatures, a rarity-weighted
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
144
|
+
random indexing, truncated SVD, posting-list signatures, a rarity-weighted hash,
|
|
145
|
+
call-graph diffusion, postings expansion, authored-fields-only. None beat using
|
|
146
|
+
no vector: **a vector computed from the same words cannot know anything the
|
|
147
|
+
words do not already say.** Making it useful takes a model, which is an
|
|
150
148
|
installable option and is measured below.
|
|
151
149
|
|
|
152
150
|
### 3.3 · Retrieval reaches only what somebody wrote down
|
|
153
151
|
|
|
154
152
|
`retry_charge` tokenizes to one opaque term, not to *retry* and *charge*.
|
|
155
153
|
Splitting identifiers was measured with query and stored vectors rebuilt
|
|
156
|
-
together: equal or worse on every ruler, because the pieces are `get`, `find
|
|
157
|
-
`check`,
|
|
154
|
+
together: equal or worse on every ruler, because the pieces are `get`, `find`
|
|
155
|
+
and `check`, which rarity weighting discounts.
|
|
158
156
|
|
|
159
157
|
So the vocabulary ladder is the answer, cheapest rung first:
|
|
160
158
|
|
|
@@ -167,8 +165,7 @@ So the vocabulary ladder is the answer, cheapest rung first:
|
|
|
167
165
|
|
|
168
166
|
## 4 · How it works
|
|
169
167
|
|
|
170
|
-
|
|
171
|
-
**[docs/FLOW.md](docs/FLOW.md)**.
|
|
168
|
+
Drawn out, with the refusal path and the surfaces: **[docs/FLOW.md](docs/FLOW.md)**.
|
|
172
169
|
|
|
173
170
|
```
|
|
174
171
|
your repository
|
|
@@ -196,13 +193,12 @@ swallow the ones after it. A 530-byte JavaScript file that took 12.6 s to parse
|
|
|
196
193
|
now takes 0.37 ms.
|
|
197
194
|
|
|
198
195
|
Qualified names come from the spans the closer already produced: nested inside
|
|
199
|
-
another's span *is* nested in it,
|
|
200
|
-
|
|
196
|
+
another's span *is* nested in it. One mechanism, so there is no second one to
|
|
197
|
+
disagree with it.
|
|
201
198
|
|
|
202
199
|
**Ranking.** BM25F with per-field length normalisation, which is the part that
|
|
203
200
|
matters: against one length for the whole unit, a body repeating a word forty
|
|
204
|
-
times
|
|
205
|
-
advantage cancelled its length penalty.
|
|
201
|
+
times beat the declaration named after it, raw count cancelling length penalty.
|
|
206
202
|
|
|
207
203
|
| field | weight | why |
|
|
208
204
|
|---|---|---|
|
|
@@ -214,13 +210,13 @@ advantage cancelled its length penalty.
|
|
|
214
210
|
|
|
215
211
|
**Rarity comes from your corpus, not a stopword list.** `the` and `calls` earn
|
|
216
212
|
their low weight the same way a Chinese bigram does — by being everywhere — so
|
|
217
|
-
|
|
213
|
+
no list is maintained and an unanticipated language works. It is also where the
|
|
214
|
+
design degrades: see the refusal table in section 6.
|
|
218
215
|
|
|
219
216
|
**Safety.** A scanned repository is untrusted input, including any
|
|
220
217
|
`.rag-your-code/index.json` it ships, so nothing read out of an index may name
|
|
221
|
-
a path to act on: superseded
|
|
222
|
-
|
|
223
|
-
in-tree file and report success.
|
|
218
|
+
a path to act on: superseded sidecars are enumerated from the writer's own
|
|
219
|
+
naming scheme. A crafted index once made `index` delete an in-tree file.
|
|
224
220
|
|
|
225
221
|
## 5 · Before and after
|
|
226
222
|
|
|
@@ -228,7 +224,7 @@ A question with no lexical shortcut, asked of this repository:
|
|
|
228
224
|
|
|
229
225
|
````console
|
|
230
226
|
$ rag-your-code search "where does it decide whether to answer at all" --limit 1
|
|
231
|
-
[src/ragyourcode/search.py:117:Evidence] score=0.
|
|
227
|
+
[src/ragyourcode/search.py:117:Evidence] score=0.447
|
|
232
228
|
The verdict on whether a question reached this index at all, kept separate from
|
|
233
229
|
how results rank. ... 中文:判定一个提问究竟有没有够到索引的结论。...
|
|
234
230
|
```python
|
|
@@ -239,13 +235,12 @@ class Evidence:
|
|
|
239
235
|
|
|
240
236
|
There is no string here to grep for: *decide* occurs nowhere in that
|
|
241
237
|
declaration and matched nothing. What ranked it first is ordinary words —
|
|
242
|
-
*answer*, *whether*, *where* —
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
238
|
+
*answer*, *whether*, *where* — rare enough in this corpus to tell declarations
|
|
239
|
+
apart. What the agent-written description adds is the other language:
|
|
240
|
+
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.395,
|
|
241
|
+
sharing not one character with its source.
|
|
246
242
|
|
|
247
|
-
Now the case that motivated 1.0.0
|
|
248
|
-
answer to at all:
|
|
243
|
+
Now the case that motivated 1.0.0 — a question with no answer here at all:
|
|
249
244
|
|
|
250
245
|
```console
|
|
251
246
|
$ rag-your-code search "why does the print spooler leave a duplex job stuck"
|
|
@@ -274,7 +269,7 @@ it happens to use elsewhere.
|
|
|
274
269
|
Read `coverage: 0.5` against `concentration: 0.1691`. Half the distinctive
|
|
275
270
|
words are here — `job`, `leave`, `print` — and spread thin enough that no
|
|
276
271
|
declaration holds a fifth of what was asked, against a bar of 0.28. Before
|
|
277
|
-
1.1.0
|
|
272
|
+
1.1.0 it came back with a confident-looking result.
|
|
278
273
|
|
|
279
274
|
Four reasons, because each is recovered by a different move:
|
|
280
275
|
|
|
@@ -283,116 +278,124 @@ Four reasons, because each is recovered by a different move:
|
|
|
283
278
|
| `no_query_term_in_index` | no word of the question occurs anywhere | ask in the code's vocabulary |
|
|
284
279
|
| `only_ubiquitous_terms_matched` | only words the repository uses throughout | add a distinctive term |
|
|
285
280
|
| `too_little_of_the_query_matched` | most of the question is absent | rephrase, or write descriptions |
|
|
286
|
-
| `matched_terms_are_scattered` | the words are here, never together | the subject is probably not
|
|
281
|
+
| `matched_terms_are_scattered` | the words are here, never together | the subject is probably not here |
|
|
287
282
|
|
|
288
283
|
## 6 · Benchmark dashboard
|
|
289
284
|
|
|
290
|
-
|
|
291
|
-
one of them runs against
|
|
292
|
-
**found**; the
|
|
293
|
-
|
|
294
|
-
|
|
285
|
+
Five rulers, 175 distinct questions in English and Chinese, graded 305 times —
|
|
286
|
+
one of them runs against all three corpora. Four grade whether the answer is
|
|
287
|
+
**found**; the fifth grades whether silence is **kept**. Every report carries a
|
|
288
|
+
fingerprint of the corpus it graded, because between two runs of an unchanged
|
|
289
|
+
`search.py` the foreign ruler moved 0.257 → 0.229 purely because that
|
|
295
290
|
repository had grown by ninety units.
|
|
296
291
|
|
|
297
292
|
**Accuracy — default embedder, zero dependencies**
|
|
298
293
|
|
|
299
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
300
295
|
|---|---|---|---|---|---|
|
|
301
|
-
| **
|
|
302
|
-
| **
|
|
303
|
-
| **
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
`
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
|
341
|
-
|
|
296
|
+
| **E** cobra v1.9.1, Go, no descriptions | a foreign repo in a foreign language | 40 | 0.075 | 0.150 | 0.108 |
|
|
297
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time Python user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
298
|
+
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.381 |
|
|
299
|
+
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.429 | 0.600 | 0.498 |
|
|
300
|
+
|
|
301
|
+
The corpora, without which none of the above is reproducible — **E** 602 units,
|
|
302
|
+
`3eabaa705477`; **A** 1,572 units, `5fd51169eacc`; **B** 604 units,
|
|
303
|
+
`81e47eb0a50c`; **C** 604 units, `f84556ba7881`. Both foreign subjects are
|
|
304
|
+
carried in this repository at pinned tags, under
|
|
305
|
+
[`benchmarks/corpus/`](benchmarks/corpus/), and CI runs both as ordinary jobs.
|
|
306
|
+
|
|
307
|
+
**The spread across those four rows is the honest headline.** The same code
|
|
308
|
+
scores 0.075 and 0.429 depending on nothing but which repository it is asked
|
|
309
|
+
about and whether anyone described it. Ruler E is 1.5.0's third corpus and the
|
|
310
|
+
first that is not Python: Go documents *above* the declaration in one terse
|
|
311
|
+
sentence beginning with the identifier, which the parser picks up correctly and
|
|
312
|
+
which shares almost nothing with the words a user asks in. Prose density, not
|
|
313
|
+
language, is what a cold number tracks.
|
|
314
|
+
|
|
315
|
+
**Refusal — the fifth ruler, 30 questions with no answer anywhere**
|
|
316
|
+
|
|
317
|
+
| | this repo | Flask | cobra |
|
|
318
|
+
|---|---|---|---|
|
|
319
|
+
| correctly met with silence | **0.967** | **0.833** | **0.900** |
|
|
320
|
+
| English only | **0.933** | 0.667 | 0.800 |
|
|
321
|
+
| Chinese only | **1.000** | **1.000** | **1.000** |
|
|
322
|
+
| results resting on no lexical evidence | **0.000** | **0.000** | **0.000** |
|
|
323
|
+
|
|
324
|
+
Silence is lower on both foreign corpora than on this one, and the cause is a
|
|
325
|
+
limit of the design rather than a defect. A word counts as evidence unless it
|
|
326
|
+
occurs in more than 5% of units — a stopword list derived from the corpus, so
|
|
327
|
+
that it needs no list and works in any language. Here `how`, `when`, `does` and
|
|
328
|
+
`are` are everywhere, because 317 units carry written English prose. Across a
|
|
329
|
+
corpus of short, tersely documented declarations they occur in 1–5% of them and
|
|
330
|
+
start counting as evidence: on cobra, three English questions get through on
|
|
331
|
+
sets like `[a, is, the, how, after]`.
|
|
332
|
+
|
|
333
|
+
**What each bar costs and buys** — corpora stamped above, gate varied alone:
|
|
334
|
+
|
|
335
|
+
| gate | A | B | C | E | silence own / Flask / cobra |
|
|
336
|
+
|---|---|---|---|---|---|
|
|
337
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.388 | 0.486/0.686/0.567 | 0.100/0.200/0.146 | 0.000 / 0.000 / 0.000 |
|
|
338
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.381 | 0.486/0.671/0.559 | 0.075/0.175/0.121 | 0.500 / 0.733 / 0.767 |
|
|
339
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.381 | 0.429/0.600/0.498 | 0.075/0.150/0.108 | 0.967 / 0.800 / 0.833 |
|
|
340
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.381** | 0.429/0.600/0.498 | 0.075/0.150/0.108 | **0.967 / 0.833 / 0.900** |
|
|
342
341
|
|
|
343
342
|
Ruler A is **unmoved by either bar**; B loses one hit@3 question to either bar
|
|
344
|
-
alone and nothing further when both apply. The rest
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
this
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
Raising the
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
| | | across the five |
|
|
343
|
+
alone and nothing further when both apply. The rest is four of seventy at hit@1
|
|
344
|
+
on the warmest ruler and six at hit@3, plus one and two of forty on the Go one.
|
|
345
|
+
|
|
346
|
+
**Both bars together give the best silence on all three corpora.** Through
|
|
347
|
+
1.3.0 this section said concentration subsumes coverage; on Flask it does not
|
|
348
|
+
(0.833 against 0.800), and the third corpus, which arrived long after the
|
|
349
|
+
defaults were fixed, says the same (0.900 against 0.833). Four constants fitted
|
|
350
|
+
on two repositories, tested on a third in a language nobody here chose, and not
|
|
351
|
+
one of them moved.
|
|
352
|
+
|
|
353
|
+
Raising the bar buys the remaining silence out of the answers, and is refused:
|
|
354
|
+
at 0.50 the foreign absent ruler is silent on all thirty while A falls to 0.086
|
|
355
|
+
hit@1, B to 0.214 and C to 0.329. **0.28 was chosen before either foreign
|
|
356
|
+
corpus existed and survived meeting both**, which is the only kind of evidence
|
|
357
|
+
a default can have.
|
|
358
|
+
|
|
359
|
+
**Latency** — warm corpus, 604 units `f84556ba7881`, one
|
|
360
|
+
`python -m benchmarks.query_latency` (5 repeats × 420 samples), idle machine:
|
|
361
|
+
|
|
362
|
+
| | | across the repeats |
|
|
366
363
|
|---|---|---|
|
|
367
|
-
| query, median | **0.
|
|
368
|
-
| query, p95 | 1.
|
|
369
|
-
| refusing an unanswerable query | **0.
|
|
370
|
-
| refusal cheaper than answering by | **~
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
figures
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
364
|
+
| query, median | **0.73 ms** | 0.70 – 0.78 |
|
|
365
|
+
| query, p95 | 1.30 ms | 1.17 – 1.39 |
|
|
366
|
+
| refusing an unanswerable query | **0.016 ms** | 0.015 – 0.018 |
|
|
367
|
+
| refusal cheaper than answering by | **~44×** | 40 – 47 |
|
|
368
|
+
|
|
369
|
+
*Idle* is load-bearing: the same corpus at the same commit measured 0.99 ms
|
|
370
|
+
median while a coverage run was in progress and 0.49 ms once it finished.
|
|
371
|
+
This row read 0.49 ms in 1.5.0 at 601 units; a corpus 0.5% larger cannot
|
|
372
|
+
explain 40%, and a repeat run here read 0.69 ms — the machine moved, not the code.
|
|
373
|
+
|
|
374
|
+
Two significant figures and a spread, because that is the precision this has.
|
|
375
|
+
Across five releases on the same idle machine the median has landed from 0.49
|
|
376
|
+
to 1.44 ms and p95 from 0.85 to 7.34 ms — a band wider than any change the code
|
|
377
|
+
has made to it. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to
|
|
378
|
+
three figures from an uncommitted script; both sit inside that band, which is
|
|
379
|
+
the point — unfalsifiable rather than wrong. Refusal is cheap structurally: an
|
|
380
|
+
unanswerable query touches only its distinctive words' posting lists, never ranking.
|
|
381
|
+
|
|
382
|
+
**Scale**, synthetic 10,000-unit repository (500 files), re-measured in 1.5.0
|
|
383
|
+
— the previous row of figures was optimistic by more than noise:
|
|
384
|
+
|
|
385
|
+
| | | previously published |
|
|
386
|
+
|---|---|---|
|
|
387
|
+
| full build | 3.45 s | 1.84 s |
|
|
388
|
+
| incremental rebuild after one file changes | 0.286 s (**12.1×**) | 0.207 s |
|
|
389
|
+
| compact storage vs readable JSON | 35.6% | 35.6% |
|
|
390
|
+
| index load, fresh process | 79.3 ms | 45.4 ms |
|
|
391
|
+
| resident memory | 72.2 MiB | 58.7 MiB |
|
|
392
|
+
| mean query, full recall | 15.6 ms | 3.90 ms |
|
|
385
393
|
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
| incremental rebuild after one file changes | 0.207 s (**8.9×**) |
|
|
390
|
-
| compact storage vs readable JSON | 35.6% |
|
|
391
|
-
| index load, fresh process | 45.4 ms |
|
|
392
|
-
| resident memory | 58.7 MiB |
|
|
394
|
+
That last row had been carried since before BM25F replaced the scoring it was
|
|
395
|
+
taken under. Six releases, a committed script, and nothing that made anyone run
|
|
396
|
+
it again.
|
|
393
397
|
|
|
394
|
-
**Parsing**, against source-controlled fixtures (15 files, 237 negative cases
|
|
395
|
-
89 constructs the spec deliberately excludes):
|
|
398
|
+
**Parsing**, against source-controlled fixtures (15 files, 237 negative cases):
|
|
396
399
|
|
|
397
400
|
| | |
|
|
398
401
|
|---|---|
|
|
@@ -401,10 +404,10 @@ reaches ranking at all.
|
|
|
401
404
|
| with a usable signature | **91 / 91** |
|
|
402
405
|
| units invented that do not exist | **0** |
|
|
403
406
|
|
|
404
|
-
Directional local measurements, not service levels — but
|
|
405
|
-
|
|
407
|
+
Directional local measurements, not service levels — but each is a command
|
|
408
|
+
rather than a memory, which two of them were not before. Each
|
|
406
409
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
407
|
-
[`benchmarks/README.md`](benchmarks/README.md) lists the
|
|
410
|
+
[`benchmarks/README.md`](benchmarks/README.md) lists the seven scripts and what
|
|
408
411
|
each is for, and the corpus one of them grades is now carried here too.
|
|
409
412
|
|
|
410
413
|
## 7 · `rag-your-code search` vs a Grep loop
|
|
@@ -415,107 +418,97 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
415
418
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
416
419
|
declaration spans.
|
|
417
420
|
|
|
418
|
-
**
|
|
419
|
-
|
|
420
|
-
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
421
|
-
answerable by matching identifiers. Swapping the subject for a public web
|
|
422
|
-
framework reversed it. The honest claim is narrower than either table alone:
|
|
421
|
+
**On an undescribed repository, which side wins depends on the repository.**
|
|
422
|
+
Three subjects, three answers:
|
|
423
423
|
|
|
424
|
-
|
|
|
425
|
-
|
|
426
|
-
|
|
|
427
|
-
|
|
|
428
|
-
|
|
|
429
|
-
|
|
430
|
-
|
|
424
|
+
| undescribed · Grep loop → rag-your-code | first | top 3 | answered | characters |
|
|
425
|
+
|---|---|---|---|---|
|
|
426
|
+
| **Flask** 35 q · 1,572 units `5fd51169eacc` | 22.9% → **37.1%** | 45.7% → **57.1%** | 30 → 30 | 1,415,656 → **258,236** |
|
|
427
|
+
| **cobra** 40 q · 602 units `3eabaa705477` | 17.5% → 17.5% | 20.0% → **30.0%** | 22 → 17 | 799,475 → **156,336** |
|
|
428
|
+
| a hook-heavy tool, retired in 1.4.0 | 34.3% → 31.4% | — | — | — |
|
|
429
|
+
|
|
430
|
+
Flask wins for this side, cobra ties on first place, and the retired subject
|
|
431
|
+
lost. A cold index retrieves against a generated sentence plus whatever the
|
|
432
|
+
author documented, so the outcome is set by how much prose the repository
|
|
433
|
+
already carries — Flask documents most public methods in paragraphs, cobra in
|
|
434
|
+
one terse line, the retired tool barely at all. What holds on all three is the
|
|
435
|
+
payload: this side hands back a fifth to a half of the text, ranked and spanned.
|
|
431
436
|
|
|
432
437
|
**Once the vocabulary exists, it is not close.**
|
|
433
438
|
|
|
434
|
-
| this repository · 70 questions ·
|
|
439
|
+
| this repository · 70 questions · 604 units `f84556ba7881` · 317 described | Grep loop | rag-your-code |
|
|
435
440
|
|---|---|---|
|
|
436
441
|
| right file first | 22.9% | **58.6%** |
|
|
437
|
-
| right file in top 3 |
|
|
438
|
-
| lines it hands back, all questions |
|
|
439
|
-
| characters returned, all questions | 1,
|
|
442
|
+
| right file in top 3 | 52.9% | **78.6%** |
|
|
443
|
+
| lines it hands back, all questions | 12,540 | — |
|
|
444
|
+
| characters returned, all questions | 1,190,816 | **611,859** |
|
|
440
445
|
| questions it answers | **61** | 60 |
|
|
441
446
|
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
Until then this section — the strongest claim the project makes — came from a
|
|
456
|
-
script that was never committed, so nothing here could be checked and "Grep
|
|
457
|
-
loop" had no precise meaning. The committed version defines it: take the
|
|
458
|
-
query's words, drop the ones the corpus itself shows are everywhere, run one
|
|
459
|
-
substring search per remaining word over exactly the files the index was built
|
|
460
|
-
from, rank each file by how many distinct words hit it, break ties on path.
|
|
461
|
-
Reconstructing it reproduced this side's figures exactly and moved Grep's —
|
|
462
|
-
the expected shape, since the ranked arm was always a call into shipped code.
|
|
463
|
-
|
|
464
|
-
Four qualifications, because the table would otherwise flatter both sides:
|
|
447
|
+
That is section 3.3's argument measured rather than asserted, and it is the one
|
|
448
|
+
thing the subject does not change: first-place accuracy more than doubles
|
|
449
|
+
Grep's.
|
|
450
|
+
|
|
451
|
+
**Every table comes from `python -m benchmarks.grep_baseline`**, new in 1.3.0.
|
|
452
|
+
Until then this section — the strongest claim the project makes — came from an
|
|
453
|
+
uncommitted script, so nothing here could be checked and "Grep loop" had no
|
|
454
|
+
precise meaning. The committed version defines it: take the query's words, drop
|
|
455
|
+
the ones the corpus itself shows are everywhere, run one substring search per
|
|
456
|
+
remaining word over exactly the files the index was built from, rank each file
|
|
457
|
+
by how many distinct words hit it, break ties on path.
|
|
458
|
+
|
|
459
|
+
Qualifications, because the tables would otherwise flatter both sides:
|
|
465
460
|
|
|
466
461
|
- **Scored at file granularity**, which understates this side. A Grep hit is a
|
|
467
462
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
468
463
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
469
|
-
- **Dropping the corpus-common words is generous to Grep**, and
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
decline the same five of
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
cross-language case is not this tool's failure but the corpus's.
|
|
464
|
+
- **Dropping the corpus-common words is generous to Grep**, and is what makes
|
|
465
|
+
the baseline fair rather than a straw man: an agent that greps `the` gets
|
|
466
|
+
every file back in no order. It is also why Grep declines nine of the seventy
|
|
467
|
+
questions here — no word was left that this corpus does not use everywhere.
|
|
468
|
+
- **Payload is counted in characters on both sides.** Grep hands back roughly
|
|
469
|
+
19,300 characters per question it answers here, unranked and without spans,
|
|
470
|
+
against 10,300 ranked and capped by `search.max_chars` — a factor of 1.9,
|
|
471
|
+
5.5 on Flask and 5.1 on cobra, where a framework repeats its vocabulary
|
|
472
|
+
across files and Grep cannot rank what it finds. 1.4.1 changed what fits in
|
|
473
|
+
that cap: the block had been reprinting the docstring the code below already
|
|
474
|
+
showed, so the same budget now carries **119 declarations instead of 92** on
|
|
475
|
+
Flask.
|
|
476
|
+
- **Chinese is the corpus's limit, not the tool's, when a corpus is
|
|
477
|
+
monolingual.** Both sides decline the same five of Flask's 35 — every Chinese
|
|
478
|
+
one. A Chinese word is neither a substring of English source nor a token in
|
|
479
|
+
an index built from it.
|
|
486
480
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
487
481
|
exact, instant and complete, and nothing here replaces it.
|
|
488
482
|
|
|
489
483
|
The two are complementary, and the honest summary is narrow: this earns its
|
|
490
|
-
place on questions phrased as questions, over a repository somebody
|
|
491
|
-
the time to describe.
|
|
484
|
+
place on questions phrased as questions, over a repository somebody described.
|
|
492
485
|
|
|
493
486
|
## 8 · Design principles
|
|
494
487
|
|
|
495
488
|
**Build the ruler before reshaping the thing measured.** Four candidate scoring
|
|
496
489
|
changes once landed between five and six correct over an eight-question set —
|
|
497
|
-
the instrument's resolution limit, not a ranking. There are
|
|
498
|
-
across
|
|
490
|
+
the instrument's resolution limit, not a ranking. There are 175 questions now
|
|
491
|
+
across five rulers, and every claim here is a number from one of them.
|
|
499
492
|
|
|
500
493
|
**Measure somewhere it can fail.** Every ruler this project had once graded a
|
|
501
494
|
repository its own authors wrote; cold against a foreign one the same code
|
|
502
|
-
scored 0.086 hit@1 against a self-reported 0.457.
|
|
503
|
-
|
|
504
|
-
|
|
495
|
+
scored 0.086 hit@1 against a self-reported 0.457. Rulers A and E exist so that
|
|
496
|
+
cannot be comfortable again — stemming, which helps both own-repo rulers, was
|
|
497
|
+
rejected on A, and E publishes a 0.075 this project would rather not print.
|
|
505
498
|
|
|
506
499
|
**Make the error structurally impossible rather than checking for it.** A line
|
|
507
|
-
number that *is* the loop index cannot drift
|
|
508
|
-
of its own code cannot outlive it.
|
|
509
|
-
|
|
510
|
-
**A ratio inside the query, never a threshold on a score.** Scales move;
|
|
511
|
-
ratios do not.
|
|
500
|
+
number that *is* the loop index cannot drift; a description keyed by a digest
|
|
501
|
+
of its own code cannot outlive it. **And a ratio inside the query, never a
|
|
502
|
+
threshold on a score** — scales move, ratios do not.
|
|
512
503
|
|
|
513
504
|
**Derive figures from data; a hand-maintained number is a claim nobody checks.**
|
|
514
|
-
|
|
515
|
-
|
|
505
|
+
This README's settings table is asserted against `config.py` in both directions
|
|
506
|
+
— it had drifted nine settings behind before that test existed — and the four
|
|
507
|
+
diagrams in `docs/FLOW.md` are parsed by a test that refuses a label mermaid
|
|
508
|
+
would silently fail to render.
|
|
516
509
|
|
|
517
510
|
**The contract does not move.** `CodeUnit`, index schema 2 and the JSON-lines
|
|
518
|
-
protocol are unchanged across every release
|
|
511
|
+
protocol are unchanged across every release: new information arrives in new
|
|
519
512
|
fields, never by widening an enumeration callers branch on.
|
|
520
513
|
|
|
521
514
|
**Publish what was measured and rejected.** Twelve changes were implemented,
|
|
@@ -539,14 +532,14 @@ dimensions = 384
|
|
|
539
532
|
pip install "rag-your-code[sentence-transformers]"
|
|
540
533
|
```
|
|
541
534
|
|
|
542
|
-
The extra is optional by construction: `dependencies = []` is
|
|
543
|
-
install
|
|
535
|
+
The extra is optional by construction: `dependencies = []` is a default
|
|
536
|
+
install, the import happens inside the constructor, and a test asserts the
|
|
544
537
|
default provider imports none of it.
|
|
545
538
|
|
|
546
|
-
**Measured on the
|
|
547
|
-
published this comparison and read it as a win
|
|
548
|
-
foreign ruler, whose two arms turned out to have been taken
|
|
549
|
-
|
|
539
|
+
**Measured on the four rulers that then existed, both arms against one
|
|
540
|
+
corpus.** 1.1.0 published this comparison and read it as a win; its largest
|
|
541
|
+
gain was on the foreign ruler, whose two arms turned out to have been taken
|
|
542
|
+
against two states of a repository being edited while the script ran. Repeated
|
|
550
543
|
against a pinned corpus:
|
|
551
544
|
|
|
552
545
|
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
@@ -558,8 +551,8 @@ against a pinned corpus:
|
|
|
558
551
|
|
|
559
552
|
**Worse or identical on every ruler.** The 581-unit stamps are the corpus both
|
|
560
553
|
arms shared, kept rather than refreshed — that is what a stamp is for. It ships
|
|
561
|
-
anyway
|
|
562
|
-
|
|
554
|
+
anyway for the one thing the hash cannot do and these rulers cannot see: reach
|
|
555
|
+
a unit sharing no word with the question. The pairs it scores zero on:
|
|
563
556
|
|
|
564
557
|
| pair | signed hash | MiniLM |
|
|
565
558
|
|---|---|---|
|
|
@@ -568,18 +561,16 @@ reach a unit sharing no word with the question. The pairs it scores zero on:
|
|
|
568
561
|
| `计算两个数的和` vs `sum two numbers` | **0.000** | **0.822** |
|
|
569
562
|
| `刷新索引` vs `rebuild the index` | **0.000** | **0.684** |
|
|
570
563
|
|
|
571
|
-
A semantic embedder is **not** exempt from the evidence bars,
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
564
|
+
A semantic embedder is **not** exempt from the evidence bars, correcting
|
|
565
|
+
1.0.0. Exempting one was reasoned — a paraphrase sharing no word with its
|
|
566
|
+
answer is exactly what a model is for — and wrong: exempt and asked no other
|
|
567
|
+
question, the model answered all sixty unanswerable questions. Two vector-space
|
|
568
|
+
replacements were then measured and rejected: a similarity floor is a threshold
|
|
569
|
+
on a score and the distributions overlap (0.469 vs 0.418 median), and a
|
|
570
|
+
scale-free standout metric took ruler B from 0.329 to 0.186 for two thirds of
|
|
571
|
+
the silence. Applying the lexical bars costs ruler A nothing.
|
|
579
572
|
|
|
580
|
-
|
|
581
|
-
it for the cross-language and paraphrase cases in the table above, which is
|
|
582
|
-
where the difference between the two columns actually lives.
|
|
573
|
+
Install it for the cross-language and paraphrase cases above, not for the hit rates.
|
|
583
574
|
|
|
584
575
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
585
576
|
source anywhere:
|
|
@@ -592,17 +583,17 @@ dimensions = 1536
|
|
|
592
583
|
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
593
584
|
```
|
|
594
585
|
|
|
595
|
-
The key is never a setting
|
|
596
|
-
everyone who clones sees what shaped the index
|
|
597
|
-
with the opposite requirement. Sending a key over plain `http://` to
|
|
598
|
-
but your own machine is refused rather than warned about
|
|
599
|
-
build rather than falling back
|
|
600
|
-
a cosine across them is a meaningless number ranking would act on
|
|
586
|
+
The key is never a setting: `rag-your-code.toml` is meant to be committed so
|
|
587
|
+
everyone who clones sees what shaped the index, and a credential is the one
|
|
588
|
+
value with the opposite requirement. Sending a key over plain `http://` to
|
|
589
|
+
anything but your own machine is refused rather than warned about, and a
|
|
590
|
+
failure stops the build rather than falling back — a mixed index is two vector
|
|
591
|
+
spaces, and a cosine across them is a meaningless number ranking would act on.
|
|
601
592
|
|
|
602
593
|
With a semantic embedder, similarity may also **add** candidates rather than
|
|
603
594
|
only reorder them (`search.vector_recall`) — the one thing that can reach a
|
|
604
|
-
unit sharing no word with the question. Under the hash
|
|
605
|
-
|
|
595
|
+
unit sharing no word with the question. Under the hash it measured worse, so it
|
|
596
|
+
stays off there.
|
|
606
597
|
|
|
607
598
|
## 10 · Install and use
|
|
608
599
|
|
|
@@ -650,14 +641,13 @@ rag-your-code describe promote | git apply # move descriptions into the cod
|
|
|
650
641
|
`bootstrap` exists because indexing is not the same as being searchable: a
|
|
651
642
|
fresh index retrieves against a generated sentence that adds no word the source
|
|
652
643
|
did not have. It reports which rung the repository is on and hands over that
|
|
653
|
-
rung's work; run it again after each round.
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
modified**. `describe promote` emits a diff to review — it never writes source.
|
|
644
|
+
rung's work; run it again after each round. The index is written under
|
|
645
|
+
`.rag-your-code/` and **your source files are never modified** — `describe
|
|
646
|
+
promote` emits a diff to review rather than writing source.
|
|
657
647
|
|
|
658
648
|
### Configuration
|
|
659
649
|
|
|
660
|
-
|
|
650
|
+
23 settings in `rag-your-code.toml`:
|
|
661
651
|
|
|
662
652
|
| section | settings |
|
|
663
653
|
|---|---|
|
|
@@ -665,7 +655,7 @@ modified**. `describe promote` emits a diff to review — it never writes source
|
|
|
665
655
|
| `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
|
|
666
656
|
| `[search]` | `min_coverage`, `min_concentration`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
|
|
667
657
|
| `[agent]` | `max_open_bytes`, `max_open_chars` |
|
|
668
|
-
| `[describe]` | `languages`, `batch`, `max_chars` |
|
|
658
|
+
| `[describe]` | `languages`, `batch`, `max_chars`, `skip` |
|
|
669
659
|
|
|
670
660
|
This table is asserted against `config.py` in both directions by
|
|
671
661
|
`tests/test_metadata.py`. Resolution is CLI flag > file > built-in default;
|
|
@@ -724,37 +714,45 @@ questions: it is the corpus's limit, not this tool's.
|
|
|
724
714
|
|
|
725
715
|
**There is no stemming.** `catastrophic backtracking` does not reach
|
|
726
716
|
`backtracks catastrophically`. A light suffix stripper was implemented and
|
|
727
|
-
measured on
|
|
728
|
-
the foreign one 3 of 35 hit@3, so it was rejected.
|
|
729
|
-
|
|
730
|
-
**A test declaration sometimes outranks real code** — 9 of
|
|
731
|
-
|
|
732
|
-
none of them
|
|
733
|
-
code it *tests*, is wrong: five of the
|
|
734
|
-
prose. A callee-before-caller rerank fires
|
|
735
|
-
field weight moves nothing because an
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
717
|
+
measured on every ruler that then existed: it improves both own-repository
|
|
718
|
+
rulers and costs the foreign one 3 of 35 hit@3, so it was rejected.
|
|
719
|
+
|
|
720
|
+
**A test declaration sometimes outranks real code** — 9 of 215 questions across
|
|
721
|
+
all four positive rulers (`benchmarks/displacement.py`), a test at rank 1
|
|
722
|
+
displacing an accepted answer at rank 2–3, **none of them foreign**. The long-standing
|
|
723
|
+
explanation, that a test outranks the code it *tests*, is wrong: five of the
|
|
724
|
+
nine are unrelated tests winning on prose. A callee-before-caller rerank fires
|
|
725
|
+
on zero questions, and the `name` field weight moves nothing because an
|
|
726
|
+
underscored test name is one token.
|
|
727
|
+
|
|
728
|
+
**Describing a declaration that already has a good docstring loses ground.** An
|
|
729
|
+
authored description *replaces* the generated sentence, which is the only route
|
|
730
|
+
by which the author's docstring reaches the weight-3 description field — so
|
|
731
|
+
writing one demotes it to the weight-1 body. On `parser.py::_generic_units` a
|
|
732
|
+
long description cost one graded question and a short one cost three;
|
|
733
|
+
appending the docstring to every description instead cost the 1.5.0 corpus
|
|
734
|
+
0.443 → 0.414 hit@1. `describe.skip` records the decision.
|
|
735
|
+
|
|
736
|
+
**The vectors are 72.1% of the index and earn at most two questions** under the
|
|
737
|
+
default embedder — 74.8% Flask, 79.7% cobra. Ablating costs A one, gains B one
|
|
738
|
+
and C two at hit@3, moves E none; that storage is what an optional model needs.
|
|
739
739
|
|
|
740
740
|
**`search.vector_recall` scans every vector per query** — under a semantic
|
|
741
741
|
embedder. The default hash never widens at all. Affordable at the measured
|
|
742
742
|
envelope, and exactly the work an ANN index would replace.
|
|
743
743
|
|
|
744
|
-
**Tree-sitter parsing and a SQLite/ANN storage layer are not here.** Both
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
**Whether the skill fires unprompted is not measured**, and
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
nothing. Since 1.2.0 the four commands give an entry path that does not depend
|
|
757
|
-
on it.
|
|
744
|
+
**Tree-sitter parsing and a SQLite/ANN storage layer are not here.** Both need
|
|
745
|
+
a dependency, and the policy is settled: they follow the embedding provider's
|
|
746
|
+
pattern — optional, user-selected, never in a default install. Full reasoning
|
|
747
|
+
in [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
748
|
+
|
|
749
|
+
**Whether the skill fires unprompted is not measured**, and it is the only
|
|
750
|
+
claim here with no command behind it. `claude plugin eval` grades exactly this
|
|
751
|
+
— `tool_used: Skill` as the indicator, against a no-plugin baseline arm — but
|
|
752
|
+
it is gated behind an account-level early access this project does not have,
|
|
753
|
+
and a suite written from `--help` fragments could not be run once to see
|
|
754
|
+
whether it loads. Since 1.2.0 the four commands give an entry path that does
|
|
755
|
+
not depend on it.
|
|
758
756
|
|
|
759
757
|
## 12 · Development
|
|
760
758
|
|
|
@@ -763,17 +761,20 @@ python -m pip install -e ".[dev]"
|
|
|
763
761
|
pytest -q
|
|
764
762
|
```
|
|
765
763
|
|
|
766
|
-
Per-release test counts
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
764
|
+
Per-release test counts and line coverage are in
|
|
765
|
+
[CHANGELOG.md](CHANGELOG.md); a bare figure in a living document is a claim
|
|
766
|
+
that rots. `pytest --cov=ragyourcode` is the command behind the coverage one.
|
|
767
|
+
CI runs Python 3.10–3.13 on Linux and Windows, installs the built wheel into a
|
|
768
|
+
clean environment and runs the documented CLI end to end — `bootstrap` through
|
|
769
|
+
`describe promote` — plus the skill's own install line verbatim, and grades
|
|
770
|
+
every ruler including both vendored corpora.
|
|
771
771
|
|
|
772
772
|
- [docs/FLOW.md](docs/FLOW.md) — the whole thing in four diagrams
|
|
773
773
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) — how each stage works and why
|
|
774
|
-
- [docs/TESTING.md](docs/TESTING.md) — what the suites protect
|
|
774
|
+
- [docs/TESTING.md](docs/TESTING.md) — what the suites protect, and how covered
|
|
775
775
|
- [docs/ROADMAP.md](docs/ROADMAP.md) — what shipped, what was rejected and why
|
|
776
776
|
- [CONTRIBUTING.md](CONTRIBUTING.md) — ground rules, and how to add a language
|
|
777
777
|
- [CHANGELOG.md](CHANGELOG.md) — every release with its measurements
|
|
778
|
+
- [benchmarks/corpus/](benchmarks/corpus/) — the two vendored repositories
|
|
778
779
|
|
|
779
780
|
MIT licensed.
|