rag-your-code 1.4.3__tar.gz → 1.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.4.3/src/rag_your_code.egg-info → rag_your_code-1.5.0}/PKG-INFO +273 -272
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/README.md +271 -271
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/pyproject.toml +5 -1
- {rag_your_code-1.4.3 → rag_your_code-1.5.0/src/rag_your_code.egg-info}/PKG-INFO +273 -272
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/rag_your_code.egg-info/SOURCES.txt +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/rag_your_code.egg-info/requires.txt +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/cli.py +1 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/config.py +14 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/descriptions.py +41 -3
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/workflow.py +13 -4
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_absent_queries.py +10 -2
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_descriptions.py +83 -0
- rag_your_code-1.5.0/tests/test_diagrams.py +151 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_repo_queries.py +20 -3
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/LICENSE +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/setup.cfg +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/src/ragyourcode/search.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_agentic.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_config.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_document.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_evidence.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_golden.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_local_model.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_metadata.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_providers.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_ranking.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_resilience.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.4.3 → rag_your_code-1.5.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.5.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -22,6 +22,7 @@ Description-Content-Type: text/markdown
|
|
|
22
22
|
License-File: LICENSE
|
|
23
23
|
Provides-Extra: dev
|
|
24
24
|
Requires-Dist: pytest>=7; extra == "dev"
|
|
25
|
+
Requires-Dist: pytest-cov>=4; extra == "dev"
|
|
25
26
|
Requires-Dist: tomli>=2.0; python_version < "3.11" and extra == "dev"
|
|
26
27
|
Provides-Extra: sentence-transformers
|
|
27
28
|
Requires-Dist: sentence-transformers>=2.2; extra == "sentence-transformers"
|
|
@@ -57,14 +58,12 @@ An agent looking for something in an unfamiliar codebase has two bad options.
|
|
|
57
58
|
|
|
58
59
|
**Grep** is fast and exact, and it only finds the string you already guessed.
|
|
59
60
|
Ask "where does it decide whether to answer at all" and there is no string to
|
|
60
|
-
grep for. **Reading whole files** is thorough and blows the context budget
|
|
61
|
-
five files of a real repository is tens of thousands of tokens, most of them
|
|
62
|
-
irrelevant.
|
|
61
|
+
grep for. **Reading whole files** is thorough and blows the context budget.
|
|
63
62
|
|
|
64
|
-
Retrieval sits in between
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
63
|
+
Retrieval sits in between and brings a third problem neither has: Grep can say
|
|
64
|
+
it found nothing, and **a ranking cannot.** It always produces a least-bad
|
|
65
|
+
candidate and returns it with a score and rank that read exactly like an answer,
|
|
66
|
+
whether or not the repository holds anything relevant.
|
|
68
67
|
|
|
69
68
|
## 2 · What it does
|
|
70
69
|
|
|
@@ -72,15 +71,14 @@ exactly like an answer, whether or not the repository holds anything relevant.
|
|
|
72
71
|
|---|---|
|
|
73
72
|
| **Index** | Every function, method and class in 15 languages becomes one `CodeUnit`: id, signature, exact line range, source, calls, imports, description. |
|
|
74
73
|
| **Retrieve** | BM25F over five weighted fields, blended with vector similarity. Results carry the terms they matched on. |
|
|
75
|
-
| **Refuse** | Two evidence tests decide whether *any* result is an answer
|
|
76
|
-
| **Expand** | Optional bounded walk over `calls` / `imports` / `contains`,
|
|
74
|
+
| **Refuse** | Two evidence tests decide whether *any* result is an answer; when neither is met, retrieval returns nothing plus a machine-readable diagnosis. |
|
|
75
|
+
| **Expand** | Optional bounded walk over `calls` / `imports` / `contains`, each hop carrying its edge path as evidence. |
|
|
77
76
|
| **Describe** | Your agent writes the vocabulary the source never contained, stored in a committed sidecar or promoted into the code as a reviewable diff. |
|
|
78
77
|
| **Serve** | A CLI, and a JSON-lines protocol for a long-lived agent subprocess. |
|
|
79
78
|
|
|
80
|
-
**Scope.** Retrieval over source declarations
|
|
79
|
+
**Scope.** Retrieval over source declarations — not a code-understanding model,
|
|
81
80
|
not a generation step, not an IDE index. Questions are answered in vocabulary
|
|
82
|
-
somebody wrote down: in the code, its documentation, or
|
|
83
|
-
added.
|
|
81
|
+
somebody wrote down: in the code, its documentation, or an agent's description.
|
|
84
82
|
|
|
85
83
|
## 3 · What is actually hard here
|
|
86
84
|
|
|
@@ -90,14 +88,14 @@ Three things, and all three are measured rather than argued.
|
|
|
90
88
|
|
|
91
89
|
Eight releases measured how well retrieval *finds* the answer. None could see
|
|
92
90
|
what it does when there is none, because every question graded had one. A
|
|
93
|
-
|
|
91
|
+
ruler of its own — thirty questions about subjects no graded repository
|
|
94
92
|
implements — settled it in one run: **all thirty answered**, both languages,
|
|
95
|
-
|
|
93
|
+
every corpus.
|
|
96
94
|
|
|
97
95
|
| asked of a repository containing no such code | answered with | on the evidence of |
|
|
98
96
|
|---|---|---|
|
|
99
97
|
| `where are CUDA kernels dispatched to the device` | a test about word counting | `are` `the` `to` `where` |
|
|
100
|
-
|
|
|
98
|
+
| `准入钩子为什么会拒绝没有资源限额的工作负载` | the UTF-8 console setup | `拒绝` `没有` `为什么` |
|
|
101
99
|
| `how is the OAuth refresh token rotated` | a description-store method | `before` `is` `refresh` `the` |
|
|
102
100
|
|
|
103
101
|
Not a Chinese problem and not a ranking problem — a **missing question**: nothing
|
|
@@ -113,14 +111,14 @@ until you notice which half.
|
|
|
113
111
|
|
|
114
112
|
**Concentration** — what share of the query's *rarity* lands inside a single
|
|
115
113
|
declaration. Coverage alone asks whether each word occurs somewhere, which a
|
|
116
|
-
question about
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
114
|
+
question about an unimplemented subject can satisfy entirely out of unrelated
|
|
115
|
+
units: four of six words in four declarations with nothing to do with the
|
|
116
|
+
question or with one another. Rarity-weighted rather than counted, because two
|
|
117
|
+
ordinary words are not better evidence than the rare word asked about.
|
|
120
118
|
|
|
121
119
|
Both are **ratios inside the query**, never thresholds on a score: a score
|
|
122
|
-
threshold is tied to whatever scale the ranking produces, and
|
|
123
|
-
|
|
120
|
+
threshold is tied to whatever scale the ranking produces, and one here silently
|
|
121
|
+
stopped existing the moment BM25F changed that scale.
|
|
124
122
|
|
|
125
123
|
### 3.2 · The vector was carrying nothing, and here is why
|
|
126
124
|
|
|
@@ -130,12 +128,12 @@ occupy **72.1%** of the index. That was known since 0.6.0 and left unexplained.
|
|
|
130
128
|
The explanation, measured here:
|
|
131
129
|
|
|
132
130
|
- **Not saturation.** Median 56 distinct tokens per unit into 384 buckets, 0.4%
|
|
133
|
-
|
|
134
|
-
|
|
131
|
+
over the width; widening to 16,384 raises fidelity from r=0.40 to r=0.56 and
|
|
132
|
+
buys no ranking.
|
|
135
133
|
- **Not redundancy.** Its cosine correlates only **+0.45** with BM25F over
|
|
136
134
|
26,490 scored candidates, so it does carry variance of its own.
|
|
137
135
|
- **The variance is the wrong variance.** A signed hash counts every token
|
|
138
|
-
equally
|
|
136
|
+
equally, so the independent part of what it measures is precisely the
|
|
139
137
|
contribution of words that are everywhere — the part rarity weighting exists
|
|
140
138
|
to discard. Independent *noise*, not independent signal.
|
|
141
139
|
- **And it can only reorder.** Candidates come from the lexical half, so a
|
|
@@ -143,18 +141,18 @@ The explanation, measured here:
|
|
|
143
141
|
questions have an accepted answer sharing **no token at all** with the query.
|
|
144
142
|
|
|
145
143
|
Eight replacement schemes were measured across releases — character n-grams,
|
|
146
|
-
random indexing, truncated SVD, posting-list signatures, a rarity-weighted
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
144
|
+
random indexing, truncated SVD, posting-list signatures, a rarity-weighted hash,
|
|
145
|
+
call-graph diffusion, postings expansion, authored-fields-only. None beat using
|
|
146
|
+
no vector: **a vector computed from the same words cannot know anything the
|
|
147
|
+
words do not already say.** Making it useful takes a model, which is an
|
|
150
148
|
installable option and is measured below.
|
|
151
149
|
|
|
152
150
|
### 3.3 · Retrieval reaches only what somebody wrote down
|
|
153
151
|
|
|
154
152
|
`retry_charge` tokenizes to one opaque term, not to *retry* and *charge*.
|
|
155
153
|
Splitting identifiers was measured with query and stored vectors rebuilt
|
|
156
|
-
together: equal or worse on every ruler, because the pieces are `get`, `find
|
|
157
|
-
`check`,
|
|
154
|
+
together: equal or worse on every ruler, because the pieces are `get`, `find`
|
|
155
|
+
and `check`, which rarity weighting discounts.
|
|
158
156
|
|
|
159
157
|
So the vocabulary ladder is the answer, cheapest rung first:
|
|
160
158
|
|
|
@@ -167,8 +165,7 @@ So the vocabulary ladder is the answer, cheapest rung first:
|
|
|
167
165
|
|
|
168
166
|
## 4 · How it works
|
|
169
167
|
|
|
170
|
-
|
|
171
|
-
**[docs/FLOW.md](docs/FLOW.md)**.
|
|
168
|
+
Drawn out, with the refusal path and the surfaces: **[docs/FLOW.md](docs/FLOW.md)**.
|
|
172
169
|
|
|
173
170
|
```
|
|
174
171
|
your repository
|
|
@@ -196,13 +193,12 @@ swallow the ones after it. A 530-byte JavaScript file that took 12.6 s to parse
|
|
|
196
193
|
now takes 0.37 ms.
|
|
197
194
|
|
|
198
195
|
Qualified names come from the spans the closer already produced: nested inside
|
|
199
|
-
another's span *is* nested in it,
|
|
200
|
-
|
|
196
|
+
another's span *is* nested in it. One mechanism, so there is no second one to
|
|
197
|
+
disagree with it.
|
|
201
198
|
|
|
202
199
|
**Ranking.** BM25F with per-field length normalisation, which is the part that
|
|
203
200
|
matters: against one length for the whole unit, a body repeating a word forty
|
|
204
|
-
times
|
|
205
|
-
advantage cancelled its length penalty.
|
|
201
|
+
times beat the declaration named after it, raw count cancelling length penalty.
|
|
206
202
|
|
|
207
203
|
| field | weight | why |
|
|
208
204
|
|---|---|---|
|
|
@@ -214,13 +210,13 @@ advantage cancelled its length penalty.
|
|
|
214
210
|
|
|
215
211
|
**Rarity comes from your corpus, not a stopword list.** `the` and `calls` earn
|
|
216
212
|
their low weight the same way a Chinese bigram does — by being everywhere — so
|
|
217
|
-
|
|
213
|
+
no list is maintained and an unanticipated language works. It is also where the
|
|
214
|
+
design degrades: see the refusal table in section 6.
|
|
218
215
|
|
|
219
216
|
**Safety.** A scanned repository is untrusted input, including any
|
|
220
217
|
`.rag-your-code/index.json` it ships, so nothing read out of an index may name
|
|
221
|
-
a path to act on: superseded
|
|
222
|
-
|
|
223
|
-
in-tree file and report success.
|
|
218
|
+
a path to act on: superseded sidecars are enumerated from the writer's own
|
|
219
|
+
naming scheme. A crafted index once made `index` delete an in-tree file.
|
|
224
220
|
|
|
225
221
|
## 5 · Before and after
|
|
226
222
|
|
|
@@ -228,7 +224,7 @@ A question with no lexical shortcut, asked of this repository:
|
|
|
228
224
|
|
|
229
225
|
````console
|
|
230
226
|
$ rag-your-code search "where does it decide whether to answer at all" --limit 1
|
|
231
|
-
[src/ragyourcode/search.py:117:Evidence] score=0.
|
|
227
|
+
[src/ragyourcode/search.py:117:Evidence] score=0.447
|
|
232
228
|
The verdict on whether a question reached this index at all, kept separate from
|
|
233
229
|
how results rank. ... 中文:判定一个提问究竟有没有够到索引的结论。...
|
|
234
230
|
```python
|
|
@@ -239,13 +235,12 @@ class Evidence:
|
|
|
239
235
|
|
|
240
236
|
There is no string here to grep for: *decide* occurs nowhere in that
|
|
241
237
|
declaration and matched nothing. What ranked it first is ordinary words —
|
|
242
|
-
*answer*, *whether*, *where* —
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
238
|
+
*answer*, *whether*, *where* — rare enough in this corpus to tell declarations
|
|
239
|
+
apart. What the agent-written description adds is the other language:
|
|
240
|
+
「在哪里判定一个提问有没有答案」 returns the same declaration first, at 0.395,
|
|
241
|
+
sharing not one character with its source.
|
|
246
242
|
|
|
247
|
-
Now the case that motivated 1.0.0
|
|
248
|
-
answer to at all:
|
|
243
|
+
Now the case that motivated 1.0.0 — a question with no answer here at all:
|
|
249
244
|
|
|
250
245
|
```console
|
|
251
246
|
$ rag-your-code search "why does the print spooler leave a duplex job stuck"
|
|
@@ -274,7 +269,7 @@ it happens to use elsewhere.
|
|
|
274
269
|
Read `coverage: 0.5` against `concentration: 0.1691`. Half the distinctive
|
|
275
270
|
words are here — `job`, `leave`, `print` — and spread thin enough that no
|
|
276
271
|
declaration holds a fifth of what was asked, against a bar of 0.28. Before
|
|
277
|
-
1.1.0
|
|
272
|
+
1.1.0 it came back with a confident-looking result.
|
|
278
273
|
|
|
279
274
|
Four reasons, because each is recovered by a different move:
|
|
280
275
|
|
|
@@ -283,116 +278,125 @@ Four reasons, because each is recovered by a different move:
|
|
|
283
278
|
| `no_query_term_in_index` | no word of the question occurs anywhere | ask in the code's vocabulary |
|
|
284
279
|
| `only_ubiquitous_terms_matched` | only words the repository uses throughout | add a distinctive term |
|
|
285
280
|
| `too_little_of_the_query_matched` | most of the question is absent | rephrase, or write descriptions |
|
|
286
|
-
| `matched_terms_are_scattered` | the words are here, never together | the subject is probably not
|
|
281
|
+
| `matched_terms_are_scattered` | the words are here, never together | the subject is probably not here |
|
|
287
282
|
|
|
288
283
|
## 6 · Benchmark dashboard
|
|
289
284
|
|
|
290
|
-
|
|
291
|
-
one of them runs against
|
|
292
|
-
**found**; the
|
|
293
|
-
|
|
294
|
-
|
|
285
|
+
Five rulers, 175 distinct questions in English and Chinese, graded 305 times —
|
|
286
|
+
one of them runs against all three corpora. Four grade whether the answer is
|
|
287
|
+
**found**; the fifth grades whether silence is **kept**. Every report carries a
|
|
288
|
+
fingerprint of the corpus it graded, because between two runs of an unchanged
|
|
289
|
+
`search.py` the foreign ruler moved 0.257 → 0.229 purely because that
|
|
295
290
|
repository had grown by ninety units.
|
|
296
291
|
|
|
297
292
|
**Accuracy — default embedder, zero dependencies**
|
|
298
293
|
|
|
299
294
|
| ruler | what it represents | n | hit@1 | hit@3 | MRR |
|
|
300
295
|
|---|---|---|---|---|---|
|
|
301
|
-
| **
|
|
302
|
-
| **
|
|
303
|
-
| **
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
`
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
|
341
|
-
|
|
296
|
+
| **E** cobra v1.9.1, Go, no descriptions | a foreign repo in a foreign language | 40 | 0.075 | 0.150 | 0.108 |
|
|
297
|
+
| **A** Flask 3.1.3, no descriptions | what a first-time Python user gets | 35 | 0.200 | 0.286 | 0.238 |
|
|
298
|
+
| **B** this repo, generated descriptions only | a cold index of familiar code | 70 | 0.314 | 0.471 | 0.381 |
|
|
299
|
+
| **C** this repo, agent-written descriptions | the warmest case supported | 70 | 0.429 | 0.600 | 0.498 |
|
|
300
|
+
|
|
301
|
+
The corpora, without which none of the above is reproducible — **E** 602 units,
|
|
302
|
+
`3eabaa705477`; **A** 1,572 units, `5fd51169eacc`; **B** 601 units,
|
|
303
|
+
`566616fbe1e7`; **C** 601 units, `ac3ae43a33e7`. Both foreign subjects are
|
|
304
|
+
carried in this repository at pinned tags, under
|
|
305
|
+
[`benchmarks/corpus/`](benchmarks/corpus/), and CI runs both as ordinary jobs.
|
|
306
|
+
|
|
307
|
+
**The spread across those four rows is the honest headline.** The same code
|
|
308
|
+
scores 0.075 and 0.429 depending on nothing but which repository it is asked
|
|
309
|
+
about and whether anyone described it. Ruler E is 1.5.0's third corpus and the
|
|
310
|
+
first that is not Python: Go documents *above* the declaration in one terse
|
|
311
|
+
sentence beginning with the identifier, which the parser picks up correctly and
|
|
312
|
+
which shares almost nothing with the words a user asks in. Prose density, not
|
|
313
|
+
language, is what a cold number tracks.
|
|
314
|
+
|
|
315
|
+
**Refusal — the fifth ruler, 30 questions with no answer anywhere**
|
|
316
|
+
|
|
317
|
+
| | this repo | Flask | cobra |
|
|
318
|
+
|---|---|---|---|
|
|
319
|
+
| correctly met with silence | **0.967** | **0.833** | **0.900** |
|
|
320
|
+
| English only | **0.933** | 0.667 | 0.800 |
|
|
321
|
+
| Chinese only | **1.000** | **1.000** | **1.000** |
|
|
322
|
+
| results resting on no lexical evidence | **0.000** | **0.000** | **0.000** |
|
|
323
|
+
|
|
324
|
+
Silence is lower on both foreign corpora than on this one, and the cause is a
|
|
325
|
+
limit of the design rather than a defect. A word counts as evidence unless it
|
|
326
|
+
occurs in more than 5% of units — a stopword list derived from the corpus, so
|
|
327
|
+
that it needs no list and works in any language. Here `how`, `when`, `does` and
|
|
328
|
+
`are` are everywhere, because 314 units carry written English prose. Across a
|
|
329
|
+
corpus of short, tersely documented declarations they occur in 1–5% of them and
|
|
330
|
+
start counting as evidence: on cobra, three English questions get through on
|
|
331
|
+
sets like `[a, is, the, how, after]`.
|
|
332
|
+
|
|
333
|
+
**What each bar costs and buys** — every corpus, gate varied alone:
|
|
334
|
+
|
|
335
|
+
| gate | A | B | C | E | silence own / Flask / cobra |
|
|
336
|
+
|---|---|---|---|---|---|
|
|
337
|
+
| neither (pre-1.0.0) | 0.200/0.286/0.238 | 0.314/0.486/0.388 | 0.486/0.686/0.567 | 0.100/0.200/0.146 | 0.000 / 0.000 / 0.000 |
|
|
338
|
+
| coverage only (1.0.0) | 0.200/0.286/0.238 | 0.314/0.471/0.381 | 0.486/0.671/0.559 | 0.075/0.175/0.121 | 0.500 / 0.733 / 0.767 |
|
|
339
|
+
| concentration only | 0.200/0.286/0.238 | 0.314/0.471/0.381 | 0.429/0.600/0.498 | 0.075/0.150/0.108 | 0.967 / 0.800 / 0.833 |
|
|
340
|
+
| **both (1.1.0)** | **0.200/0.286/0.238** | **0.314/0.471/0.381** | 0.429/0.600/0.498 | 0.075/0.150/0.108 | **0.967 / 0.833 / 0.900** |
|
|
342
341
|
|
|
343
342
|
Ruler A is **unmoved by either bar**; B loses one hit@3 question to either bar
|
|
344
|
-
alone and nothing further when both apply. The rest of the cost is
|
|
345
|
-
seventy at hit@1 on the warmest ruler
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
silent on all thirty while
|
|
357
|
-
|
|
358
|
-
existed and survived meeting
|
|
359
|
-
can have.
|
|
360
|
-
|
|
361
|
-
**Latency** — warm corpus,
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
| | | across the five |
|
|
343
|
+
alone and nothing further when both apply. The rest of the cost is four of
|
|
344
|
+
seventy at hit@1 on the warmest ruler and six at hit@3, plus one of forty and
|
|
345
|
+
two of forty on the Go one.
|
|
346
|
+
|
|
347
|
+
**Both bars together give the best silence on all three corpora.** Through
|
|
348
|
+
1.3.0 this section said concentration subsumes coverage; on Flask it does not
|
|
349
|
+
(0.833 against 0.800), and the third corpus, which arrived long after the
|
|
350
|
+
defaults were fixed, says the same (0.900 against 0.833). Four constants fitted
|
|
351
|
+
on two repositories, tested on a third in a language nobody here chose, and not
|
|
352
|
+
one of them moved.
|
|
353
|
+
|
|
354
|
+
Raising the bar buys the remaining silence out of the answers, and is refused:
|
|
355
|
+
at 0.50 the foreign absent ruler is silent on all thirty while A falls to 0.086
|
|
356
|
+
hit@1, B to 0.214 and C to 0.329. **0.28 was chosen before either foreign
|
|
357
|
+
corpus existed and survived meeting both**, which is the only kind of evidence
|
|
358
|
+
a default can have.
|
|
359
|
+
|
|
360
|
+
**Latency** — warm corpus, 601 units `ac3ae43a33e7`, one
|
|
361
|
+
`python -m benchmarks.query_latency` (5 repeats × 420 samples), idle machine:
|
|
362
|
+
|
|
363
|
+
| | | across the repeats |
|
|
366
364
|
|---|---|---|
|
|
367
|
-
| query, median | **0.
|
|
368
|
-
| query, p95 |
|
|
369
|
-
| refusing an unanswerable query | **0.
|
|
370
|
-
| refusal cheaper than answering by | **~
|
|
371
|
-
|
|
372
|
-
Two significant figures and a spread, because that is the precision the
|
|
373
|
-
measurement has. Across twenty invocations over three releases on the same idle
|
|
374
|
-
machine the median has landed anywhere from 0.51 to 1.44 ms and p95 from 0.85
|
|
375
|
-
to 7.34 ms — a band wider than any change the code has ever made to this
|
|
376
|
-
number. Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three
|
|
377
|
-
figures from a script that was never committed; both values sit inside that
|
|
378
|
-
band, which is the point: they were unfalsifiable rather than wrong.
|
|
365
|
+
| query, median | **0.49 ms** | 0.45 – 0.58 |
|
|
366
|
+
| query, p95 | 0.85 ms | 0.72 – 1.07 |
|
|
367
|
+
| refusing an unanswerable query | **0.016 ms** | 0.015 – 0.016 |
|
|
368
|
+
| refusal cheaper than answering by | **~30×** | 29 – 37 |
|
|
379
369
|
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
reaches ranking at all.
|
|
370
|
+
*Idle* is load-bearing: the same corpus at the same commit measured 0.99 ms
|
|
371
|
+
median while a coverage run was in progress and 0.49 ms once it finished.
|
|
383
372
|
|
|
384
|
-
|
|
373
|
+
Two significant figures and a spread, because that is the precision the
|
|
374
|
+
measurement has. Across twenty invocations over four releases on the same idle
|
|
375
|
+
machine the median has landed anywhere from 0.49 to 1.44 ms and p95 from 0.85
|
|
376
|
+
to 7.34 ms — a band wider than any change the code has made to this number.
|
|
377
|
+
Releases before 1.3.0 published `0.83 ms / p95 1.68 ms` to three figures from a
|
|
378
|
+
script that was never committed; both sit inside that band, which is the point:
|
|
379
|
+
they were unfalsifiable rather than wrong. Refusal is cheap structurally rather
|
|
380
|
+
than by tuning — an unanswerable query touches only the posting lists of its
|
|
381
|
+
own distinctive words and never reaches ranking.
|
|
382
|
+
|
|
383
|
+
**Scale**, synthetic 10,000-unit repository (500 files), re-measured in 1.5.0
|
|
384
|
+
— the previous row of figures was optimistic by more than noise:
|
|
385
|
+
|
|
386
|
+
| | | previously published |
|
|
387
|
+
|---|---|---|
|
|
388
|
+
| full build | 3.45 s | 1.84 s |
|
|
389
|
+
| incremental rebuild after one file changes | 0.286 s (**12.1×**) | 0.207 s |
|
|
390
|
+
| compact storage vs readable JSON | 35.6% | 35.6% |
|
|
391
|
+
| index load, fresh process | 79.3 ms | 45.4 ms |
|
|
392
|
+
| resident memory | 72.2 MiB | 58.7 MiB |
|
|
393
|
+
| mean query, full recall | 15.6 ms | 3.90 ms |
|
|
385
394
|
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
| incremental rebuild after one file changes | 0.207 s (**8.9×**) |
|
|
390
|
-
| compact storage vs readable JSON | 35.6% |
|
|
391
|
-
| index load, fresh process | 45.4 ms |
|
|
392
|
-
| resident memory | 58.7 MiB |
|
|
395
|
+
That last row had been carried since before BM25F replaced the scoring it was
|
|
396
|
+
taken under. Six releases, a committed script, and nothing that made anyone run
|
|
397
|
+
it again.
|
|
393
398
|
|
|
394
|
-
**Parsing**, against source-controlled fixtures (15 files, 237 negative cases
|
|
395
|
-
89 constructs the spec deliberately excludes):
|
|
399
|
+
**Parsing**, against source-controlled fixtures (15 files, 237 negative cases):
|
|
396
400
|
|
|
397
401
|
| | |
|
|
398
402
|
|---|---|
|
|
@@ -401,8 +405,8 @@ reaches ranking at all.
|
|
|
401
405
|
| with a usable signature | **91 / 91** |
|
|
402
406
|
| units invented that do not exist | **0** |
|
|
403
407
|
|
|
404
|
-
Directional local measurements, not service levels — but
|
|
405
|
-
|
|
408
|
+
Directional local measurements, not service levels — but each is a command
|
|
409
|
+
rather than a memory, which two of them were not before. Each
|
|
406
410
|
prints the corpus fingerprint beside its score; quote both or neither.
|
|
407
411
|
[`benchmarks/README.md`](benchmarks/README.md) lists the six scripts and what
|
|
408
412
|
each is for, and the corpus one of them grades is now carried here too.
|
|
@@ -415,107 +419,97 @@ many hit. That is what this reproduces — same corpus, same questions, same
|
|
|
415
419
|
ruler, scored at **file** granularity so Grep is not penalised for lacking
|
|
416
420
|
declaration spans.
|
|
417
421
|
|
|
418
|
-
**
|
|
419
|
-
|
|
420
|
-
undescribed repository ever measured was a hook-heavy tool whose questions were
|
|
421
|
-
answerable by matching identifiers. Swapping the subject for a public web
|
|
422
|
-
framework reversed it. The honest claim is narrower than either table alone:
|
|
422
|
+
**On an undescribed repository, which side wins depends on the repository.**
|
|
423
|
+
Three subjects, three answers:
|
|
423
424
|
|
|
424
|
-
|
|
|
425
|
-
|
|
426
|
-
|
|
|
427
|
-
|
|
|
428
|
-
|
|
|
429
|
-
|
|
430
|
-
|
|
425
|
+
| undescribed · Grep loop → rag-your-code | first | top 3 | answered | characters |
|
|
426
|
+
|---|---|---|---|---|
|
|
427
|
+
| **Flask** 35 q · 1,572 units `5fd51169eacc` | 22.9% → **37.1%** | 45.7% → **57.1%** | 30 → 30 | 1,415,656 → **258,236** |
|
|
428
|
+
| **cobra** 40 q · 602 units `3eabaa705477` | 17.5% → 17.5% | 20.0% → **30.0%** | 22 → 17 | 799,475 → **156,336** |
|
|
429
|
+
| a hook-heavy tool, retired in 1.4.0 | 34.3% → 31.4% | — | — | — |
|
|
430
|
+
|
|
431
|
+
Flask wins for this side, cobra ties on first place, and the retired subject
|
|
432
|
+
lost. A cold index retrieves against a generated sentence plus whatever the
|
|
433
|
+
author documented, so the outcome is set by how much prose the repository
|
|
434
|
+
already carries — Flask documents most public methods in paragraphs, cobra in
|
|
435
|
+
one terse line, the retired tool barely at all. What holds on all three is the
|
|
436
|
+
payload: this side hands back a fifth to a half of the text, ranked and spanned.
|
|
431
437
|
|
|
432
438
|
**Once the vocabulary exists, it is not close.**
|
|
433
439
|
|
|
434
|
-
| this repository · 70 questions ·
|
|
440
|
+
| this repository · 70 questions · 601 units `ac3ae43a33e7` · 314 described | Grep loop | rag-your-code |
|
|
435
441
|
|---|---|---|
|
|
436
442
|
| right file first | 22.9% | **58.6%** |
|
|
437
|
-
| right file in top 3 | 54.3% | **
|
|
438
|
-
| lines it hands back, all questions |
|
|
439
|
-
| characters returned, all questions | 1,
|
|
443
|
+
| right file in top 3 | 54.3% | **78.6%** |
|
|
444
|
+
| lines it hands back, all questions | 12,421 | — |
|
|
445
|
+
| characters returned, all questions | 1,179,431 | **617,305** |
|
|
440
446
|
| questions it answers | **61** | 60 |
|
|
441
447
|
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
Until then this section — the strongest claim the project makes — came from a
|
|
456
|
-
script that was never committed, so nothing here could be checked and "Grep
|
|
457
|
-
loop" had no precise meaning. The committed version defines it: take the
|
|
458
|
-
query's words, drop the ones the corpus itself shows are everywhere, run one
|
|
459
|
-
substring search per remaining word over exactly the files the index was built
|
|
460
|
-
from, rank each file by how many distinct words hit it, break ties on path.
|
|
461
|
-
Reconstructing it reproduced this side's figures exactly and moved Grep's —
|
|
462
|
-
the expected shape, since the ranked arm was always a call into shipped code.
|
|
463
|
-
|
|
464
|
-
Four qualifications, because the table would otherwise flatter both sides:
|
|
448
|
+
That is section 3.3's argument measured rather than asserted, and it is the one
|
|
449
|
+
thing the subject does not change: first-place accuracy more than doubles
|
|
450
|
+
Grep's.
|
|
451
|
+
|
|
452
|
+
**Every table comes from `python -m benchmarks.grep_baseline`**, new in 1.3.0.
|
|
453
|
+
Until then this section — the strongest claim the project makes — came from an
|
|
454
|
+
uncommitted script, so nothing here could be checked and "Grep loop" had no
|
|
455
|
+
precise meaning. The committed version defines it: take the query's words, drop
|
|
456
|
+
the ones the corpus itself shows are everywhere, run one substring search per
|
|
457
|
+
remaining word over exactly the files the index was built from, rank each file
|
|
458
|
+
by how many distinct words hit it, break ties on path.
|
|
459
|
+
|
|
460
|
+
Qualifications, because the tables would otherwise flatter both sides:
|
|
465
461
|
|
|
466
462
|
- **Scored at file granularity**, which understates this side. A Grep hit is a
|
|
467
463
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
468
464
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
469
|
-
- **Dropping the corpus-common words is generous to Grep**, and
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
decline the same five of
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
cross-language case is not this tool's failure but the corpus's.
|
|
465
|
+
- **Dropping the corpus-common words is generous to Grep**, and is what makes
|
|
466
|
+
the baseline fair rather than a straw man: an agent that greps `the` gets
|
|
467
|
+
every file back in no order. It is also why Grep declines nine of the seventy
|
|
468
|
+
questions here — no word was left that this corpus does not use everywhere.
|
|
469
|
+
- **Payload is counted in characters on both sides.** Grep hands back roughly
|
|
470
|
+
19,300 characters per question it answers here, unranked and without spans,
|
|
471
|
+
against 10,300 ranked and capped by `search.max_chars` — a factor of 1.9,
|
|
472
|
+
5.5 on Flask and 5.1 on cobra, where a framework repeats its vocabulary
|
|
473
|
+
across files and Grep cannot rank what it finds. 1.4.1 changed what fits in
|
|
474
|
+
that cap: the block had been reprinting the docstring the code below already
|
|
475
|
+
showed, so the same budget now carries **119 declarations instead of 92** on
|
|
476
|
+
Flask.
|
|
477
|
+
- **Chinese is the corpus's limit, not the tool's, when a corpus is
|
|
478
|
+
monolingual.** Both sides decline the same five of Flask's 35 — every Chinese
|
|
479
|
+
one. A Chinese word is neither a substring of English source nor a token in
|
|
480
|
+
an index built from it.
|
|
486
481
|
- **Grep wins outright when you know the string.** `grep -rn "COMMON_TERM"` is
|
|
487
482
|
exact, instant and complete, and nothing here replaces it.
|
|
488
483
|
|
|
489
484
|
The two are complementary, and the honest summary is narrow: this earns its
|
|
490
|
-
place on questions phrased as questions, over a repository somebody
|
|
491
|
-
the time to describe.
|
|
485
|
+
place on questions phrased as questions, over a repository somebody described.
|
|
492
486
|
|
|
493
487
|
## 8 · Design principles
|
|
494
488
|
|
|
495
489
|
**Build the ruler before reshaping the thing measured.** Four candidate scoring
|
|
496
490
|
changes once landed between five and six correct over an eight-question set —
|
|
497
|
-
the instrument's resolution limit, not a ranking. There are
|
|
498
|
-
across
|
|
491
|
+
the instrument's resolution limit, not a ranking. There are 175 questions now
|
|
492
|
+
across five rulers, and every claim here is a number from one of them.
|
|
499
493
|
|
|
500
494
|
**Measure somewhere it can fail.** Every ruler this project had once graded a
|
|
501
495
|
repository its own authors wrote; cold against a foreign one the same code
|
|
502
|
-
scored 0.086 hit@1 against a self-reported 0.457.
|
|
503
|
-
|
|
504
|
-
|
|
496
|
+
scored 0.086 hit@1 against a self-reported 0.457. Rulers A and E exist so that
|
|
497
|
+
cannot be comfortable again — stemming, which helps both own-repo rulers, was
|
|
498
|
+
rejected on A, and E publishes a 0.075 this project would rather not print.
|
|
505
499
|
|
|
506
500
|
**Make the error structurally impossible rather than checking for it.** A line
|
|
507
|
-
number that *is* the loop index cannot drift
|
|
508
|
-
of its own code cannot outlive it.
|
|
509
|
-
|
|
510
|
-
**A ratio inside the query, never a threshold on a score.** Scales move;
|
|
511
|
-
ratios do not.
|
|
501
|
+
number that *is* the loop index cannot drift; a description keyed by a digest
|
|
502
|
+
of its own code cannot outlive it. **And a ratio inside the query, never a
|
|
503
|
+
threshold on a score** — scales move, ratios do not.
|
|
512
504
|
|
|
513
505
|
**Derive figures from data; a hand-maintained number is a claim nobody checks.**
|
|
514
|
-
|
|
515
|
-
|
|
506
|
+
This README's settings table is asserted against `config.py` in both directions
|
|
507
|
+
— it had drifted nine settings behind before that test existed — and the four
|
|
508
|
+
diagrams in `docs/FLOW.md` are parsed by a test that refuses a label mermaid
|
|
509
|
+
would silently fail to render.
|
|
516
510
|
|
|
517
511
|
**The contract does not move.** `CodeUnit`, index schema 2 and the JSON-lines
|
|
518
|
-
protocol are unchanged across every release
|
|
512
|
+
protocol are unchanged across every release: new information arrives in new
|
|
519
513
|
fields, never by widening an enumeration callers branch on.
|
|
520
514
|
|
|
521
515
|
**Publish what was measured and rejected.** Twelve changes were implemented,
|
|
@@ -539,14 +533,14 @@ dimensions = 384
|
|
|
539
533
|
pip install "rag-your-code[sentence-transformers]"
|
|
540
534
|
```
|
|
541
535
|
|
|
542
|
-
The extra is optional by construction: `dependencies = []` is
|
|
543
|
-
install
|
|
536
|
+
The extra is optional by construction: `dependencies = []` is a default
|
|
537
|
+
install, the import happens inside the constructor, and a test asserts the
|
|
544
538
|
default provider imports none of it.
|
|
545
539
|
|
|
546
|
-
**Measured on the
|
|
547
|
-
published this comparison and read it as a win
|
|
548
|
-
foreign ruler, whose two arms turned out to have been taken
|
|
549
|
-
|
|
540
|
+
**Measured on the four rulers that then existed, both arms against one
|
|
541
|
+
corpus.** 1.1.0 published this comparison and read it as a win; its largest
|
|
542
|
+
gain was on the foreign ruler, whose two arms turned out to have been taken
|
|
543
|
+
against two states of a repository being edited while the script ran. Repeated
|
|
550
544
|
against a pinned corpus:
|
|
551
545
|
|
|
552
546
|
| ruler | corpus | signed hash (default) | MiniLM, local |
|
|
@@ -558,8 +552,8 @@ against a pinned corpus:
|
|
|
558
552
|
|
|
559
553
|
**Worse or identical on every ruler.** The 581-unit stamps are the corpus both
|
|
560
554
|
arms shared, kept rather than refreshed — that is what a stamp is for. It ships
|
|
561
|
-
anyway
|
|
562
|
-
|
|
555
|
+
anyway for the one thing the hash cannot do and these rulers cannot see: reach
|
|
556
|
+
a unit sharing no word with the question. The pairs it scores zero on:
|
|
563
557
|
|
|
564
558
|
| pair | signed hash | MiniLM |
|
|
565
559
|
|---|---|---|
|
|
@@ -568,18 +562,16 @@ reach a unit sharing no word with the question. The pairs it scores zero on:
|
|
|
568
562
|
| `计算两个数的和` vs `sum two numbers` | **0.000** | **0.822** |
|
|
569
563
|
| `刷新索引` vs `rebuild the index` | **0.000** | **0.684** |
|
|
570
564
|
|
|
571
|
-
A semantic embedder is **not** exempt from the evidence bars,
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
565
|
+
A semantic embedder is **not** exempt from the evidence bars, correcting
|
|
566
|
+
1.0.0. Exempting one was reasoned — a paraphrase sharing no word with its
|
|
567
|
+
answer is exactly what a model is for — and wrong: exempt and asked no other
|
|
568
|
+
question, the model answered all sixty unanswerable questions. Two vector-space
|
|
569
|
+
replacements were then measured and rejected: a similarity floor is a threshold
|
|
570
|
+
on a score and the distributions overlap (0.469 vs 0.418 median), and a
|
|
571
|
+
scale-free standout metric took ruler B from 0.329 to 0.186 for two thirds of
|
|
572
|
+
the silence. Applying the lexical bars costs ruler A nothing.
|
|
579
573
|
|
|
580
|
-
|
|
581
|
-
it for the cross-language and paraphrase cases in the table above, which is
|
|
582
|
-
where the difference between the two columns actually lives.
|
|
574
|
+
Install it for the cross-language and paraphrase cases above, not for the hit rates.
|
|
583
575
|
|
|
584
576
|
**A hosted endpoint** is the third option, and the only one that sends your
|
|
585
577
|
source anywhere:
|
|
@@ -592,17 +584,17 @@ dimensions = 1536
|
|
|
592
584
|
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
593
585
|
```
|
|
594
586
|
|
|
595
|
-
The key is never a setting
|
|
596
|
-
everyone who clones sees what shaped the index
|
|
597
|
-
with the opposite requirement. Sending a key over plain `http://` to
|
|
598
|
-
but your own machine is refused rather than warned about
|
|
599
|
-
build rather than falling back
|
|
600
|
-
a cosine across them is a meaningless number ranking would act on
|
|
587
|
+
The key is never a setting: `rag-your-code.toml` is meant to be committed so
|
|
588
|
+
everyone who clones sees what shaped the index, and a credential is the one
|
|
589
|
+
value with the opposite requirement. Sending a key over plain `http://` to
|
|
590
|
+
anything but your own machine is refused rather than warned about, and a
|
|
591
|
+
failure stops the build rather than falling back — a mixed index is two vector
|
|
592
|
+
spaces, and a cosine across them is a meaningless number ranking would act on.
|
|
601
593
|
|
|
602
594
|
With a semantic embedder, similarity may also **add** candidates rather than
|
|
603
595
|
only reorder them (`search.vector_recall`) — the one thing that can reach a
|
|
604
|
-
unit sharing no word with the question. Under the hash
|
|
605
|
-
|
|
596
|
+
unit sharing no word with the question. Under the hash it measured worse, so it
|
|
597
|
+
stays off there.
|
|
606
598
|
|
|
607
599
|
## 10 · Install and use
|
|
608
600
|
|
|
@@ -650,14 +642,13 @@ rag-your-code describe promote | git apply # move descriptions into the cod
|
|
|
650
642
|
`bootstrap` exists because indexing is not the same as being searchable: a
|
|
651
643
|
fresh index retrieves against a generated sentence that adds no word the source
|
|
652
644
|
did not have. It reports which rung the repository is on and hands over that
|
|
653
|
-
rung's work; run it again after each round.
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
modified**. `describe promote` emits a diff to review — it never writes source.
|
|
645
|
+
rung's work; run it again after each round. The index is written under
|
|
646
|
+
`.rag-your-code/` and **your source files are never modified** — `describe
|
|
647
|
+
promote` emits a diff to review rather than writing source.
|
|
657
648
|
|
|
658
649
|
### Configuration
|
|
659
650
|
|
|
660
|
-
|
|
651
|
+
23 settings in `rag-your-code.toml`:
|
|
661
652
|
|
|
662
653
|
| section | settings |
|
|
663
654
|
|---|---|
|
|
@@ -665,7 +656,7 @@ modified**. `describe promote` emits a diff to review — it never writes source
|
|
|
665
656
|
| `[embedding]` | `dimensions`, `provider`, `endpoint`, `model`, `api_key_env`, `batch`, `timeout`, `retries` |
|
|
666
657
|
| `[search]` | `min_coverage`, `min_concentration`, `vector_weight`, `vector_recall`, `limit`, `max_chars` |
|
|
667
658
|
| `[agent]` | `max_open_bytes`, `max_open_chars` |
|
|
668
|
-
| `[describe]` | `languages`, `batch`, `max_chars` |
|
|
659
|
+
| `[describe]` | `languages`, `batch`, `max_chars`, `skip` |
|
|
669
660
|
|
|
670
661
|
This table is asserted against `config.py` in both directions by
|
|
671
662
|
`tests/test_metadata.py`. Resolution is CLI flag > file > built-in default;
|
|
@@ -724,8 +715,8 @@ questions: it is the corpus's limit, not this tool's.
|
|
|
724
715
|
|
|
725
716
|
**There is no stemming.** `catastrophic backtracking` does not reach
|
|
726
717
|
`backtracks catastrophically`. A light suffix stripper was implemented and
|
|
727
|
-
measured on
|
|
728
|
-
the foreign one 3 of 35 hit@3, so it was rejected.
|
|
718
|
+
measured on every ruler that then existed: it improves both own-repository
|
|
719
|
+
rulers and costs the foreign one 3 of 35 hit@3, so it was rejected.
|
|
729
720
|
|
|
730
721
|
**A test declaration sometimes outranks real code** — 9 of 175 questions across
|
|
731
722
|
three rulers, a test at rank 1 displacing an accepted answer at rank 2–3, and
|
|
@@ -734,27 +725,34 @@ code it *tests*, is wrong: five of the nine are unrelated tests winning on
|
|
|
734
725
|
prose. A callee-before-caller rerank fires on zero questions, and the `name`
|
|
735
726
|
field weight moves nothing because an underscored test name is one token.
|
|
736
727
|
|
|
728
|
+
**Describing a declaration that already has a good docstring loses ground.** An
|
|
729
|
+
authored description *replaces* the generated sentence, which is the only route
|
|
730
|
+
by which the author's own docstring reaches the weight-3 description field — so
|
|
731
|
+
writing one demotes that docstring to the weight-1 body. Measured on
|
|
732
|
+
`parser.py::_generic_units`: a long description cost one graded question, a
|
|
733
|
+
short one cost three, and appending the docstring to all 314 descriptions
|
|
734
|
+
instead cost 0.443 → 0.414 hit@1. `describe.skip` records the decision.
|
|
735
|
+
|
|
737
736
|
**The vectors are 72.1% of the index and earn ±1 question** under the default
|
|
738
|
-
embedder. Kept: the same storage is what
|
|
737
|
+
embedder — 74.8% on Flask and 79.7% on cobra. Kept: the same storage is what
|
|
738
|
+
makes an optional model work.
|
|
739
739
|
|
|
740
740
|
**`search.vector_recall` scans every vector per query** — under a semantic
|
|
741
741
|
embedder. The default hash never widens at all. Affordable at the measured
|
|
742
742
|
envelope, and exactly the work an ANN index would replace.
|
|
743
743
|
|
|
744
|
-
**Tree-sitter parsing and a SQLite/ANN storage layer are not here.** Both
|
|
745
|
-
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
749
|
-
**Whether the skill fires unprompted is not measured**, and
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
nothing. Since 1.2.0 the four commands give an entry path that does not depend
|
|
757
|
-
on it.
|
|
744
|
+
**Tree-sitter parsing and a SQLite/ANN storage layer are not here.** Both need
|
|
745
|
+
a dependency, and the policy is settled: they follow the embedding provider's
|
|
746
|
+
pattern — optional, user-selected, never in a default install. Full reasoning
|
|
747
|
+
in [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
748
|
+
|
|
749
|
+
**Whether the skill fires unprompted is not measured**, and it is the only
|
|
750
|
+
claim here with no command behind it. `claude plugin eval` grades exactly this
|
|
751
|
+
— `tool_used: Skill` as the indicator, against a no-plugin baseline arm — but
|
|
752
|
+
it is gated behind an account-level early access this project does not have,
|
|
753
|
+
and a suite written from `--help` fragments could not be run once to see
|
|
754
|
+
whether it loads. Since 1.2.0 the four commands give an entry path that does
|
|
755
|
+
not depend on it.
|
|
758
756
|
|
|
759
757
|
## 12 · Development
|
|
760
758
|
|
|
@@ -763,17 +761,20 @@ python -m pip install -e ".[dev]"
|
|
|
763
761
|
pytest -q
|
|
764
762
|
```
|
|
765
763
|
|
|
766
|
-
Per-release test counts
|
|
767
|
-
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
764
|
+
Per-release test counts and line coverage are in
|
|
765
|
+
[CHANGELOG.md](CHANGELOG.md); a bare figure in a living document is a claim
|
|
766
|
+
that rots. `pytest --cov=ragyourcode` is the command behind the coverage one.
|
|
767
|
+
CI runs Python 3.10–3.13 on Linux and Windows, installs the built wheel into a
|
|
768
|
+
clean environment and runs the documented CLI end to end — `bootstrap` through
|
|
769
|
+
`describe promote` — plus the skill's own install line verbatim, and grades
|
|
770
|
+
every ruler including both vendored corpora.
|
|
771
771
|
|
|
772
772
|
- [docs/FLOW.md](docs/FLOW.md) — the whole thing in four diagrams
|
|
773
773
|
- [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) — how each stage works and why
|
|
774
|
-
- [docs/TESTING.md](docs/TESTING.md) — what the suites protect
|
|
774
|
+
- [docs/TESTING.md](docs/TESTING.md) — what the suites protect, and how covered
|
|
775
775
|
- [docs/ROADMAP.md](docs/ROADMAP.md) — what shipped, what was rejected and why
|
|
776
776
|
- [CONTRIBUTING.md](CONTRIBUTING.md) — ground rules, and how to add a language
|
|
777
777
|
- [CHANGELOG.md](CHANGELOG.md) — every release with its measurements
|
|
778
|
+
- [benchmarks/corpus/](benchmarks/corpus/) — the two vendored repositories
|
|
778
779
|
|
|
779
780
|
MIT licensed.
|