rag-your-code 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-1.1.0/src/rag_your_code.egg-info → rag_your_code-1.2.0}/PKG-INFO +29 -17
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/README.md +28 -16
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/pyproject.toml +1 -1
- {rag_your_code-1.1.0 → rag_your_code-1.2.0/src/rag_your_code.egg-info}/PKG-INFO +29 -17
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/cli.py +41 -2
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_e2e_cli.py +74 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_metadata.py +46 -2
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/LICENSE +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/setup.cfg +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/agentic.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/config.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/embeddings.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/graph.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/indexer.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/providers.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/search.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/workflow.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_absent_queries.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_agentic.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_config.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_document.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_evidence.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_golden.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_local_model.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_providers.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_ranking.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_resilience.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -129,7 +129,7 @@ the scale.
|
|
|
129
129
|
|
|
130
130
|
The default embedder is a signed feature hash. Ablating it entirely moves the
|
|
131
131
|
three positive rulers by **±1 question in either direction** while the vectors
|
|
132
|
-
occupy **65.
|
|
132
|
+
occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
|
|
133
133
|
The explanation, measured here:
|
|
134
134
|
|
|
135
135
|
- **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
|
|
@@ -323,9 +323,9 @@ it answers a different question and names a different diagnosis.
|
|
|
323
323
|
|
|
324
324
|
| | |
|
|
325
325
|
|---|---|
|
|
326
|
-
| query, median | **0.
|
|
327
|
-
| query, p95 |
|
|
328
|
-
| refusing an unanswerable query | **0.
|
|
326
|
+
| query, median | **0.83 ms** |
|
|
327
|
+
| query, p95 | 1.68 ms |
|
|
328
|
+
| refusing an unanswerable query | **0.03 ms** |
|
|
329
329
|
|
|
330
330
|
Refusal is cheaper than answering by a factor of forty: an unanswerable query
|
|
331
331
|
touches only the posting lists of its own distinctive words, never the corpus.
|
|
@@ -364,22 +364,22 @@ declaration spans.
|
|
|
364
364
|
**On a repository nobody has described, Grep wins.** That is the measured
|
|
365
365
|
result and it is not softened here.
|
|
366
366
|
|
|
367
|
-
| foreign repository · 35 questions · 1,
|
|
367
|
+
| foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
|
|
368
368
|
|---|---|---|
|
|
369
369
|
| right file first | **34.3%** | 31.4% |
|
|
370
370
|
| right file in top 3 | **60.0%** | 48.6% |
|
|
371
|
-
| lines matched across the repo, all questions | 33,
|
|
372
|
-
| characters returned, all questions | — | **163,
|
|
371
|
+
| lines matched across the repo, all questions | 33,213 | — |
|
|
372
|
+
| characters returned, all questions | — | **163,294** |
|
|
373
373
|
| questions it answers | 35 | 28 |
|
|
374
374
|
|
|
375
375
|
**Once the vocabulary exists, it is not close.**
|
|
376
376
|
|
|
377
|
-
| this repository · 70 questions ·
|
|
377
|
+
| this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
|
|
378
378
|
|---|---|---|
|
|
379
|
-
| right file first | 25.7% | **
|
|
379
|
+
| right file first | 25.7% | **58.6%** |
|
|
380
380
|
| right file in top 3 | 64.3% | **75.7%** |
|
|
381
|
-
| lines matched across the repo, all questions |
|
|
382
|
-
| characters returned, all questions | — | **
|
|
381
|
+
| lines matched across the repo, all questions | 40,150 | — |
|
|
382
|
+
| characters returned, all questions | — | **277,327** |
|
|
383
383
|
| questions it answers | 70 | 60 |
|
|
384
384
|
|
|
385
385
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
@@ -396,7 +396,7 @@ Three qualifications, because the table would otherwise flatter both sides:
|
|
|
396
396
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
397
397
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
398
398
|
- **Grep answers everything.** It never declines, which is why it hands back
|
|
399
|
-
33,
|
|
399
|
+
33,213 matching lines for 35 questions — about 950 lines per question, no
|
|
400
400
|
ranking, no spans, no indication which match is the definition. This returns
|
|
401
401
|
roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
|
|
402
402
|
questions come back empty here instead, with a reason.
|
|
@@ -527,9 +527,21 @@ measured worse, so it stays off there.
|
|
|
527
527
|
/reload-plugins
|
|
528
528
|
```
|
|
529
529
|
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
530
|
+
Four commands and one skill. No hooks, no agents, no MCP server:
|
|
531
|
+
|
|
532
|
+
| | |
|
|
533
|
+
|---|---|
|
|
534
|
+
| `/rag-your-code:index` | index, and say which rung this repository is on |
|
|
535
|
+
| `/rag-your-code:search` | ask in plain language; cite `path:line` |
|
|
536
|
+
| `/rag-your-code:describe` | write the vocabulary the source does not contain |
|
|
537
|
+
| `/rag-your-code:status` | stale? coverage? which embedder? what next? |
|
|
538
|
+
|
|
539
|
+
Measured with `claude plugin details` on an installed copy: **~249 tokens added
|
|
540
|
+
to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
|
|
541
|
+
one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
|
|
542
|
+
of being findable — a skill fires only when a model decides it should, which
|
|
543
|
+
left the whole plugin with no entry point a person could discover. The commands
|
|
544
|
+
install the Python package on first use.
|
|
533
545
|
|
|
534
546
|
**As a CLI:**
|
|
535
547
|
|
|
@@ -627,7 +639,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
|
|
|
627
639
|
prose. A callee-before-caller rerank fires on zero questions and the `name`
|
|
628
640
|
field weight moves nothing, because an underscored test name is a single token.
|
|
629
641
|
|
|
630
|
-
**The vectors are 65.
|
|
642
|
+
**The vectors are 65.3% of the index and earn ±1 question** under the default
|
|
631
643
|
embedder. Not removed: the same storage is what makes an optional model work,
|
|
632
644
|
and the schema stays one shape.
|
|
633
645
|
|
|
@@ -100,7 +100,7 @@ the scale.
|
|
|
100
100
|
|
|
101
101
|
The default embedder is a signed feature hash. Ablating it entirely moves the
|
|
102
102
|
three positive rulers by **±1 question in either direction** while the vectors
|
|
103
|
-
occupy **65.
|
|
103
|
+
occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
|
|
104
104
|
The explanation, measured here:
|
|
105
105
|
|
|
106
106
|
- **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
|
|
@@ -294,9 +294,9 @@ it answers a different question and names a different diagnosis.
|
|
|
294
294
|
|
|
295
295
|
| | |
|
|
296
296
|
|---|---|
|
|
297
|
-
| query, median | **0.
|
|
298
|
-
| query, p95 |
|
|
299
|
-
| refusing an unanswerable query | **0.
|
|
297
|
+
| query, median | **0.83 ms** |
|
|
298
|
+
| query, p95 | 1.68 ms |
|
|
299
|
+
| refusing an unanswerable query | **0.03 ms** |
|
|
300
300
|
|
|
301
301
|
Refusal is cheaper than answering by a factor of forty: an unanswerable query
|
|
302
302
|
touches only the posting lists of its own distinctive words, never the corpus.
|
|
@@ -335,22 +335,22 @@ declaration spans.
|
|
|
335
335
|
**On a repository nobody has described, Grep wins.** That is the measured
|
|
336
336
|
result and it is not softened here.
|
|
337
337
|
|
|
338
|
-
| foreign repository · 35 questions · 1,
|
|
338
|
+
| foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
|
|
339
339
|
|---|---|---|
|
|
340
340
|
| right file first | **34.3%** | 31.4% |
|
|
341
341
|
| right file in top 3 | **60.0%** | 48.6% |
|
|
342
|
-
| lines matched across the repo, all questions | 33,
|
|
343
|
-
| characters returned, all questions | — | **163,
|
|
342
|
+
| lines matched across the repo, all questions | 33,213 | — |
|
|
343
|
+
| characters returned, all questions | — | **163,294** |
|
|
344
344
|
| questions it answers | 35 | 28 |
|
|
345
345
|
|
|
346
346
|
**Once the vocabulary exists, it is not close.**
|
|
347
347
|
|
|
348
|
-
| this repository · 70 questions ·
|
|
348
|
+
| this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
|
|
349
349
|
|---|---|---|
|
|
350
|
-
| right file first | 25.7% | **
|
|
350
|
+
| right file first | 25.7% | **58.6%** |
|
|
351
351
|
| right file in top 3 | 64.3% | **75.7%** |
|
|
352
|
-
| lines matched across the repo, all questions |
|
|
353
|
-
| characters returned, all questions | — | **
|
|
352
|
+
| lines matched across the repo, all questions | 40,150 | — |
|
|
353
|
+
| characters returned, all questions | — | **277,327** |
|
|
354
354
|
| questions it answers | 70 | 60 |
|
|
355
355
|
|
|
356
356
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
@@ -367,7 +367,7 @@ Three qualifications, because the table would otherwise flatter both sides:
|
|
|
367
367
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
368
368
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
369
369
|
- **Grep answers everything.** It never declines, which is why it hands back
|
|
370
|
-
33,
|
|
370
|
+
33,213 matching lines for 35 questions — about 950 lines per question, no
|
|
371
371
|
ranking, no spans, no indication which match is the definition. This returns
|
|
372
372
|
roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
|
|
373
373
|
questions come back empty here instead, with a reason.
|
|
@@ -498,9 +498,21 @@ measured worse, so it stays off there.
|
|
|
498
498
|
/reload-plugins
|
|
499
499
|
```
|
|
500
500
|
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
501
|
+
Four commands and one skill. No hooks, no agents, no MCP server:
|
|
502
|
+
|
|
503
|
+
| | |
|
|
504
|
+
|---|---|
|
|
505
|
+
| `/rag-your-code:index` | index, and say which rung this repository is on |
|
|
506
|
+
| `/rag-your-code:search` | ask in plain language; cite `path:line` |
|
|
507
|
+
| `/rag-your-code:describe` | write the vocabulary the source does not contain |
|
|
508
|
+
| `/rag-your-code:status` | stale? coverage? which embedder? what next? |
|
|
509
|
+
|
|
510
|
+
Measured with `claude plugin details` on an installed copy: **~249 tokens added
|
|
511
|
+
to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
|
|
512
|
+
one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
|
|
513
|
+
of being findable — a skill fires only when a model decides it should, which
|
|
514
|
+
left the whole plugin with no entry point a person could discover. The commands
|
|
515
|
+
install the Python package on first use.
|
|
504
516
|
|
|
505
517
|
**As a CLI:**
|
|
506
518
|
|
|
@@ -598,7 +610,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
|
|
|
598
610
|
prose. A callee-before-caller rerank fires on zero questions and the `name`
|
|
599
611
|
field weight moves nothing, because an underscored test name is a single token.
|
|
600
612
|
|
|
601
|
-
**The vectors are 65.
|
|
613
|
+
**The vectors are 65.3% of the index and earn ±1 question** under the default
|
|
602
614
|
embedder. Not removed: the same storage is what makes an optional model work,
|
|
603
615
|
and the schema stays one shape.
|
|
604
616
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -129,7 +129,7 @@ the scale.
|
|
|
129
129
|
|
|
130
130
|
The default embedder is a signed feature hash. Ablating it entirely moves the
|
|
131
131
|
three positive rulers by **±1 question in either direction** while the vectors
|
|
132
|
-
occupy **65.
|
|
132
|
+
occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
|
|
133
133
|
The explanation, measured here:
|
|
134
134
|
|
|
135
135
|
- **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
|
|
@@ -323,9 +323,9 @@ it answers a different question and names a different diagnosis.
|
|
|
323
323
|
|
|
324
324
|
| | |
|
|
325
325
|
|---|---|
|
|
326
|
-
| query, median | **0.
|
|
327
|
-
| query, p95 |
|
|
328
|
-
| refusing an unanswerable query | **0.
|
|
326
|
+
| query, median | **0.83 ms** |
|
|
327
|
+
| query, p95 | 1.68 ms |
|
|
328
|
+
| refusing an unanswerable query | **0.03 ms** |
|
|
329
329
|
|
|
330
330
|
Refusal is cheaper than answering by a factor of forty: an unanswerable query
|
|
331
331
|
touches only the posting lists of its own distinctive words, never the corpus.
|
|
@@ -364,22 +364,22 @@ declaration spans.
|
|
|
364
364
|
**On a repository nobody has described, Grep wins.** That is the measured
|
|
365
365
|
result and it is not softened here.
|
|
366
366
|
|
|
367
|
-
| foreign repository · 35 questions · 1,
|
|
367
|
+
| foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
|
|
368
368
|
|---|---|---|
|
|
369
369
|
| right file first | **34.3%** | 31.4% |
|
|
370
370
|
| right file in top 3 | **60.0%** | 48.6% |
|
|
371
|
-
| lines matched across the repo, all questions | 33,
|
|
372
|
-
| characters returned, all questions | — | **163,
|
|
371
|
+
| lines matched across the repo, all questions | 33,213 | — |
|
|
372
|
+
| characters returned, all questions | — | **163,294** |
|
|
373
373
|
| questions it answers | 35 | 28 |
|
|
374
374
|
|
|
375
375
|
**Once the vocabulary exists, it is not close.**
|
|
376
376
|
|
|
377
|
-
| this repository · 70 questions ·
|
|
377
|
+
| this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
|
|
378
378
|
|---|---|---|
|
|
379
|
-
| right file first | 25.7% | **
|
|
379
|
+
| right file first | 25.7% | **58.6%** |
|
|
380
380
|
| right file in top 3 | 64.3% | **75.7%** |
|
|
381
|
-
| lines matched across the repo, all questions |
|
|
382
|
-
| characters returned, all questions | — | **
|
|
381
|
+
| lines matched across the repo, all questions | 40,150 | — |
|
|
382
|
+
| characters returned, all questions | — | **277,327** |
|
|
383
383
|
| questions it answers | 70 | 60 |
|
|
384
384
|
|
|
385
385
|
Those two tables are the whole argument of section 3.3, measured against a real
|
|
@@ -396,7 +396,7 @@ Three qualifications, because the table would otherwise flatter both sides:
|
|
|
396
396
|
file; a hit here is a declaration with an exact span, a score, and the words
|
|
397
397
|
it matched on. The agent that reads the result opens 40 lines, not a file.
|
|
398
398
|
- **Grep answers everything.** It never declines, which is why it hands back
|
|
399
|
-
33,
|
|
399
|
+
33,213 matching lines for 35 questions — about 950 lines per question, no
|
|
400
400
|
ranking, no spans, no indication which match is the definition. This returns
|
|
401
401
|
roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
|
|
402
402
|
questions come back empty here instead, with a reason.
|
|
@@ -527,9 +527,21 @@ measured worse, so it stays off there.
|
|
|
527
527
|
/reload-plugins
|
|
528
528
|
```
|
|
529
529
|
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
530
|
+
Four commands and one skill. No hooks, no agents, no MCP server:
|
|
531
|
+
|
|
532
|
+
| | |
|
|
533
|
+
|---|---|
|
|
534
|
+
| `/rag-your-code:index` | index, and say which rung this repository is on |
|
|
535
|
+
| `/rag-your-code:search` | ask in plain language; cite `path:line` |
|
|
536
|
+
| `/rag-your-code:describe` | write the vocabulary the source does not contain |
|
|
537
|
+
| `/rag-your-code:status` | stale? coverage? which embedder? what next? |
|
|
538
|
+
|
|
539
|
+
Measured with `claude plugin details` on an installed copy: **~249 tokens added
|
|
540
|
+
to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
|
|
541
|
+
one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
|
|
542
|
+
of being findable — a skill fires only when a model decides it should, which
|
|
543
|
+
left the whole plugin with no entry point a person could discover. The commands
|
|
544
|
+
install the Python package on first use.
|
|
533
545
|
|
|
534
546
|
**As a CLI:**
|
|
535
547
|
|
|
@@ -627,7 +639,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
|
|
|
627
639
|
prose. A callee-before-caller rerank fires on zero questions and the `name`
|
|
628
640
|
field weight moves nothing, because an underscored test name is a single token.
|
|
629
641
|
|
|
630
|
-
**The vectors are 65.
|
|
642
|
+
**The vectors are 65.3% of the index and earn ±1 question** under the default
|
|
631
643
|
embedder. Not removed: the same storage is what makes an optional model work,
|
|
632
644
|
and the schema stays one shape.
|
|
633
645
|
|
|
@@ -174,6 +174,42 @@ def _load(args: argparse.Namespace):
|
|
|
174
174
|
return payload, units, graph_from_dict(units, payload.get("graph")), cfg, store
|
|
175
175
|
|
|
176
176
|
|
|
177
|
+
# Refusal reasons that writing descriptions can actually fix. `scattered` is
|
|
178
|
+
# deliberately absent: it means the words are here but never together, which is
|
|
179
|
+
# what a question about a subject the repository does not implement looks like,
|
|
180
|
+
# and telling somebody to describe more code would be advice that cannot work.
|
|
181
|
+
_DESCRIBABLE_REASONS = frozenset({
|
|
182
|
+
"no_query_term_in_index",
|
|
183
|
+
"only_ubiquitous_terms_matched",
|
|
184
|
+
"too_little_of_the_query_matched",
|
|
185
|
+
})
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _describe_nudge(store, units, reason: str) -> str:
|
|
189
|
+
"""The one line worth printing when a question came back empty and the
|
|
190
|
+
vocabulary that would have answered it has not been written yet.
|
|
191
|
+
|
|
192
|
+
Deliberately tied to a refusal rather than to a low score. "The results
|
|
193
|
+
looked weak" would need a threshold on a score, which is the failure this
|
|
194
|
+
project has already had once -- a constant tied to whatever scale the
|
|
195
|
+
ranking currently produces. A refusal is a fact, not a judgement, and it is
|
|
196
|
+
also the only moment the reader has actually lost something.
|
|
197
|
+
"""
|
|
198
|
+
if reason not in _DESCRIBABLE_REASONS:
|
|
199
|
+
return ""
|
|
200
|
+
groups = store.classify(units)
|
|
201
|
+
undescribed = len(groups["missing"]) + len(groups["superseded"])
|
|
202
|
+
if not undescribed:
|
|
203
|
+
return ""
|
|
204
|
+
total = len(units) or 1
|
|
205
|
+
return (
|
|
206
|
+
f"\n{undescribed} of {total} declarations carry only the sentence the parser generated, "
|
|
207
|
+
f"which adds no word the source did not already have.\n"
|
|
208
|
+
f"Writing descriptions is what makes a question phrased in your own words reachable: "
|
|
209
|
+
f"run `rag-your-code bootstrap .` for the next batch."
|
|
210
|
+
)
|
|
211
|
+
|
|
212
|
+
|
|
177
213
|
def _cmd_search(args: argparse.Namespace) -> int:
|
|
178
214
|
"""The search command: retrieves the code units most relevant to a
|
|
179
215
|
question, optionally following relationships outward, and prints either
|
|
@@ -182,7 +218,7 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
182
218
|
similarity all fall back to the repository settings when no flag
|
|
183
219
|
overrides them. Warns when the index no longer describes the repository.
|
|
184
220
|
"""
|
|
185
|
-
payload, units, graph, cfg,
|
|
221
|
+
payload, units, graph, cfg, store = _load(args)
|
|
186
222
|
limit = args.limit if args.limit is not None else cfg["search.limit"]
|
|
187
223
|
max_chars = args.max_chars if args.max_chars is not None else cfg["search.max_chars"]
|
|
188
224
|
weight = cfg["search.vector_weight"]
|
|
@@ -209,7 +245,10 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
209
245
|
if payload.get("stale"):
|
|
210
246
|
print("Warning: index is stale; run `rag-your-code index` to refresh.", file=sys.stderr)
|
|
211
247
|
if report:
|
|
212
|
-
|
|
248
|
+
# The JSON reply is unchanged: a machine reads `diagnosis` and does
|
|
249
|
+
# not need prose about it. This line exists for the person watching,
|
|
250
|
+
# who otherwise has no way to learn that the lever exists at all.
|
|
251
|
+
print(f"No matching code units.\n{report['hint']}{_describe_nudge(store, units, report['reason'])}")
|
|
213
252
|
else:
|
|
214
253
|
print(context(results, max_chars))
|
|
215
254
|
return 0
|
|
@@ -66,6 +66,80 @@ def test_search_without_index_is_a_concise_error(tmp_path: Path):
|
|
|
66
66
|
assert "Traceback" not in proc.stderr
|
|
67
67
|
|
|
68
68
|
|
|
69
|
+
def _undescribed_repository(root: Path, count: int) -> Path:
|
|
70
|
+
"""Enough declarations for the evidence bars to be at full strength, none of
|
|
71
|
+
them carrying a written description.
|
|
72
|
+
"""
|
|
73
|
+
for index in range(count):
|
|
74
|
+
(root / f"ledger_{index}.py").write_text(
|
|
75
|
+
f"def post_ledger_entry_{index}(entry, ledger):\n"
|
|
76
|
+
' """Post an accounting entry to the ledger and return it."""\n'
|
|
77
|
+
" return ledger\n",
|
|
78
|
+
encoding="utf-8",
|
|
79
|
+
)
|
|
80
|
+
return root
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_a_refused_search_says_that_descriptions_are_the_missing_piece(tmp_path: Path):
|
|
84
|
+
"""The lever nobody can discover on their own.
|
|
85
|
+
|
|
86
|
+
Writing descriptions is the largest single move available on retrieval
|
|
87
|
+
quality, and until now the only place that said so was a skill that fires
|
|
88
|
+
when a model decides it should, or a command a user has to already know
|
|
89
|
+
exists. A refusal is the moment somebody has actually lost something, so it
|
|
90
|
+
is the moment worth spending a line on.
|
|
91
|
+
"""
|
|
92
|
+
_undescribed_repository(tmp_path, 220)
|
|
93
|
+
run_cli("index", str(tmp_path), cwd=Path.cwd())
|
|
94
|
+
refused = run_cli("search", "how is the mooring winch tension calibrated", "--root", str(tmp_path), cwd=Path.cwd())
|
|
95
|
+
assert "No matching code units." in refused.stdout
|
|
96
|
+
assert "declarations carry only the sentence the parser generated" in refused.stdout
|
|
97
|
+
assert "bootstrap" in refused.stdout
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def test_the_machine_readable_reply_gains_no_prose(tmp_path: Path):
|
|
101
|
+
"""The nudge is for the person watching. An agent reads `diagnosis` and
|
|
102
|
+
branches on `reason`; prose in the JSON would be a second, softer copy of a
|
|
103
|
+
field it already has.
|
|
104
|
+
"""
|
|
105
|
+
_undescribed_repository(tmp_path, 220)
|
|
106
|
+
run_cli("index", str(tmp_path), cwd=Path.cwd())
|
|
107
|
+
reply = json.loads(
|
|
108
|
+
run_cli("search", "how is the mooring winch tension calibrated", "--root", str(tmp_path), "--json", cwd=Path.cwd()).stdout
|
|
109
|
+
)
|
|
110
|
+
assert reply["results"] == []
|
|
111
|
+
assert reply["diagnosis"]["reason"]
|
|
112
|
+
assert "declarations carry only" not in json.dumps(reply)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def test_no_nudge_when_the_subject_is_simply_not_here(tmp_path: Path):
|
|
116
|
+
"""`matched_terms_are_scattered` means the words are present but never
|
|
117
|
+
together, which is what a question about a subject the repository does not
|
|
118
|
+
implement looks like. Describing more code cannot fix that, so advising it
|
|
119
|
+
would be advice that cannot work.
|
|
120
|
+
"""
|
|
121
|
+
from ragyourcode.cli import _describe_nudge
|
|
122
|
+
|
|
123
|
+
class _Store:
|
|
124
|
+
def classify(self, units):
|
|
125
|
+
return {"described": [], "superseded": [], "missing": list(units)}
|
|
126
|
+
|
|
127
|
+
units = [object()] * 10
|
|
128
|
+
assert _describe_nudge(_Store(), units, "matched_terms_are_scattered") == ""
|
|
129
|
+
assert _describe_nudge(_Store(), units, "too_little_of_the_query_matched")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def test_no_nudge_once_everything_carries_a_description(tmp_path: Path):
|
|
133
|
+
"""A repository that has done the work is not told to do it again."""
|
|
134
|
+
from ragyourcode.cli import _describe_nudge
|
|
135
|
+
|
|
136
|
+
class _Store:
|
|
137
|
+
def classify(self, units):
|
|
138
|
+
return {"described": list(units), "superseded": [], "missing": []}
|
|
139
|
+
|
|
140
|
+
assert _describe_nudge(_Store(), [object()] * 10, "no_query_term_in_index") == ""
|
|
141
|
+
|
|
142
|
+
|
|
69
143
|
SAMPLE_WITH_NON_ASCII = '''def 重试请求(url):
|
|
70
144
|
"重试失败的 HTTP 请求 🚀 with backoff."
|
|
71
145
|
return url
|
|
@@ -49,7 +49,13 @@ def test_the_manifests_point_at_a_repository_that_exists():
|
|
|
49
49
|
|
|
50
50
|
# --- the documentation an agent is told to follow must be executable --------
|
|
51
51
|
|
|
52
|
-
|
|
52
|
+
COMMANDS = tuple(sorted(str(path.relative_to(ROOT)).replace("\\", "/") for path in (ROOT / "commands").glob("*.md")))
|
|
53
|
+
# Every document that hands somebody a command to run. The command files are
|
|
54
|
+
# included by discovery rather than by name: a fifth command added without a
|
|
55
|
+
# line here would otherwise be the one file nothing checks, which is exactly how
|
|
56
|
+
# an install line naming a package index this project does not publish to
|
|
57
|
+
# shipped twice.
|
|
58
|
+
DOCS = ("skills/rag-your-code/SKILL.md", "README.md") + COMMANDS
|
|
53
59
|
|
|
54
60
|
|
|
55
61
|
def _subcommands() -> set[str]:
|
|
@@ -83,12 +89,18 @@ def test_every_documented_subcommand_exists():
|
|
|
83
89
|
|
|
84
90
|
def test_every_documented_protocol_action_is_handled():
|
|
85
91
|
known = _protocol_actions()
|
|
92
|
+
everywhere: set[str] = set()
|
|
86
93
|
for name in DOCS:
|
|
87
94
|
text = (ROOT / name).read_text(encoding="utf-8")
|
|
88
95
|
used = set(re.findall(r'"action"\s*:\s*"([a-z_]+)"', text))
|
|
89
96
|
unknown = used - known
|
|
90
97
|
assert not unknown, f"{name} documents actions the agent loop ignores: {sorted(unknown)}"
|
|
91
|
-
|
|
98
|
+
everywhere |= used
|
|
99
|
+
# The anti-vacuity guard belongs to the set, not to each file. A command
|
|
100
|
+
# file documents the command line and has no reason to mention the
|
|
101
|
+
# subprocess protocol at all; requiring one from every document would make
|
|
102
|
+
# this pass only by forcing irrelevant JSON into user-facing pages.
|
|
103
|
+
assert everywhere, "no document shows a protocol action; this guard would then pass vacuously"
|
|
92
104
|
|
|
93
105
|
|
|
94
106
|
def test_the_documented_list_of_actions_is_the_real_list():
|
|
@@ -159,6 +171,38 @@ def test_no_document_claims_this_package_is_on_an_index_it_is_not_on():
|
|
|
159
171
|
assert seen, "the install instructions vanished; this guard would then pass vacuously"
|
|
160
172
|
|
|
161
173
|
|
|
174
|
+
def test_every_command_is_loadable_and_describes_itself():
|
|
175
|
+
"""A command file is a plugin's user-facing surface, and Claude Code reads
|
|
176
|
+
its frontmatter to list it. A missing or empty `description` makes the
|
|
177
|
+
command invisible in the very place a user goes to discover it -- which is
|
|
178
|
+
the whole reason these exist, since a skill only fires when a model decides
|
|
179
|
+
it should.
|
|
180
|
+
"""
|
|
181
|
+
assert COMMANDS, "the plugin ships no commands; this guard would pass vacuously"
|
|
182
|
+
for name in COMMANDS:
|
|
183
|
+
text = (ROOT / name).read_text(encoding="utf-8")
|
|
184
|
+
assert text.startswith("---\n"), f"{name}: no frontmatter block"
|
|
185
|
+
front = text.split("---\n", 2)[1]
|
|
186
|
+
described = re.search(r"^description:\s*(\S.*)$", front, re.M)
|
|
187
|
+
assert described, f"{name}: frontmatter carries no description"
|
|
188
|
+
assert len(described.group(1)) <= 200, f"{name}: description is a paragraph, not a listing line"
|
|
189
|
+
assert re.search(r"^# /rag-your-code:", text, re.M), f"{name}: body does not name the command it is"
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def test_every_command_a_document_offers_actually_exists():
|
|
193
|
+
"""Both directions, for the same reason the settings table is checked both
|
|
194
|
+
ways: a document offering `/rag-your-code:describe` when no such command
|
|
195
|
+
exists is a dead end at the exact moment somebody took the advice, and a
|
|
196
|
+
command nobody is told about is not a feature.
|
|
197
|
+
"""
|
|
198
|
+
real = {Path(name).stem for name in COMMANDS}
|
|
199
|
+
offered: set[str] = set()
|
|
200
|
+
for name in ("README.md", "skills/rag-your-code/SKILL.md", *COMMANDS):
|
|
201
|
+
offered.update(re.findall(r"/rag-your-code:([a-z-]+)", (ROOT / name).read_text(encoding="utf-8")))
|
|
202
|
+
assert offered - real == set(), f"documents offer commands that do not exist: {sorted(offered - real)}"
|
|
203
|
+
assert real - offered == set(), f"commands nobody is told about: {sorted(real - offered)}"
|
|
204
|
+
|
|
205
|
+
|
|
162
206
|
def test_the_documented_fixture_counts_are_the_real_ones():
|
|
163
207
|
"""Numbers stated in prose, checked against the data they describe.
|
|
164
208
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|