rag-your-code 1.1.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {rag_your_code-1.1.0/src/rag_your_code.egg-info → rag_your_code-1.2.0}/PKG-INFO +29 -17
  2. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/README.md +28 -16
  3. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/pyproject.toml +1 -1
  4. {rag_your_code-1.1.0 → rag_your_code-1.2.0/src/rag_your_code.egg-info}/PKG-INFO +29 -17
  5. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/__init__.py +1 -1
  6. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/cli.py +41 -2
  7. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_e2e_cli.py +74 -0
  8. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_metadata.py +46 -2
  9. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/LICENSE +0 -0
  10. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/setup.cfg +0 -0
  11. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/SOURCES.txt +0 -0
  12. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  13. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  14. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  15. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  16. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/agentic.py +0 -0
  17. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/annotate.py +0 -0
  18. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/config.py +0 -0
  19. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/descriptions.py +0 -0
  20. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/document.py +0 -0
  21. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/embeddings.py +0 -0
  22. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/graph.py +0 -0
  23. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/indexer.py +0 -0
  24. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/models.py +0 -0
  25. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/parser.py +0 -0
  26. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/providers.py +0 -0
  27. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/py.typed +0 -0
  28. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/search.py +0 -0
  29. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/src/ragyourcode/workflow.py +0 -0
  30. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_absent_queries.py +0 -0
  31. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_agent_protocol.py +0 -0
  32. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_agentic.py +0 -0
  33. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_config.py +0 -0
  34. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_descriptions.py +0 -0
  35. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_doc_comments.py +0 -0
  36. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_document.py +0 -0
  37. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_evidence.py +0 -0
  38. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_golden.py +0 -0
  39. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_graph_incremental.py +0 -0
  40. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_language_fixtures.py +0 -0
  41. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_large_repo.py +0 -0
  42. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_local_model.py +0 -0
  43. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_multilanguage.py +0 -0
  44. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_parser_edges.py +0 -0
  45. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_providers.py +0 -0
  46. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_ragyourcode.py +0 -0
  47. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_ranking.py +0 -0
  48. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_repo_queries.py +0 -0
  49. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_resilience.py +0 -0
  50. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_retrieval_correctness.py +0 -0
  51. {rag_your_code-1.1.0 → rag_your_code-1.2.0}/tests/test_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -129,7 +129,7 @@ the scale.
129
129
 
130
130
  The default embedder is a signed feature hash. Ablating it entirely moves the
131
131
  three positive rulers by **±1 question in either direction** while the vectors
132
- occupy **65.4%** of the index. That was known since 0.6.0 and left unexplained.
132
+ occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
133
133
  The explanation, measured here:
134
134
 
135
135
  - **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
@@ -323,9 +323,9 @@ it answers a different question and names a different diagnosis.
323
323
 
324
324
  | | |
325
325
  |---|---|
326
- | query, median | **0.41 ms** |
327
- | query, p95 | 0.60 ms |
328
- | refusing an unanswerable query | **0.01 ms** |
326
+ | query, median | **0.83 ms** |
327
+ | query, p95 | 1.68 ms |
328
+ | refusing an unanswerable query | **0.03 ms** |
329
329
 
330
330
  Refusal is cheaper than answering by a factor of forty: an unanswerable query
331
331
  touches only the posting lists of its own distinctive words, never the corpus.
@@ -364,22 +364,22 @@ declaration spans.
364
364
  **On a repository nobody has described, Grep wins.** That is the measured
365
365
  result and it is not softened here.
366
366
 
367
- | foreign repository · 35 questions · 1,257 units · no descriptions | Grep loop | rag-your-code |
367
+ | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
368
368
  |---|---|---|
369
369
  | right file first | **34.3%** | 31.4% |
370
370
  | right file in top 3 | **60.0%** | 48.6% |
371
- | lines matched across the repo, all questions | 33,115 | — |
372
- | characters returned, all questions | — | **163,521** |
371
+ | lines matched across the repo, all questions | 33,213 | — |
372
+ | characters returned, all questions | — | **163,294** |
373
373
  | questions it answers | 35 | 28 |
374
374
 
375
375
  **Once the vocabulary exists, it is not close.**
376
376
 
377
- | this repository · 70 questions · 557 units · 303 described | Grep loop | rag-your-code |
377
+ | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
378
378
  |---|---|---|
379
- | right file first | 25.7% | **57.1%** |
379
+ | right file first | 25.7% | **58.6%** |
380
380
  | right file in top 3 | 64.3% | **75.7%** |
381
- | lines matched across the repo, all questions | 39,550 | — |
382
- | characters returned, all questions | — | **278,929** |
381
+ | lines matched across the repo, all questions | 40,150 | — |
382
+ | characters returned, all questions | — | **277,327** |
383
383
  | questions it answers | 70 | 60 |
384
384
 
385
385
  Those two tables are the whole argument of section 3.3, measured against a real
@@ -396,7 +396,7 @@ Three qualifications, because the table would otherwise flatter both sides:
396
396
  file; a hit here is a declaration with an exact span, a score, and the words
397
397
  it matched on. The agent that reads the result opens 40 lines, not a file.
398
398
  - **Grep answers everything.** It never declines, which is why it hands back
399
- 33,115 matching lines for 35 questions — about 950 lines per question, no
399
+ 33,213 matching lines for 35 questions — about 950 lines per question, no
400
400
  ranking, no spans, no indication which match is the definition. This returns
401
401
  roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
402
402
  questions come back empty here instead, with a reason.
@@ -527,9 +527,21 @@ measured worse, so it stays off there.
527
527
  /reload-plugins
528
528
  ```
529
529
 
530
- One skill, no hooks, no agents, no MCP server: **~39 tokens added to every
531
- session**, ~1.4k only when it fires. The skill installs the package on first
532
- use.
530
+ Four commands and one skill. No hooks, no agents, no MCP server:
531
+
532
+ | | |
533
+ |---|---|
534
+ | `/rag-your-code:index` | index, and say which rung this repository is on |
535
+ | `/rag-your-code:search` | ask in plain language; cite `path:line` |
536
+ | `/rag-your-code:describe` | write the vocabulary the source does not contain |
537
+ | `/rag-your-code:status` | stale? coverage? which embedder? what next? |
538
+
539
+ Measured with `claude plugin details` on an installed copy: **~249 tokens added
540
+ to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
541
+ one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
542
+ of being findable — a skill fires only when a model decides it should, which
543
+ left the whole plugin with no entry point a person could discover. The commands
544
+ install the Python package on first use.
533
545
 
534
546
  **As a CLI:**
535
547
 
@@ -627,7 +639,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
627
639
  prose. A callee-before-caller rerank fires on zero questions and the `name`
628
640
  field weight moves nothing, because an underscored test name is a single token.
629
641
 
630
- **The vectors are 65.4% of the index and earn ±1 question** under the default
642
+ **The vectors are 65.3% of the index and earn ±1 question** under the default
631
643
  embedder. Not removed: the same storage is what makes an optional model work,
632
644
  and the schema stays one shape.
633
645
 
@@ -100,7 +100,7 @@ the scale.
100
100
 
101
101
  The default embedder is a signed feature hash. Ablating it entirely moves the
102
102
  three positive rulers by **±1 question in either direction** while the vectors
103
- occupy **65.4%** of the index. That was known since 0.6.0 and left unexplained.
103
+ occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
104
104
  The explanation, measured here:
105
105
 
106
106
  - **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
@@ -294,9 +294,9 @@ it answers a different question and names a different diagnosis.
294
294
 
295
295
  | | |
296
296
  |---|---|
297
- | query, median | **0.41 ms** |
298
- | query, p95 | 0.60 ms |
299
- | refusing an unanswerable query | **0.01 ms** |
297
+ | query, median | **0.83 ms** |
298
+ | query, p95 | 1.68 ms |
299
+ | refusing an unanswerable query | **0.03 ms** |
300
300
 
301
301
  Refusal is cheaper than answering by a factor of forty: an unanswerable query
302
302
  touches only the posting lists of its own distinctive words, never the corpus.
@@ -335,22 +335,22 @@ declaration spans.
335
335
  **On a repository nobody has described, Grep wins.** That is the measured
336
336
  result and it is not softened here.
337
337
 
338
- | foreign repository · 35 questions · 1,257 units · no descriptions | Grep loop | rag-your-code |
338
+ | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
339
339
  |---|---|---|
340
340
  | right file first | **34.3%** | 31.4% |
341
341
  | right file in top 3 | **60.0%** | 48.6% |
342
- | lines matched across the repo, all questions | 33,115 | — |
343
- | characters returned, all questions | — | **163,521** |
342
+ | lines matched across the repo, all questions | 33,213 | — |
343
+ | characters returned, all questions | — | **163,294** |
344
344
  | questions it answers | 35 | 28 |
345
345
 
346
346
  **Once the vocabulary exists, it is not close.**
347
347
 
348
- | this repository · 70 questions · 557 units · 303 described | Grep loop | rag-your-code |
348
+ | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
349
349
  |---|---|---|
350
- | right file first | 25.7% | **57.1%** |
350
+ | right file first | 25.7% | **58.6%** |
351
351
  | right file in top 3 | 64.3% | **75.7%** |
352
- | lines matched across the repo, all questions | 39,550 | — |
353
- | characters returned, all questions | — | **278,929** |
352
+ | lines matched across the repo, all questions | 40,150 | — |
353
+ | characters returned, all questions | — | **277,327** |
354
354
  | questions it answers | 70 | 60 |
355
355
 
356
356
  Those two tables are the whole argument of section 3.3, measured against a real
@@ -367,7 +367,7 @@ Three qualifications, because the table would otherwise flatter both sides:
367
367
  file; a hit here is a declaration with an exact span, a score, and the words
368
368
  it matched on. The agent that reads the result opens 40 lines, not a file.
369
369
  - **Grep answers everything.** It never declines, which is why it hands back
370
- 33,115 matching lines for 35 questions — about 950 lines per question, no
370
+ 33,213 matching lines for 35 questions — about 950 lines per question, no
371
371
  ranking, no spans, no indication which match is the definition. This returns
372
372
  roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
373
373
  questions come back empty here instead, with a reason.
@@ -498,9 +498,21 @@ measured worse, so it stays off there.
498
498
  /reload-plugins
499
499
  ```
500
500
 
501
- One skill, no hooks, no agents, no MCP server: **~39 tokens added to every
502
- session**, ~1.4k only when it fires. The skill installs the package on first
503
- use.
501
+ Four commands and one skill. No hooks, no agents, no MCP server:
502
+
503
+ | | |
504
+ |---|---|
505
+ | `/rag-your-code:index` | index, and say which rung this repository is on |
506
+ | `/rag-your-code:search` | ask in plain language; cite `path:line` |
507
+ | `/rag-your-code:describe` | write the vocabulary the source does not contain |
508
+ | `/rag-your-code:status` | stale? coverage? which embedder? what next? |
509
+
510
+ Measured with `claude plugin details` on an installed copy: **~249 tokens added
511
+ to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
512
+ one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
513
+ of being findable — a skill fires only when a model decides it should, which
514
+ left the whole plugin with no entry point a person could discover. The commands
515
+ install the Python package on first use.
504
516
 
505
517
  **As a CLI:**
506
518
 
@@ -598,7 +610,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
598
610
  prose. A callee-before-caller rerank fires on zero questions and the `name`
599
611
  field weight moves nothing, because an underscored test name is a single token.
600
612
 
601
- **The vectors are 65.4% of the index and earn ±1 question** under the default
613
+ **The vectors are 65.3% of the index and earn ±1 question** under the default
602
614
  embedder. Not removed: the same storage is what makes an optional model work,
603
615
  and the schema stays one shape.
604
616
 
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "1.1.0"
9
+ version = "1.2.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 1.1.0
3
+ Version: 1.2.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -129,7 +129,7 @@ the scale.
129
129
 
130
130
  The default embedder is a signed feature hash. Ablating it entirely moves the
131
131
  three positive rulers by **±1 question in either direction** while the vectors
132
- occupy **65.4%** of the index. That was known since 0.6.0 and left unexplained.
132
+ occupy **65.3%** of the index. That was known since 0.6.0 and left unexplained.
133
133
  The explanation, measured here:
134
134
 
135
135
  - **Not saturation.** Median 56 distinct tokens per unit into 384 buckets;
@@ -323,9 +323,9 @@ it answers a different question and names a different diagnosis.
323
323
 
324
324
  | | |
325
325
  |---|---|
326
- | query, median | **0.41 ms** |
327
- | query, p95 | 0.60 ms |
328
- | refusing an unanswerable query | **0.01 ms** |
326
+ | query, median | **0.83 ms** |
327
+ | query, p95 | 1.68 ms |
328
+ | refusing an unanswerable query | **0.03 ms** |
329
329
 
330
330
  Refusal is cheaper than answering by a factor of forty: an unanswerable query
331
331
  touches only the posting lists of its own distinctive words, never the corpus.
@@ -364,22 +364,22 @@ declaration spans.
364
364
  **On a repository nobody has described, Grep wins.** That is the measured
365
365
  result and it is not softened here.
366
366
 
367
- | foreign repository · 35 questions · 1,257 units · no descriptions | Grep loop | rag-your-code |
367
+ | foreign repository · 35 questions · 1,267 units · no descriptions | Grep loop | rag-your-code |
368
368
  |---|---|---|
369
369
  | right file first | **34.3%** | 31.4% |
370
370
  | right file in top 3 | **60.0%** | 48.6% |
371
- | lines matched across the repo, all questions | 33,115 | — |
372
- | characters returned, all questions | — | **163,521** |
371
+ | lines matched across the repo, all questions | 33,213 | — |
372
+ | characters returned, all questions | — | **163,294** |
373
373
  | questions it answers | 35 | 28 |
374
374
 
375
375
  **Once the vocabulary exists, it is not close.**
376
376
 
377
- | this repository · 70 questions · 557 units · 303 described | Grep loop | rag-your-code |
377
+ | this repository · 70 questions · 569 units · 304 described | Grep loop | rag-your-code |
378
378
  |---|---|---|
379
- | right file first | 25.7% | **57.1%** |
379
+ | right file first | 25.7% | **58.6%** |
380
380
  | right file in top 3 | 64.3% | **75.7%** |
381
- | lines matched across the repo, all questions | 39,550 | — |
382
- | characters returned, all questions | — | **278,929** |
381
+ | lines matched across the repo, all questions | 40,150 | — |
382
+ | characters returned, all questions | — | **277,327** |
383
383
  | questions it answers | 70 | 60 |
384
384
 
385
385
  Those two tables are the whole argument of section 3.3, measured against a real
@@ -396,7 +396,7 @@ Three qualifications, because the table would otherwise flatter both sides:
396
396
  file; a hit here is a declaration with an exact span, a score, and the words
397
397
  it matched on. The agent that reads the result opens 40 lines, not a file.
398
398
  - **Grep answers everything.** It never declines, which is why it hands back
399
- 33,115 matching lines for 35 questions — about 950 lines per question, no
399
+ 33,213 matching lines for 35 questions — about 950 lines per question, no
400
400
  ranking, no spans, no indication which match is the definition. This returns
401
401
  roughly 5,800 characters per question, ranked. Seven of 35 and ten of 70
402
402
  questions come back empty here instead, with a reason.
@@ -527,9 +527,21 @@ measured worse, so it stays off there.
527
527
  /reload-plugins
528
528
  ```
529
529
 
530
- One skill, no hooks, no agents, no MCP server: **~39 tokens added to every
531
- session**, ~1.4k only when it fires. The skill installs the package on first
532
- use.
530
+ Four commands and one skill. No hooks, no agents, no MCP server:
531
+
532
+ | | |
533
+ |---|---|
534
+ | `/rag-your-code:index` | index, and say which rung this repository is on |
535
+ | `/rag-your-code:search` | ask in plain language; cite `path:line` |
536
+ | `/rag-your-code:describe` | write the vocabulary the source does not contain |
537
+ | `/rag-your-code:status` | stale? coverage? which embedder? what next? |
538
+
539
+ Measured with `claude plugin details` on an installed copy: **~249 tokens added
540
+ to every session** (skill ~30, each command ~50–60), and 590–2,400 only when
541
+ one of them fires. That is up from ~39 in 1.1.0, and the increase is the price
542
+ of being findable — a skill fires only when a model decides it should, which
543
+ left the whole plugin with no entry point a person could discover. The commands
544
+ install the Python package on first use.
533
545
 
534
546
  **As a CLI:**
535
547
 
@@ -627,7 +639,7 @@ wrong: of the eight inspected, seven are unrelated tests winning on
627
639
  prose. A callee-before-caller rerank fires on zero questions and the `name`
628
640
  field weight moves nothing, because an underscored test name is a single token.
629
641
 
630
- **The vectors are 65.4% of the index and earn ±1 question** under the default
642
+ **The vectors are 65.3% of the index and earn ±1 question** under the default
631
643
  embedder. Not removed: the same storage is what makes an optional model work,
632
644
  and the schema stays one shape.
633
645
 
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "1.1.0"
6
+ __version__ = "1.2.0"
@@ -174,6 +174,42 @@ def _load(args: argparse.Namespace):
174
174
  return payload, units, graph_from_dict(units, payload.get("graph")), cfg, store
175
175
 
176
176
 
177
+ # Refusal reasons that writing descriptions can actually fix. `scattered` is
178
+ # deliberately absent: it means the words are here but never together, which is
179
+ # what a question about a subject the repository does not implement looks like,
180
+ # and telling somebody to describe more code would be advice that cannot work.
181
+ _DESCRIBABLE_REASONS = frozenset({
182
+ "no_query_term_in_index",
183
+ "only_ubiquitous_terms_matched",
184
+ "too_little_of_the_query_matched",
185
+ })
186
+
187
+
188
+ def _describe_nudge(store, units, reason: str) -> str:
189
+ """The one line worth printing when a question came back empty and the
190
+ vocabulary that would have answered it has not been written yet.
191
+
192
+ Deliberately tied to a refusal rather than to a low score. "The results
193
+ looked weak" would need a threshold on a score, which is the failure this
194
+ project has already had once -- a constant tied to whatever scale the
195
+ ranking currently produces. A refusal is a fact, not a judgement, and it is
196
+ also the only moment the reader has actually lost something.
197
+ """
198
+ if reason not in _DESCRIBABLE_REASONS:
199
+ return ""
200
+ groups = store.classify(units)
201
+ undescribed = len(groups["missing"]) + len(groups["superseded"])
202
+ if not undescribed:
203
+ return ""
204
+ total = len(units) or 1
205
+ return (
206
+ f"\n{undescribed} of {total} declarations carry only the sentence the parser generated, "
207
+ f"which adds no word the source did not already have.\n"
208
+ f"Writing descriptions is what makes a question phrased in your own words reachable: "
209
+ f"run `rag-your-code bootstrap .` for the next batch."
210
+ )
211
+
212
+
177
213
  def _cmd_search(args: argparse.Namespace) -> int:
178
214
  """The search command: retrieves the code units most relevant to a
179
215
  question, optionally following relationships outward, and prints either
@@ -182,7 +218,7 @@ def _cmd_search(args: argparse.Namespace) -> int:
182
218
  similarity all fall back to the repository settings when no flag
183
219
  overrides them. Warns when the index no longer describes the repository.
184
220
  """
185
- payload, units, graph, cfg, _ = _load(args)
221
+ payload, units, graph, cfg, store = _load(args)
186
222
  limit = args.limit if args.limit is not None else cfg["search.limit"]
187
223
  max_chars = args.max_chars if args.max_chars is not None else cfg["search.max_chars"]
188
224
  weight = cfg["search.vector_weight"]
@@ -209,7 +245,10 @@ def _cmd_search(args: argparse.Namespace) -> int:
209
245
  if payload.get("stale"):
210
246
  print("Warning: index is stale; run `rag-your-code index` to refresh.", file=sys.stderr)
211
247
  if report:
212
- print(f"No matching code units.\n{report['hint']}")
248
+ # The JSON reply is unchanged: a machine reads `diagnosis` and does
249
+ # not need prose about it. This line exists for the person watching,
250
+ # who otherwise has no way to learn that the lever exists at all.
251
+ print(f"No matching code units.\n{report['hint']}{_describe_nudge(store, units, report['reason'])}")
213
252
  else:
214
253
  print(context(results, max_chars))
215
254
  return 0
@@ -66,6 +66,80 @@ def test_search_without_index_is_a_concise_error(tmp_path: Path):
66
66
  assert "Traceback" not in proc.stderr
67
67
 
68
68
 
69
+ def _undescribed_repository(root: Path, count: int) -> Path:
70
+ """Enough declarations for the evidence bars to be at full strength, none of
71
+ them carrying a written description.
72
+ """
73
+ for index in range(count):
74
+ (root / f"ledger_{index}.py").write_text(
75
+ f"def post_ledger_entry_{index}(entry, ledger):\n"
76
+ ' """Post an accounting entry to the ledger and return it."""\n'
77
+ " return ledger\n",
78
+ encoding="utf-8",
79
+ )
80
+ return root
81
+
82
+
83
+ def test_a_refused_search_says_that_descriptions_are_the_missing_piece(tmp_path: Path):
84
+ """The lever nobody can discover on their own.
85
+
86
+ Writing descriptions is the largest single move available on retrieval
87
+ quality, and until now the only place that said so was a skill that fires
88
+ when a model decides it should, or a command a user has to already know
89
+ exists. A refusal is the moment somebody has actually lost something, so it
90
+ is the moment worth spending a line on.
91
+ """
92
+ _undescribed_repository(tmp_path, 220)
93
+ run_cli("index", str(tmp_path), cwd=Path.cwd())
94
+ refused = run_cli("search", "how is the mooring winch tension calibrated", "--root", str(tmp_path), cwd=Path.cwd())
95
+ assert "No matching code units." in refused.stdout
96
+ assert "declarations carry only the sentence the parser generated" in refused.stdout
97
+ assert "bootstrap" in refused.stdout
98
+
99
+
100
+ def test_the_machine_readable_reply_gains_no_prose(tmp_path: Path):
101
+ """The nudge is for the person watching. An agent reads `diagnosis` and
102
+ branches on `reason`; prose in the JSON would be a second, softer copy of a
103
+ field it already has.
104
+ """
105
+ _undescribed_repository(tmp_path, 220)
106
+ run_cli("index", str(tmp_path), cwd=Path.cwd())
107
+ reply = json.loads(
108
+ run_cli("search", "how is the mooring winch tension calibrated", "--root", str(tmp_path), "--json", cwd=Path.cwd()).stdout
109
+ )
110
+ assert reply["results"] == []
111
+ assert reply["diagnosis"]["reason"]
112
+ assert "declarations carry only" not in json.dumps(reply)
113
+
114
+
115
+ def test_no_nudge_when_the_subject_is_simply_not_here(tmp_path: Path):
116
+ """`matched_terms_are_scattered` means the words are present but never
117
+ together, which is what a question about a subject the repository does not
118
+ implement looks like. Describing more code cannot fix that, so advising it
119
+ would be advice that cannot work.
120
+ """
121
+ from ragyourcode.cli import _describe_nudge
122
+
123
+ class _Store:
124
+ def classify(self, units):
125
+ return {"described": [], "superseded": [], "missing": list(units)}
126
+
127
+ units = [object()] * 10
128
+ assert _describe_nudge(_Store(), units, "matched_terms_are_scattered") == ""
129
+ assert _describe_nudge(_Store(), units, "too_little_of_the_query_matched")
130
+
131
+
132
+ def test_no_nudge_once_everything_carries_a_description(tmp_path: Path):
133
+ """A repository that has done the work is not told to do it again."""
134
+ from ragyourcode.cli import _describe_nudge
135
+
136
+ class _Store:
137
+ def classify(self, units):
138
+ return {"described": list(units), "superseded": [], "missing": []}
139
+
140
+ assert _describe_nudge(_Store(), [object()] * 10, "no_query_term_in_index") == ""
141
+
142
+
69
143
  SAMPLE_WITH_NON_ASCII = '''def 重试请求(url):
70
144
  "重试失败的 HTTP 请求 🚀 with backoff."
71
145
  return url
@@ -49,7 +49,13 @@ def test_the_manifests_point_at_a_repository_that_exists():
49
49
 
50
50
  # --- the documentation an agent is told to follow must be executable --------
51
51
 
52
- DOCS = ("skills/rag-your-code/SKILL.md", "README.md")
52
+ COMMANDS = tuple(sorted(str(path.relative_to(ROOT)).replace("\\", "/") for path in (ROOT / "commands").glob("*.md")))
53
+ # Every document that hands somebody a command to run. The command files are
54
+ # included by discovery rather than by name: a fifth command added without a
55
+ # line here would otherwise be the one file nothing checks, which is exactly how
56
+ # an install line naming a package index this project does not publish to
57
+ # shipped twice.
58
+ DOCS = ("skills/rag-your-code/SKILL.md", "README.md") + COMMANDS
53
59
 
54
60
 
55
61
  def _subcommands() -> set[str]:
@@ -83,12 +89,18 @@ def test_every_documented_subcommand_exists():
83
89
 
84
90
  def test_every_documented_protocol_action_is_handled():
85
91
  known = _protocol_actions()
92
+ everywhere: set[str] = set()
86
93
  for name in DOCS:
87
94
  text = (ROOT / name).read_text(encoding="utf-8")
88
95
  used = set(re.findall(r'"action"\s*:\s*"([a-z_]+)"', text))
89
96
  unknown = used - known
90
97
  assert not unknown, f"{name} documents actions the agent loop ignores: {sorted(unknown)}"
91
- assert used, f"{name} should show at least one protocol action"
98
+ everywhere |= used
99
+ # The anti-vacuity guard belongs to the set, not to each file. A command
100
+ # file documents the command line and has no reason to mention the
101
+ # subprocess protocol at all; requiring one from every document would make
102
+ # this pass only by forcing irrelevant JSON into user-facing pages.
103
+ assert everywhere, "no document shows a protocol action; this guard would then pass vacuously"
92
104
 
93
105
 
94
106
  def test_the_documented_list_of_actions_is_the_real_list():
@@ -159,6 +171,38 @@ def test_no_document_claims_this_package_is_on_an_index_it_is_not_on():
159
171
  assert seen, "the install instructions vanished; this guard would then pass vacuously"
160
172
 
161
173
 
174
+ def test_every_command_is_loadable_and_describes_itself():
175
+ """A command file is a plugin's user-facing surface, and Claude Code reads
176
+ its frontmatter to list it. A missing or empty `description` makes the
177
+ command invisible in the very place a user goes to discover it -- which is
178
+ the whole reason these exist, since a skill only fires when a model decides
179
+ it should.
180
+ """
181
+ assert COMMANDS, "the plugin ships no commands; this guard would pass vacuously"
182
+ for name in COMMANDS:
183
+ text = (ROOT / name).read_text(encoding="utf-8")
184
+ assert text.startswith("---\n"), f"{name}: no frontmatter block"
185
+ front = text.split("---\n", 2)[1]
186
+ described = re.search(r"^description:\s*(\S.*)$", front, re.M)
187
+ assert described, f"{name}: frontmatter carries no description"
188
+ assert len(described.group(1)) <= 200, f"{name}: description is a paragraph, not a listing line"
189
+ assert re.search(r"^# /rag-your-code:", text, re.M), f"{name}: body does not name the command it is"
190
+
191
+
192
+ def test_every_command_a_document_offers_actually_exists():
193
+ """Both directions, for the same reason the settings table is checked both
194
+ ways: a document offering `/rag-your-code:describe` when no such command
195
+ exists is a dead end at the exact moment somebody took the advice, and a
196
+ command nobody is told about is not a feature.
197
+ """
198
+ real = {Path(name).stem for name in COMMANDS}
199
+ offered: set[str] = set()
200
+ for name in ("README.md", "skills/rag-your-code/SKILL.md", *COMMANDS):
201
+ offered.update(re.findall(r"/rag-your-code:([a-z-]+)", (ROOT / name).read_text(encoding="utf-8")))
202
+ assert offered - real == set(), f"documents offer commands that do not exist: {sorted(offered - real)}"
203
+ assert real - offered == set(), f"commands nobody is told about: {sorted(real - offered)}"
204
+
205
+
162
206
  def test_the_documented_fixture_counts_are_the_real_ones():
163
207
  """Numbers stated in prose, checked against the data they describe.
164
208
 
File without changes
File without changes