rag-your-code 0.7.0__tar.gz → 0.8.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. {rag_your_code-0.7.0/src/rag_your_code.egg-info → rag_your_code-0.8.0}/PKG-INFO +60 -8
  2. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/README.md +60 -8
  3. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/pyproject.toml +1 -1
  4. {rag_your_code-0.7.0 → rag_your_code-0.8.0/src/rag_your_code.egg-info}/PKG-INFO +60 -8
  5. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/SOURCES.txt +2 -0
  6. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/__init__.py +1 -1
  7. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/agentic.py +4 -3
  8. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/cli.py +26 -11
  9. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/config.py +77 -1
  10. rag_your_code-0.8.0/src/ragyourcode/embeddings.py +194 -0
  11. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/graph.py +3 -2
  12. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/indexer.py +17 -6
  13. rag_your_code-0.8.0/src/ragyourcode/providers.py +174 -0
  14. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/search.py +45 -4
  15. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/workflow.py +12 -5
  16. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_config.py +1 -0
  17. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_metadata.py +27 -0
  18. rag_your_code-0.8.0/tests/test_providers.py +363 -0
  19. rag_your_code-0.7.0/src/ragyourcode/embeddings.py +0 -82
  20. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/LICENSE +0 -0
  21. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/setup.cfg +0 -0
  22. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
  23. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
  24. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/requires.txt +0 -0
  25. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
  26. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/annotate.py +0 -0
  27. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/descriptions.py +0 -0
  28. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/document.py +0 -0
  29. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/models.py +0 -0
  30. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/parser.py +0 -0
  31. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/py.typed +0 -0
  32. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_agent_protocol.py +0 -0
  33. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_agentic.py +0 -0
  34. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_descriptions.py +0 -0
  35. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_doc_comments.py +0 -0
  36. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_document.py +0 -0
  37. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_e2e_cli.py +0 -0
  38. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_golden.py +0 -0
  39. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_graph_incremental.py +0 -0
  40. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_language_fixtures.py +0 -0
  41. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_large_repo.py +0 -0
  42. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_multilanguage.py +0 -0
  43. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_parser_edges.py +0 -0
  44. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_ragyourcode.py +0 -0
  45. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_ranking.py +0 -0
  46. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_repo_queries.py +0 -0
  47. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_resilience.py +0 -0
  48. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_retrieval_correctness.py +0 -0
  49. {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_workflow.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -407,14 +407,66 @@ it stopped.
407
407
  Nothing authored lives under `.rag-your-code/` — that directory is what people
408
408
  delete to clear the cache.
409
409
 
410
- ## Not here yet
410
+ ## Bringing your own model
411
411
 
412
- Provider-backed embeddings, Tree-sitter parsing, and a SQLite/ANN storage layer
413
- for repositories past the measured JSON envelope. Agent-authored descriptions
414
- are deliberately the cheaper answer to the same problem provider embeddings
415
- solve: they keep the zero-dependency, offline, reproducible-index properties,
416
- and produce text a human can read and correct rather than opaque floats. See
417
- [docs/ROADMAP.md](docs/ROADMAP.md).
412
+ Everything above works with no model at all. If you would rather have real
413
+ semantics, point the index at any OpenAI-compatible embeddings endpoint —
414
+ which includes a model server on your own machine:
415
+
416
+ ```toml
417
+ # rag-your-code.toml
418
+ [embedding]
419
+ provider = "openai-compatible"
420
+ endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
421
+ model = "nomic-embed-text"
422
+ dimensions = 768 # must match the model
423
+ ```
424
+
425
+ A hosted service is the same three lines with an `https://` endpoint, plus the
426
+ name of the environment variable holding your key:
427
+
428
+ ```toml
429
+ api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
430
+ ```
431
+
432
+ **The key is never a setting.** `rag-your-code.toml` is meant to be committed
433
+ so everyone who clones can see what shaped the index; a credential is the one
434
+ value with the opposite requirement, so the file only ever names the variable
435
+ it lives in. Sending a key over plain `http://` to anything but your own
436
+ machine is refused rather than warned about.
437
+
438
+ Three things follow from turning this on, and it is worth knowing all three
439
+ before you do:
440
+
441
+ - **Your source leaves the machine**, unless the endpoint is local. That is
442
+ the whole reason the local case is written first here.
443
+ - **Similarity may now find things, not just order them.** With the local
444
+ hash a cosine shortlist is measurably noise, so it is confined to
445
+ re-ranking. A real model earns the right to add candidates the words never
446
+ reached, which is the one gap no amount of ranking closes:
447
+ `search.vector_recall` sets how many. Lexical evidence still dominates — a
448
+ unit found by similarity alone scores at most `search.vector_weight`.
449
+ - **A failure stops the build.** Falling back to the local hash would leave an
450
+ index whose vectors come from two incompatible spaces, and ranking would act
451
+ on the meaningless cosine between them with full confidence.
452
+
453
+ Switching provider, model or width discards the old vectors and rebuilds, so
454
+ an index can never be a mixture. An incremental run over unchanged files makes
455
+ no request at all.
456
+
457
+ **What is not measured:** whether this helps *your* repository, and by how
458
+ much. No number here is from a real model — this project has no key, and a
459
+ figure produced by a stub would be fiction. The instrument ships instead:
460
+ point `benchmarks/repo_queries.py --index` at your own index and grade it. You
461
+ will probably also want a higher `search.vector_weight` than the 0.15 tuned
462
+ for a hash that carries no meaning.
463
+
464
+ ## Still not here
465
+
466
+ Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
467
+ measured JSON envelope. Note also that `search.vector_recall` scans every
468
+ unit's vector on every query, which is fine at the measured envelope and is
469
+ the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
418
470
 
419
471
  ## Development
420
472
 
@@ -380,14 +380,66 @@ it stopped.
380
380
  Nothing authored lives under `.rag-your-code/` — that directory is what people
381
381
  delete to clear the cache.
382
382
 
383
- ## Not here yet
384
-
385
- Provider-backed embeddings, Tree-sitter parsing, and a SQLite/ANN storage layer
386
- for repositories past the measured JSON envelope. Agent-authored descriptions
387
- are deliberately the cheaper answer to the same problem provider embeddings
388
- solve: they keep the zero-dependency, offline, reproducible-index properties,
389
- and produce text a human can read and correct rather than opaque floats. See
390
- [docs/ROADMAP.md](docs/ROADMAP.md).
383
+ ## Bringing your own model
384
+
385
+ Everything above works with no model at all. If you would rather have real
386
+ semantics, point the index at any OpenAI-compatible embeddings endpoint —
387
+ which includes a model server on your own machine:
388
+
389
+ ```toml
390
+ # rag-your-code.toml
391
+ [embedding]
392
+ provider = "openai-compatible"
393
+ endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
394
+ model = "nomic-embed-text"
395
+ dimensions = 768 # must match the model
396
+ ```
397
+
398
+ A hosted service is the same three lines with an `https://` endpoint, plus the
399
+ name of the environment variable holding your key:
400
+
401
+ ```toml
402
+ api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
403
+ ```
404
+
405
+ **The key is never a setting.** `rag-your-code.toml` is meant to be committed
406
+ so everyone who clones can see what shaped the index; a credential is the one
407
+ value with the opposite requirement, so the file only ever names the variable
408
+ it lives in. Sending a key over plain `http://` to anything but your own
409
+ machine is refused rather than warned about.
410
+
411
+ Three things follow from turning this on, and it is worth knowing all three
412
+ before you do:
413
+
414
+ - **Your source leaves the machine**, unless the endpoint is local. That is
415
+ the whole reason the local case is written first here.
416
+ - **Similarity may now find things, not just order them.** With the local
417
+ hash a cosine shortlist is measurably noise, so it is confined to
418
+ re-ranking. A real model earns the right to add candidates the words never
419
+ reached, which is the one gap no amount of ranking closes:
420
+ `search.vector_recall` sets how many. Lexical evidence still dominates — a
421
+ unit found by similarity alone scores at most `search.vector_weight`.
422
+ - **A failure stops the build.** Falling back to the local hash would leave an
423
+ index whose vectors come from two incompatible spaces, and ranking would act
424
+ on the meaningless cosine between them with full confidence.
425
+
426
+ Switching provider, model or width discards the old vectors and rebuilds, so
427
+ an index can never be a mixture. An incremental run over unchanged files makes
428
+ no request at all.
429
+
430
+ **What is not measured:** whether this helps *your* repository, and by how
431
+ much. No number here is from a real model — this project has no key, and a
432
+ figure produced by a stub would be fiction. The instrument ships instead:
433
+ point `benchmarks/repo_queries.py --index` at your own index and grade it. You
434
+ will probably also want a higher `search.vector_weight` than the 0.15 tuned
435
+ for a hash that carries no meaning.
436
+
437
+ ## Still not here
438
+
439
+ Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
440
+ measured JSON envelope. Note also that `search.vector_recall` scans every
441
+ unit's vector on every query, which is fine at the measured envelope and is
442
+ the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
391
443
 
392
444
  ## Development
393
445
 
@@ -6,7 +6,7 @@ build-backend = "setuptools.build_meta"
6
6
 
7
7
  [project]
8
8
  name = "rag-your-code"
9
- version = "0.7.0"
9
+ version = "0.8.0"
10
10
  description = "A local, explainable RAG index for codebases and coding agents"
11
11
  readme = "README.md"
12
12
  requires-python = ">=3.10"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: rag-your-code
3
- Version: 0.7.0
3
+ Version: 0.8.0
4
4
  Summary: A local, explainable RAG index for codebases and coding agents
5
5
  Author: rag-your-code contributors
6
6
  License-Expression: MIT
@@ -407,14 +407,66 @@ it stopped.
407
407
  Nothing authored lives under `.rag-your-code/` — that directory is what people
408
408
  delete to clear the cache.
409
409
 
410
- ## Not here yet
410
+ ## Bringing your own model
411
411
 
412
- Provider-backed embeddings, Tree-sitter parsing, and a SQLite/ANN storage layer
413
- for repositories past the measured JSON envelope. Agent-authored descriptions
414
- are deliberately the cheaper answer to the same problem provider embeddings
415
- solve: they keep the zero-dependency, offline, reproducible-index properties,
416
- and produce text a human can read and correct rather than opaque floats. See
417
- [docs/ROADMAP.md](docs/ROADMAP.md).
412
+ Everything above works with no model at all. If you would rather have real
413
+ semantics, point the index at any OpenAI-compatible embeddings endpoint —
414
+ which includes a model server on your own machine:
415
+
416
+ ```toml
417
+ # rag-your-code.toml
418
+ [embedding]
419
+ provider = "openai-compatible"
420
+ endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
421
+ model = "nomic-embed-text"
422
+ dimensions = 768 # must match the model
423
+ ```
424
+
425
+ A hosted service is the same three lines with an `https://` endpoint, plus the
426
+ name of the environment variable holding your key:
427
+
428
+ ```toml
429
+ api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
430
+ ```
431
+
432
+ **The key is never a setting.** `rag-your-code.toml` is meant to be committed
433
+ so everyone who clones can see what shaped the index; a credential is the one
434
+ value with the opposite requirement, so the file only ever names the variable
435
+ it lives in. Sending a key over plain `http://` to anything but your own
436
+ machine is refused rather than warned about.
437
+
438
+ Three things follow from turning this on, and it is worth knowing all three
439
+ before you do:
440
+
441
+ - **Your source leaves the machine**, unless the endpoint is local. That is
442
+ the whole reason the local case is written first here.
443
+ - **Similarity may now find things, not just order them.** With the local
444
+ hash a cosine shortlist is measurably noise, so it is confined to
445
+ re-ranking. A real model earns the right to add candidates the words never
446
+ reached, which is the one gap no amount of ranking closes:
447
+ `search.vector_recall` sets how many. Lexical evidence still dominates — a
448
+ unit found by similarity alone scores at most `search.vector_weight`.
449
+ - **A failure stops the build.** Falling back to the local hash would leave an
450
+ index whose vectors come from two incompatible spaces, and ranking would act
451
+ on the meaningless cosine between them with full confidence.
452
+
453
+ Switching provider, model or width discards the old vectors and rebuilds, so
454
+ an index can never be a mixture. An incremental run over unchanged files makes
455
+ no request at all.
456
+
457
+ **What is not measured:** whether this helps *your* repository, and by how
458
+ much. No number here is from a real model — this project has no key, and a
459
+ figure produced by a stub would be fiction. The instrument ships instead:
460
+ point `benchmarks/repo_queries.py --index` at your own index and grade it. You
461
+ will probably also want a higher `search.vector_weight` than the 0.15 tuned
462
+ for a hash that carries no meaning.
463
+
464
+ ## Still not here
465
+
466
+ Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
467
+ measured JSON envelope. Note also that `search.vector_recall` scans every
468
+ unit's vector on every query, which is fine at the measured envelope and is
469
+ the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
418
470
 
419
471
  ## Development
420
472
 
@@ -19,6 +19,7 @@ src/ragyourcode/graph.py
19
19
  src/ragyourcode/indexer.py
20
20
  src/ragyourcode/models.py
21
21
  src/ragyourcode/parser.py
22
+ src/ragyourcode/providers.py
22
23
  src/ragyourcode/py.typed
23
24
  src/ragyourcode/search.py
24
25
  src/ragyourcode/workflow.py
@@ -36,6 +37,7 @@ tests/test_large_repo.py
36
37
  tests/test_metadata.py
37
38
  tests/test_multilanguage.py
38
39
  tests/test_parser_edges.py
40
+ tests/test_providers.py
39
41
  tests/test_ragyourcode.py
40
42
  tests/test_ranking.py
41
43
  tests/test_repo_queries.py
@@ -3,4 +3,4 @@
3
3
  from .models import CodeUnit, SearchResult
4
4
 
5
5
  __all__ = ["CodeUnit", "SearchResult"]
6
- __version__ = "0.7.0"
6
+ __version__ = "0.8.0"
@@ -4,7 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  from .graph import CodeGraph, graph_search
6
6
  from .models import CodeUnit, SearchResult
7
- from .search import DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
7
+ from .search import DEFAULT_VECTOR_RECALL, DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
8
8
 
9
9
 
10
10
  def _result_ids(results: list[SearchResult]) -> set[str]:
@@ -80,6 +80,7 @@ def research(
80
80
  graph: CodeGraph | None = None,
81
81
  search_index: SearchIndex | None = None,
82
82
  vector_weight: float = DEFAULT_VECTOR_WEIGHT,
83
+ vector_recall: int = DEFAULT_VECTOR_RECALL,
83
84
  max_chars: int = 12000,
84
85
  ) -> dict:
85
86
  """Run at most two deterministic retrieval steps and explain the stop.
@@ -96,7 +97,7 @@ def research(
96
97
  """
97
98
  max_steps = min(2, max(1, max_steps))
98
99
  steps: list[dict] = []
99
- initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight)
100
+ initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
100
101
  steps.append({"action": "search", "query": query, "results": _trace(initial)})
101
102
  if not initial:
102
103
  return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
@@ -105,7 +106,7 @@ def research(
105
106
  kept = initial[:limit]
106
107
  return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
107
108
 
108
- expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight)
109
+ expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
109
110
  steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
110
111
  merged = {result.unit.id: result for result in initial}
111
112
  for result in expanded:
@@ -15,10 +15,11 @@ from .agentic import DEFAULT_DOMINANCE, research
15
15
  from .config import BY_PATH, SETTINGS, Config, ConfigError
16
16
  from .descriptions import index_descriptions_fingerprint
17
17
  from .document import plan as plan_documentation, render_patch, summarise as summarise_documentation
18
- from .embeddings import embed, embedding_metadata
18
+ from .embeddings import embedder
19
19
  from .graph import build_graph, graph_from_dict, graph_search
20
20
  from .indexer import StaleMonitor, build_fingerprint, build_units, fingerprint, index_build_fingerprint, read_index, snapshot_repository, write_index
21
21
  from .models import SearchResult
22
+ from .providers import ProviderError
22
23
  from .search import build_search_index, context, search, within_budget
23
24
  from .workflow import apply_descriptions, bootstrap, describe_batch, store_descriptions
24
25
 
@@ -48,6 +49,10 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
48
49
  units still have none.
49
50
  """
50
51
  cfg = cfg if cfg is not None else config_module.load(root)
52
+ # Built once and handed to every stage of this run, so the vectors stored,
53
+ # the metadata published and the queries later asked all come from the
54
+ # same scheme by construction rather than by three call sites agreeing.
55
+ embed_with = embedder(cfg)
51
56
  previous_payload: dict = {}
52
57
  previous_units = []
53
58
  if output.exists() and not full:
@@ -55,7 +60,7 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
55
60
  previous_payload, previous_units = read_index(output)
56
61
  except (OSError, TypeError, ValueError, json.JSONDecodeError):
57
62
  previous_payload, previous_units = {}, []
58
- if previous_payload.get("embedding") != embedding_metadata(cfg["embedding.dimensions"]):
63
+ if previous_payload.get("embedding") != embed_with.metadata:
59
64
  for unit in previous_units:
60
65
  unit.vector = []
61
66
  if compact is None:
@@ -83,9 +88,10 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
83
88
  cfg=cfg,
84
89
  previous_build=None if inputs_changed else previous_build,
85
90
  descriptions=store,
91
+ embed_with=embed_with,
86
92
  )
87
93
  graph = build_graph(units)
88
- write_index(output, root, units, graph.to_dict(), compact=compact, diagnostics=diagnostics, snapshot=snapshot, cfg=cfg, descriptions_fingerprint=store.fingerprint)
94
+ write_index(output, root, units, graph.to_dict(), compact=compact, diagnostics=diagnostics, snapshot=snapshot, cfg=cfg, descriptions_fingerprint=store.fingerprint, embed_with=embed_with)
89
95
  groups = store.classify(units)
90
96
  return {
91
97
  "indexed_units": len(units),
@@ -175,11 +181,12 @@ def _cmd_search(args: argparse.Namespace) -> int:
175
181
  limit = args.limit if args.limit is not None else cfg["search.limit"]
176
182
  max_chars = args.max_chars if args.max_chars is not None else cfg["search.max_chars"]
177
183
  weight = cfg["search.vector_weight"]
178
- search_index = build_search_index(units)
184
+ search_index = build_search_index(units, embedder(cfg))
185
+ recall = cfg["search.vector_recall"]
179
186
  results = (
180
- graph_search(units, args.query, limit, args.hops, graph, search_index, vector_weight=weight)
187
+ graph_search(units, args.query, limit, args.hops, graph, search_index, vector_weight=weight, vector_recall=recall)
181
188
  if args.graph
182
- else search(units, args.query, limit, search_index=search_index, vector_weight=weight)
189
+ else search(units, args.query, limit, search_index=search_index, vector_weight=weight, vector_recall=recall)
183
190
  )
184
191
  if args.json:
185
192
  # Results are navigation and cost almost nothing, so every one that was
@@ -390,11 +397,12 @@ def _request_int(request: dict, key: str, default: int, minimum: int, maximum: i
390
397
  def _cmd_agent(args: argparse.Namespace) -> int:
391
398
  """Serve one JSON request per line, suitable for a plugin subprocess."""
392
399
  payload, units, graph, cfg, store = _load(args)
393
- search_index = build_search_index(units)
400
+ search_index = build_search_index(units, embedder(cfg))
394
401
  root = Path(args.root).resolve()
395
402
  weight = cfg["search.vector_weight"]
396
403
  default_limit = cfg["search.limit"]
397
404
  default_chars = cfg["search.max_chars"]
405
+ recall = cfg["search.vector_recall"]
398
406
  stale_monitor = StaleMonitor(root, payload, assume_checked=True, cfg=cfg, descriptions_fingerprint=store.fingerprint)
399
407
  # Descriptions stored this session reach the live units immediately but not
400
408
  # the published index, which is a different thing from the index being
@@ -420,9 +428,9 @@ def _cmd_agent(args: argparse.Namespace) -> int:
420
428
  hops = _request_int(request, "hops", 1, 0, 3)
421
429
  use_graph = bool(request.get("graph", False))
422
430
  results = (
423
- graph_search(units, query, limit, hops, graph, search_index, vector_weight=weight)
431
+ graph_search(units, query, limit, hops, graph, search_index, vector_weight=weight, vector_recall=recall)
424
432
  if use_graph
425
- else search(units, query, limit, search_index=search_index, vector_weight=weight)
433
+ else search(units, query, limit, search_index=search_index, vector_weight=weight, vector_recall=recall)
426
434
  )
427
435
  budget = _request_int(request, "max_chars", default_chars, 0, 100000)
428
436
  shown = within_budget(results, budget)
@@ -438,6 +446,7 @@ def _cmd_agent(args: argparse.Namespace) -> int:
438
446
  graph,
439
447
  search_index,
440
448
  vector_weight=weight,
449
+ vector_recall=recall,
441
450
  max_chars=_request_int(request, "max_chars", default_chars, 0, 100000),
442
451
  )
443
452
  response["stale"] = payload.get("stale", True)
@@ -479,13 +488,13 @@ def _cmd_agent(args: argparse.Namespace) -> int:
479
488
  # waiting for a refresh the agent has no reason to expect.
480
489
  response["applied"] = apply_descriptions(units, store, cfg)
481
490
  if response["applied"]:
482
- search_index = build_search_index(units)
491
+ search_index = build_search_index(units, embedder(cfg))
483
492
  index_behind = index_behind or response["reindex_required"]
484
493
  elif action == "refresh":
485
494
  output = Path(args.index) if args.index else _default_index(root)
486
495
  response = _refresh_index(root, output, cfg=cfg)
487
496
  payload, units, graph, cfg, store = _load(args)
488
- search_index = build_search_index(units)
497
+ search_index = build_search_index(units, embedder(cfg))
489
498
  stale_monitor = StaleMonitor(root, payload, assume_checked=True, cfg=cfg, descriptions_fingerprint=store.fingerprint)
490
499
  index_behind = False
491
500
  elif action == "stats":
@@ -609,6 +618,12 @@ def main(argv: list[str] | None = None) -> int:
609
618
  args = build_parser().parse_args(argv)
610
619
  try:
611
620
  return args.func(args)
621
+ except ProviderError as exc:
622
+ # A provider that cannot be reached or cannot be trusted stops the run
623
+ # rather than falling back: an index whose vectors come from two
624
+ # spaces would rank confidently on a number that means nothing.
625
+ print(f"error: embedding provider: {exc}", file=sys.stderr)
626
+ return 2
612
627
  except ConfigError as exc:
613
628
  # Surfaced separately from the generic handler because the fix is
614
629
  # always in one named file: say which, so the message is actionable.
@@ -117,7 +117,74 @@ SETTINGS: tuple[Setting, ...] = (
117
117
  affects_build=True,
118
118
  minimum=32,
119
119
  maximum=4096,
120
- help="feature-hash vector width; changing it invalidates every vector",
120
+ help="vector width; changing it invalidates every vector",
121
+ ),
122
+ # The default keeps the whole pipeline offline and dependency-free. The
123
+ # other value sends each unit's text to an OpenAI-compatible embeddings
124
+ # endpoint, which may be a vendor or a model server on localhost -- one
125
+ # request shape covers both, and the local one keeps source on the machine.
126
+ Setting(
127
+ "embedding.provider",
128
+ "str",
129
+ "signed-feature-hash",
130
+ affects_build=True,
131
+ members=frozenset({"signed-feature-hash", "openai-compatible"}),
132
+ help="who computes vectors; the default never opens a socket",
133
+ ),
134
+ Setting(
135
+ "embedding.endpoint",
136
+ "str",
137
+ "",
138
+ affects_build=True,
139
+ help="OpenAI-compatible embeddings URL, e.g. http://localhost:11434/v1/embeddings",
140
+ ),
141
+ Setting(
142
+ "embedding.model",
143
+ "str",
144
+ "",
145
+ affects_build=True,
146
+ help="model name the endpoint expects; part of what an index records",
147
+ ),
148
+ # The NAME of an environment variable, never a key. Every other setting
149
+ # here is meant to be committed so everyone who clones sees what shaped the
150
+ # index; a credential is the one value with the opposite requirement, so it
151
+ # is the one value this file only points at. `config list` prints whether
152
+ # the variable is set, never what it holds.
153
+ Setting(
154
+ "embedding.api_key_env",
155
+ "str",
156
+ "RAG_YOUR_CODE_API_KEY",
157
+ help="environment variable holding the endpoint's key; empty means no auth header",
158
+ ),
159
+ Setting(
160
+ "embedding.batch",
161
+ "int",
162
+ 64,
163
+ minimum=1,
164
+ maximum=512,
165
+ help="units per embeddings request; one request per unit is unusable at scale",
166
+ ),
167
+ Setting("embedding.timeout", "int", 60, minimum=1, maximum=600, help="seconds to wait for one embeddings request"),
168
+ Setting(
169
+ "embedding.retries",
170
+ "int",
171
+ 3,
172
+ minimum=0,
173
+ maximum=10,
174
+ help="attempts per request before the build aborts rather than mixing schemes",
175
+ ),
176
+ # Only consulted when the vectors carry real semantics. Under the feature
177
+ # hash a cosine shortlist is noise, and letting it add candidates would
178
+ # dilute a ranking that measured better without it; with a trained model
179
+ # it is the one thing that can make a unit retrievable that shares no word
180
+ # with the query.
181
+ Setting(
182
+ "search.vector_recall",
183
+ "int",
184
+ 50,
185
+ minimum=0,
186
+ maximum=500,
187
+ help="units a semantic provider may add to the candidate set by similarity alone",
121
188
  ),
122
189
  Setting(
123
190
  "search.vector_weight",
@@ -186,6 +253,15 @@ def _coerce(setting: Setting, value: Any) -> Any:
186
253
  f"Adding a language means adding a rule table entry in parser.py."
187
254
  )
188
255
  return items
256
+ if setting.kind == "str":
257
+ if not isinstance(value, str):
258
+ raise ConfigError(f"{setting.path} must be a string")
259
+ text = value.strip()
260
+ if setting.members is not None and text not in setting.members:
261
+ raise ConfigError(
262
+ f"{setting.path}: {text!r} is not one of {', '.join(sorted(setting.members))}"
263
+ )
264
+ return text
189
265
  if setting.kind == "int":
190
266
  if isinstance(value, bool) or not isinstance(value, int):
191
267
  raise ConfigError(f"{setting.path} must be an integer")