rag-your-code 0.7.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rag_your_code-0.7.0/src/rag_your_code.egg-info → rag_your_code-0.8.0}/PKG-INFO +60 -8
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/README.md +60 -8
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/pyproject.toml +1 -1
- {rag_your_code-0.7.0 → rag_your_code-0.8.0/src/rag_your_code.egg-info}/PKG-INFO +60 -8
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/SOURCES.txt +2 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/__init__.py +1 -1
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/agentic.py +4 -3
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/cli.py +26 -11
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/config.py +77 -1
- rag_your_code-0.8.0/src/ragyourcode/embeddings.py +194 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/graph.py +3 -2
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/indexer.py +17 -6
- rag_your_code-0.8.0/src/ragyourcode/providers.py +174 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/search.py +45 -4
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/workflow.py +12 -5
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_config.py +1 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_metadata.py +27 -0
- rag_your_code-0.8.0/tests/test_providers.py +363 -0
- rag_your_code-0.7.0/src/ragyourcode/embeddings.py +0 -82
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/LICENSE +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/setup.cfg +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/dependency_links.txt +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/entry_points.txt +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/requires.txt +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/rag_your_code.egg-info/top_level.txt +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/annotate.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/descriptions.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/document.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/models.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/parser.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/src/ragyourcode/py.typed +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_agent_protocol.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_agentic.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_descriptions.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_doc_comments.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_document.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_e2e_cli.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_golden.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_graph_incremental.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_language_fixtures.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_large_repo.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_multilanguage.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_parser_edges.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_ragyourcode.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_ranking.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_repo_queries.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_resilience.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_retrieval_correctness.py +0 -0
- {rag_your_code-0.7.0 → rag_your_code-0.8.0}/tests/test_workflow.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -407,14 +407,66 @@ it stopped.
|
|
|
407
407
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
408
408
|
delete to clear the cache.
|
|
409
409
|
|
|
410
|
-
##
|
|
410
|
+
## Bringing your own model
|
|
411
411
|
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
412
|
+
Everything above works with no model at all. If you would rather have real
|
|
413
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
414
|
+
which includes a model server on your own machine:
|
|
415
|
+
|
|
416
|
+
```toml
|
|
417
|
+
# rag-your-code.toml
|
|
418
|
+
[embedding]
|
|
419
|
+
provider = "openai-compatible"
|
|
420
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
421
|
+
model = "nomic-embed-text"
|
|
422
|
+
dimensions = 768 # must match the model
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
426
|
+
name of the environment variable holding your key:
|
|
427
|
+
|
|
428
|
+
```toml
|
|
429
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
430
|
+
```
|
|
431
|
+
|
|
432
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
433
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
434
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
435
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
436
|
+
machine is refused rather than warned about.
|
|
437
|
+
|
|
438
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
439
|
+
before you do:
|
|
440
|
+
|
|
441
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
442
|
+
the whole reason the local case is written first here.
|
|
443
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
444
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
445
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
446
|
+
reached, which is the one gap no amount of ranking closes:
|
|
447
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
448
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
449
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
450
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
451
|
+
on the meaningless cosine between them with full confidence.
|
|
452
|
+
|
|
453
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
454
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
455
|
+
no request at all.
|
|
456
|
+
|
|
457
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
458
|
+
much. No number here is from a real model — this project has no key, and a
|
|
459
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
460
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
461
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
462
|
+
for a hash that carries no meaning.
|
|
463
|
+
|
|
464
|
+
## Still not here
|
|
465
|
+
|
|
466
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
467
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
468
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
469
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
418
470
|
|
|
419
471
|
## Development
|
|
420
472
|
|
|
@@ -380,14 +380,66 @@ it stopped.
|
|
|
380
380
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
381
381
|
delete to clear the cache.
|
|
382
382
|
|
|
383
|
-
##
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
383
|
+
## Bringing your own model
|
|
384
|
+
|
|
385
|
+
Everything above works with no model at all. If you would rather have real
|
|
386
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
387
|
+
which includes a model server on your own machine:
|
|
388
|
+
|
|
389
|
+
```toml
|
|
390
|
+
# rag-your-code.toml
|
|
391
|
+
[embedding]
|
|
392
|
+
provider = "openai-compatible"
|
|
393
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
394
|
+
model = "nomic-embed-text"
|
|
395
|
+
dimensions = 768 # must match the model
|
|
396
|
+
```
|
|
397
|
+
|
|
398
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
399
|
+
name of the environment variable holding your key:
|
|
400
|
+
|
|
401
|
+
```toml
|
|
402
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
403
|
+
```
|
|
404
|
+
|
|
405
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
406
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
407
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
408
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
409
|
+
machine is refused rather than warned about.
|
|
410
|
+
|
|
411
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
412
|
+
before you do:
|
|
413
|
+
|
|
414
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
415
|
+
the whole reason the local case is written first here.
|
|
416
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
417
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
418
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
419
|
+
reached, which is the one gap no amount of ranking closes:
|
|
420
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
421
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
422
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
423
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
424
|
+
on the meaningless cosine between them with full confidence.
|
|
425
|
+
|
|
426
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
427
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
428
|
+
no request at all.
|
|
429
|
+
|
|
430
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
431
|
+
much. No number here is from a real model — this project has no key, and a
|
|
432
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
433
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
434
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
435
|
+
for a hash that carries no meaning.
|
|
436
|
+
|
|
437
|
+
## Still not here
|
|
438
|
+
|
|
439
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
440
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
441
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
442
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
391
443
|
|
|
392
444
|
## Development
|
|
393
445
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: rag-your-code
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: A local, explainable RAG index for codebases and coding agents
|
|
5
5
|
Author: rag-your-code contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -407,14 +407,66 @@ it stopped.
|
|
|
407
407
|
Nothing authored lives under `.rag-your-code/` — that directory is what people
|
|
408
408
|
delete to clear the cache.
|
|
409
409
|
|
|
410
|
-
##
|
|
410
|
+
## Bringing your own model
|
|
411
411
|
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
412
|
+
Everything above works with no model at all. If you would rather have real
|
|
413
|
+
semantics, point the index at any OpenAI-compatible embeddings endpoint —
|
|
414
|
+
which includes a model server on your own machine:
|
|
415
|
+
|
|
416
|
+
```toml
|
|
417
|
+
# rag-your-code.toml
|
|
418
|
+
[embedding]
|
|
419
|
+
provider = "openai-compatible"
|
|
420
|
+
endpoint = "http://localhost:11434/v1/embeddings" # ollama, LM Studio, vLLM…
|
|
421
|
+
model = "nomic-embed-text"
|
|
422
|
+
dimensions = 768 # must match the model
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
A hosted service is the same three lines with an `https://` endpoint, plus the
|
|
426
|
+
name of the environment variable holding your key:
|
|
427
|
+
|
|
428
|
+
```toml
|
|
429
|
+
api_key_env = "OPENAI_API_KEY" # the NAME of the variable, never the key
|
|
430
|
+
```
|
|
431
|
+
|
|
432
|
+
**The key is never a setting.** `rag-your-code.toml` is meant to be committed
|
|
433
|
+
so everyone who clones can see what shaped the index; a credential is the one
|
|
434
|
+
value with the opposite requirement, so the file only ever names the variable
|
|
435
|
+
it lives in. Sending a key over plain `http://` to anything but your own
|
|
436
|
+
machine is refused rather than warned about.
|
|
437
|
+
|
|
438
|
+
Three things follow from turning this on, and it is worth knowing all three
|
|
439
|
+
before you do:
|
|
440
|
+
|
|
441
|
+
- **Your source leaves the machine**, unless the endpoint is local. That is
|
|
442
|
+
the whole reason the local case is written first here.
|
|
443
|
+
- **Similarity may now find things, not just order them.** With the local
|
|
444
|
+
hash a cosine shortlist is measurably noise, so it is confined to
|
|
445
|
+
re-ranking. A real model earns the right to add candidates the words never
|
|
446
|
+
reached, which is the one gap no amount of ranking closes:
|
|
447
|
+
`search.vector_recall` sets how many. Lexical evidence still dominates — a
|
|
448
|
+
unit found by similarity alone scores at most `search.vector_weight`.
|
|
449
|
+
- **A failure stops the build.** Falling back to the local hash would leave an
|
|
450
|
+
index whose vectors come from two incompatible spaces, and ranking would act
|
|
451
|
+
on the meaningless cosine between them with full confidence.
|
|
452
|
+
|
|
453
|
+
Switching provider, model or width discards the old vectors and rebuilds, so
|
|
454
|
+
an index can never be a mixture. An incremental run over unchanged files makes
|
|
455
|
+
no request at all.
|
|
456
|
+
|
|
457
|
+
**What is not measured:** whether this helps *your* repository, and by how
|
|
458
|
+
much. No number here is from a real model — this project has no key, and a
|
|
459
|
+
figure produced by a stub would be fiction. The instrument ships instead:
|
|
460
|
+
point `benchmarks/repo_queries.py --index` at your own index and grade it. You
|
|
461
|
+
will probably also want a higher `search.vector_weight` than the 0.15 tuned
|
|
462
|
+
for a hash that carries no meaning.
|
|
463
|
+
|
|
464
|
+
## Still not here
|
|
465
|
+
|
|
466
|
+
Tree-sitter parsing, and a SQLite/ANN storage layer for repositories past the
|
|
467
|
+
measured JSON envelope. Note also that `search.vector_recall` scans every
|
|
468
|
+
unit's vector on every query, which is fine at the measured envelope and is
|
|
469
|
+
the thing an ANN index would replace. See [docs/ROADMAP.md](docs/ROADMAP.md).
|
|
418
470
|
|
|
419
471
|
## Development
|
|
420
472
|
|
|
@@ -19,6 +19,7 @@ src/ragyourcode/graph.py
|
|
|
19
19
|
src/ragyourcode/indexer.py
|
|
20
20
|
src/ragyourcode/models.py
|
|
21
21
|
src/ragyourcode/parser.py
|
|
22
|
+
src/ragyourcode/providers.py
|
|
22
23
|
src/ragyourcode/py.typed
|
|
23
24
|
src/ragyourcode/search.py
|
|
24
25
|
src/ragyourcode/workflow.py
|
|
@@ -36,6 +37,7 @@ tests/test_large_repo.py
|
|
|
36
37
|
tests/test_metadata.py
|
|
37
38
|
tests/test_multilanguage.py
|
|
38
39
|
tests/test_parser_edges.py
|
|
40
|
+
tests/test_providers.py
|
|
39
41
|
tests/test_ragyourcode.py
|
|
40
42
|
tests/test_ranking.py
|
|
41
43
|
tests/test_repo_queries.py
|
|
@@ -4,7 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
from .graph import CodeGraph, graph_search
|
|
6
6
|
from .models import CodeUnit, SearchResult
|
|
7
|
-
from .search import DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
|
|
7
|
+
from .search import DEFAULT_VECTOR_RECALL, DEFAULT_VECTOR_WEIGHT, SearchIndex, context, search, within_budget
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
def _result_ids(results: list[SearchResult]) -> set[str]:
|
|
@@ -80,6 +80,7 @@ def research(
|
|
|
80
80
|
graph: CodeGraph | None = None,
|
|
81
81
|
search_index: SearchIndex | None = None,
|
|
82
82
|
vector_weight: float = DEFAULT_VECTOR_WEIGHT,
|
|
83
|
+
vector_recall: int = DEFAULT_VECTOR_RECALL,
|
|
83
84
|
max_chars: int = 12000,
|
|
84
85
|
) -> dict:
|
|
85
86
|
"""Run at most two deterministic retrieval steps and explain the stop.
|
|
@@ -96,7 +97,7 @@ def research(
|
|
|
96
97
|
"""
|
|
97
98
|
max_steps = min(2, max(1, max_steps))
|
|
98
99
|
steps: list[dict] = []
|
|
99
|
-
initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight)
|
|
100
|
+
initial = search(units, query, max(limit, 1), search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
|
|
100
101
|
steps.append({"action": "search", "query": query, "results": _trace(initial)})
|
|
101
102
|
if not initial:
|
|
102
103
|
return {"query": query, "results": [], "steps": steps, "stop_reason": "no_results", "context": ""}
|
|
@@ -105,7 +106,7 @@ def research(
|
|
|
105
106
|
kept = initial[:limit]
|
|
106
107
|
return {"query": query, "results": _serialize(kept), "steps": steps, "stop_reason": "high_confidence", "context": context(within_budget(kept, max_chars), max_chars)}
|
|
107
108
|
|
|
108
|
-
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight)
|
|
109
|
+
expanded = graph_search(units, query, limit=max(limit * 2, 8), hops=hops, graph=graph, search_index=search_index, vector_weight=vector_weight, vector_recall=vector_recall)
|
|
109
110
|
steps.append({"action": "graph_expand", "hops": hops, "results": _trace(expanded[:limit])})
|
|
110
111
|
merged = {result.unit.id: result for result in initial}
|
|
111
112
|
for result in expanded:
|
|
@@ -15,10 +15,11 @@ from .agentic import DEFAULT_DOMINANCE, research
|
|
|
15
15
|
from .config import BY_PATH, SETTINGS, Config, ConfigError
|
|
16
16
|
from .descriptions import index_descriptions_fingerprint
|
|
17
17
|
from .document import plan as plan_documentation, render_patch, summarise as summarise_documentation
|
|
18
|
-
from .embeddings import
|
|
18
|
+
from .embeddings import embedder
|
|
19
19
|
from .graph import build_graph, graph_from_dict, graph_search
|
|
20
20
|
from .indexer import StaleMonitor, build_fingerprint, build_units, fingerprint, index_build_fingerprint, read_index, snapshot_repository, write_index
|
|
21
21
|
from .models import SearchResult
|
|
22
|
+
from .providers import ProviderError
|
|
22
23
|
from .search import build_search_index, context, search, within_budget
|
|
23
24
|
from .workflow import apply_descriptions, bootstrap, describe_batch, store_descriptions
|
|
24
25
|
|
|
@@ -48,6 +49,10 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
|
|
|
48
49
|
units still have none.
|
|
49
50
|
"""
|
|
50
51
|
cfg = cfg if cfg is not None else config_module.load(root)
|
|
52
|
+
# Built once and handed to every stage of this run, so the vectors stored,
|
|
53
|
+
# the metadata published and the queries later asked all come from the
|
|
54
|
+
# same scheme by construction rather than by three call sites agreeing.
|
|
55
|
+
embed_with = embedder(cfg)
|
|
51
56
|
previous_payload: dict = {}
|
|
52
57
|
previous_units = []
|
|
53
58
|
if output.exists() and not full:
|
|
@@ -55,7 +60,7 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
|
|
|
55
60
|
previous_payload, previous_units = read_index(output)
|
|
56
61
|
except (OSError, TypeError, ValueError, json.JSONDecodeError):
|
|
57
62
|
previous_payload, previous_units = {}, []
|
|
58
|
-
if previous_payload.get("embedding") !=
|
|
63
|
+
if previous_payload.get("embedding") != embed_with.metadata:
|
|
59
64
|
for unit in previous_units:
|
|
60
65
|
unit.vector = []
|
|
61
66
|
if compact is None:
|
|
@@ -83,9 +88,10 @@ def _refresh_index(root: Path, output: Path, full: bool = False, compact: bool |
|
|
|
83
88
|
cfg=cfg,
|
|
84
89
|
previous_build=None if inputs_changed else previous_build,
|
|
85
90
|
descriptions=store,
|
|
91
|
+
embed_with=embed_with,
|
|
86
92
|
)
|
|
87
93
|
graph = build_graph(units)
|
|
88
|
-
write_index(output, root, units, graph.to_dict(), compact=compact, diagnostics=diagnostics, snapshot=snapshot, cfg=cfg, descriptions_fingerprint=store.fingerprint)
|
|
94
|
+
write_index(output, root, units, graph.to_dict(), compact=compact, diagnostics=diagnostics, snapshot=snapshot, cfg=cfg, descriptions_fingerprint=store.fingerprint, embed_with=embed_with)
|
|
89
95
|
groups = store.classify(units)
|
|
90
96
|
return {
|
|
91
97
|
"indexed_units": len(units),
|
|
@@ -175,11 +181,12 @@ def _cmd_search(args: argparse.Namespace) -> int:
|
|
|
175
181
|
limit = args.limit if args.limit is not None else cfg["search.limit"]
|
|
176
182
|
max_chars = args.max_chars if args.max_chars is not None else cfg["search.max_chars"]
|
|
177
183
|
weight = cfg["search.vector_weight"]
|
|
178
|
-
search_index = build_search_index(units)
|
|
184
|
+
search_index = build_search_index(units, embedder(cfg))
|
|
185
|
+
recall = cfg["search.vector_recall"]
|
|
179
186
|
results = (
|
|
180
|
-
graph_search(units, args.query, limit, args.hops, graph, search_index, vector_weight=weight)
|
|
187
|
+
graph_search(units, args.query, limit, args.hops, graph, search_index, vector_weight=weight, vector_recall=recall)
|
|
181
188
|
if args.graph
|
|
182
|
-
else search(units, args.query, limit, search_index=search_index, vector_weight=weight)
|
|
189
|
+
else search(units, args.query, limit, search_index=search_index, vector_weight=weight, vector_recall=recall)
|
|
183
190
|
)
|
|
184
191
|
if args.json:
|
|
185
192
|
# Results are navigation and cost almost nothing, so every one that was
|
|
@@ -390,11 +397,12 @@ def _request_int(request: dict, key: str, default: int, minimum: int, maximum: i
|
|
|
390
397
|
def _cmd_agent(args: argparse.Namespace) -> int:
|
|
391
398
|
"""Serve one JSON request per line, suitable for a plugin subprocess."""
|
|
392
399
|
payload, units, graph, cfg, store = _load(args)
|
|
393
|
-
search_index = build_search_index(units)
|
|
400
|
+
search_index = build_search_index(units, embedder(cfg))
|
|
394
401
|
root = Path(args.root).resolve()
|
|
395
402
|
weight = cfg["search.vector_weight"]
|
|
396
403
|
default_limit = cfg["search.limit"]
|
|
397
404
|
default_chars = cfg["search.max_chars"]
|
|
405
|
+
recall = cfg["search.vector_recall"]
|
|
398
406
|
stale_monitor = StaleMonitor(root, payload, assume_checked=True, cfg=cfg, descriptions_fingerprint=store.fingerprint)
|
|
399
407
|
# Descriptions stored this session reach the live units immediately but not
|
|
400
408
|
# the published index, which is a different thing from the index being
|
|
@@ -420,9 +428,9 @@ def _cmd_agent(args: argparse.Namespace) -> int:
|
|
|
420
428
|
hops = _request_int(request, "hops", 1, 0, 3)
|
|
421
429
|
use_graph = bool(request.get("graph", False))
|
|
422
430
|
results = (
|
|
423
|
-
graph_search(units, query, limit, hops, graph, search_index, vector_weight=weight)
|
|
431
|
+
graph_search(units, query, limit, hops, graph, search_index, vector_weight=weight, vector_recall=recall)
|
|
424
432
|
if use_graph
|
|
425
|
-
else search(units, query, limit, search_index=search_index, vector_weight=weight)
|
|
433
|
+
else search(units, query, limit, search_index=search_index, vector_weight=weight, vector_recall=recall)
|
|
426
434
|
)
|
|
427
435
|
budget = _request_int(request, "max_chars", default_chars, 0, 100000)
|
|
428
436
|
shown = within_budget(results, budget)
|
|
@@ -438,6 +446,7 @@ def _cmd_agent(args: argparse.Namespace) -> int:
|
|
|
438
446
|
graph,
|
|
439
447
|
search_index,
|
|
440
448
|
vector_weight=weight,
|
|
449
|
+
vector_recall=recall,
|
|
441
450
|
max_chars=_request_int(request, "max_chars", default_chars, 0, 100000),
|
|
442
451
|
)
|
|
443
452
|
response["stale"] = payload.get("stale", True)
|
|
@@ -479,13 +488,13 @@ def _cmd_agent(args: argparse.Namespace) -> int:
|
|
|
479
488
|
# waiting for a refresh the agent has no reason to expect.
|
|
480
489
|
response["applied"] = apply_descriptions(units, store, cfg)
|
|
481
490
|
if response["applied"]:
|
|
482
|
-
search_index = build_search_index(units)
|
|
491
|
+
search_index = build_search_index(units, embedder(cfg))
|
|
483
492
|
index_behind = index_behind or response["reindex_required"]
|
|
484
493
|
elif action == "refresh":
|
|
485
494
|
output = Path(args.index) if args.index else _default_index(root)
|
|
486
495
|
response = _refresh_index(root, output, cfg=cfg)
|
|
487
496
|
payload, units, graph, cfg, store = _load(args)
|
|
488
|
-
search_index = build_search_index(units)
|
|
497
|
+
search_index = build_search_index(units, embedder(cfg))
|
|
489
498
|
stale_monitor = StaleMonitor(root, payload, assume_checked=True, cfg=cfg, descriptions_fingerprint=store.fingerprint)
|
|
490
499
|
index_behind = False
|
|
491
500
|
elif action == "stats":
|
|
@@ -609,6 +618,12 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
609
618
|
args = build_parser().parse_args(argv)
|
|
610
619
|
try:
|
|
611
620
|
return args.func(args)
|
|
621
|
+
except ProviderError as exc:
|
|
622
|
+
# A provider that cannot be reached or cannot be trusted stops the run
|
|
623
|
+
# rather than falling back: an index whose vectors come from two
|
|
624
|
+
# spaces would rank confidently on a number that means nothing.
|
|
625
|
+
print(f"error: embedding provider: {exc}", file=sys.stderr)
|
|
626
|
+
return 2
|
|
612
627
|
except ConfigError as exc:
|
|
613
628
|
# Surfaced separately from the generic handler because the fix is
|
|
614
629
|
# always in one named file: say which, so the message is actionable.
|
|
@@ -117,7 +117,74 @@ SETTINGS: tuple[Setting, ...] = (
|
|
|
117
117
|
affects_build=True,
|
|
118
118
|
minimum=32,
|
|
119
119
|
maximum=4096,
|
|
120
|
-
help="
|
|
120
|
+
help="vector width; changing it invalidates every vector",
|
|
121
|
+
),
|
|
122
|
+
# The default keeps the whole pipeline offline and dependency-free. The
|
|
123
|
+
# other value sends each unit's text to an OpenAI-compatible embeddings
|
|
124
|
+
# endpoint, which may be a vendor or a model server on localhost -- one
|
|
125
|
+
# request shape covers both, and the local one keeps source on the machine.
|
|
126
|
+
Setting(
|
|
127
|
+
"embedding.provider",
|
|
128
|
+
"str",
|
|
129
|
+
"signed-feature-hash",
|
|
130
|
+
affects_build=True,
|
|
131
|
+
members=frozenset({"signed-feature-hash", "openai-compatible"}),
|
|
132
|
+
help="who computes vectors; the default never opens a socket",
|
|
133
|
+
),
|
|
134
|
+
Setting(
|
|
135
|
+
"embedding.endpoint",
|
|
136
|
+
"str",
|
|
137
|
+
"",
|
|
138
|
+
affects_build=True,
|
|
139
|
+
help="OpenAI-compatible embeddings URL, e.g. http://localhost:11434/v1/embeddings",
|
|
140
|
+
),
|
|
141
|
+
Setting(
|
|
142
|
+
"embedding.model",
|
|
143
|
+
"str",
|
|
144
|
+
"",
|
|
145
|
+
affects_build=True,
|
|
146
|
+
help="model name the endpoint expects; part of what an index records",
|
|
147
|
+
),
|
|
148
|
+
# The NAME of an environment variable, never a key. Every other setting
|
|
149
|
+
# here is meant to be committed so everyone who clones sees what shaped the
|
|
150
|
+
# index; a credential is the one value with the opposite requirement, so it
|
|
151
|
+
# is the one value this file only points at. `config list` prints whether
|
|
152
|
+
# the variable is set, never what it holds.
|
|
153
|
+
Setting(
|
|
154
|
+
"embedding.api_key_env",
|
|
155
|
+
"str",
|
|
156
|
+
"RAG_YOUR_CODE_API_KEY",
|
|
157
|
+
help="environment variable holding the endpoint's key; empty means no auth header",
|
|
158
|
+
),
|
|
159
|
+
Setting(
|
|
160
|
+
"embedding.batch",
|
|
161
|
+
"int",
|
|
162
|
+
64,
|
|
163
|
+
minimum=1,
|
|
164
|
+
maximum=512,
|
|
165
|
+
help="units per embeddings request; one request per unit is unusable at scale",
|
|
166
|
+
),
|
|
167
|
+
Setting("embedding.timeout", "int", 60, minimum=1, maximum=600, help="seconds to wait for one embeddings request"),
|
|
168
|
+
Setting(
|
|
169
|
+
"embedding.retries",
|
|
170
|
+
"int",
|
|
171
|
+
3,
|
|
172
|
+
minimum=0,
|
|
173
|
+
maximum=10,
|
|
174
|
+
help="attempts per request before the build aborts rather than mixing schemes",
|
|
175
|
+
),
|
|
176
|
+
# Only consulted when the vectors carry real semantics. Under the feature
|
|
177
|
+
# hash a cosine shortlist is noise, and letting it add candidates would
|
|
178
|
+
# dilute a ranking that measured better without it; with a trained model
|
|
179
|
+
# it is the one thing that can make a unit retrievable that shares no word
|
|
180
|
+
# with the query.
|
|
181
|
+
Setting(
|
|
182
|
+
"search.vector_recall",
|
|
183
|
+
"int",
|
|
184
|
+
50,
|
|
185
|
+
minimum=0,
|
|
186
|
+
maximum=500,
|
|
187
|
+
help="units a semantic provider may add to the candidate set by similarity alone",
|
|
121
188
|
),
|
|
122
189
|
Setting(
|
|
123
190
|
"search.vector_weight",
|
|
@@ -186,6 +253,15 @@ def _coerce(setting: Setting, value: Any) -> Any:
|
|
|
186
253
|
f"Adding a language means adding a rule table entry in parser.py."
|
|
187
254
|
)
|
|
188
255
|
return items
|
|
256
|
+
if setting.kind == "str":
|
|
257
|
+
if not isinstance(value, str):
|
|
258
|
+
raise ConfigError(f"{setting.path} must be a string")
|
|
259
|
+
text = value.strip()
|
|
260
|
+
if setting.members is not None and text not in setting.members:
|
|
261
|
+
raise ConfigError(
|
|
262
|
+
f"{setting.path}: {text!r} is not one of {', '.join(sorted(setting.members))}"
|
|
263
|
+
)
|
|
264
|
+
return text
|
|
189
265
|
if setting.kind == "int":
|
|
190
266
|
if isinstance(value, bool) or not isinstance(value, int):
|
|
191
267
|
raise ConfigError(f"{setting.path} must be an integer")
|