ws-rag-26 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ws_rag_26-0.1.0/.gitignore +29 -0
- ws_rag_26-0.1.0/CHANGELOG.md +144 -0
- ws_rag_26-0.1.0/LICENSE +21 -0
- ws_rag_26-0.1.0/PKG-INFO +437 -0
- ws_rag_26-0.1.0/README.md +366 -0
- ws_rag_26-0.1.0/docker-compose.yml +72 -0
- ws_rag_26-0.1.0/evals/README.md +73 -0
- ws_rag_26-0.1.0/evals/dataset.py +260 -0
- ws_rag_26-0.1.0/evals/run.py +172 -0
- ws_rag_26-0.1.0/pyproject.toml +187 -0
- ws_rag_26-0.1.0/src/wsrag/__init__.py +139 -0
- ws_rag_26-0.1.0/src/wsrag/cli.py +482 -0
- ws_rag_26-0.1.0/src/wsrag/config.py +397 -0
- ws_rag_26-0.1.0/src/wsrag/core.py +978 -0
- ws_rag_26-0.1.0/src/wsrag/embeddings/__init__.py +5 -0
- ws_rag_26-0.1.0/src/wsrag/embeddings/factory.py +207 -0
- ws_rag_26-0.1.0/src/wsrag/exceptions.py +62 -0
- ws_rag_26-0.1.0/src/wsrag/llm/__init__.py +5 -0
- ws_rag_26-0.1.0/src/wsrag/llm/factory.py +146 -0
- ws_rag_26-0.1.0/src/wsrag/loaders/__init__.py +111 -0
- ws_rag_26-0.1.0/src/wsrag/loaders/base.py +83 -0
- ws_rag_26-0.1.0/src/wsrag/loaders/markitdown_loader.py +62 -0
- ws_rag_26-0.1.0/src/wsrag/loaders/unstructured_loader.py +163 -0
- ws_rag_26-0.1.0/src/wsrag/mcp/__init__.py +12 -0
- ws_rag_26-0.1.0/src/wsrag/mcp/__main__.py +42 -0
- ws_rag_26-0.1.0/src/wsrag/mcp/server.py +357 -0
- ws_rag_26-0.1.0/src/wsrag/metadata/__init__.py +26 -0
- ws_rag_26-0.1.0/src/wsrag/metadata/extractor.py +188 -0
- ws_rag_26-0.1.0/src/wsrag/metadata/schema.py +192 -0
- ws_rag_26-0.1.0/src/wsrag/processing/__init__.py +26 -0
- ws_rag_26-0.1.0/src/wsrag/processing/contextual.py +111 -0
- ws_rag_26-0.1.0/src/wsrag/processing/hashing.py +56 -0
- ws_rag_26-0.1.0/src/wsrag/processing/profiles.py +69 -0
- ws_rag_26-0.1.0/src/wsrag/processing/splitter.py +123 -0
- ws_rag_26-0.1.0/src/wsrag/py.typed +0 -0
- ws_rag_26-0.1.0/src/wsrag/rerank/__init__.py +11 -0
- ws_rag_26-0.1.0/src/wsrag/rerank/factory.py +121 -0
- ws_rag_26-0.1.0/src/wsrag/retrieval/__init__.py +26 -0
- ws_rag_26-0.1.0/src/wsrag/retrieval/filters.py +257 -0
- ws_rag_26-0.1.0/src/wsrag/retrieval/fusion.py +87 -0
- ws_rag_26-0.1.0/src/wsrag/retrieval/query.py +96 -0
- ws_rag_26-0.1.0/src/wsrag/retrieval/strategies.py +154 -0
- ws_rag_26-0.1.0/src/wsrag/skill_template/SKILL.md +108 -0
- ws_rag_26-0.1.0/src/wsrag/skill_template/__init__.py +15 -0
- ws_rag_26-0.1.0/src/wsrag/skill_template/commands/ingest-document.md +16 -0
- ws_rag_26-0.1.0/src/wsrag/skill_template/commands/rag-search.md +18 -0
- ws_rag_26-0.1.0/src/wsrag/skill_template/references/field-reference.md +91 -0
- ws_rag_26-0.1.0/src/wsrag/types.py +91 -0
- ws_rag_26-0.1.0/src/wsrag/utils/__init__.py +12 -0
- ws_rag_26-0.1.0/src/wsrag/utils/logging.py +86 -0
- ws_rag_26-0.1.0/src/wsrag/utils/results.py +43 -0
- ws_rag_26-0.1.0/src/wsrag/vectorstores/__init__.py +6 -0
- ws_rag_26-0.1.0/src/wsrag/vectorstores/base.py +74 -0
- ws_rag_26-0.1.0/src/wsrag/vectorstores/qdrant_store.py +614 -0
- ws_rag_26-0.1.0/tests/__init__.py +0 -0
- ws_rag_26-0.1.0/tests/conftest.py +100 -0
- ws_rag_26-0.1.0/tests/test_cli.py +418 -0
- ws_rag_26-0.1.0/tests/test_config.py +214 -0
- ws_rag_26-0.1.0/tests/test_contextual.py +148 -0
- ws_rag_26-0.1.0/tests/test_core.py +1263 -0
- ws_rag_26-0.1.0/tests/test_core_integration.py +404 -0
- ws_rag_26-0.1.0/tests/test_hashing.py +55 -0
- ws_rag_26-0.1.0/tests/test_install.py +234 -0
- ws_rag_26-0.1.0/tests/test_loaders.py +173 -0
- ws_rag_26-0.1.0/tests/test_logging.py +89 -0
- ws_rag_26-0.1.0/tests/test_mcp_server.py +390 -0
- ws_rag_26-0.1.0/tests/test_metadata.py +578 -0
- ws_rag_26-0.1.0/tests/test_packaging.py +193 -0
- ws_rag_26-0.1.0/tests/test_processing.py +77 -0
- ws_rag_26-0.1.0/tests/test_provider.py +223 -0
- ws_rag_26-0.1.0/tests/test_query.py +117 -0
- ws_rag_26-0.1.0/tests/test_rerank.py +249 -0
- ws_rag_26-0.1.0/tests/test_results.py +85 -0
- ws_rag_26-0.1.0/tests/test_retrieval.py +315 -0
- ws_rag_26-0.1.0/tests/test_strategies.py +88 -0
- ws_rag_26-0.1.0/tests/test_vectorstores.py +825 -0
- ws_rag_26-0.1.0/tests/test_vectorstores_integration.py +354 -0
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
.venv/
|
|
6
|
+
venv/
|
|
7
|
+
build/
|
|
8
|
+
dist/
|
|
9
|
+
rag-core
|
|
10
|
+
|
|
11
|
+
# Tooling caches
|
|
12
|
+
.pytest_cache/
|
|
13
|
+
.ruff_cache/
|
|
14
|
+
.mypy_cache/
|
|
15
|
+
.graphify-out/
|
|
16
|
+
.coverage
|
|
17
|
+
htmlcov/
|
|
18
|
+
|
|
19
|
+
# Secrets
|
|
20
|
+
.env
|
|
21
|
+
.env.local
|
|
22
|
+
|
|
23
|
+
# Embedded Qdrant storage — never commit a vector database
|
|
24
|
+
qdrant_data/
|
|
25
|
+
*.qdrant/
|
|
26
|
+
|
|
27
|
+
# Credentials — twine reads ~/.pypirc, never keep a token inside the repo
|
|
28
|
+
.pypirc
|
|
29
|
+
*.pypirc
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and this project
|
|
5
|
+
adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
6
|
+
|
|
7
|
+
## [0.1.0] — unreleased
|
|
8
|
+
|
|
9
|
+
First release. Composable RAG primitives on Qdrant, reachable three ways: as a
|
|
10
|
+
Python library, as the `wsrag` command, and as an MCP server. The design bet is
|
|
11
|
+
that no service has to be running first — Qdrant runs embedded, MarkItDown
|
|
12
|
+
parses without a container, and every provider is an optional extra.
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
|
|
16
|
+
#### Library
|
|
17
|
+
|
|
18
|
+
- `WsRag`, the pipeline: `ingest`, `ingest_bytes`, `ingest_directory`, `search`,
|
|
19
|
+
`stats`, `list_domains`, `field_values`, `extract_filters`,
|
|
20
|
+
`get_filter_context`, `delete_by_filter`, `delete_by_file_hash`, `reset`.
|
|
21
|
+
Usable as a context manager, which matters in embedded mode — the storage
|
|
22
|
+
directory stays locked until `close()` runs.
|
|
23
|
+
- Every component `WsRag` uses is also exported from `wsrag`, so a caller who
|
|
24
|
+
needs a different arrangement can assemble one: `QdrantStore`, `get_embedding`,
|
|
25
|
+
`get_llm`, `build_loader`, `run_strategy`, `rrf_fuse`, `split_text`,
|
|
26
|
+
`MetadataExtractor` and the rest — 48 names in all.
|
|
27
|
+
- **Two-level deduplication.** An unchanged file is recognised by content hash
|
|
28
|
+
and skipped before it is parsed, chunked or embedded. A document that changed
|
|
29
|
+
in one paragraph re-embeds that paragraph, not the file. Chunks that vanished
|
|
30
|
+
from a new version are deleted, so edited-away text stops being answerable.
|
|
31
|
+
- **Layered configuration**: built-in defaults < environment < YAML < keyword
|
|
32
|
+
arguments. Environment alone is sufficient, which is what makes an MCP
|
|
33
|
+
deployment possible — a host can inject nothing else.
|
|
34
|
+
- Retrieval strategies `basic`, `hybrid`, `multi-query-vector` and
|
|
35
|
+
`multi-query-hybrid`, the multi-query pair fusing rephrasings with Reciprocal
|
|
36
|
+
Rank Fusion. Without an LLM configured they degrade to a single query rather
|
|
37
|
+
than failing.
|
|
38
|
+
- Optional Cohere reranking, off by default. Runtime failures — a rate limit, a
|
|
39
|
+
dropped connection, an uninstalled SDK — degrade to truncation rather than
|
|
40
|
+
turning a working search into an error.
|
|
41
|
+
- Provider extras: `openai`, `anthropic`, `google`, `ollama`, `groq`,
|
|
42
|
+
`openrouter`, `huggingface`, `fastembed`, `cohere`, `unstructured`, `hybrid`,
|
|
43
|
+
`mcp`, and `all`. Importing `wsrag` touches none of them; every provider SDK
|
|
44
|
+
is imported inside the function that needs it.
|
|
45
|
+
- **Contextual retrieval**, off by default. With `contextual.enabled`, each
|
|
46
|
+
chunk is embedded together with a short LLM-written preamble describing where
|
|
47
|
+
it sits in its document, so the vector is of a self-contained passage rather
|
|
48
|
+
than a fragment. It costs one LLM call per *chunk* at ingestion, which is why
|
|
49
|
+
it is opt-in; `evals/` exists to decide whether it earns that on your corpus.
|
|
50
|
+
Chunk identity still comes from the original text, so turning it on does not
|
|
51
|
+
by itself re-embed anything — pass `force=True` after changing it.
|
|
52
|
+
- `py.typed`, so the annotations survive installation.
|
|
53
|
+
|
|
54
|
+
#### Command line
|
|
55
|
+
|
|
56
|
+
- `wsrag` console script: `search`, `ingest`, `list-domains`, `field-values`,
|
|
57
|
+
`stats`, `install`. Every run prints exactly one JSON object to stdout and
|
|
58
|
+
exits non-zero on failure; logs stay on stderr. Both signals are emitted
|
|
59
|
+
because callers check different ones — a shell reads `$?`, a model parses the
|
|
60
|
+
JSON.
|
|
61
|
+
- `--filter KEY=VALUE`, repeatable, values read as JSON scalars so a field's
|
|
62
|
+
type reaches Qdrant intact. Repeating a key ORs its values.
|
|
63
|
+
- stdout is forced to UTF-8, so retrieved CJK text does not crash a redirected
|
|
64
|
+
command on Windows.
|
|
65
|
+
|
|
66
|
+
#### MCP server
|
|
67
|
+
|
|
68
|
+
- `wsrag-mcp` console script and six tools: `list_domains`, `get_field_values`,
|
|
69
|
+
`search`, `ingest_document`, `ingest_text`, `get_stats`. Thin and
|
|
70
|
+
deterministic on purpose — routing, query reformulation and synthesis stay
|
|
71
|
+
with the host model rather than being wrapped in a tool an agent calls.
|
|
72
|
+
- Tools return errors as results rather than raising, so a failure is something
|
|
73
|
+
the model can act on instead of a transport fault it can only report.
|
|
74
|
+
- A misconfigured server still starts and reports the problem through the first
|
|
75
|
+
tool call, naming the missing variable.
|
|
76
|
+
- Requires the `mcp` extra: someone using the Python API should not have to
|
|
77
|
+
install an MCP SDK they will never import.
|
|
78
|
+
|
|
79
|
+
#### Claude Code
|
|
80
|
+
|
|
81
|
+
- `wsrag install [--skill] [--commands] [--user] [--force]` writes
|
|
82
|
+
`.claude/skills/wsrag-search/` and `.claude/commands/` into a project (the
|
|
83
|
+
default) or a home directory. Needs no optional extra and no configured
|
|
84
|
+
Qdrant, so a bare install is enough to bootstrap.
|
|
85
|
+
- The skill carries the working rules the MCP tool descriptions have no room
|
|
86
|
+
for — call `list_domains` before filtering, never guess a domain, what to try
|
|
87
|
+
when a search returns nothing, why `score` is `null` and must not be reported
|
|
88
|
+
as confidence. Its body loads on trigger and `references/field-reference.md`
|
|
89
|
+
only when the model reaches for it.
|
|
90
|
+
- `/ingest-document <path>` and `/rag-search <question>` as deterministic manual
|
|
91
|
+
entry points.
|
|
92
|
+
|
|
93
|
+
### Fixed
|
|
94
|
+
|
|
95
|
+
Found by running the three surfaces against live services rather than fakes —
|
|
96
|
+
none of these were reachable from the test suite.
|
|
97
|
+
|
|
98
|
+
- `force` reached only the file-hash level of deduplication. The chunk-hash
|
|
99
|
+
filter then removed every document anyway, so a forced run loaded, split, paid
|
|
100
|
+
for an LLM metadata extraction and stored nothing — defeating the two cases
|
|
101
|
+
that call for it, a changed metadata schema and a first run where the LLM was
|
|
102
|
+
unreachable.
|
|
103
|
+
- `extract_images` defaulted to on, sending every unstructured call down the
|
|
104
|
+
base64 image-extraction path. Measured on one markdown file: 3m13s with it,
|
|
105
|
+
0.06s without. The load it produced also timed out unrelated Qdrant calls.
|
|
106
|
+
- `unstructured_strategy` and `extract_images` existed only as inline fallbacks
|
|
107
|
+
inside `build_loader`, absent from `DEFAULTS` and unreachable by env var.
|
|
108
|
+
- `setup_logging` left propagation on, so anything installing a root handler —
|
|
109
|
+
constructing an MCP server does — printed every record twice.
|
|
110
|
+
- The test suite read the developer's own `.env`: `load_dotenv()` restored the
|
|
111
|
+
variables that the isolation fixture had just deleted, process-wide.
|
|
112
|
+
|
|
113
|
+
### Notes for users
|
|
114
|
+
|
|
115
|
+
Things worth knowing before the first run, because each one is silent rather
|
|
116
|
+
than loud:
|
|
117
|
+
|
|
118
|
+
- **Embedded Qdrant holds an exclusive lock** on its directory. The MCP server
|
|
119
|
+
and the `wsrag` CLI cannot both use one `QDRANT_PATH`; whichever opens it
|
|
120
|
+
second fails. Use `QDRANT_URL` to run them side by side, and for any
|
|
121
|
+
multi-worker deployment.
|
|
122
|
+
- **A metadata filter is not a security boundary.** Anyone who can call `search`
|
|
123
|
+
can pass any filter, so every domain in a collection is readable by every
|
|
124
|
+
caller. Isolate sensitive material in a separate collection.
|
|
125
|
+
- **A filter value that does not exist matches nothing** and reports zero
|
|
126
|
+
results rather than an error — indistinguishable from an empty collection.
|
|
127
|
+
This is why `list_domains` comes first.
|
|
128
|
+
- **`vectorstore.distance` is fixed when the collection is created** and cannot
|
|
129
|
+
be changed afterwards.
|
|
130
|
+
- **`search_type="mmr"` is dense-only, even on a hybrid store.**
|
|
131
|
+
langchain-qdrant's MMR path embeds the query and goes straight to the dense
|
|
132
|
+
vector with no retrieval-mode check, so the sparse half is silently dropped.
|
|
133
|
+
Verified by blinding the dense vector: `similarity` still ranked documents
|
|
134
|
+
correctly while `mmr` returned the same one for every query. A warning is
|
|
135
|
+
logged; dense MMR is still a real capability, it is just not hybrid.
|
|
136
|
+
- **Qdrant's client timeout defaults to 5 seconds**, which is generous for a
|
|
137
|
+
search and tight for ingestion. `vectorstore.timeout` (30s) and
|
|
138
|
+
`QDRANT_TIMEOUT` exist because a timeout mid-ingest is recorded as a failed
|
|
139
|
+
*file*, not a retried request.
|
|
140
|
+
- **Deleting is not exposed** to the CLI or the MCP server. It is irreversible
|
|
141
|
+
and the usual caller there is a model acting on instructions it may have
|
|
142
|
+
misread, so it stays on the Python API.
|
|
143
|
+
|
|
144
|
+
[0.1.0]: https://github.com/tenPro4/ws-rag-26/releases/tag/v0.1.0
|
ws_rag_26-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 tenPro4
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
ws_rag_26-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,437 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: ws-rag-26
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Composable RAG primitives on Qdrant — a library, a CLI, and an MCP server
|
|
5
|
+
Project-URL: Homepage, https://github.com/tenPro4/ws-rag-26
|
|
6
|
+
Project-URL: Repository, https://github.com/tenPro4/ws-rag-26
|
|
7
|
+
Project-URL: Changelog, https://github.com/tenPro4/ws-rag-26/blob/main/CHANGELOG.md
|
|
8
|
+
Project-URL: Issues, https://github.com/tenPro4/ws-rag-26/issues
|
|
9
|
+
Author: tenPro4
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: embeddings,hybrid-search,mcp,qdrant,rag,retrieval
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Environment :: Console
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
21
|
+
Classifier: Typing :: Typed
|
|
22
|
+
Requires-Python: >=3.13
|
|
23
|
+
Requires-Dist: langchain-qdrant>=1.1.0
|
|
24
|
+
Requires-Dist: langchain-text-splitters>=1.1.2
|
|
25
|
+
Requires-Dist: markitdown[pdf]>=0.1.7
|
|
26
|
+
Requires-Dist: pydantic>=2.13.4
|
|
27
|
+
Requires-Dist: python-dotenv>=1.2.2
|
|
28
|
+
Requires-Dist: pyyaml>=6.0.3
|
|
29
|
+
Requires-Dist: qdrant-client>=1.18.0
|
|
30
|
+
Provides-Extra: all
|
|
31
|
+
Requires-Dist: cohere>=5.0.0; extra == 'all'
|
|
32
|
+
Requires-Dist: fastembed>=0.8.0; extra == 'all'
|
|
33
|
+
Requires-Dist: langchain-anthropic>=1.0; extra == 'all'
|
|
34
|
+
Requires-Dist: langchain-community>=0.4.0; extra == 'all'
|
|
35
|
+
Requires-Dist: langchain-google-genai>=4.0; extra == 'all'
|
|
36
|
+
Requires-Dist: langchain-groq>=1.0; extra == 'all'
|
|
37
|
+
Requires-Dist: langchain-huggingface>=1.0; extra == 'all'
|
|
38
|
+
Requires-Dist: langchain-ollama>=1.0; extra == 'all'
|
|
39
|
+
Requires-Dist: langchain-openai>=1.0; extra == 'all'
|
|
40
|
+
Requires-Dist: mcp>=2.0.0; extra == 'all'
|
|
41
|
+
Requires-Dist: unstructured-client>=0.40.0; extra == 'all'
|
|
42
|
+
Requires-Dist: unstructured>=0.20.0; extra == 'all'
|
|
43
|
+
Provides-Extra: anthropic
|
|
44
|
+
Requires-Dist: langchain-anthropic>=1.0; extra == 'anthropic'
|
|
45
|
+
Provides-Extra: cohere
|
|
46
|
+
Requires-Dist: cohere>=5.0.0; extra == 'cohere'
|
|
47
|
+
Provides-Extra: fastembed
|
|
48
|
+
Requires-Dist: fastembed>=0.8.0; extra == 'fastembed'
|
|
49
|
+
Requires-Dist: langchain-community>=0.4.0; extra == 'fastembed'
|
|
50
|
+
Provides-Extra: google
|
|
51
|
+
Requires-Dist: langchain-google-genai>=4.0; extra == 'google'
|
|
52
|
+
Provides-Extra: groq
|
|
53
|
+
Requires-Dist: langchain-groq>=1.0; extra == 'groq'
|
|
54
|
+
Provides-Extra: huggingface
|
|
55
|
+
Requires-Dist: langchain-huggingface>=1.0; extra == 'huggingface'
|
|
56
|
+
Provides-Extra: hybrid
|
|
57
|
+
Requires-Dist: fastembed>=0.8.0; extra == 'hybrid'
|
|
58
|
+
Provides-Extra: mcp
|
|
59
|
+
Requires-Dist: mcp>=2.0.0; extra == 'mcp'
|
|
60
|
+
Provides-Extra: ollama
|
|
61
|
+
Requires-Dist: langchain-ollama>=1.0; extra == 'ollama'
|
|
62
|
+
Provides-Extra: openai
|
|
63
|
+
Requires-Dist: langchain-openai>=1.0; extra == 'openai'
|
|
64
|
+
Provides-Extra: openrouter
|
|
65
|
+
Requires-Dist: langchain-openrouter>=0.2.4; extra == 'openrouter'
|
|
66
|
+
Requires-Dist: openrouter<1.0.0,>=0.9.2; extra == 'openrouter'
|
|
67
|
+
Provides-Extra: unstructured
|
|
68
|
+
Requires-Dist: unstructured-client>=0.40.0; extra == 'unstructured'
|
|
69
|
+
Requires-Dist: unstructured>=0.20.0; extra == 'unstructured'
|
|
70
|
+
Description-Content-Type: text/markdown
|
|
71
|
+
|
|
72
|
+
# ws-rag-26
|
|
73
|
+
|
|
74
|
+
Composable RAG primitives on Qdrant, with three ways in: a Python library, a
|
|
75
|
+
`wsrag` command, and an MCP server. Ingest documents, search them with metadata
|
|
76
|
+
filters, and get plain dicts back.
|
|
77
|
+
|
|
78
|
+
The design bet: **no service you have to run first**. Qdrant runs embedded in
|
|
79
|
+
your process, MarkItDown parses files without a container, and everything else
|
|
80
|
+
is optional. A single `pip install` gets you from nothing to a working index.
|
|
81
|
+
|
|
82
|
+
## Install
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install ws-rag-26 # core: embedded Qdrant + MarkItDown
|
|
86
|
+
pip install "ws-rag-26[openai]" # plus an embedding/LLM provider
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
Optional extras: `openai`, `anthropic`, `google`, `ollama`, `groq`,
|
|
90
|
+
`openrouter`, `huggingface`, `hybrid` (sparse vectors via fastembed),
|
|
91
|
+
`unstructured` (title-aware chunking via an unstructured-api container), `all`.
|
|
92
|
+
|
|
93
|
+
## Configure
|
|
94
|
+
|
|
95
|
+
Three environment variables are enough:
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
QDRANT_PATH=./qdrant_data # embedded mode — no server, no Docker
|
|
99
|
+
AI_PROVIDER=openai # applies to both embeddings and the LLM
|
|
100
|
+
AI_KEY=sk-...
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`AI_PROVIDER` and `AI_KEY` are fallbacks; `AI_EMBEDDING_PROVIDER` /
|
|
104
|
+
`AI_LLM_PROVIDER` override them per role when you want a different model for
|
|
105
|
+
each. A `.env` file is loaded automatically.
|
|
106
|
+
|
|
107
|
+
Set **exactly one** of `QDRANT_PATH` and `QDRANT_URL`. Both together is refused
|
|
108
|
+
rather than resolved: the URL would win and the path — along with everything
|
|
109
|
+
already stored in it — would be silently ignored, which reads as data loss.
|
|
110
|
+
|
|
111
|
+
Anything beyond the three variables goes in YAML or in keyword arguments:
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
rag = WsRag(
|
|
115
|
+
vectorstore={"distance": "cosine"}, # or euclid / dot / manhattan
|
|
116
|
+
splitter={"chunk_size": 1200},
|
|
117
|
+
retriever={"strategy": "multi-query-hybrid", "top_k": 8},
|
|
118
|
+
)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
`distance` is fixed when the collection is created and cannot be changed
|
|
122
|
+
afterwards — pick it before your first ingestion.
|
|
123
|
+
|
|
124
|
+
Configuration merges in four layers, each winning over the one before:
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
built-in defaults < environment < YAML file < keyword arguments
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
Environment alone must be enough, because an MCP host can only inject env vars.
|
|
131
|
+
A checked-in YAML file expresses a stronger intent than the ambient environment,
|
|
132
|
+
so it wins; an argument passed in code wins over everything.
|
|
133
|
+
|
|
134
|
+
## Use
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
from wsrag import WsRag
|
|
138
|
+
|
|
139
|
+
with WsRag() as rag:
|
|
140
|
+
rag.ingest(["report.pdf", "notes.md"], domain="finance")
|
|
141
|
+
|
|
142
|
+
hits = rag.search("What was Q3 revenue?", domain="finance", top_k=5)
|
|
143
|
+
for hit in hits:
|
|
144
|
+
print(hit["metadata"]["file_name"], "→", hit["content"][:120])
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
Results are plain dicts (`content`, `metadata`, and `score` when a strategy
|
|
148
|
+
produces one), so they serialise to JSON without LangChain on the other end.
|
|
149
|
+
|
|
150
|
+
Use `with` — or call `close()`. In embedded mode Qdrant holds an **exclusive
|
|
151
|
+
lock** on its directory until the client is released, including against the next
|
|
152
|
+
run of the same script.
|
|
153
|
+
|
|
154
|
+
### Ingesting
|
|
155
|
+
|
|
156
|
+
```python
|
|
157
|
+
rag.ingest(["a.pdf"], domain="finance", sub_domain="payroll")
|
|
158
|
+
rag.ingest(["a.pdf"], force=True) # re-ingest regardless
|
|
159
|
+
rag.ingest_directory("./docs", recursive=True) # filtered by loader.extensions
|
|
160
|
+
rag.ingest_bytes(uploaded, "report.pdf") # for uploads and MCP clients
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Every call returns a JSON-serialisable summary:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
{"total": 2, "processed": 1, "skipped": 1, "failed": 0, "chunks_created": 12, "errors": []}
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
**Re-ingestion is cheap and safe.** Two levels of deduplication:
|
|
170
|
+
|
|
171
|
+
1. The **file hash** — an unchanged file is hashed, recognised, and skipped
|
|
172
|
+
before it is parsed, chunked or embedded. The read is not free; everything
|
|
173
|
+
after it is.
|
|
174
|
+
2. **Chunk hashes** — a document that changed in one paragraph re-embeds that
|
|
175
|
+
paragraph, not the whole file. Chunks that vanished from the new version are
|
|
176
|
+
deleted, so edited-away text stops being answerable.
|
|
177
|
+
|
|
178
|
+
Chunk identity keys on the document's `source`, which is why `ingest_bytes`
|
|
179
|
+
records the file name you pass rather than the temp path it writes: a random
|
|
180
|
+
path each time would make every re-upload look like a brand new document.
|
|
181
|
+
|
|
182
|
+
### Searching
|
|
183
|
+
|
|
184
|
+
```python
|
|
185
|
+
rag.search("query") # config defaults
|
|
186
|
+
rag.search("query", top_k=10, strategy="multi-query-hybrid")
|
|
187
|
+
rag.search("query", filters={"domain": ["finance", "ops"], "fiscal_year": 2024})
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
Different fields AND together; a list within one field ORs. `domain=` and
|
|
191
|
+
`sub_domain=` are shorthands that merge into `filters`.
|
|
192
|
+
|
|
193
|
+
Four strategies: `basic`, `hybrid`, `multi-query-vector`, `multi-query-hybrid`.
|
|
194
|
+
The multi-query ones ask the LLM for rephrasings and fuse the results with
|
|
195
|
+
Reciprocal Rank Fusion; without an LLM configured they degrade to a single
|
|
196
|
+
query rather than failing. Note that dense-vs-hybrid is decided when the store
|
|
197
|
+
is built (`use_sparse` plus fastembed), not per query — see the note in
|
|
198
|
+
`wsrag/retrieval/strategies.py`.
|
|
199
|
+
|
|
200
|
+
### Reranking (optional)
|
|
201
|
+
|
|
202
|
+
A cross-encoder reorders the retrieved candidates far more accurately than
|
|
203
|
+
vector similarity can. Off by default — it costs an extra API call per search:
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
pip install "ws-rag-26[cohere]"
|
|
207
|
+
RERANK_KEY=... # or COHERE_API_KEY
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
```python
|
|
211
|
+
rag = WsRag(rerank={"enabled": True})
|
|
212
|
+
rag.search("query", top_k=5) # fetches 20, returns the best 5
|
|
213
|
+
rag.search("query", top_k=5, rerank=False) # skip it for this call
|
|
214
|
+
rag.search("query", top_k=5, rerank=True) # rerank even with enabled=False
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
With reranking on, the vector search over-fetches `top_k × 4` candidates. That
|
|
218
|
+
multiplier is the whole mechanism: a reranker handed exactly `top_k` hits can
|
|
219
|
+
only permute them, so no document outside the original `top_k` could ever reach
|
|
220
|
+
the answer.
|
|
221
|
+
|
|
222
|
+
**Reranking never breaks retrieval.** A rate limit, a network failure, an
|
|
223
|
+
uninstalled SDK — each degrades to plain truncation, which is exactly what you
|
|
224
|
+
would have got without it. It is a quality layer over results that are already
|
|
225
|
+
correct, so it is never allowed to turn a working search into an error.
|
|
226
|
+
|
|
227
|
+
The one exception is `rerank=True`. Naming it explicitly is a request, not a
|
|
228
|
+
preference, so if no reranker can be built for it — no API key, an unknown
|
|
229
|
+
provider — you get a `ConfigError` saying which. Silence there would be
|
|
230
|
+
indistinguishable from success. Configuration is the line: weather is not.
|
|
231
|
+
|
|
232
|
+
### Metadata
|
|
233
|
+
|
|
234
|
+
Every chunk carries pipeline-owned system fields (`source`, `file_name`,
|
|
235
|
+
`file_hash`, `chunk_index`, `total_chunks`, …) plus domain fields you declare.
|
|
236
|
+
With an LLM configured, domain fields are extracted from the document; anything
|
|
237
|
+
you pass explicitly is never re-derived.
|
|
238
|
+
|
|
239
|
+
Declare your own schema in YAML and point `WS_RAG_METADATA_CONFIG` at it:
|
|
240
|
+
|
|
241
|
+
```yaml
|
|
242
|
+
fields:
|
|
243
|
+
- name: domain
|
|
244
|
+
type: string
|
|
245
|
+
values: [finance, production, hr]
|
|
246
|
+
- name: fiscal_year
|
|
247
|
+
type: integer
|
|
248
|
+
- name: parties
|
|
249
|
+
type: list
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
Fields are deliberately **flat**. A hierarchy is expressed as parallel fields
|
|
253
|
+
(`domain` + `sub_domain`), never as a path string — a path can only be
|
|
254
|
+
prefix-matched, whereas parallel fields filter at any level independently.
|
|
255
|
+
|
|
256
|
+
### Contextual retrieval (optional)
|
|
257
|
+
|
|
258
|
+
A chunk pulled out of a document loses what made it findable. Turning this on
|
|
259
|
+
embeds each chunk together with a short LLM-written preamble saying what it is
|
|
260
|
+
and where it sits, so the vector describes a self-contained passage:
|
|
261
|
+
|
|
262
|
+
```python
|
|
263
|
+
rag = WsRag(contextual={"enabled": True})
|
|
264
|
+
rag.ingest(["handbook.pdf"], force=True) # see below for why force
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
**It costs one LLM call per chunk**, not per document, which is why it is off by
|
|
268
|
+
default — switching it on multiplies the price of loading a corpus by its chunk
|
|
269
|
+
count. `evals/` exists to measure whether that buys you anything on your own
|
|
270
|
+
data before you pay for it.
|
|
271
|
+
|
|
272
|
+
Chunk identity is still taken from the original text, deliberately: an LLM
|
|
273
|
+
preamble is not reproducible, and hashing it would make every re-ingestion look
|
|
274
|
+
like a changed document. The consequence is that flipping this setting does not
|
|
275
|
+
re-embed anything by itself — pass `force=True` once after changing it.
|
|
276
|
+
|
|
277
|
+
### Introspection and maintenance
|
|
278
|
+
|
|
279
|
+
```python
|
|
280
|
+
rag.stats() # collection + active component configuration
|
|
281
|
+
rag.list_domains() # {"domain": [...], "sub_domain": [...]}
|
|
282
|
+
rag.field_values("domain") # unique values, for building a filter UI
|
|
283
|
+
rag.extract_filters(query) # what the LLM reads out of a query, no search
|
|
284
|
+
rag.get_filter_context(query) # the same, rendered as a prompt block
|
|
285
|
+
rag.delete_by_filter({"domain": "finance"})
|
|
286
|
+
rag.delete_by_file_hash(file_hash)
|
|
287
|
+
rag.reset() # drop and recreate — irreversible
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
`delete_by_filter` refuses an empty filter: Qdrant reads an empty condition list
|
|
291
|
+
as *match everything*, so forwarding one would wipe the collection while looking
|
|
292
|
+
like a targeted delete.
|
|
293
|
+
|
|
294
|
+
## Command line
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
wsrag ingest ./docs --recursive --domain finance
|
|
298
|
+
wsrag list-domains
|
|
299
|
+
wsrag search "Q3 revenue" --domain finance --top-k 5
|
|
300
|
+
wsrag field-values domain
|
|
301
|
+
wsrag stats
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
**stdout is JSON, always** — one object per run, `{"ok": true, ...}` or
|
|
305
|
+
`{"ok": false, "error": ...}`. Logs and tracebacks go to stderr, and a failure
|
|
306
|
+
also exits non-zero. Both signals are emitted because callers check different
|
|
307
|
+
ones: a shell script reads `$?`, a model parses the JSON.
|
|
308
|
+
|
|
309
|
+
`--filter KEY=VALUE` is repeatable, and repeating a key ORs its values. Values
|
|
310
|
+
are read as JSON scalars, which is how a field's type reaches Qdrant intact:
|
|
311
|
+
|
|
312
|
+
```bash
|
|
313
|
+
wsrag search "budget" --filter fiscal_year=2024 --filter domain=finance --filter domain=ops
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
`fiscal_year=2024` is the number 2024, `domain=finance` is the string. This is
|
|
317
|
+
not cosmetic — Qdrant's equality match compares by type, so filtering on the
|
|
318
|
+
string `"2024"` against a payload holding the number `2024` matches nothing and
|
|
319
|
+
reports it as zero results rather than as an error. When a field really does
|
|
320
|
+
hold digits as text, quote inside the value: `--filter code='"2024"'`.
|
|
321
|
+
|
|
322
|
+
`reset`, `delete-by-filter` and `delete-by-file-hash` are deliberately **not**
|
|
323
|
+
commands. The usual caller here is a model shelling out, and an irreversible
|
|
324
|
+
operation whose blast radius is the whole collection is not something to hand
|
|
325
|
+
one. They stay on the Python API.
|
|
326
|
+
|
|
327
|
+
## MCP server
|
|
328
|
+
|
|
329
|
+
```bash
|
|
330
|
+
pip install "ws-rag-26[mcp]"
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
```json
|
|
334
|
+
{
|
|
335
|
+
"mcpServers": {
|
|
336
|
+
"wsrag": {
|
|
337
|
+
"command": "wsrag-mcp",
|
|
338
|
+
"env": { "QDRANT_PATH": "/abs/path/qdrant_data", "AI_PROVIDER": "openai", "AI_KEY": "sk-..." }
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
Six tools: `list_domains`, `get_field_values`, `search`, `ingest_document`,
|
|
345
|
+
`ingest_text`, `get_stats`. They are thin and deterministic on purpose — routing,
|
|
346
|
+
query reformulation and synthesis stay with the host model, which is already an
|
|
347
|
+
LLM and already has the conversation. Wrapping an agent inside a tool that an
|
|
348
|
+
agent calls means two models negotiating through a JSON boundary: more latency,
|
|
349
|
+
more cost, and a decision the user cannot see.
|
|
350
|
+
|
|
351
|
+
Configuration comes from environment variables only, because that is all an MCP
|
|
352
|
+
host can inject. Embedded Qdrant locks its directory, so nothing else may hold
|
|
353
|
+
the same path while the server runs — including the `wsrag` CLI. Run both
|
|
354
|
+
against one `QDRANT_PATH` and whichever starts second fails; use `QDRANT_URL`
|
|
355
|
+
if you need them side by side.
|
|
356
|
+
|
|
357
|
+
### Claude Code
|
|
358
|
+
|
|
359
|
+
```bash
|
|
360
|
+
wsrag install # ./.claude/ for this project (default)
|
|
361
|
+
wsrag install --user # ~/.claude/ for every project
|
|
362
|
+
wsrag install --skill # or --commands, to install just one
|
|
363
|
+
```
|
|
364
|
+
|
|
365
|
+
Two things land, and they do different jobs:
|
|
366
|
+
|
|
367
|
+
- **`/wsrag-search`** — a skill carrying the *working rules* the tool
|
|
368
|
+
descriptions have no room for: call `list_domains` before filtering, never
|
|
369
|
+
guess a domain, what to try when a search comes back empty, why `score` is
|
|
370
|
+
`null` and must never be reported as confidence. Its body loads only when
|
|
371
|
+
triggered, and `references/field-reference.md` only when the model reaches
|
|
372
|
+
for it, so none of it costs context until it is needed.
|
|
373
|
+
- **`/ingest-document <path>`** and **`/search <question>`** — slash commands.
|
|
374
|
+
Deterministic manual entry points for when you would rather type the
|
|
375
|
+
operation than describe it.
|
|
376
|
+
|
|
377
|
+
They do not compete. The MCP tool descriptions say *what exists*, the skill says
|
|
378
|
+
*how to judge*, the commands say *do exactly this*. Give a command a
|
|
379
|
+
`description` in its frontmatter and the model can invoke it too; leave it out
|
|
380
|
+
to keep it a human-only path.
|
|
381
|
+
|
|
382
|
+
A misconfigured server still starts. The problem is reported through the first
|
|
383
|
+
tool call the model makes, naming the missing variable — a host that cannot
|
|
384
|
+
start a server shows the user far less than a tool result can.
|
|
385
|
+
|
|
386
|
+
## Embedded vs remote
|
|
387
|
+
|
|
388
|
+
| | Embedded (`QDRANT_PATH`) | Remote (`QDRANT_URL`) |
|
|
389
|
+
|---|---|---|
|
|
390
|
+
| Setup | none | a Qdrant server |
|
|
391
|
+
| Processes | exactly one at a time | many |
|
|
392
|
+
| Payload indexes | ignored (local Qdrant scans) | used |
|
|
393
|
+
| Suits | MCP server, notebook, CLI | FastAPI with >1 worker |
|
|
394
|
+
|
|
395
|
+
Multi-worker deployments must use `QDRANT_URL` — the embedded directory lock is
|
|
396
|
+
per-process and there is no way around it.
|
|
397
|
+
|
|
398
|
+
`docker-compose.yml` brings up both optional services:
|
|
399
|
+
|
|
400
|
+
```bash
|
|
401
|
+
docker compose up -d qdrant # a Qdrant server on :6333
|
|
402
|
+
docker compose up -d # plus unstructured-api on :8000
|
|
403
|
+
```
|
|
404
|
+
|
|
405
|
+
Then set `QDRANT_URL=http://localhost:6333` and **remove `QDRANT_PATH`** —
|
|
406
|
+
wsrag refuses a configuration with both rather than silently picking one,
|
|
407
|
+
because the loser takes every document stored in it out of sight.
|
|
408
|
+
|
|
409
|
+
## Security note
|
|
410
|
+
|
|
411
|
+
A metadata filter is **not** a security boundary. Anyone who can call `search`
|
|
412
|
+
can pass any filter, so every domain in a collection is readable by every
|
|
413
|
+
caller. Isolate sensitive material in a separate collection, not behind a
|
|
414
|
+
`domain` value.
|
|
415
|
+
|
|
416
|
+
## Development
|
|
417
|
+
|
|
418
|
+
```bash
|
|
419
|
+
uv sync
|
|
420
|
+
uv run pytest # unit tests
|
|
421
|
+
uv run pytest -m integration # real embedded Qdrant on a temp directory
|
|
422
|
+
uv run ruff check src tests
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
Integration tests are excluded from the default run. They spin up a real Qdrant
|
|
426
|
+
and use deterministic hash-based embeddings, so they need no network and no
|
|
427
|
+
model download.
|
|
428
|
+
|
|
429
|
+
```bash
|
|
430
|
+
uv run mypy # the suite runs clean; py.typed ships the types
|
|
431
|
+
uv build # sdist + wheel into dist/
|
|
432
|
+
```
|
|
433
|
+
|
|
434
|
+
## License
|
|
435
|
+
|
|
436
|
+
MIT — see [LICENSE](LICENSE). Changes are recorded in
|
|
437
|
+
[CHANGELOG.md](CHANGELOG.md).
|