vector-graph-rag 0.1.5__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/PKG-INFO +16 -7
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/README.md +15 -6
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/faq.md +5 -7
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/getting-started.md +9 -7
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/how-it-works.md +1 -1
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/index.md +1 -1
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/python-api.md +39 -22
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/rest-api.md +3 -3
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/pyproject.toml +1 -1
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/__init__.py +1 -1
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/rag.py +258 -78
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/milvus.py +132 -35
- vector_graph_rag-0.2.1/tests/test_rag_incremental.py +784 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/uv.lock +1 -1
- vector_graph_rag-0.1.5/tests/test_rag_incremental.py +0 -279
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/.dockerignore +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/.env.example +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/.github/workflows/docs.yml +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/.github/workflows/release.yml +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/.gitignore +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/AGENTS.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/CLAUDE.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/Dockerfile +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/LICENSE +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/api/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/api/main.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/api/schemas.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/assets/logo.png +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/assets/logo.svg +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/design-philosophy.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/evaluation.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/frontend.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/stylesheets/custom.css +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/docs/use-cases.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/README.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/2wikimultihopqa.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/2wikimultihopqa_corpus.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/hotpotqa.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/hotpotqa_corpus.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/musique.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/musique_corpus.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/2wikimultihopqa_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/hotpotqa_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/hotpotqa_train_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/musique_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/sample_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/openie_2wikimultihopqa_results_ner_gpt-3.5-turbo-1106_6119.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/openie_hotpotqa_results_ner_gpt-3.5-turbo-1106_9221.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/openie_musique_results_ner_gpt-3.5-turbo-1106_11656.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/openie_test_sample_results_ner_gpt-3.5-turbo-1106_20.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/test_sample.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/data/test_sample_corpus.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/evaluation/evaluate.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/.gitignore +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/README.md +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/eslint.config.js +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/index.html +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/package-lock.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/package.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/public/vite.svg +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/App.css +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/App.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/api/client.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/api/queries.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/assets/react.svg +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/graph/EntityNode.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/graph/GraphCanvas.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/graph/GraphLegend.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/graph/RelationEdge.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/graph/index.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/CreateGraphDialog.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/FileUploader.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/GraphSelector.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/GraphStatsPreview.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportDialog.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportProgress.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportSettings.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/import/UrlInput.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/panels/AnswerPanel.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/panels/NodeDetailPanel.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/panels/index.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/search/SearchInput.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/search/index.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/settings/SettingsDialog.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/timeline/ProcessTimeline.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/timeline/index.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/Header.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/button.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/dialog.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/input.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/label.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/progress.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/switch.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/components/ui/tabs.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/hooks/use-toast.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/index.css +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/main.tsx +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/stores/datasetStore.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/stores/graphStore.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/stores/searchStore.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/types/api.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/utils/cn.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/src/utils/graphLayout.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/tsconfig.app.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/tsconfig.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/tsconfig.node.json +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/frontend/vite.config.ts +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/mkdocs.yml +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/api/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/api/app.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/config.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/builder.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/graph.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/knowledge_graph.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/retriever.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/cache.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/extractor.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/reranker.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/chunker.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/converter.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/url_fetcher.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/models.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/google.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/huggingface.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/jina.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/local.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/mistral.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/ollama.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/onnx.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/openai.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/utils.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/voyage.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embeddings.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/__init__.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/conftest.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/test_api.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/test_embedding_providers.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/test_graph.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/test_milvus_store.py +0 -0
- {vector_graph_rag-0.1.5 → vector_graph_rag-0.2.1}/tests/test_rag_metadata_filter.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: vector-graph-rag
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A Graph RAG implementation using pure vector search with Milvus
|
|
5
5
|
Project-URL: Homepage, https://github.com/zilliztech/vector-graph-rag
|
|
6
6
|
Project-URL: Documentation, https://zilliztech.github.io/vector-graph-rag/
|
|
@@ -195,18 +195,23 @@ rag.rebuild_documents_with_triplets([
|
|
|
195
195
|
<details>
|
|
196
196
|
<summary>🔄 <b>Incremental document updates</b> — click to expand</summary>
|
|
197
197
|
|
|
198
|
-
Use `
|
|
199
|
-
|
|
198
|
+
Use `upsert_documents_by_source()` when a source file, message, or page is
|
|
199
|
+
created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
|
|
200
|
+
source object is identified by `metadata["source"]` or the explicit `source`
|
|
201
|
+
argument. The method replaces only that source's chunks and graph references.
|
|
202
|
+
Source-level writes are not transactionally atomic, but the same upsert/delete
|
|
203
|
+
operation can be retried after an interruption to converge the source back to a
|
|
204
|
+
consistent state.
|
|
200
205
|
|
|
201
206
|
```python
|
|
202
207
|
from langchain_core.documents import Document
|
|
203
208
|
|
|
204
|
-
rag.
|
|
205
|
-
document_id="sharepoint:file-123",
|
|
209
|
+
rag.upsert_documents_by_source(
|
|
206
210
|
documents=[
|
|
207
211
|
Document(
|
|
208
212
|
page_content="Einstein developed relativity at Princeton.",
|
|
209
213
|
metadata={
|
|
214
|
+
"source": "sharepoint:file-123",
|
|
210
215
|
"triplets": [
|
|
211
216
|
["Einstein", "developed", "relativity"],
|
|
212
217
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -214,13 +219,17 @@ rag.upsert_documents(
|
|
|
214
219
|
},
|
|
215
220
|
),
|
|
216
221
|
],
|
|
217
|
-
metadata={"source": "sharepoint"},
|
|
218
222
|
extract_triplets=False,
|
|
219
223
|
)
|
|
220
224
|
|
|
221
|
-
rag.
|
|
225
|
+
rag.delete_documents_by_source("sharepoint:file-123")
|
|
222
226
|
```
|
|
223
227
|
|
|
228
|
+
> **Migration note:** v0.1.5 exposed `upsert_documents(document_id=...)` and
|
|
229
|
+
> `delete_documents(document_id)`. These names were removed in v0.2.0 because
|
|
230
|
+
> `Document` means passage/chunk in this project. Use the `*_by_source()` APIs
|
|
231
|
+
> shown above.
|
|
232
|
+
|
|
224
233
|
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned
|
|
225
234
|
for removal in v1.0.0. For explicit full refreshes, use `rebuild_texts()`,
|
|
226
235
|
`rebuild_documents()`, or `rebuild_documents_with_triplets()`.
|
|
@@ -105,18 +105,23 @@ rag.rebuild_documents_with_triplets([
|
|
|
105
105
|
<details>
|
|
106
106
|
<summary>🔄 <b>Incremental document updates</b> — click to expand</summary>
|
|
107
107
|
|
|
108
|
-
Use `
|
|
109
|
-
|
|
108
|
+
Use `upsert_documents_by_source()` when a source file, message, or page is
|
|
109
|
+
created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
|
|
110
|
+
source object is identified by `metadata["source"]` or the explicit `source`
|
|
111
|
+
argument. The method replaces only that source's chunks and graph references.
|
|
112
|
+
Source-level writes are not transactionally atomic, but the same upsert/delete
|
|
113
|
+
operation can be retried after an interruption to converge the source back to a
|
|
114
|
+
consistent state.
|
|
110
115
|
|
|
111
116
|
```python
|
|
112
117
|
from langchain_core.documents import Document
|
|
113
118
|
|
|
114
|
-
rag.
|
|
115
|
-
document_id="sharepoint:file-123",
|
|
119
|
+
rag.upsert_documents_by_source(
|
|
116
120
|
documents=[
|
|
117
121
|
Document(
|
|
118
122
|
page_content="Einstein developed relativity at Princeton.",
|
|
119
123
|
metadata={
|
|
124
|
+
"source": "sharepoint:file-123",
|
|
120
125
|
"triplets": [
|
|
121
126
|
["Einstein", "developed", "relativity"],
|
|
122
127
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -124,13 +129,17 @@ rag.upsert_documents(
|
|
|
124
129
|
},
|
|
125
130
|
),
|
|
126
131
|
],
|
|
127
|
-
metadata={"source": "sharepoint"},
|
|
128
132
|
extract_triplets=False,
|
|
129
133
|
)
|
|
130
134
|
|
|
131
|
-
rag.
|
|
135
|
+
rag.delete_documents_by_source("sharepoint:file-123")
|
|
132
136
|
```
|
|
133
137
|
|
|
138
|
+
> **Migration note:** v0.1.5 exposed `upsert_documents(document_id=...)` and
|
|
139
|
+
> `delete_documents(document_id)`. These names were removed in v0.2.0 because
|
|
140
|
+
> `Document` means passage/chunk in this project. Use the `*_by_source()` APIs
|
|
141
|
+
> shown above.
|
|
142
|
+
|
|
134
143
|
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned
|
|
135
144
|
for removal in v1.0.0. For explicit full refreshes, use `rebuild_texts()`,
|
|
136
145
|
`rebuild_documents()`, or `rebuild_documents_with_triplets()`.
|
|
@@ -43,27 +43,25 @@ Frequently asked questions about Vector Graph RAG — covering when to use it, c
|
|
|
43
43
|
For very large corpora, consider using a remote Milvus instance rather than Milvus Lite for better performance and scalability.
|
|
44
44
|
|
|
45
45
|
??? note "How do I update or delete one source document?"
|
|
46
|
-
Use `
|
|
46
|
+
Use `upsert_documents_by_source()` for source create/update and `delete_documents_by_source()` for delete. A source can be a file, web page, message, or another business-level object. Its parsed chunks are passed as LangChain `Document` objects, and each chunk should carry the same stable `metadata["source"]` value.
|
|
47
47
|
|
|
48
48
|
```python
|
|
49
49
|
from langchain_core.documents import Document
|
|
50
50
|
|
|
51
|
-
rag.
|
|
52
|
-
document_id="file-123",
|
|
51
|
+
rag.upsert_documents_by_source(
|
|
53
52
|
documents=[
|
|
54
53
|
Document(
|
|
55
54
|
page_content="Updated document chunk text.",
|
|
56
|
-
metadata={"page": 1},
|
|
55
|
+
metadata={"source": "file-123", "page": 1},
|
|
57
56
|
)
|
|
58
57
|
],
|
|
59
|
-
metadata={"source": "sharepoint"},
|
|
60
58
|
extract_triplets=True,
|
|
61
59
|
)
|
|
62
60
|
|
|
63
|
-
rag.
|
|
61
|
+
rag.delete_documents_by_source("file-123")
|
|
64
62
|
```
|
|
65
63
|
|
|
66
|
-
Incremental updates replace only the chunks and graph references for that
|
|
64
|
+
Incremental updates replace only the chunks and graph references for that source value. They are not transactionally atomic, so avoid interrupting or concurrently mutating the same collection prefix during an update.
|
|
67
65
|
|
|
68
66
|
??? note "Can I use local/open-source LLMs?"
|
|
69
67
|
Yes. Vector Graph RAG uses the OpenAI-compatible API format, so any LLM that exposes an OpenAI-compatible endpoint will work. This includes local models served via [Ollama](https://ollama.com/), [vLLM](https://github.com/vllm-project/vllm), [LM Studio](https://lmstudio.ai/), or any other OpenAI-compatible server. You can configure the base URL and model name when initializing the RAG instance. Keep in mind that triplet extraction and reranking quality depend heavily on the LLM's capability — weaker models may produce incomplete or inaccurate triplets, which directly affects retrieval quality. For best results, use a model with strong instruction-following and reasoning abilities.
|
|
@@ -134,7 +134,7 @@ finance_rag = VectorGraphRAG(milvus_uri="./data.db", collection_prefix="finance"
|
|
|
134
134
|
## Adding Documents
|
|
135
135
|
|
|
136
136
|
!!! warning "Full rebuild vs incremental updates"
|
|
137
|
-
`add_texts()`, `add_documents()`, and `add_documents_with_triplets()` rebuild the full knowledge base for the current collection prefix. These legacy convenience APIs are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes. For source
|
|
137
|
+
`add_texts()`, `add_documents()`, and `add_documents_with_triplets()` rebuild the full knowledge base for the current collection prefix. These legacy convenience APIs are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes. For source create/update/delete flows, use `upsert_documents_by_source()` and `delete_documents_by_source()`.
|
|
138
138
|
|
|
139
139
|
### From Text Strings
|
|
140
140
|
|
|
@@ -184,17 +184,17 @@ rag.rebuild_documents(result.documents, extract_triplets=True)
|
|
|
184
184
|
|
|
185
185
|
### Incremental Updates
|
|
186
186
|
|
|
187
|
-
Use `
|
|
187
|
+
Use `upsert_documents_by_source()` when a source file, page, or message is created or modified. In Vector Graph RAG, a `Document` is a passage/chunk, and `metadata["source"]` identifies the source object whose chunks should be replaced.
|
|
188
188
|
|
|
189
189
|
```python
|
|
190
190
|
from langchain_core.documents import Document
|
|
191
191
|
|
|
192
|
-
rag.
|
|
193
|
-
document_id="sharepoint:file-123",
|
|
192
|
+
rag.upsert_documents_by_source(
|
|
194
193
|
documents=[
|
|
195
194
|
Document(
|
|
196
195
|
page_content="Einstein developed relativity at Princeton.",
|
|
197
196
|
metadata={
|
|
197
|
+
"source": "sharepoint:file-123",
|
|
198
198
|
"triplets": [
|
|
199
199
|
["Einstein", "developed", "relativity"],
|
|
200
200
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -202,15 +202,17 @@ rag.upsert_documents(
|
|
|
202
202
|
},
|
|
203
203
|
)
|
|
204
204
|
],
|
|
205
|
-
metadata={"source": "sharepoint"},
|
|
206
205
|
extract_triplets=False,
|
|
207
206
|
)
|
|
208
207
|
|
|
209
|
-
rag.
|
|
208
|
+
rag.delete_documents_by_source("sharepoint:file-123")
|
|
210
209
|
```
|
|
211
210
|
|
|
211
|
+
!!! warning "v0.2.0 migration"
|
|
212
|
+
`upsert_documents(document_id=...)` and `delete_documents(document_id)` were removed because `Document` means passage/chunk in this project. Use `upsert_documents_by_source()` and `delete_documents_by_source()` with a stable `metadata["source"]` value instead.
|
|
213
|
+
|
|
212
214
|
!!! note "Consistency model"
|
|
213
|
-
Incremental updates perform a
|
|
215
|
+
Incremental updates perform a source-level cascade across passages, entities, and relations. They are not transactionally atomic, and queries may observe intermediate state if a write is interrupted. If an upsert or delete fails, rerun the same operation for the same source to converge the source back to a consistent state. Avoid concurrently mutating the same collection prefix during an update.
|
|
214
216
|
|
|
215
217
|
## Querying
|
|
216
218
|
|
|
@@ -223,4 +223,4 @@ flowchart TD
|
|
|
223
223
|
Triplet extraction and reranking quality depends heavily on the underlying LLM's capability. Weaker models may produce incomplete or inaccurate triplets, which directly affects retrieval quality.
|
|
224
224
|
|
|
225
225
|
!!! warning "Graph Consistency"
|
|
226
|
-
The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g.,
|
|
226
|
+
The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g., source-level cascade updates) are not atomic, and queries may observe intermediate state if a write is interrupted. Source-level `upsert_documents_by_source()` and `delete_documents_by_source()` operations are designed to be retryable: rerun the same operation for the same source after a failure to converge the source back to a consistent state. Use `rebuild_documents()` for initial bulk indexing or full refreshes.
|
|
@@ -114,7 +114,7 @@ print(result.answer)
|
|
|
114
114
|
```
|
|
115
115
|
|
|
116
116
|
!!! note "Ingestion semantics"
|
|
117
|
-
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes, and use `
|
|
117
|
+
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes, and use `upsert_documents_by_source()` / `delete_documents_by_source()` when a source file, page, or message changes later.
|
|
118
118
|
|
|
119
119
|
!!! tip "Getting Started"
|
|
120
120
|
See the [Getting Started](getting-started.md) guide for installation and configuration options.
|
|
@@ -177,7 +177,7 @@ Legacy compatibility wrapper for rebuilding the graph from plain text strings.
|
|
|
177
177
|
|
|
178
178
|
!!! warning "Full rebuild"
|
|
179
179
|
This method delegates to `rebuild_texts()` and rebuilds the full knowledge base.
|
|
180
|
-
This legacy convenience API is planned for removal in v1.0.0. Use [`rebuild_texts`](#rebuild_texts) for full refreshes, or [`
|
|
180
|
+
This legacy convenience API is planned for removal in v1.0.0. Use [`rebuild_texts`](#rebuild_texts) for full refreshes, or [`upsert_documents_by_source`](#upsert_documents_by_source) for source-level incremental updates.
|
|
181
181
|
|
|
182
182
|
```python
|
|
183
183
|
def add_texts(
|
|
@@ -251,7 +251,7 @@ result = rag.rebuild_texts([
|
|
|
251
251
|
Legacy compatibility wrapper for rebuilding the graph from [LangChain `Document`](https://python.langchain.com/docs/modules/data_connection/document_loaders/) objects.
|
|
252
252
|
|
|
253
253
|
!!! warning "Full rebuild"
|
|
254
|
-
`add_documents()` keeps its original behavior for backward compatibility: it drops and recreates the Milvus collections for this graph before indexing the provided documents. This legacy API is planned for removal in v1.0.0. Use [`rebuild_documents`](#rebuild_documents) when you want this behavior explicitly, or [`
|
|
254
|
+
`add_documents()` keeps its original behavior for backward compatibility: it drops and recreates the Milvus collections for this graph before indexing the provided documents. This legacy API is planned for removal in v1.0.0. Use [`rebuild_documents`](#rebuild_documents) when you want this behavior explicitly, or [`upsert_documents_by_source`](#upsert_documents_by_source) for source-level incremental create/update.
|
|
255
255
|
|
|
256
256
|
```python
|
|
257
257
|
def add_documents(
|
|
@@ -311,7 +311,7 @@ Use this for initial bulk indexing, benchmark rebuilds, or when you intentionall
|
|
|
311
311
|
Legacy compatibility wrapper for rebuilding the graph from documents where triplets have already been extracted externally.
|
|
312
312
|
|
|
313
313
|
!!! warning "Full rebuild"
|
|
314
|
-
This method delegates to `rebuild_documents_with_triplets()` and rebuilds the full knowledge base. This legacy convenience API is planned for removal in v1.0.0. For pre-extracted triplets in an incremental update, put the triplets in each chunk's `metadata["triplets"]` and call [`
|
|
314
|
+
This method delegates to `rebuild_documents_with_triplets()` and rebuilds the full knowledge base. This legacy convenience API is planned for removal in v1.0.0. For pre-extracted triplets in an incremental update, put the triplets in each chunk's `metadata["triplets"]` and call [`upsert_documents_by_source`](#upsert_documents_by_source) with `extract_triplets=False`.
|
|
315
315
|
|
|
316
316
|
```python
|
|
317
317
|
def add_documents_with_triplets(
|
|
@@ -375,14 +375,15 @@ Optional `metadata` is stored on the passage and can be used by query filters.
|
|
|
375
375
|
|
|
376
376
|
---
|
|
377
377
|
|
|
378
|
-
#### `
|
|
378
|
+
#### `upsert_documents_by_source`
|
|
379
379
|
|
|
380
|
-
Incrementally create or replace one source
|
|
380
|
+
Incrementally create or replace all chunks that belong to one source. A source can be a file, page, message, SharePoint item, business record, or any other stable external object. In Vector Graph RAG, a LangChain `Document` is a passage/chunk; source ownership is represented by `metadata["source"]` by default.
|
|
381
381
|
|
|
382
382
|
```python
|
|
383
|
-
def
|
|
384
|
-
document_id: str,
|
|
383
|
+
def upsert_documents_by_source(
|
|
385
384
|
documents: List[Document],
|
|
385
|
+
source: Optional[str] = None,
|
|
386
|
+
source_field: str = "source",
|
|
386
387
|
metadata: Optional[Dict[str, Any]] = None,
|
|
387
388
|
extract_triplets: bool = True,
|
|
388
389
|
show_progress: bool = True,
|
|
@@ -391,15 +392,20 @@ def upsert_documents(
|
|
|
391
392
|
|
|
392
393
|
| Parameter | Description |
|
|
393
394
|
|---|---|
|
|
394
|
-
| `
|
|
395
|
-
| `
|
|
396
|
-
| `
|
|
395
|
+
| `documents` | Parsed chunks/passages for one source. If a chunk has `Document.id`, it is used as the passage ID; otherwise a deterministic passage ID is generated from `source` and chunk index. |
|
|
396
|
+
| `source` | Optional stable source value. If omitted, all documents must include the same `Document.metadata[source_field]` value. |
|
|
397
|
+
| `source_field` | Metadata field used to group chunks by source. Defaults to `"source"`. |
|
|
398
|
+
| `metadata` | Source-level metadata merged into every chunk. It must not conflict with `source` or the chunks' `source_field` values. |
|
|
397
399
|
| `extract_triplets` | If `True`, extract triplets from each chunk. If triplets are already in `metadata["triplets"]`, set this to `False`. |
|
|
398
400
|
| `show_progress` | Show a progress bar. |
|
|
399
401
|
|
|
400
402
|
**Returns:** [`ExtractionResult`](#extractionresult)
|
|
401
403
|
|
|
402
|
-
If
|
|
404
|
+
If the source does not exist, the method inserts a new source. If it already exists, the method deletes the previous chunks for that source, updates graph references, and inserts the new chunks without rebuilding unrelated sources.
|
|
405
|
+
|
|
406
|
+
Source-level writes are not transactionally atomic. If an upsert fails during the multi-step cascade, rerun the same `upsert_documents_by_source()` call for the same source to converge the source back to the requested state. Queries may observe intermediate state until the retry succeeds.
|
|
407
|
+
|
|
408
|
+
The method accepts exactly one source per call. If `source` is not provided and the documents contain multiple `metadata[source_field]` values, it raises `ValueError`.
|
|
403
409
|
|
|
404
410
|
```python
|
|
405
411
|
from langchain_core.documents import Document
|
|
@@ -407,40 +413,51 @@ from langchain_core.documents import Document
|
|
|
407
413
|
chunks = [
|
|
408
414
|
Document(
|
|
409
415
|
page_content="Alpha owns the blue database.",
|
|
410
|
-
metadata={
|
|
416
|
+
metadata={
|
|
417
|
+
"source": "sharepoint:file-123",
|
|
418
|
+
"triplets": [["Alpha", "owns", "blue database"]],
|
|
419
|
+
},
|
|
411
420
|
)
|
|
412
421
|
]
|
|
413
422
|
|
|
414
|
-
rag.
|
|
415
|
-
document_id="sharepoint:file-123",
|
|
423
|
+
rag.upsert_documents_by_source(
|
|
416
424
|
documents=chunks,
|
|
417
|
-
metadata={"tenant_id": "team_a"
|
|
425
|
+
metadata={"tenant_id": "team_a"},
|
|
418
426
|
extract_triplets=False,
|
|
419
427
|
)
|
|
420
428
|
```
|
|
421
429
|
|
|
422
430
|
---
|
|
423
431
|
|
|
424
|
-
#### `
|
|
432
|
+
#### `delete_documents_by_source`
|
|
425
433
|
|
|
426
|
-
Incrementally delete one source
|
|
434
|
+
Incrementally delete all chunks that belong to one source and remove graph references that only belonged to that source.
|
|
427
435
|
|
|
428
436
|
```python
|
|
429
|
-
def
|
|
437
|
+
def delete_documents_by_source(
|
|
438
|
+
source: str,
|
|
439
|
+
source_field: str = "source",
|
|
440
|
+
) -> bool
|
|
430
441
|
```
|
|
431
442
|
|
|
432
443
|
| Parameter | Description |
|
|
433
444
|
|---|---|
|
|
434
|
-
| `
|
|
445
|
+
| `source` | Stable source value previously used with `upsert_documents_by_source()`. |
|
|
446
|
+
| `source_field` | Metadata field used to group chunks by source. Defaults to `"source"`. |
|
|
435
447
|
|
|
436
448
|
**Returns:** `True` if at least one passage was deleted; otherwise `False`.
|
|
437
449
|
|
|
438
|
-
Shared entities and relations are preserved when other
|
|
450
|
+
Shared entities and relations are preserved when other sources still reference them. Orphaned relations and entities are removed.
|
|
451
|
+
|
|
452
|
+
Source-level deletes are retryable. If a delete fails after partially cleaning graph records, rerun `delete_documents_by_source()` with the same `source` and `source_field` to finish the cascade.
|
|
439
453
|
|
|
440
454
|
```python
|
|
441
|
-
deleted = rag.
|
|
455
|
+
deleted = rag.delete_documents_by_source("sharepoint:file-123")
|
|
442
456
|
```
|
|
443
457
|
|
|
458
|
+
!!! warning "v0.2.0 migration"
|
|
459
|
+
`upsert_documents(document_id=...)` and `delete_documents(document_id)` were removed because `Document` means passage/chunk in this project. These methods now raise `RuntimeError` with migration guidance. Use `upsert_documents_by_source()` and `delete_documents_by_source()` with a stable `metadata["source"]` value.
|
|
460
|
+
|
|
444
461
|
---
|
|
445
462
|
|
|
446
463
|
#### `query`
|
|
@@ -654,7 +671,7 @@ print(f"Final passages: {len(result.passages)}")
|
|
|
654
671
|
|
|
655
672
|
## ExtractionResult
|
|
656
673
|
|
|
657
|
-
A data class returned by document ingestion methods, including legacy `add_*`, `rebuild_*`, and `
|
|
674
|
+
A data class returned by document ingestion methods, including legacy `add_*`, `rebuild_*`, and `upsert_documents_by_source()`. It summarises what was ingested and extracted.
|
|
658
675
|
|
|
659
676
|
```python
|
|
660
677
|
from vector_graph_rag import ExtractionResult
|
|
@@ -161,7 +161,7 @@ curl -X DELETE http://localhost:8000/graph/my_graph
|
|
|
161
161
|
Indexes documents by rebuilding the selected graph. Optionally extracts knowledge graph triplets from the text using the configured LLM.
|
|
162
162
|
|
|
163
163
|
!!! warning "Full rebuild"
|
|
164
|
-
This endpoint uses the Python full-rebuild ingestion path. Calling it replaces the current graph contents for the selected `graph_name`.
|
|
164
|
+
This endpoint uses the Python full-rebuild ingestion path. Calling it replaces the current graph contents for the selected `graph_name`. Source-level incremental upsert/delete is currently available through the Python API via `upsert_documents_by_source()` and `delete_documents_by_source()`.
|
|
165
165
|
|
|
166
166
|
**Query Parameters**
|
|
167
167
|
|
|
@@ -230,7 +230,7 @@ curl -X POST "http://localhost:8000/add_documents?graph_name=my_graph" \
|
|
|
230
230
|
Imports documents from URLs or local file paths. Supports chunking for large documents.
|
|
231
231
|
|
|
232
232
|
!!! warning "Full rebuild"
|
|
233
|
-
Imported documents are indexed through the full-rebuild ingestion path for the target graph. Use the Python API for
|
|
233
|
+
Imported documents are indexed through the full-rebuild ingestion path for the target graph. Use the Python API for source-level incremental updates after parsing files externally.
|
|
234
234
|
|
|
235
235
|
**Request Body**
|
|
236
236
|
|
|
@@ -402,7 +402,7 @@ Retrieves a single document by its ID.
|
|
|
402
402
|
|
|
403
403
|
| Parameter | Type | Description |
|
|
404
404
|
|---|---|---|
|
|
405
|
-
| `document_id` | `string` | The document ID. |
|
|
405
|
+
| `document_id` | `string` | The passage/document ID used by the REST document CRUD endpoints. This is not the source key used by `upsert_documents_by_source()`. |
|
|
406
406
|
|
|
407
407
|
**Query Parameters**
|
|
408
408
|
|
|
@@ -18,7 +18,7 @@ from vector_graph_rag.rag import VectorGraphRAG, create_rag
|
|
|
18
18
|
from vector_graph_rag.storage.embeddings import EmbeddingModel
|
|
19
19
|
from vector_graph_rag.storage.milvus import MilvusStore
|
|
20
20
|
|
|
21
|
-
__version__ = "0.1
|
|
21
|
+
__version__ = "0.2.1"
|
|
22
22
|
|
|
23
23
|
__all__ = [
|
|
24
24
|
"Settings",
|