vector-graph-rag 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/PKG-INFO +61 -3
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/README.md +46 -2
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/faq.md +1 -1
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/getting-started.md +50 -4
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/how-it-works.md +1 -1
- vector_graph_rag-0.2.2/docs/incremental-updates.md +245 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/index.md +1 -1
- vector_graph_rag-0.2.2/docs/observability.md +174 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/python-api.md +143 -3
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/rest-api.md +1 -1
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/mkdocs.yml +2 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/pyproject.toml +22 -2
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/__init__.py +4 -1
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/api/app.py +38 -1
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/graph/knowledge_graph.py +55 -23
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/graph/retriever.py +199 -139
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/llm/extractor.py +74 -47
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/llm/reranker.py +60 -40
- vector_graph_rag-0.2.2/src/vector_graph_rag/loaders/__init__.py +217 -0
- vector_graph_rag-0.2.2/src/vector_graph_rag/loaders/converter.py +125 -0
- vector_graph_rag-0.2.2/src/vector_graph_rag/loaders/docling.py +150 -0
- vector_graph_rag-0.2.2/src/vector_graph_rag/loaders/mineru.py +284 -0
- vector_graph_rag-0.2.2/src/vector_graph_rag/loaders/url_fetcher.py +179 -0
- vector_graph_rag-0.2.2/src/vector_graph_rag/observability.py +134 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/rag.py +568 -367
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embeddings.py +29 -7
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/milvus.py +487 -191
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/conftest.py +17 -4
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_api.py +71 -3
- vector_graph_rag-0.2.2/tests/test_loaders.py +430 -0
- vector_graph_rag-0.2.2/tests/test_observability.py +254 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_rag_incremental.py +344 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/uv.lock +1669 -14
- vector_graph_rag-0.2.0/src/vector_graph_rag/loaders/__init__.py +0 -167
- vector_graph_rag-0.2.0/src/vector_graph_rag/loaders/converter.py +0 -99
- vector_graph_rag-0.2.0/src/vector_graph_rag/loaders/url_fetcher.py +0 -153
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/.dockerignore +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/.env.example +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/.github/workflows/docs.yml +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/.github/workflows/release.yml +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/.gitignore +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/AGENTS.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/CLAUDE.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/Dockerfile +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/LICENSE +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/api/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/api/main.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/api/schemas.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/assets/logo.png +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/assets/logo.svg +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/design-philosophy.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/evaluation.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/frontend.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/stylesheets/custom.css +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/docs/use-cases.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/README.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/2wikimultihopqa.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/2wikimultihopqa_corpus.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/hotpotqa.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/hotpotqa_corpus.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/musique.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/musique_corpus.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/ner_cache/2wikimultihopqa_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/ner_cache/hotpotqa_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/ner_cache/hotpotqa_train_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/ner_cache/musique_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/ner_cache/sample_queries.named_entity_output.tsv +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/openie_2wikimultihopqa_results_ner_gpt-3.5-turbo-1106_6119.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/openie_hotpotqa_results_ner_gpt-3.5-turbo-1106_9221.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/openie_musique_results_ner_gpt-3.5-turbo-1106_11656.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/openie_test_sample_results_ner_gpt-3.5-turbo-1106_20.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/test_sample.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/data/test_sample_corpus.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/evaluation/evaluate.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/.gitignore +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/README.md +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/eslint.config.js +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/index.html +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/package-lock.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/package.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/public/vite.svg +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/App.css +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/App.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/api/client.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/api/queries.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/assets/react.svg +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/graph/EntityNode.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/graph/GraphCanvas.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/graph/GraphLegend.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/graph/RelationEdge.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/graph/index.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/CreateGraphDialog.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/FileUploader.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/GraphSelector.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/GraphStatsPreview.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/ImportDialog.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/ImportProgress.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/ImportSettings.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/import/UrlInput.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/panels/AnswerPanel.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/panels/NodeDetailPanel.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/panels/index.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/search/SearchInput.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/search/index.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/settings/SettingsDialog.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/timeline/ProcessTimeline.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/timeline/index.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/Header.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/button.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/dialog.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/input.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/label.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/progress.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/switch.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/components/ui/tabs.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/hooks/use-toast.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/index.css +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/main.tsx +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/stores/datasetStore.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/stores/graphStore.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/stores/searchStore.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/types/api.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/utils/cn.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/src/utils/graphLayout.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/tsconfig.app.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/tsconfig.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/tsconfig.node.json +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/frontend/vite.config.ts +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/api/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/config.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/graph/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/graph/builder.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/graph/graph.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/llm/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/llm/cache.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/loaders/chunker.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/models.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/google.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/huggingface.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/jina.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/local.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/mistral.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/ollama.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/onnx.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/openai.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/utils.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/src/vector_graph_rag/storage/embedding_providers/voyage.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/__init__.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_embedding_providers.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_graph.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_milvus_store.py +0 -0
- {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.2}/tests/test_rag_metadata_filter.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: vector-graph-rag
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: A Graph RAG implementation using pure vector search with Milvus
|
|
5
5
|
Project-URL: Homepage, https://github.com/zilliztech/vector-graph-rag
|
|
6
6
|
Project-URL: Documentation, https://zilliztech.github.io/vector-graph-rag/
|
|
@@ -42,6 +42,8 @@ Requires-Dist: mistralai>=1.0; extra == 'all'
|
|
|
42
42
|
Requires-Dist: ollama>=0.4; extra == 'all'
|
|
43
43
|
Requires-Dist: onnxruntime<1.24,>=1.17; (python_version == '3.10') and extra == 'all'
|
|
44
44
|
Requires-Dist: onnxruntime>=1.17; (python_version >= '3.11') and extra == 'all'
|
|
45
|
+
Requires-Dist: opentelemetry-api>=1.0; extra == 'all'
|
|
46
|
+
Requires-Dist: opentelemetry-sdk>=1.0; extra == 'all'
|
|
45
47
|
Requires-Dist: pytest-asyncio>=0.21.0; extra == 'all'
|
|
46
48
|
Requires-Dist: pytest>=7.0.0; extra == 'all'
|
|
47
49
|
Requires-Dist: ruff>=0.1.0; extra == 'all'
|
|
@@ -55,9 +57,14 @@ Provides-Extra: api
|
|
|
55
57
|
Requires-Dist: fastapi>=0.109.0; extra == 'api'
|
|
56
58
|
Requires-Dist: uvicorn[standard]>=0.27.0; extra == 'api'
|
|
57
59
|
Provides-Extra: dev
|
|
60
|
+
Requires-Dist: opentelemetry-sdk>=1.0; extra == 'dev'
|
|
58
61
|
Requires-Dist: pytest-asyncio>=0.21.0; extra == 'dev'
|
|
59
62
|
Requires-Dist: pytest>=7.0.0; extra == 'dev'
|
|
60
63
|
Requires-Dist: ruff>=0.1.0; extra == 'dev'
|
|
64
|
+
Provides-Extra: docling
|
|
65
|
+
Requires-Dist: docling>=2.0.0; extra == 'docling'
|
|
66
|
+
Requires-Dist: lxml-html-clean; extra == 'docling'
|
|
67
|
+
Requires-Dist: trafilatura>=2.0.0; extra == 'docling'
|
|
61
68
|
Provides-Extra: docs
|
|
62
69
|
Requires-Dist: mkdocs-material>=9.0.0; extra == 'docs'
|
|
63
70
|
Provides-Extra: google
|
|
@@ -75,8 +82,15 @@ Provides-Extra: local
|
|
|
75
82
|
Requires-Dist: einops>=0.8.2; extra == 'local'
|
|
76
83
|
Requires-Dist: sentence-transformers>=3.0; extra == 'local'
|
|
77
84
|
Requires-Dist: torch>=2.0.0; extra == 'local'
|
|
85
|
+
Provides-Extra: mineru
|
|
86
|
+
Requires-Dist: lxml-html-clean; extra == 'mineru'
|
|
87
|
+
Requires-Dist: markitdown[docx,pdf]>=0.1.4; extra == 'mineru'
|
|
88
|
+
Requires-Dist: mineru[pipeline]>=3.0.0; extra == 'mineru'
|
|
89
|
+
Requires-Dist: trafilatura>=2.0.0; extra == 'mineru'
|
|
78
90
|
Provides-Extra: mistral
|
|
79
91
|
Requires-Dist: mistralai>=1.0; extra == 'mistral'
|
|
92
|
+
Provides-Extra: observability
|
|
93
|
+
Requires-Dist: opentelemetry-api>=1.0; extra == 'observability'
|
|
80
94
|
Provides-Extra: ollama
|
|
81
95
|
Requires-Dist: ollama>=0.4; extra == 'ollama'
|
|
82
96
|
Provides-Extra: onnx
|
|
@@ -155,6 +169,17 @@ uv add "vector-graph-rag[all]"
|
|
|
155
169
|
|
|
156
170
|
</details>
|
|
157
171
|
|
|
172
|
+
<details>
|
|
173
|
+
<summary><b>With OpenTelemetry tracing</b></summary>
|
|
174
|
+
|
|
175
|
+
```bash
|
|
176
|
+
pip install "vector-graph-rag[observability]"
|
|
177
|
+
# or
|
|
178
|
+
uv add "vector-graph-rag[observability]"
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
</details>
|
|
182
|
+
|
|
158
183
|
## 🚀 Quick Start
|
|
159
184
|
|
|
160
185
|
```python
|
|
@@ -199,6 +224,11 @@ Use `upsert_documents_by_source()` when a source file, message, or page is
|
|
|
199
224
|
created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
|
|
200
225
|
source object is identified by `metadata["source"]` or the explicit `source`
|
|
201
226
|
argument. The method replaces only that source's chunks and graph references.
|
|
227
|
+
Source-level writes are not transactionally atomic, but the same upsert/delete
|
|
228
|
+
operation can be retried after an interruption to converge the source back to a
|
|
229
|
+
consistent state.
|
|
230
|
+
See the [Incremental Updates guide](docs/incremental-updates.md) for parser
|
|
231
|
+
integration, source key design, and retry recommendations.
|
|
202
232
|
|
|
203
233
|
```python
|
|
204
234
|
from langchain_core.documents import Document
|
|
@@ -208,7 +238,7 @@ rag.upsert_documents_by_source(
|
|
|
208
238
|
Document(
|
|
209
239
|
page_content="Einstein developed relativity at Princeton.",
|
|
210
240
|
metadata={
|
|
211
|
-
"source": "
|
|
241
|
+
"source": "file:file-123",
|
|
212
242
|
"triplets": [
|
|
213
243
|
["Einstein", "developed", "relativity"],
|
|
214
244
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -219,7 +249,7 @@ rag.upsert_documents_by_source(
|
|
|
219
249
|
extract_triplets=False,
|
|
220
250
|
)
|
|
221
251
|
|
|
222
|
-
rag.delete_documents_by_source("
|
|
252
|
+
rag.delete_documents_by_source("file:file-123")
|
|
223
253
|
```
|
|
224
254
|
|
|
225
255
|
> **Migration note:** v0.1.5 exposed `upsert_documents(document_id=...)` and
|
|
@@ -233,6 +263,34 @@ for removal in v1.0.0. For explicit full refreshes, use `rebuild_texts()`,
|
|
|
233
263
|
|
|
234
264
|
</details>
|
|
235
265
|
|
|
266
|
+
<details>
|
|
267
|
+
<summary>📈 <b>OpenTelemetry tracing</b> — click to expand</summary>
|
|
268
|
+
|
|
269
|
+
Vector Graph RAG can emit OpenTelemetry spans for ingestion, loaders,
|
|
270
|
+
embeddings, Milvus operations, retrieval, and generation. Configure the
|
|
271
|
+
OpenTelemetry SDK/exporter in your application, then attach request context
|
|
272
|
+
around RAG calls:
|
|
273
|
+
|
|
274
|
+
```python
|
|
275
|
+
from vector_graph_rag import VectorGraphRAG, observability_context
|
|
276
|
+
|
|
277
|
+
rag = VectorGraphRAG(collection_prefix="my_project")
|
|
278
|
+
|
|
279
|
+
with observability_context(
|
|
280
|
+
request_id="req-123",
|
|
281
|
+
tenant_id="tenant-a",
|
|
282
|
+
graph_name="my_project",
|
|
283
|
+
source="file-123",
|
|
284
|
+
):
|
|
285
|
+
rag.upsert_documents_by_source(chunks, source="file-123")
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
The built-in spans avoid document text, prompts, query text, generated answers,
|
|
289
|
+
filters, and full URLs by default. See the
|
|
290
|
+
[Observability guide](docs/observability.md) for setup details.
|
|
291
|
+
|
|
292
|
+
</details>
|
|
293
|
+
|
|
236
294
|
<details>
|
|
237
295
|
<summary>🌐 <b>Import from URLs and files</b> — click to expand</summary>
|
|
238
296
|
|
|
@@ -65,6 +65,17 @@ uv add "vector-graph-rag[all]"
|
|
|
65
65
|
|
|
66
66
|
</details>
|
|
67
67
|
|
|
68
|
+
<details>
|
|
69
|
+
<summary><b>With OpenTelemetry tracing</b></summary>
|
|
70
|
+
|
|
71
|
+
```bash
|
|
72
|
+
pip install "vector-graph-rag[observability]"
|
|
73
|
+
# or
|
|
74
|
+
uv add "vector-graph-rag[observability]"
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
</details>
|
|
78
|
+
|
|
68
79
|
## 🚀 Quick Start
|
|
69
80
|
|
|
70
81
|
```python
|
|
@@ -109,6 +120,11 @@ Use `upsert_documents_by_source()` when a source file, message, or page is
|
|
|
109
120
|
created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
|
|
110
121
|
source object is identified by `metadata["source"]` or the explicit `source`
|
|
111
122
|
argument. The method replaces only that source's chunks and graph references.
|
|
123
|
+
Source-level writes are not transactionally atomic, but the same upsert/delete
|
|
124
|
+
operation can be retried after an interruption to converge the source back to a
|
|
125
|
+
consistent state.
|
|
126
|
+
See the [Incremental Updates guide](docs/incremental-updates.md) for parser
|
|
127
|
+
integration, source key design, and retry recommendations.
|
|
112
128
|
|
|
113
129
|
```python
|
|
114
130
|
from langchain_core.documents import Document
|
|
@@ -118,7 +134,7 @@ rag.upsert_documents_by_source(
|
|
|
118
134
|
Document(
|
|
119
135
|
page_content="Einstein developed relativity at Princeton.",
|
|
120
136
|
metadata={
|
|
121
|
-
"source": "
|
|
137
|
+
"source": "file:file-123",
|
|
122
138
|
"triplets": [
|
|
123
139
|
["Einstein", "developed", "relativity"],
|
|
124
140
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -129,7 +145,7 @@ rag.upsert_documents_by_source(
|
|
|
129
145
|
extract_triplets=False,
|
|
130
146
|
)
|
|
131
147
|
|
|
132
|
-
rag.delete_documents_by_source("
|
|
148
|
+
rag.delete_documents_by_source("file:file-123")
|
|
133
149
|
```
|
|
134
150
|
|
|
135
151
|
> **Migration note:** v0.1.5 exposed `upsert_documents(document_id=...)` and
|
|
@@ -143,6 +159,34 @@ for removal in v1.0.0. For explicit full refreshes, use `rebuild_texts()`,
|
|
|
143
159
|
|
|
144
160
|
</details>
|
|
145
161
|
|
|
162
|
+
<details>
|
|
163
|
+
<summary>📈 <b>OpenTelemetry tracing</b> — click to expand</summary>
|
|
164
|
+
|
|
165
|
+
Vector Graph RAG can emit OpenTelemetry spans for ingestion, loaders,
|
|
166
|
+
embeddings, Milvus operations, retrieval, and generation. Configure the
|
|
167
|
+
OpenTelemetry SDK/exporter in your application, then attach request context
|
|
168
|
+
around RAG calls:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
from vector_graph_rag import VectorGraphRAG, observability_context
|
|
172
|
+
|
|
173
|
+
rag = VectorGraphRAG(collection_prefix="my_project")
|
|
174
|
+
|
|
175
|
+
with observability_context(
|
|
176
|
+
request_id="req-123",
|
|
177
|
+
tenant_id="tenant-a",
|
|
178
|
+
graph_name="my_project",
|
|
179
|
+
source="file-123",
|
|
180
|
+
):
|
|
181
|
+
rag.upsert_documents_by_source(chunks, source="file-123")
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
The built-in spans avoid document text, prompts, query text, generated answers,
|
|
185
|
+
filters, and full URLs by default. See the
|
|
186
|
+
[Observability guide](docs/observability.md) for setup details.
|
|
187
|
+
|
|
188
|
+
</details>
|
|
189
|
+
|
|
146
190
|
<details>
|
|
147
191
|
<summary>🌐 <b>Import from URLs and files</b> — click to expand</summary>
|
|
148
192
|
|
|
@@ -61,7 +61,7 @@ Frequently asked questions about Vector Graph RAG — covering when to use it, c
|
|
|
61
61
|
rag.delete_documents_by_source("file-123")
|
|
62
62
|
```
|
|
63
63
|
|
|
64
|
-
Incremental updates replace only the chunks and graph references for that source value. They are not transactionally atomic
|
|
64
|
+
Incremental updates replace only the chunks and graph references for that source value. They are not transactionally atomic. If a write is interrupted, rerun the same source operation to converge the source back to a consistent state. See [Incremental Updates](incremental-updates.md) for the full parser, CUD, and retry pattern.
|
|
65
65
|
|
|
66
66
|
??? note "Can I use local/open-source LLMs?"
|
|
67
67
|
Yes. Vector Graph RAG uses the OpenAI-compatible API format, so any LLM that exposes an OpenAI-compatible endpoint will work. This includes local models served via [Ollama](https://ollama.com/), [vLLM](https://github.com/vllm-project/vllm), [LM Studio](https://lmstudio.ai/), or any other OpenAI-compatible server. You can configure the base URL and model name when initializing the RAG instance. Keep in mind that triplet extraction and reranking quality depend heavily on the LLM's capability — weaker models may produce incomplete or inaccurate triplets, which directly affects retrieval quality. For best results, use a model with strong instruction-following and reasoning abilities.
|
|
@@ -32,6 +32,12 @@
|
|
|
32
32
|
pip install "vector-graph-rag[all]"
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
+
=== "With observability"
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install "vector-graph-rag[observability]"
|
|
39
|
+
```
|
|
40
|
+
|
|
35
41
|
!!! note "Prerequisites"
|
|
36
42
|
- Python 3.9+
|
|
37
43
|
- An OpenAI API key (set `OPENAI_API_KEY` environment variable)
|
|
@@ -180,12 +186,32 @@ rag.rebuild_documents(result.documents, extract_triplets=True)
|
|
|
180
186
|
```
|
|
181
187
|
|
|
182
188
|
!!! warning "Loader dependencies"
|
|
183
|
-
Install with `
|
|
189
|
+
Install with `uv add "vector-graph-rag[loaders]"` to enable URL fetching and document conversion.
|
|
190
|
+
|
|
191
|
+
To use a different local parser, pass a converter into `DocumentImporter`:
|
|
192
|
+
|
|
193
|
+
```python
|
|
194
|
+
from vector_graph_rag.loaders import DoclingConverter, DocumentImporter
|
|
195
|
+
|
|
196
|
+
importer = DocumentImporter(
|
|
197
|
+
converter=DoclingConverter(),
|
|
198
|
+
chunk_size=1000,
|
|
199
|
+
chunk_overlap=200,
|
|
200
|
+
)
|
|
201
|
+
result = importer.import_sources(["/path/to/document.pdf"])
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
Install Docling support with `uv add "vector-graph-rag[docling]"`.
|
|
184
205
|
|
|
185
206
|
### Incremental Updates
|
|
186
207
|
|
|
187
208
|
Use `upsert_documents_by_source()` when a source file, page, or message is created or modified. In Vector Graph RAG, a `Document` is a passage/chunk, and `metadata["source"]` identifies the source object whose chunks should be replaced.
|
|
188
209
|
|
|
210
|
+
For parser and loader workflows, attach the same stable source value to every
|
|
211
|
+
chunk produced from one file, page, or message. See
|
|
212
|
+
[Incremental Updates](incremental-updates.md) for the full source-level CUD and
|
|
213
|
+
retry pattern.
|
|
214
|
+
|
|
189
215
|
```python
|
|
190
216
|
from langchain_core.documents import Document
|
|
191
217
|
|
|
@@ -194,7 +220,7 @@ rag.upsert_documents_by_source(
|
|
|
194
220
|
Document(
|
|
195
221
|
page_content="Einstein developed relativity at Princeton.",
|
|
196
222
|
metadata={
|
|
197
|
-
"source": "
|
|
223
|
+
"source": "file:file-123",
|
|
198
224
|
"triplets": [
|
|
199
225
|
["Einstein", "developed", "relativity"],
|
|
200
226
|
["Einstein", "worked at", "Princeton"],
|
|
@@ -205,14 +231,34 @@ rag.upsert_documents_by_source(
|
|
|
205
231
|
extract_triplets=False,
|
|
206
232
|
)
|
|
207
233
|
|
|
208
|
-
rag.delete_documents_by_source("
|
|
234
|
+
rag.delete_documents_by_source("file:file-123")
|
|
209
235
|
```
|
|
210
236
|
|
|
211
237
|
!!! warning "v0.2.0 migration"
|
|
212
238
|
`upsert_documents(document_id=...)` and `delete_documents(document_id)` were removed because `Document` means passage/chunk in this project. Use `upsert_documents_by_source()` and `delete_documents_by_source()` with a stable `metadata["source"]` value instead.
|
|
213
239
|
|
|
214
240
|
!!! note "Consistency model"
|
|
215
|
-
Incremental updates perform a source-level cascade across passages, entities, and relations. They are not transactionally atomic,
|
|
241
|
+
Incremental updates perform a source-level cascade across passages, entities, and relations. They are not transactionally atomic, and queries may observe intermediate state if a write is interrupted. If an upsert or delete fails, rerun the same operation for the same source to converge the source back to a consistent state. Avoid concurrently mutating the same collection prefix during an update.
|
|
242
|
+
|
|
243
|
+
## Observability
|
|
244
|
+
|
|
245
|
+
Install `vector-graph-rag[observability]` to enable OpenTelemetry trace instrumentation. The library creates spans for ingestion, loading, embeddings, Milvus operations, retrieval, and generation, while your application configures the OpenTelemetry SDK and exporter.
|
|
246
|
+
|
|
247
|
+
```python
|
|
248
|
+
from vector_graph_rag import VectorGraphRAG, observability_context
|
|
249
|
+
|
|
250
|
+
rag = VectorGraphRAG(collection_prefix="my_project")
|
|
251
|
+
|
|
252
|
+
with observability_context(
|
|
253
|
+
request_id="req-123",
|
|
254
|
+
tenant_id="tenant-a",
|
|
255
|
+
graph_name="my_project",
|
|
256
|
+
source="file-123",
|
|
257
|
+
):
|
|
258
|
+
rag.upsert_documents_by_source(chunks, source="file-123")
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
See [Observability](observability.md) for setup, span coverage, and data-safety notes.
|
|
216
262
|
|
|
217
263
|
## Querying
|
|
218
264
|
|
|
@@ -223,4 +223,4 @@ flowchart TD
|
|
|
223
223
|
Triplet extraction and reranking quality depends heavily on the underlying LLM's capability. Weaker models may produce incomplete or inaccurate triplets, which directly affects retrieval quality.
|
|
224
224
|
|
|
225
225
|
!!! warning "Graph Consistency"
|
|
226
|
-
The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g., source-level cascade updates) are not atomic and may
|
|
226
|
+
The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g., source-level cascade updates) are not atomic, and queries may observe intermediate state if a write is interrupted. Source-level `upsert_documents_by_source()` and `delete_documents_by_source()` operations are designed to be retryable: rerun the same operation for the same source after a failure to converge the source back to a consistent state. Use `rebuild_documents()` for initial bulk indexing or full refreshes.
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
# Incremental Updates
|
|
2
|
+
|
|
3
|
+
Use source-level incremental updates when your upstream system can tell you
|
|
4
|
+
which external object changed. A source can be a file, web page, email message,
|
|
5
|
+
database row, or any other stable business object.
|
|
6
|
+
|
|
7
|
+
In Vector Graph RAG, a LangChain `Document` represents one passage or chunk.
|
|
8
|
+
The source object is tracked through metadata, not through `Document.id`.
|
|
9
|
+
|
|
10
|
+
## Core Contract
|
|
11
|
+
|
|
12
|
+
Each incremental update call must contain chunks from exactly one source.
|
|
13
|
+
|
|
14
|
+
```python
|
|
15
|
+
from langchain_core.documents import Document
|
|
16
|
+
|
|
17
|
+
chunks = [
|
|
18
|
+
Document(
|
|
19
|
+
page_content="Q2 revenue increased in the enterprise segment.",
|
|
20
|
+
metadata={
|
|
21
|
+
"source": "file:file-123",
|
|
22
|
+
"page": 1,
|
|
23
|
+
"chunk_index": 0,
|
|
24
|
+
},
|
|
25
|
+
),
|
|
26
|
+
Document(
|
|
27
|
+
page_content="Renewal rates improved after support response times dropped.",
|
|
28
|
+
metadata={
|
|
29
|
+
"source": "file:file-123",
|
|
30
|
+
"page": 2,
|
|
31
|
+
"chunk_index": 1,
|
|
32
|
+
},
|
|
33
|
+
),
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
rag.upsert_documents_by_source(chunks)
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
The default grouping field is `metadata["source"]`. You can also pass the
|
|
40
|
+
source explicitly:
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
rag.upsert_documents_by_source(
|
|
44
|
+
documents=chunks,
|
|
45
|
+
source="file:file-123",
|
|
46
|
+
)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Use a custom field when your application already has a different metadata name:
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
rag.upsert_documents_by_source(
|
|
53
|
+
documents=chunks,
|
|
54
|
+
source="file-123",
|
|
55
|
+
source_field="file_id",
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
rag.delete_documents_by_source("file-123", source_field="file_id")
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Parser Or Loader Output
|
|
62
|
+
|
|
63
|
+
Document parsing is intentionally outside the core incremental API. Use your
|
|
64
|
+
existing loader or parser to produce text chunks, then attach the same stable
|
|
65
|
+
source value to every chunk from the same external object.
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from langchain_core.documents import Document
|
|
69
|
+
from vector_graph_rag import VectorGraphRAG
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def parse_file_to_chunks(path: str) -> list[str]:
|
|
73
|
+
"""Return text chunks from your parser or document processing pipeline."""
|
|
74
|
+
return [
|
|
75
|
+
"The handbook describes the support escalation policy.",
|
|
76
|
+
"Escalated issues must include the customer impact and owner.",
|
|
77
|
+
]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def index_file(rag: VectorGraphRAG, path: str, file_id: str) -> None:
|
|
81
|
+
chunks = [
|
|
82
|
+
Document(
|
|
83
|
+
page_content=text,
|
|
84
|
+
metadata={
|
|
85
|
+
"source": f"file:{file_id}",
|
|
86
|
+
"chunk_index": index,
|
|
87
|
+
"parser": "internal",
|
|
88
|
+
},
|
|
89
|
+
)
|
|
90
|
+
for index, text in enumerate(parse_file_to_chunks(path))
|
|
91
|
+
]
|
|
92
|
+
|
|
93
|
+
rag.upsert_documents_by_source(chunks, extract_triplets=True)
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
If your parser also extracts knowledge graph triplets, store them in each
|
|
97
|
+
chunk's `metadata["triplets"]` and disable LLM triplet extraction:
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
chunks = [
|
|
101
|
+
Document(
|
|
102
|
+
page_content="The handbook describes the support escalation policy.",
|
|
103
|
+
metadata={
|
|
104
|
+
"source": "file:file-123",
|
|
105
|
+
"triplets": [
|
|
106
|
+
["handbook", "describes", "support escalation policy"],
|
|
107
|
+
["escalated issues", "include", "customer impact"],
|
|
108
|
+
],
|
|
109
|
+
},
|
|
110
|
+
)
|
|
111
|
+
]
|
|
112
|
+
|
|
113
|
+
rag.upsert_documents_by_source(chunks, extract_triplets=False)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## Create, Update, Delete
|
|
117
|
+
|
|
118
|
+
Use the same upsert API for both create and update.
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
# Create or replace one source.
|
|
122
|
+
rag.upsert_documents_by_source(
|
|
123
|
+
documents=chunks,
|
|
124
|
+
source="file:file-123",
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
# Delete one source.
|
|
128
|
+
rag.delete_documents_by_source("file:file-123")
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
When a source already exists, `upsert_documents_by_source()` removes the old
|
|
132
|
+
chunks and graph references for that source, then inserts the new chunks. Other
|
|
133
|
+
sources under the same collection prefix are preserved.
|
|
134
|
+
|
|
135
|
+
For multi-source updates, call the API once per source:
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
for source, chunks in changed_sources:
|
|
139
|
+
rag.upsert_documents_by_source(chunks, source=source)
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
## Retry Behavior
|
|
143
|
+
|
|
144
|
+
Source-level updates perform a multi-step cascade across passages, relations,
|
|
145
|
+
and entities. Milvus does not provide a transaction that spans those logical
|
|
146
|
+
records, so these writes are not transactionally atomic. If the process is
|
|
147
|
+
interrupted, queries may observe intermediate state until the update is retried.
|
|
148
|
+
|
|
149
|
+
The source-level APIs are designed to be retryable. If an upsert or delete
|
|
150
|
+
raises an exception, rerun the same operation for the same source.
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
try:
|
|
154
|
+
rag.upsert_documents_by_source(
|
|
155
|
+
documents=chunks,
|
|
156
|
+
source="file:file-123",
|
|
157
|
+
)
|
|
158
|
+
except Exception:
|
|
159
|
+
# Log the failed source in your application job state, then retry it.
|
|
160
|
+
rag.upsert_documents_by_source(
|
|
161
|
+
documents=chunks,
|
|
162
|
+
source="file:file-123",
|
|
163
|
+
)
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
For delete:
|
|
167
|
+
|
|
168
|
+
```python
|
|
169
|
+
try:
|
|
170
|
+
rag.delete_documents_by_source("file:file-123")
|
|
171
|
+
except Exception:
|
|
172
|
+
rag.delete_documents_by_source("file:file-123")
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
A production ingestion job should track which source keys succeeded or failed.
|
|
176
|
+
If a retry sees no matching passages during delete, the source may already have
|
|
177
|
+
been removed by the previous attempt.
|
|
178
|
+
|
|
179
|
+
## ID Guidance
|
|
180
|
+
|
|
181
|
+
Use `source` for the external object identity. Use `Document.id` only when you
|
|
182
|
+
need to control the passage ID.
|
|
183
|
+
|
|
184
|
+
If `Document.id` is omitted, `upsert_documents_by_source()` generates stable
|
|
185
|
+
passage IDs from the source value and chunk index. If you provide `Document.id`,
|
|
186
|
+
make sure it is unique to that passage and does not belong to another source.
|
|
187
|
+
|
|
188
|
+
Good source values are stable and globally meaningful in your application:
|
|
189
|
+
|
|
190
|
+
```text
|
|
191
|
+
file:file-123
|
|
192
|
+
url:https://example.com/docs/pricing
|
|
193
|
+
message:message-456
|
|
194
|
+
record:account-789
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
Avoid using a chunk ID, page number, or transient parser output path as the
|
|
198
|
+
source. Those values identify a passage or a local processing artifact, not the
|
|
199
|
+
external object that should be replaced as a unit.
|
|
200
|
+
|
|
201
|
+
## Metadata And Filtering
|
|
202
|
+
|
|
203
|
+
Additional metadata is stored on passages and can be used for filtering.
|
|
204
|
+
|
|
205
|
+
```python
|
|
206
|
+
rag.upsert_documents_by_source(
|
|
207
|
+
documents=chunks,
|
|
208
|
+
source="file:file-123",
|
|
209
|
+
metadata={
|
|
210
|
+
"tenant_id": "tenant-a",
|
|
211
|
+
"workspace_id": "finance",
|
|
212
|
+
},
|
|
213
|
+
)
|
|
214
|
+
|
|
215
|
+
result = rag.query(
|
|
216
|
+
"What should escalated issues include?",
|
|
217
|
+
filter='tenant_id == "tenant-a" and workspace_id == "finance"',
|
|
218
|
+
)
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
The source metadata is also stored on each passage, so you can filter by source
|
|
222
|
+
when needed:
|
|
223
|
+
|
|
224
|
+
```python
|
|
225
|
+
result = rag.query(
|
|
226
|
+
"What should escalated issues include?",
|
|
227
|
+
filter='source == "file:file-123"',
|
|
228
|
+
)
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
## Initial Bulk Load
|
|
232
|
+
|
|
233
|
+
For a first-time corpus load, you can either use `rebuild_documents()` or call
|
|
234
|
+
`upsert_documents_by_source()` once per source. Use `rebuild_documents()` when
|
|
235
|
+
you intentionally want a full refresh of the collection prefix. Use
|
|
236
|
+
`upsert_documents_by_source()` when you want the same ingestion path for initial
|
|
237
|
+
load and later CUD events.
|
|
238
|
+
|
|
239
|
+
```python
|
|
240
|
+
for source, chunks in all_sources:
|
|
241
|
+
rag.upsert_documents_by_source(chunks, source=source)
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
Avoid the legacy `add_*` helpers for new ingestion code. They keep their old
|
|
245
|
+
full-rebuild behavior and are planned for removal in v1.0.0.
|
|
@@ -114,7 +114,7 @@ print(result.answer)
|
|
|
114
114
|
```
|
|
115
115
|
|
|
116
116
|
!!! note "Ingestion semantics"
|
|
117
|
-
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes, and use `upsert_documents_by_source()` / `delete_documents_by_source()` when a source file, page, or message changes later.
|
|
117
|
+
The legacy `add_*` ingestion helpers rebuild the full knowledge base and are planned for removal in v1.0.0. Use `rebuild_texts()`, `rebuild_documents()`, or `rebuild_documents_with_triplets()` for full refreshes, and use `upsert_documents_by_source()` / `delete_documents_by_source()` when a source file, page, or message changes later. See [Incremental Updates](incremental-updates.md) for parser integration, source key design, and retry recommendations.
|
|
118
118
|
|
|
119
119
|
!!! tip "Getting Started"
|
|
120
120
|
See the [Getting Started](getting-started.md) guide for installation and configuration options.
|