vector-graph-rag 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/PKG-INFO +4 -1
  2. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/README.md +3 -0
  3. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/getting-started.md +1 -1
  4. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/how-it-works.md +1 -1
  5. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/python-api.md +4 -0
  6. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/pyproject.toml +1 -1
  7. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/__init__.py +1 -1
  8. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/rag.py +108 -32
  9. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/milvus.py +90 -29
  10. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_rag_incremental.py +344 -0
  11. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/uv.lock +1 -1
  12. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/.dockerignore +0 -0
  13. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/.env.example +0 -0
  14. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/.github/workflows/docs.yml +0 -0
  15. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/.github/workflows/release.yml +0 -0
  16. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/.gitignore +0 -0
  17. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/AGENTS.md +0 -0
  18. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/CLAUDE.md +0 -0
  19. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/Dockerfile +0 -0
  20. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/LICENSE +0 -0
  21. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/api/__init__.py +0 -0
  22. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/api/main.py +0 -0
  23. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/api/schemas.py +0 -0
  24. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/assets/logo.png +0 -0
  25. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/assets/logo.svg +0 -0
  26. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/design-philosophy.md +0 -0
  27. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/evaluation.md +0 -0
  28. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/faq.md +0 -0
  29. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/frontend.md +0 -0
  30. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/index.md +0 -0
  31. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/rest-api.md +0 -0
  32. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/stylesheets/custom.css +0 -0
  33. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/docs/use-cases.md +0 -0
  34. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/README.md +0 -0
  35. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/2wikimultihopqa.json +0 -0
  36. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/2wikimultihopqa_corpus.json +0 -0
  37. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/hotpotqa.json +0 -0
  38. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/hotpotqa_corpus.json +0 -0
  39. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/musique.json +0 -0
  40. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/musique_corpus.json +0 -0
  41. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/2wikimultihopqa_queries.named_entity_output.tsv +0 -0
  42. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/hotpotqa_queries.named_entity_output.tsv +0 -0
  43. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/hotpotqa_train_queries.named_entity_output.tsv +0 -0
  44. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/musique_queries.named_entity_output.tsv +0 -0
  45. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/ner_cache/sample_queries.named_entity_output.tsv +0 -0
  46. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/openie_2wikimultihopqa_results_ner_gpt-3.5-turbo-1106_6119.json +0 -0
  47. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/openie_hotpotqa_results_ner_gpt-3.5-turbo-1106_9221.json +0 -0
  48. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/openie_musique_results_ner_gpt-3.5-turbo-1106_11656.json +0 -0
  49. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/openie_test_sample_results_ner_gpt-3.5-turbo-1106_20.json +0 -0
  50. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/test_sample.json +0 -0
  51. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/data/test_sample_corpus.json +0 -0
  52. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/evaluation/evaluate.py +0 -0
  53. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/.gitignore +0 -0
  54. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/README.md +0 -0
  55. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/eslint.config.js +0 -0
  56. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/index.html +0 -0
  57. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/package-lock.json +0 -0
  58. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/package.json +0 -0
  59. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/public/vite.svg +0 -0
  60. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/App.css +0 -0
  61. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/App.tsx +0 -0
  62. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/api/client.ts +0 -0
  63. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/api/queries.ts +0 -0
  64. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/assets/react.svg +0 -0
  65. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/graph/EntityNode.tsx +0 -0
  66. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/graph/GraphCanvas.tsx +0 -0
  67. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/graph/GraphLegend.tsx +0 -0
  68. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/graph/RelationEdge.tsx +0 -0
  69. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/graph/index.ts +0 -0
  70. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/CreateGraphDialog.tsx +0 -0
  71. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/FileUploader.tsx +0 -0
  72. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/GraphSelector.tsx +0 -0
  73. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/GraphStatsPreview.tsx +0 -0
  74. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportDialog.tsx +0 -0
  75. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportProgress.tsx +0 -0
  76. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/ImportSettings.tsx +0 -0
  77. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/import/UrlInput.tsx +0 -0
  78. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/panels/AnswerPanel.tsx +0 -0
  79. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/panels/NodeDetailPanel.tsx +0 -0
  80. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/panels/index.ts +0 -0
  81. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/search/SearchInput.tsx +0 -0
  82. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/search/index.ts +0 -0
  83. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/settings/SettingsDialog.tsx +0 -0
  84. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/timeline/ProcessTimeline.tsx +0 -0
  85. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/timeline/index.ts +0 -0
  86. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/Header.tsx +0 -0
  87. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/button.tsx +0 -0
  88. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/dialog.tsx +0 -0
  89. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/input.tsx +0 -0
  90. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/label.tsx +0 -0
  91. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/progress.tsx +0 -0
  92. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/switch.tsx +0 -0
  93. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/components/ui/tabs.tsx +0 -0
  94. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/hooks/use-toast.ts +0 -0
  95. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/index.css +0 -0
  96. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/main.tsx +0 -0
  97. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/stores/datasetStore.ts +0 -0
  98. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/stores/graphStore.ts +0 -0
  99. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/stores/searchStore.ts +0 -0
  100. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/types/api.ts +0 -0
  101. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/utils/cn.ts +0 -0
  102. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/src/utils/graphLayout.ts +0 -0
  103. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/tsconfig.app.json +0 -0
  104. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/tsconfig.json +0 -0
  105. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/tsconfig.node.json +0 -0
  106. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/frontend/vite.config.ts +0 -0
  107. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/mkdocs.yml +0 -0
  108. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/api/__init__.py +0 -0
  109. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/api/app.py +0 -0
  110. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/config.py +0 -0
  111. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/__init__.py +0 -0
  112. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/builder.py +0 -0
  113. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/graph.py +0 -0
  114. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/knowledge_graph.py +0 -0
  115. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/graph/retriever.py +0 -0
  116. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/__init__.py +0 -0
  117. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/cache.py +0 -0
  118. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/extractor.py +0 -0
  119. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/llm/reranker.py +0 -0
  120. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/__init__.py +0 -0
  121. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/chunker.py +0 -0
  122. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/converter.py +0 -0
  123. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/loaders/url_fetcher.py +0 -0
  124. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/models.py +0 -0
  125. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/__init__.py +0 -0
  126. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/__init__.py +0 -0
  127. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/google.py +0 -0
  128. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/huggingface.py +0 -0
  129. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/jina.py +0 -0
  130. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/local.py +0 -0
  131. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/mistral.py +0 -0
  132. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/ollama.py +0 -0
  133. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/onnx.py +0 -0
  134. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/openai.py +0 -0
  135. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/utils.py +0 -0
  136. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embedding_providers/voyage.py +0 -0
  137. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/src/vector_graph_rag/storage/embeddings.py +0 -0
  138. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/__init__.py +0 -0
  139. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/conftest.py +0 -0
  140. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_api.py +0 -0
  141. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_embedding_providers.py +0 -0
  142. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_graph.py +0 -0
  143. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_milvus_store.py +0 -0
  144. {vector_graph_rag-0.2.0 → vector_graph_rag-0.2.1}/tests/test_rag_metadata_filter.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: vector-graph-rag
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: A Graph RAG implementation using pure vector search with Milvus
5
5
  Project-URL: Homepage, https://github.com/zilliztech/vector-graph-rag
6
6
  Project-URL: Documentation, https://zilliztech.github.io/vector-graph-rag/
@@ -199,6 +199,9 @@ Use `upsert_documents_by_source()` when a source file, message, or page is
199
199
  created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
200
200
  source object is identified by `metadata["source"]` or the explicit `source`
201
201
  argument. The method replaces only that source's chunks and graph references.
202
+ Source-level writes are not transactionally atomic, but the same upsert/delete
203
+ operation can be retried after an interruption to converge the source back to a
204
+ consistent state.
202
205
 
203
206
  ```python
204
207
  from langchain_core.documents import Document
@@ -109,6 +109,9 @@ Use `upsert_documents_by_source()` when a source file, message, or page is
109
109
  created or modified. In Vector Graph RAG, a `Document` is a passage/chunk; the
110
110
  source object is identified by `metadata["source"]` or the explicit `source`
111
111
  argument. The method replaces only that source's chunks and graph references.
112
+ Source-level writes are not transactionally atomic, but the same upsert/delete
113
+ operation can be retried after an interruption to converge the source back to a
114
+ consistent state.
112
115
 
113
116
  ```python
114
117
  from langchain_core.documents import Document
@@ -212,7 +212,7 @@ rag.delete_documents_by_source("sharepoint:file-123")
212
212
  `upsert_documents(document_id=...)` and `delete_documents(document_id)` were removed because `Document` means passage/chunk in this project. Use `upsert_documents_by_source()` and `delete_documents_by_source()` with a stable `metadata["source"]` value instead.
213
213
 
214
214
  !!! note "Consistency model"
215
- Incremental updates perform a source-level cascade across passages, entities, and relations. They are not transactionally atomic, so avoid interrupting or concurrently mutating the same collection prefix during an update.
215
+ Incremental updates perform a source-level cascade across passages, entities, and relations. They are not transactionally atomic, and queries may observe intermediate state if a write is interrupted. If an upsert or delete fails, rerun the same operation for the same source to converge the source back to a consistent state. Avoid concurrently mutating the same collection prefix during an update.
216
216
 
217
217
  ## Querying
218
218
 
@@ -223,4 +223,4 @@ flowchart TD
223
223
  Triplet extraction and reranking quality depends heavily on the underlying LLM's capability. Weaker models may produce incomplete or inaccurate triplets, which directly affects retrieval quality.
224
224
 
225
225
  !!! warning "Graph Consistency"
226
- The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g., source-level cascade updates) are not atomic and may leave inconsistent state if interrupted. Use `rebuild_documents()` for initial bulk indexing or full refreshes, and use `upsert_documents_by_source()` / `delete_documents_by_source()` for source updates when the write operation is expected to run to completion.
226
+ The knowledge graph topology is maintained as a logical layer on top of Milvus via cross-referenced ID fields. Since Milvus does not support transactions, multi-step mutations (e.g., source-level cascade updates) are not atomic, and queries may observe intermediate state if a write is interrupted. Source-level `upsert_documents_by_source()` and `delete_documents_by_source()` operations are designed to be retryable: rerun the same operation for the same source after a failure to converge the source back to a consistent state. Use `rebuild_documents()` for initial bulk indexing or full refreshes.
@@ -403,6 +403,8 @@ def upsert_documents_by_source(
403
403
 
404
404
  If the source does not exist, the method inserts a new source. If it already exists, the method deletes the previous chunks for that source, updates graph references, and inserts the new chunks without rebuilding unrelated sources.
405
405
 
406
+ Source-level writes are not transactionally atomic. If an upsert fails during the multi-step cascade, rerun the same `upsert_documents_by_source()` call for the same source to converge the source back to the requested state. Queries may observe intermediate state until the retry succeeds.
407
+
406
408
  The method accepts exactly one source per call. If `source` is not provided and the documents contain multiple `metadata[source_field]` values, it raises `ValueError`.
407
409
 
408
410
  ```python
@@ -447,6 +449,8 @@ def delete_documents_by_source(
447
449
 
448
450
  Shared entities and relations are preserved when other sources still reference them. Orphaned relations and entities are removed.
449
451
 
452
+ Source-level deletes are retryable. If a delete fails after partially cleaning graph records, rerun `delete_documents_by_source()` with the same `source` and `source_field` to finish the cascade.
453
+
450
454
  ```python
451
455
  deleted = rag.delete_documents_by_source("sharepoint:file-123")
452
456
  ```
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "vector-graph-rag"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "A Graph RAG implementation using pure vector search with Milvus"
5
5
  readme = "README.md"
6
6
  license = { text = "MIT" }
@@ -18,7 +18,7 @@ from vector_graph_rag.rag import VectorGraphRAG, create_rag
18
18
  from vector_graph_rag.storage.embeddings import EmbeddingModel
19
19
  from vector_graph_rag.storage.milvus import MilvusStore
20
20
 
21
- __version__ = "0.2.0"
21
+ __version__ = "0.2.1"
22
22
 
23
23
  __all__ = [
24
24
  "Settings",
@@ -233,7 +233,7 @@ class VectorGraphRAG:
233
233
  def _normalize_source_value(value: Any, field_name: str) -> str:
234
234
  """Validate and normalize a source metadata value."""
235
235
  if not isinstance(value, str) or not value.strip():
236
- raise ValueError(f'{field_name} must be a non-empty string.')
236
+ raise ValueError(f"{field_name} must be a non-empty string.")
237
237
  return value.strip()
238
238
 
239
239
  @staticmethod
@@ -277,7 +277,7 @@ class VectorGraphRAG:
277
277
  if conflicting_sources:
278
278
  conflicts = ", ".join(sorted(conflicting_sources))
279
279
  raise ValueError(
280
- f'documents contain {source_field} values that differ from source '
280
+ f"documents contain {source_field} values that differ from source "
281
281
  f'"{explicit_source}": {conflicts}'
282
282
  )
283
283
  return explicit_source
@@ -513,6 +513,23 @@ class VectorGraphRAG:
513
513
  rid: existing_relations.get(builder.relations[rid], {}).get("id", rid)
514
514
  for rid in builder.relation_ids
515
515
  }
516
+ current_passage_ids = set(builder.passage_ids)
517
+ known_relation_ids = set(relation_id_map.values())
518
+ suspect_existing_relation_ids = sorted(
519
+ {
520
+ relation_id
521
+ for entity in existing_entities.values()
522
+ if current_passage_ids.intersection(entity.get("passage_ids", []))
523
+ for relation_id in entity.get("relation_ids", [])
524
+ if relation_id not in known_relation_ids
525
+ }
526
+ )
527
+ live_suspect_relation_ids: set[str] = set()
528
+ if suspect_existing_relation_ids:
529
+ live_suspect_relation_ids = {
530
+ relation["id"]
531
+ for relation in self._store._get_relations_by_ids(suspect_existing_relation_ids)
532
+ }
516
533
 
517
534
  existing_passages = self._store.get_passages_by_ids(
518
535
  builder.passage_ids,
@@ -520,9 +537,7 @@ class VectorGraphRAG:
520
537
  )
521
538
  for passage in existing_passages:
522
539
  if passage.get(source_field) != source:
523
- raise ValueError(
524
- f'Passage ID "{passage["id"]}" already belongs to another source.'
525
- )
540
+ raise ValueError(f'Passage ID "{passage["id"]}" already belongs to another source.')
526
541
 
527
542
  if show_progress:
528
543
  logger.info("Upserting incremental graph into Milvus...")
@@ -534,6 +549,7 @@ class VectorGraphRAG:
534
549
  entity_insert_texts: List[str] = []
535
550
  entity_insert_embeddings: List[List[float]] = []
536
551
  entity_insert_metadatas: List[Dict[str, Any]] = []
552
+ entity_update_records: List[Dict[str, Any]] = []
537
553
 
538
554
  for index, eid in enumerate(builder.entity_ids):
539
555
  stored_id = entity_id_map[eid]
@@ -548,12 +564,28 @@ class VectorGraphRAG:
548
564
  }
549
565
  existing = existing_entities.get(entity_text_by_id[eid])
550
566
  if existing:
551
- self._store._update_entity(
552
- stored_id,
553
- text=entity_text_by_id[eid],
554
- embedding=entity_embeddings[index],
555
- relation_ids=self._merge_unique(existing.get("relation_ids", []), relation_ids),
556
- passage_ids=self._merge_unique(existing.get("passage_ids", []), passage_ids),
567
+ existing_relation_ids = existing.get("relation_ids", [])
568
+ if current_passage_ids.intersection(existing.get("passage_ids", [])):
569
+ existing_relation_ids = [
570
+ relation_id
571
+ for relation_id in existing_relation_ids
572
+ if relation_id in known_relation_ids
573
+ or relation_id in live_suspect_relation_ids
574
+ ]
575
+ entity_update_records.append(
576
+ {
577
+ "id": stored_id,
578
+ "text": entity_text_by_id[eid],
579
+ "vector": entity_embeddings[index],
580
+ "relation_ids": self._merge_unique(
581
+ existing_relation_ids,
582
+ relation_ids,
583
+ ),
584
+ "passage_ids": self._merge_unique(
585
+ existing.get("passage_ids", []),
586
+ passage_ids,
587
+ ),
588
+ }
557
589
  )
558
590
  else:
559
591
  entity_insert_ids.append(stored_id)
@@ -561,6 +593,9 @@ class VectorGraphRAG:
561
593
  entity_insert_embeddings.append(entity_embeddings[index])
562
594
  entity_insert_metadatas.append(metadata)
563
595
 
596
+ if entity_update_records:
597
+ self._store._upsert_entity_records(entity_update_records)
598
+
564
599
  if entity_insert_ids:
565
600
  self._store._insert_entities(
566
601
  entity_insert_texts,
@@ -574,6 +609,7 @@ class VectorGraphRAG:
574
609
  relation_insert_texts: List[str] = []
575
610
  relation_insert_embeddings: List[List[float]] = []
576
611
  relation_insert_metadatas: List[Dict[str, Any]] = []
612
+ relation_update_records: List[Dict[str, Any]] = []
577
613
 
578
614
  for index, rid in enumerate(builder.relation_ids):
579
615
  stored_id = relation_id_map[rid]
@@ -594,22 +630,31 @@ class VectorGraphRAG:
594
630
 
595
631
  existing = existing_relations.get(relation_text_by_id[rid])
596
632
  if existing:
597
- self._store._update_relation(
598
- stored_id,
599
- text=relation_text_by_id[rid],
600
- embedding=relation_embeddings[index],
601
- entity_ids=entity_ids,
602
- passage_ids=self._merge_unique(existing.get("passage_ids", []), passage_ids),
603
- subject=metadata.get("subject"),
604
- predicate=metadata.get("predicate"),
605
- object_=metadata.get("object"),
606
- )
633
+ update_record = {
634
+ "id": stored_id,
635
+ "text": relation_text_by_id[rid],
636
+ "vector": relation_embeddings[index],
637
+ "entity_ids": entity_ids,
638
+ "passage_ids": self._merge_unique(
639
+ existing.get("passage_ids", []),
640
+ passage_ids,
641
+ ),
642
+ }
643
+ for field in ["subject", "predicate", "object"]:
644
+ if metadata.get(field) is not None:
645
+ update_record[field] = metadata[field]
646
+ elif field in existing:
647
+ update_record[field] = existing[field]
648
+ relation_update_records.append(update_record)
607
649
  else:
608
650
  relation_insert_ids.append(stored_id)
609
651
  relation_insert_texts.append(relation_text_by_id[rid])
610
652
  relation_insert_embeddings.append(relation_embeddings[index])
611
653
  relation_insert_metadatas.append(metadata)
612
654
 
655
+ if relation_update_records:
656
+ self._store._upsert_relation_records(relation_update_records)
657
+
613
658
  if relation_insert_ids:
614
659
  self._store._insert_relations(
615
660
  relation_insert_texts,
@@ -915,6 +960,8 @@ class VectorGraphRAG:
915
960
  In this project, a LangChain Document is a passage/chunk. This method
916
961
  uses a stable source metadata value to replace all chunks that belong to
917
962
  one source object, such as a file, URL, message, or business record.
963
+ The write is not transactionally atomic, but rerunning the same source
964
+ upsert after a failure converges the source to the requested state.
918
965
 
919
966
  Args:
920
967
  documents: Parsed chunks/passages for one source.
@@ -1011,6 +1058,9 @@ class VectorGraphRAG:
1011
1058
  """
1012
1059
  Incrementally delete all documents that belong to one source.
1013
1060
 
1061
+ The write is not transactionally atomic, but rerunning the same source
1062
+ delete after a failure finishes the source-level cascade.
1063
+
1014
1064
  Args:
1015
1065
  source: Stable source value previously used with
1016
1066
  upsert_documents_by_source().
@@ -1037,7 +1087,10 @@ class VectorGraphRAG:
1037
1087
  )
1038
1088
 
1039
1089
  relations = self._store._get_relations_by_ids(relation_ids)
1040
- deleted_relation_ids: set[str] = set()
1090
+ existing_relation_ids = {relation["id"] for relation in relations}
1091
+ relation_update_records: List[Dict[str, Any]] = []
1092
+ relation_delete_ids: List[str] = []
1093
+ deleted_relation_ids: set[str] = set(relation_ids) - existing_relation_ids
1041
1094
  for relation in relations:
1042
1095
  relation_id = relation["id"]
1043
1096
  new_passage_ids = self._remove_many(
@@ -1045,15 +1098,29 @@ class VectorGraphRAG:
1045
1098
  removed_passage_ids,
1046
1099
  )
1047
1100
  if new_passage_ids:
1048
- self._store._update_relation(
1049
- relation_id,
1050
- passage_ids=new_passage_ids,
1051
- )
1101
+ update_record = {
1102
+ "id": relation_id,
1103
+ "text": relation["text"],
1104
+ "vector": self._embedding_model.embed(relation["text"]),
1105
+ "entity_ids": relation.get("entity_ids", []),
1106
+ "passage_ids": new_passage_ids,
1107
+ }
1108
+ for field in ["subject", "predicate", "object"]:
1109
+ if field in relation:
1110
+ update_record[field] = relation[field]
1111
+ relation_update_records.append(update_record)
1052
1112
  else:
1053
- self._store._delete_relation(relation_id)
1113
+ relation_delete_ids.append(relation_id)
1054
1114
  deleted_relation_ids.add(relation_id)
1055
1115
 
1116
+ if relation_update_records:
1117
+ self._store._upsert_relation_records(relation_update_records)
1118
+ if relation_delete_ids:
1119
+ self._store._delete_relations(relation_delete_ids)
1120
+
1056
1121
  entities = self._store._get_entities_by_ids(entity_ids)
1122
+ entity_update_records: List[Dict[str, Any]] = []
1123
+ entity_delete_ids: List[str] = []
1057
1124
  for entity in entities:
1058
1125
  entity_id = entity["id"]
1059
1126
  new_passage_ids = self._remove_many(
@@ -1066,13 +1133,22 @@ class VectorGraphRAG:
1066
1133
  )
1067
1134
 
1068
1135
  if new_passage_ids or new_relation_ids:
1069
- self._store._update_entity(
1070
- entity_id,
1071
- passage_ids=new_passage_ids,
1072
- relation_ids=new_relation_ids,
1136
+ entity_update_records.append(
1137
+ {
1138
+ "id": entity_id,
1139
+ "text": entity["text"],
1140
+ "vector": self._embedding_model.embed(entity["text"]),
1141
+ "passage_ids": new_passage_ids,
1142
+ "relation_ids": new_relation_ids,
1143
+ }
1073
1144
  )
1074
1145
  else:
1075
- self._store._delete_entity(entity_id)
1146
+ entity_delete_ids.append(entity_id)
1147
+
1148
+ if entity_update_records:
1149
+ self._store._upsert_entity_records(entity_update_records)
1150
+ if entity_delete_ids:
1151
+ self._store._delete_entities(entity_delete_ids)
1076
1152
 
1077
1153
  self._store.delete_passages(passage_ids)
1078
1154
  self._retriever = None
@@ -13,7 +13,7 @@ Users should interact with passages through the Graph abstraction layer.
13
13
  import logging
14
14
  import re
15
15
  import uuid
16
- from typing import Any, Dict, List, Optional
16
+ from typing import Any, Dict, Iterator, List, Optional
17
17
 
18
18
  from pymilvus import DataType, MilvusClient
19
19
  from tqdm import tqdm
@@ -97,6 +97,77 @@ class MilvusStore:
97
97
  )
98
98
  return field_name
99
99
 
100
+ def _batch_items(
101
+ self,
102
+ items: List[Any],
103
+ batch_size: Optional[int] = None,
104
+ ) -> Iterator[List[Any]]:
105
+ """Yield items in batches using the configured batch size."""
106
+ effective_batch_size = batch_size or self.settings.batch_size
107
+ for start_idx in range(0, len(items), effective_batch_size):
108
+ yield items[start_idx : start_idx + effective_batch_size]
109
+
110
+ @staticmethod
111
+ def _unique_preserve_order(values: List[str]) -> List[str]:
112
+ """Return unique strings without changing first-seen order."""
113
+ seen = set()
114
+ unique_values = []
115
+ for value in values:
116
+ if value in seen:
117
+ continue
118
+ seen.add(value)
119
+ unique_values.append(value)
120
+ return unique_values
121
+
122
+ def _query_by_texts(
123
+ self,
124
+ collection_name: str,
125
+ texts: List[str],
126
+ output_fields: List[str],
127
+ ) -> Dict[str, Dict[str, Any]]:
128
+ """Query records by exact text in batches."""
129
+ records_by_text: Dict[str, Dict[str, Any]] = {}
130
+ unique_texts = self._unique_preserve_order(texts)
131
+ if not unique_texts:
132
+ return records_by_text
133
+
134
+ for batch_texts in self._batch_items(unique_texts):
135
+ quoted_texts = ", ".join(self._quote_string(text) for text in batch_texts)
136
+ results = self.client.query(
137
+ collection_name=collection_name,
138
+ filter=f"text in [{quoted_texts}]",
139
+ output_fields=output_fields,
140
+ )
141
+ for record in results:
142
+ text = record.get("text")
143
+ if isinstance(text, str) and text not in records_by_text:
144
+ records_by_text[text] = record
145
+
146
+ return records_by_text
147
+
148
+ def _upsert_records(
149
+ self,
150
+ collection_name: str,
151
+ records: List[Dict[str, Any]],
152
+ ) -> None:
153
+ """Upsert records in batches using the configured batch size."""
154
+ if not records:
155
+ return
156
+
157
+ for batch_records in self._batch_items(records):
158
+ self.client.upsert(
159
+ collection_name=collection_name,
160
+ data=batch_records,
161
+ )
162
+
163
+ def _upsert_entity_records(self, records: List[Dict[str, Any]]) -> None:
164
+ """Upsert fully materialized entity records."""
165
+ self._upsert_records(self.entity_collection, records)
166
+
167
+ def _upsert_relation_records(self, records: List[Dict[str, Any]]) -> None:
168
+ """Upsert fully materialized relation records."""
169
+ self._upsert_records(self.relation_collection, records)
170
+
100
171
  def _create_collection(
101
172
  self,
102
173
  collection_name: str,
@@ -500,16 +571,11 @@ class MilvusStore:
500
571
  Returns:
501
572
  Mapping from entity text to the first matching entity record.
502
573
  """
503
- entities_by_text: Dict[str, Dict[str, Any]] = {}
504
- for text in entity_texts:
505
- results = self.client.query(
506
- collection_name=self.entity_collection,
507
- filter=f"text == {self._quote_string(text)}",
508
- output_fields=["id", "text", "relation_ids", "passage_ids"],
509
- )
510
- if results:
511
- entities_by_text[text] = results[0]
512
- return entities_by_text
574
+ return self._query_by_texts(
575
+ self.entity_collection,
576
+ entity_texts,
577
+ output_fields=["id", "text", "relation_ids", "passage_ids"],
578
+ )
513
579
 
514
580
  def _get_relations_by_ids(
515
581
  self,
@@ -562,24 +628,19 @@ class MilvusStore:
562
628
  Returns:
563
629
  Mapping from relation text to the first matching relation record.
564
630
  """
565
- relations_by_text: Dict[str, Dict[str, Any]] = {}
566
- for text in relation_texts:
567
- results = self.client.query(
568
- collection_name=self.relation_collection,
569
- filter=f"text == {self._quote_string(text)}",
570
- output_fields=[
571
- "id",
572
- "text",
573
- "entity_ids",
574
- "passage_ids",
575
- "subject",
576
- "predicate",
577
- "object",
578
- ],
579
- )
580
- if results:
581
- relations_by_text[text] = results[0]
582
- return relations_by_text
631
+ return self._query_by_texts(
632
+ self.relation_collection,
633
+ relation_texts,
634
+ output_fields=[
635
+ "id",
636
+ "text",
637
+ "entity_ids",
638
+ "passage_ids",
639
+ "subject",
640
+ "predicate",
641
+ "object",
642
+ ],
643
+ )
583
644
 
584
645
  def get_passages_by_ids(
585
646
  self,