goodmem-llamaindex 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- goodmem_llamaindex-0.2.3/.gitignore +13 -0
- goodmem_llamaindex-0.2.3/LICENSE +21 -0
- goodmem_llamaindex-0.2.3/PKG-INFO +108 -0
- goodmem_llamaindex-0.2.3/README.md +76 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/__init__.py +23 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/_connection.py +63 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/_ids.py +30 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/_metadata.py +28 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/base.py +422 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/filters.py +67 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/ingestion.py +202 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/py.typed +0 -0
- goodmem_llamaindex-0.2.3/llama_index/tools/goodmem/retriever.py +278 -0
- goodmem_llamaindex-0.2.3/pyproject.toml +61 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 bashareid
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: goodmem-llamaindex
|
|
3
|
+
Version: 0.2.3
|
|
4
|
+
Summary: Native LlamaIndex retrieval, Document ingestion and agent tools for GoodMem.
|
|
5
|
+
Project-URL: Homepage, https://goodmem.ai
|
|
6
|
+
Project-URL: Repository, https://github.com/PAIR-Systems-Inc/goodmem-llamaindex
|
|
7
|
+
Project-URL: Documentation, https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/main/docs/usage.md
|
|
8
|
+
Project-URL: Changelog, https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/main/CHANGELOG.md
|
|
9
|
+
Author: bashareid
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agents,goodmem,llama-index,llm,memory,semantic-search,vector-search
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: <4.0,>=3.10
|
|
24
|
+
Requires-Dist: goodmem<0.2,>=0.1.34
|
|
25
|
+
Requires-Dist: httpx<1,>=0.28
|
|
26
|
+
Requires-Dist: llama-index-core<0.15,>=0.14.24
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest-asyncio>=1.0.0; extra == 'dev'
|
|
29
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: ruff>=0.12; extra == 'dev'
|
|
31
|
+
Description-Content-Type: text/markdown
|
|
32
|
+
|
|
33
|
+
# GoodMem for LlamaIndex
|
|
34
|
+
|
|
35
|
+
Use [GoodMem](https://goodmem.ai) as a persistent document and retrieval service in LlamaIndex. GoodMem handles chunking, embeddings and optional reranking; the integration returns native `NodeWithScore` objects for query engines and agents.
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install 'goodmem-llamaindex>=0.2.3'
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
The import namespace is `llama_index.tools.goodmem`. Set `GOODMEM_BASE_URL` to your server’s REST root and `GOODMEM_API_KEY` to its API key.
|
|
42
|
+
|
|
43
|
+
## Store and retrieve Documents
|
|
44
|
+
|
|
45
|
+
Use an existing GoodMem space configured with an embedder:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
import os
|
|
49
|
+
|
|
50
|
+
from goodmem import Goodmem
|
|
51
|
+
from llama_index.core.schema import Document
|
|
52
|
+
from llama_index.tools.goodmem import (
|
|
53
|
+
GoodMemDocumentIngestor,
|
|
54
|
+
GoodMemRetriever,
|
|
55
|
+
wait_for_memories,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
with Goodmem(base_url=os.environ["GOODMEM_BASE_URL"],
|
|
59
|
+
api_key=os.environ["GOODMEM_API_KEY"]) as client:
|
|
60
|
+
space_id = os.environ["GOODMEM_SPACE_ID"]
|
|
61
|
+
ids = GoodMemDocumentIngestor(client=client, space_id=space_id).add_documents([
|
|
62
|
+
Document(text="The returns period is 30 days.",
|
|
63
|
+
metadata={"source": "https://example.com/returns"})
|
|
64
|
+
])
|
|
65
|
+
wait_for_memories(client, ids, timeout=120)
|
|
66
|
+
|
|
67
|
+
retriever = GoodMemRetriever(client=client, space_ids=[space_id])
|
|
68
|
+
for result in retriever.retrieve("How long do I have to return an item?"):
|
|
69
|
+
print(result.score, result.text, result.metadata.get("source"))
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
`add_documents` returns accepted IDs without waiting. Waiting is explicit and checks only those IDs. Ordinary empty searches return immediately.
|
|
73
|
+
|
|
74
|
+
## Connect an agent
|
|
75
|
+
|
|
76
|
+
Use LlamaIndex’s own tool wrapper. Give each collection a useful name and description:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
import os
|
|
80
|
+
|
|
81
|
+
from llama_index.core.tools import RetrieverTool
|
|
82
|
+
from llama_index.tools.goodmem import GoodMemRetriever
|
|
83
|
+
|
|
84
|
+
space_id = os.environ["GOODMEM_SPACE_ID"]
|
|
85
|
+
retriever = GoodMemRetriever(space_ids=[space_id]) # uses environment settings
|
|
86
|
+
search = RetrieverTool.from_defaults(
|
|
87
|
+
retriever,
|
|
88
|
+
name="returns_policy",
|
|
89
|
+
description="Search the company's returns and refund policies.",
|
|
90
|
+
)
|
|
91
|
+
# Pass search to a workflow-based ReActAgent or FunctionAgent.
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
The model supplies the query; the application configures spaces, filters and reranking. The same retriever works with `RetrieverQueryEngine` and standard LlamaIndex callbacks.
|
|
95
|
+
|
|
96
|
+
Pass `filters=MetadataFilters(...)` for supported scalar comparisons, membership tests and nested conditions. Pass `reranker_id=...` to rerank on the server without an LLM. Sources, custom metadata and Document metadata exclusions survive storage. Framework scores rank higher as more relevant; `raw_score` preserves the original value.
|
|
97
|
+
|
|
98
|
+
Retrieval does not raise on problems the server reports in its results; it returns what the server sent. HTTP errors, such as an unknown space, still raise the SDK's exception. If a configured reranker fails (for example, its ID does not exist), GoodMem reports `NOT_FOUND` and `RERANKING_FAILED` and still returns the vector-search hits. The retriever returns those hits as vector scores, negated like any vector score, with `score_kind="negative_inner_product"`, `partial=True` and the server's `statuses` on each result, and logs a WARNING. A reported problem with no hits returns an empty list, a `UserWarning` and a WARNING log line naming the statuses.
|
|
99
|
+
|
|
100
|
+
## Async and administrative tools
|
|
101
|
+
|
|
102
|
+
`aretrieve` and `aadd_documents` use the SDK’s `AsyncGoodmem` directly. Inject `async_client` to share its connection pool; caller-owned clients remain open.
|
|
103
|
+
|
|
104
|
+
`GoodMemToolSpec` supplies optional space and memory management tools. Its retrieval result includes chunks with the server's raw scores, `statuses`, a `partial` flag and `score_kind` (`reranker`, or `negative_inner_product` for vector scores, including when reranking failed). File uploads require an explicitly configured directory. Prefer a scoped retriever tool when an agent only needs search.
|
|
105
|
+
|
|
106
|
+
Every GoodMem ID you or a model pass in (space, memory, embedder, reranker or LLM) must be a UUID. Anything else raises `ValueError` before a request is sent, because the SDK puts IDs into URL paths, where a value such as `../spaces/<id>` would reach a different resource.
|
|
107
|
+
|
|
108
|
+
See [usage and migration](https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/v0.2.3/docs/usage.md) for async examples, supported filters and diagnostics, and the [changelog](https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/v0.2.3/CHANGELOG.md) for the changes from 0.1. Run `pip install -e '.[dev]'`, then `pytest`, `ruff check llama_index tests` and `ruff format --check llama_index tests`, as CI does. Live tests are opt-in and clean up their own spaces.
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# GoodMem for LlamaIndex
|
|
2
|
+
|
|
3
|
+
Use [GoodMem](https://goodmem.ai) as a persistent document and retrieval service in LlamaIndex. GoodMem handles chunking, embeddings and optional reranking; the integration returns native `NodeWithScore` objects for query engines and agents.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install 'goodmem-llamaindex>=0.2.3'
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
The import namespace is `llama_index.tools.goodmem`. Set `GOODMEM_BASE_URL` to your server’s REST root and `GOODMEM_API_KEY` to its API key.
|
|
10
|
+
|
|
11
|
+
## Store and retrieve Documents
|
|
12
|
+
|
|
13
|
+
Use an existing GoodMem space configured with an embedder:
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import os
|
|
17
|
+
|
|
18
|
+
from goodmem import Goodmem
|
|
19
|
+
from llama_index.core.schema import Document
|
|
20
|
+
from llama_index.tools.goodmem import (
|
|
21
|
+
GoodMemDocumentIngestor,
|
|
22
|
+
GoodMemRetriever,
|
|
23
|
+
wait_for_memories,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
with Goodmem(base_url=os.environ["GOODMEM_BASE_URL"],
|
|
27
|
+
api_key=os.environ["GOODMEM_API_KEY"]) as client:
|
|
28
|
+
space_id = os.environ["GOODMEM_SPACE_ID"]
|
|
29
|
+
ids = GoodMemDocumentIngestor(client=client, space_id=space_id).add_documents([
|
|
30
|
+
Document(text="The returns period is 30 days.",
|
|
31
|
+
metadata={"source": "https://example.com/returns"})
|
|
32
|
+
])
|
|
33
|
+
wait_for_memories(client, ids, timeout=120)
|
|
34
|
+
|
|
35
|
+
retriever = GoodMemRetriever(client=client, space_ids=[space_id])
|
|
36
|
+
for result in retriever.retrieve("How long do I have to return an item?"):
|
|
37
|
+
print(result.score, result.text, result.metadata.get("source"))
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
`add_documents` returns accepted IDs without waiting. Waiting is explicit and checks only those IDs. Ordinary empty searches return immediately.
|
|
41
|
+
|
|
42
|
+
## Connect an agent
|
|
43
|
+
|
|
44
|
+
Use LlamaIndex’s own tool wrapper. Give each collection a useful name and description:
|
|
45
|
+
|
|
46
|
+
```python
|
|
47
|
+
import os
|
|
48
|
+
|
|
49
|
+
from llama_index.core.tools import RetrieverTool
|
|
50
|
+
from llama_index.tools.goodmem import GoodMemRetriever
|
|
51
|
+
|
|
52
|
+
space_id = os.environ["GOODMEM_SPACE_ID"]
|
|
53
|
+
retriever = GoodMemRetriever(space_ids=[space_id]) # uses environment settings
|
|
54
|
+
search = RetrieverTool.from_defaults(
|
|
55
|
+
retriever,
|
|
56
|
+
name="returns_policy",
|
|
57
|
+
description="Search the company's returns and refund policies.",
|
|
58
|
+
)
|
|
59
|
+
# Pass search to a workflow-based ReActAgent or FunctionAgent.
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
The model supplies the query; the application configures spaces, filters and reranking. The same retriever works with `RetrieverQueryEngine` and standard LlamaIndex callbacks.
|
|
63
|
+
|
|
64
|
+
Pass `filters=MetadataFilters(...)` for supported scalar comparisons, membership tests and nested conditions. Pass `reranker_id=...` to rerank on the server without an LLM. Sources, custom metadata and Document metadata exclusions survive storage. Framework scores rank higher as more relevant; `raw_score` preserves the original value.
|
|
65
|
+
|
|
66
|
+
Retrieval does not raise on problems the server reports in its results; it returns what the server sent. HTTP errors, such as an unknown space, still raise the SDK's exception. If a configured reranker fails (for example, its ID does not exist), GoodMem reports `NOT_FOUND` and `RERANKING_FAILED` and still returns the vector-search hits. The retriever returns those hits as vector scores, negated like any vector score, with `score_kind="negative_inner_product"`, `partial=True` and the server's `statuses` on each result, and logs a WARNING. A reported problem with no hits returns an empty list, a `UserWarning` and a WARNING log line naming the statuses.
|
|
67
|
+
|
|
68
|
+
## Async and administrative tools
|
|
69
|
+
|
|
70
|
+
`aretrieve` and `aadd_documents` use the SDK’s `AsyncGoodmem` directly. Inject `async_client` to share its connection pool; caller-owned clients remain open.
|
|
71
|
+
|
|
72
|
+
`GoodMemToolSpec` supplies optional space and memory management tools. Its retrieval result includes chunks with the server's raw scores, `statuses`, a `partial` flag and `score_kind` (`reranker`, or `negative_inner_product` for vector scores, including when reranking failed). File uploads require an explicitly configured directory. Prefer a scoped retriever tool when an agent only needs search.
|
|
73
|
+
|
|
74
|
+
Every GoodMem ID you or a model pass in (space, memory, embedder, reranker or LLM) must be a UUID. Anything else raises `ValueError` before a request is sent, because the SDK puts IDs into URL paths, where a value such as `../spaces/<id>` would reach a different resource.
|
|
75
|
+
|
|
76
|
+
See [usage and migration](https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/v0.2.3/docs/usage.md) for async examples, supported filters and diagnostics, and the [changelog](https://github.com/PAIR-Systems-Inc/goodmem-llamaindex/blob/v0.2.3/CHANGELOG.md) for the changes from 0.1. Run `pip install -e '.[dev]'`, then `pytest`, `ruff check llama_index tests` and `ruff format --check llama_index tests`, as CI does. Live tests are opt-in and clean up their own spaces.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""GoodMem integration for LlamaIndex."""
|
|
2
|
+
|
|
3
|
+
from .base import GoodMemToolSpec
|
|
4
|
+
from .ingestion import (
|
|
5
|
+
GoodMemDocumentIngestor,
|
|
6
|
+
GoodMemIndexingError,
|
|
7
|
+
GoodMemIngestionError,
|
|
8
|
+
await_memories,
|
|
9
|
+
wait_for_memories,
|
|
10
|
+
)
|
|
11
|
+
from .retriever import GoodMemNodeWithScore, GoodMemRetrievalError, GoodMemRetriever
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"GoodMemToolSpec",
|
|
15
|
+
"GoodMemRetriever",
|
|
16
|
+
"GoodMemNodeWithScore",
|
|
17
|
+
"GoodMemRetrievalError",
|
|
18
|
+
"GoodMemDocumentIngestor",
|
|
19
|
+
"GoodMemIngestionError",
|
|
20
|
+
"GoodMemIndexingError",
|
|
21
|
+
"wait_for_memories",
|
|
22
|
+
"await_memories",
|
|
23
|
+
]
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""SDK sessions with explicit ownership and native async support."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from collections.abc import AsyncIterator, Iterator
|
|
5
|
+
from contextlib import asynccontextmanager, contextmanager
|
|
6
|
+
|
|
7
|
+
from goodmem import AsyncGoodmem, Goodmem
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class Connection:
|
|
11
|
+
"""Use injected SDK clients, or create short-lived clients from configuration.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
client: Caller-owned synchronous SDK client. With injected clients,
|
|
15
|
+
synchronous calls require this client even when environment settings exist.
|
|
16
|
+
async_client: Caller-owned asynchronous SDK client. With injected clients,
|
|
17
|
+
asynchronous calls require this client even when environment settings exist.
|
|
18
|
+
api_key: API key, defaulting to GOODMEM_API_KEY.
|
|
19
|
+
base_url: Server root, defaulting to GOODMEM_BASE_URL.
|
|
20
|
+
verify_ssl: Certificate verification, or a path to a trusted CA bundle.
|
|
21
|
+
timeout: SDK HTTP timeout in seconds.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
*,
|
|
27
|
+
client: Goodmem | None = None,
|
|
28
|
+
async_client: AsyncGoodmem | None = None,
|
|
29
|
+
api_key: str | None = None,
|
|
30
|
+
base_url: str | None = None,
|
|
31
|
+
verify_ssl: bool | str = True,
|
|
32
|
+
timeout: float = 120,
|
|
33
|
+
) -> None:
|
|
34
|
+
self.client = client
|
|
35
|
+
self.async_client = async_client
|
|
36
|
+
self.options = {
|
|
37
|
+
"api_key": api_key or os.getenv("GOODMEM_API_KEY"),
|
|
38
|
+
"base_url": base_url or os.getenv("GOODMEM_BASE_URL"),
|
|
39
|
+
"verify": verify_ssl,
|
|
40
|
+
"timeout": timeout,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
@contextmanager
|
|
44
|
+
def sync(self) -> Iterator[Goodmem]:
|
|
45
|
+
"""Yield a synchronous client without closing caller-owned connections."""
|
|
46
|
+
if self.client is not None:
|
|
47
|
+
yield self.client
|
|
48
|
+
else:
|
|
49
|
+
if self.async_client is not None:
|
|
50
|
+
raise ValueError("Pass client for sync calls with an injected async SDK client")
|
|
51
|
+
with Goodmem(**self.options) as client:
|
|
52
|
+
yield client
|
|
53
|
+
|
|
54
|
+
@asynccontextmanager
|
|
55
|
+
async def async_(self) -> AsyncIterator[AsyncGoodmem]:
|
|
56
|
+
"""Yield a native async client without using a worker thread."""
|
|
57
|
+
if self.async_client is not None:
|
|
58
|
+
yield self.async_client
|
|
59
|
+
else:
|
|
60
|
+
if self.client is not None:
|
|
61
|
+
raise ValueError("Pass async_client for async calls with an injected SDK client")
|
|
62
|
+
async with AsyncGoodmem(**self.options) as client:
|
|
63
|
+
yield client
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""GoodMem resource IDs are UUIDs; anything else is refused before a request.
|
|
2
|
+
|
|
3
|
+
The official SDK interpolates IDs into URL paths without escaping them, and httpx
|
|
4
|
+
resolves dot segments before sending. An ID such as "../spaces/<uuid>" would
|
|
5
|
+
therefore address a different resource, so every ID is checked here first.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from typing import Annotated
|
|
10
|
+
|
|
11
|
+
from pydantic import Field
|
|
12
|
+
|
|
13
|
+
UUID_PATTERN = r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$"
|
|
14
|
+
_UUID = re.compile(UUID_PATTERN)
|
|
15
|
+
|
|
16
|
+
# Tool schemas tell the model an ID is a UUID with the standard JSON-schema format.
|
|
17
|
+
# LlamaIndex keeps only json_schema_extra from a top-level annotation and prints the
|
|
18
|
+
# annotation in each tool description; a pattern there would push retrieve_memories
|
|
19
|
+
# past OpenAI's 1024-character description limit. require_uuid is the real guard.
|
|
20
|
+
UuidStr = Annotated[str, Field(json_schema_extra={"format": "uuid"})]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def require_uuid(value: object, field: str) -> str:
|
|
24
|
+
"""Return the canonical lowercase UUID, or raise ValueError naming the field."""
|
|
25
|
+
# fullmatch, unlike match with "$", also rejects a trailing newline.
|
|
26
|
+
if not isinstance(value, str) or not _UUID.fullmatch(value):
|
|
27
|
+
raise ValueError(
|
|
28
|
+
f"{field} must be a UUID (xxxxxxxx-xxxx-xxxx-xxxx-xxxxxxxxxxxx); no request was sent"
|
|
29
|
+
)
|
|
30
|
+
return value.lower()
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Persist LlamaIndex metadata visibility settings alongside user metadata."""
|
|
2
|
+
|
|
3
|
+
_DOCUMENT_METADATA = "_llamaindex_goodmem"
|
|
4
|
+
_EXCLUSIONS = ("excluded_llm_metadata_keys", "excluded_embed_metadata_keys")
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def document_metadata(document):
|
|
8
|
+
"""Encode formatting settings without overwriting a user metadata field."""
|
|
9
|
+
if _DOCUMENT_METADATA in document.metadata:
|
|
10
|
+
raise ValueError(f"Metadata key {_DOCUMENT_METADATA} is reserved for LlamaIndex settings")
|
|
11
|
+
return dict(document.metadata) | {
|
|
12
|
+
_DOCUMENT_METADATA: {key: list(getattr(document, key)) for key in _EXCLUSIONS}
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def stored_metadata(metadata):
|
|
17
|
+
"""Separate stored settings, rejecting malformed exclusions before rendering."""
|
|
18
|
+
metadata = dict(metadata or {})
|
|
19
|
+
options = metadata.pop(_DOCUMENT_METADATA, {})
|
|
20
|
+
if not isinstance(options, dict):
|
|
21
|
+
raise ValueError(f"Invalid {_DOCUMENT_METADATA} metadata settings")
|
|
22
|
+
exclusions = {}
|
|
23
|
+
for key in _EXCLUSIONS:
|
|
24
|
+
value = options.get(key, [])
|
|
25
|
+
if not isinstance(value, list) or any(not isinstance(item, str) for item in value):
|
|
26
|
+
raise ValueError(f"Invalid stored {key}: expected a list of strings")
|
|
27
|
+
exclusions[key] = list(value)
|
|
28
|
+
return metadata, exclusions
|