deepagents-graph-memory 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- deepagents_graph_memory/__init__.py +20 -0
- deepagents_graph_memory/backend.py +838 -0
- deepagents_graph_memory/errors.py +21 -0
- deepagents_graph_memory/ladybug_store.py +847 -0
- deepagents_graph_memory/paths.py +253 -0
- deepagents_graph_memory/py.typed +1 -0
- deepagents_graph_memory/recall.py +936 -0
- deepagents_graph_memory/renderers.py +199 -0
- deepagents_graph_memory/stores.py +378 -0
- deepagents_graph_memory/tools.py +193 -0
- deepagents_graph_memory/vgs.py +210 -0
- deepagents_graph_memory-0.1.0.dist-info/METADATA +770 -0
- deepagents_graph_memory-0.1.0.dist-info/RECORD +16 -0
- deepagents_graph_memory-0.1.0.dist-info/WHEEL +5 -0
- deepagents_graph_memory-0.1.0.dist-info/licenses/LICENSE +21 -0
- deepagents_graph_memory-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,838 @@
|
|
|
1
|
+
# - Lets an agent browse graph context as if it were a set of files.
|
|
2
|
+
# - Also saves connected records of what happened, why, and how it turned out.
|
|
3
|
+
# - Tests: test_backend_read.py and test_backend_ls.py cover reading and finding context;
|
|
4
|
+
# test_write_behavior.py checks that the file views stay read-only.
|
|
5
|
+
# test_trace.py, test_scope.py, and test_limits.py cover work history, separation, and size limits;
|
|
6
|
+
# test_create_and_clear.py checks how a fresh graph starts.
|
|
7
|
+
|
|
8
|
+
"""Deep Agents backend implementation for graph memory."""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import hashlib
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
from collections.abc import Callable, Sequence
|
|
16
|
+
from datetime import UTC, datetime
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Literal
|
|
19
|
+
from uuid import uuid4
|
|
20
|
+
|
|
21
|
+
from deepagents.backends.protocol import (
|
|
22
|
+
BackendProtocol,
|
|
23
|
+
EditResult,
|
|
24
|
+
FileDownloadResponse,
|
|
25
|
+
FileInfo,
|
|
26
|
+
FileUploadResponse,
|
|
27
|
+
GlobResult,
|
|
28
|
+
GrepMatch,
|
|
29
|
+
GrepResult,
|
|
30
|
+
LsResult,
|
|
31
|
+
ReadResult,
|
|
32
|
+
WriteResult,
|
|
33
|
+
)
|
|
34
|
+
from deepagents.backends.utils import create_file_data, slice_read_response
|
|
35
|
+
from wcmatch import glob as wcglob
|
|
36
|
+
|
|
37
|
+
from deepagents_graph_memory.errors import GraphMemoryError, GraphMemoryValidationError
|
|
38
|
+
from deepagents_graph_memory.ladybug_store import LadybugGraphStore
|
|
39
|
+
from deepagents_graph_memory.paths import (
|
|
40
|
+
node_path,
|
|
41
|
+
normalize_graph_path,
|
|
42
|
+
parse_graph_path,
|
|
43
|
+
validate_identifier,
|
|
44
|
+
validate_namespace,
|
|
45
|
+
validate_node_id,
|
|
46
|
+
validate_subject,
|
|
47
|
+
)
|
|
48
|
+
from deepagents_graph_memory.recall import RecallMode
|
|
49
|
+
from deepagents_graph_memory.recall import recall_graph_memory as _recall_graph_memory
|
|
50
|
+
from deepagents_graph_memory.renderers import render_index, render_node, render_schema, render_search
|
|
51
|
+
from deepagents_graph_memory.stores import GraphStoreAdapter, merge_metadata, utc_now, validate_properties
|
|
52
|
+
|
|
53
|
+
READ_ONLY_ERROR = "Graph memory views are read-only. Use graph memory tools to add or update graph facts."
|
|
54
|
+
TRACE_TEXT_ALLOWED_CONTROL_CHARS = frozenset({"\n", "\r", "\t"})
|
|
55
|
+
NamespaceFactory = Callable[[Any], tuple[str, ...]]
|
|
56
|
+
Namespace = str | Sequence[str] | NamespaceFactory | None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class GraphMemoryBackend(BackendProtocol):
|
|
60
|
+
"""Deep Agents backend that projects graph facts into graph context views."""
|
|
61
|
+
|
|
62
|
+
def __init__(
|
|
63
|
+
self,
|
|
64
|
+
store: GraphStoreAdapter,
|
|
65
|
+
*,
|
|
66
|
+
namespace: Namespace = None,
|
|
67
|
+
max_nodes: int = 50,
|
|
68
|
+
max_edges: int = 100,
|
|
69
|
+
) -> None:
|
|
70
|
+
"""Initialize a graph memory backend.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
store: Internal graph store adapter.
|
|
74
|
+
namespace: Optional Deep Agents-style namespace factory or static namespace.
|
|
75
|
+
max_nodes: Maximum nodes listed or traversed in bounded views.
|
|
76
|
+
max_edges: Maximum edges rendered in node pages.
|
|
77
|
+
"""
|
|
78
|
+
self.store = store
|
|
79
|
+
self.namespace = namespace
|
|
80
|
+
self.max_nodes = max_nodes
|
|
81
|
+
self.max_edges = max_edges
|
|
82
|
+
|
|
83
|
+
@classmethod
|
|
84
|
+
def create(
|
|
85
|
+
cls,
|
|
86
|
+
*,
|
|
87
|
+
path: str | Path | None = None,
|
|
88
|
+
namespace: Namespace = None,
|
|
89
|
+
max_nodes: int = 50,
|
|
90
|
+
max_edges: int = 100,
|
|
91
|
+
) -> GraphMemoryBackend:
|
|
92
|
+
"""Create a LadybugDB graph memory backend.
|
|
93
|
+
|
|
94
|
+
Args:
|
|
95
|
+
path: Persistent database file path; omitted for an in-memory graph.
|
|
96
|
+
namespace: Optional Deep Agents-style namespace factory or static namespace.
|
|
97
|
+
max_nodes: Maximum nodes listed or traversed in bounded views.
|
|
98
|
+
max_edges: Maximum edges rendered in node pages.
|
|
99
|
+
|
|
100
|
+
Returns:
|
|
101
|
+
Configured graph memory backend.
|
|
102
|
+
"""
|
|
103
|
+
return cls(
|
|
104
|
+
LadybugGraphStore.memory() if path is None else LadybugGraphStore.disk(path),
|
|
105
|
+
namespace=namespace,
|
|
106
|
+
max_nodes=max_nodes,
|
|
107
|
+
max_edges=max_edges,
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
def close(self) -> None:
|
|
111
|
+
"""Close resources owned by a LadybugDB store."""
|
|
112
|
+
if isinstance(self.store, LadybugGraphStore):
|
|
113
|
+
self.store.close()
|
|
114
|
+
|
|
115
|
+
def ls(self, path: str) -> LsResult:
|
|
116
|
+
"""List graph memory virtual files."""
|
|
117
|
+
try:
|
|
118
|
+
scope_key = self._scope_key()
|
|
119
|
+
normalized, had_graph_prefix = normalize_graph_path(path)
|
|
120
|
+
entries = self._ls_internal(normalized, scope_key=scope_key)
|
|
121
|
+
if had_graph_prefix:
|
|
122
|
+
entries = [_prefix_file_info(entry) for entry in entries]
|
|
123
|
+
return LsResult(entries=entries)
|
|
124
|
+
except GraphMemoryError as exc:
|
|
125
|
+
return LsResult(error=str(exc), entries=None)
|
|
126
|
+
|
|
127
|
+
def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult:
|
|
128
|
+
"""Read a graph memory virtual file."""
|
|
129
|
+
offset, limit = max(int(offset), 0), max(int(limit), 0)
|
|
130
|
+
if limit == 0:
|
|
131
|
+
metadata = {"no_lines_requested": True} if hasattr(ReadResult, "no_lines_requested") else {}
|
|
132
|
+
return ReadResult(file_data=create_file_data(""), **metadata)
|
|
133
|
+
try:
|
|
134
|
+
scope_key = self._scope_key()
|
|
135
|
+
normalized = normalize_graph_path(file_path)[0].rstrip("/")
|
|
136
|
+
if normalized in {"", "/nodes", "/search"} or (normalized.startswith("/nodes/") and normalized.count("/") == 2):
|
|
137
|
+
return ReadResult(error="is_directory")
|
|
138
|
+
parsed = parse_graph_path(file_path)
|
|
139
|
+
if parsed.kind == "index":
|
|
140
|
+
content = render_index()
|
|
141
|
+
elif parsed.kind == "schema":
|
|
142
|
+
content = render_schema(self.store.get_schema(scope_key=scope_key))
|
|
143
|
+
elif parsed.kind == "node":
|
|
144
|
+
assert parsed.label is not None and parsed.node_id is not None
|
|
145
|
+
node = self.store.get_node(parsed.label, parsed.node_id, scope_key=scope_key)
|
|
146
|
+
if node is None:
|
|
147
|
+
return ReadResult(error=f"Graph node '{parsed.label}/{parsed.node_id}' not found.")
|
|
148
|
+
neighborhood = self.store.get_neighbors(
|
|
149
|
+
parsed.label,
|
|
150
|
+
parsed.node_id,
|
|
151
|
+
scope_key=scope_key,
|
|
152
|
+
depth=1,
|
|
153
|
+
max_nodes=self.max_nodes,
|
|
154
|
+
max_edges=self.max_edges,
|
|
155
|
+
)
|
|
156
|
+
content = render_node(node, neighborhood)
|
|
157
|
+
else:
|
|
158
|
+
assert parsed.query is not None
|
|
159
|
+
content = render_search(parsed.query, self.store.search(parsed.query, scope_key=scope_key, limit=self.max_nodes))
|
|
160
|
+
result = slice_read_response(create_file_data(content), offset, limit)
|
|
161
|
+
# Deep Agents 0.6 returns text; 0.7 returns a result with pagination metadata.
|
|
162
|
+
return ReadResult(file_data=create_file_data(result)) if isinstance(result, str) else result
|
|
163
|
+
except GraphMemoryError as exc:
|
|
164
|
+
return ReadResult(error=str(exc))
|
|
165
|
+
|
|
166
|
+
def grep(self, pattern: str, path: str | None = None, glob: str | None = None, *, max_count: int | None = None) -> GrepResult:
|
|
167
|
+
"""Search rendered graph files for case-sensitive literal matching lines."""
|
|
168
|
+
try:
|
|
169
|
+
scope_key = self._scope_key()
|
|
170
|
+
base_path, had_graph_prefix = normalize_graph_path(path or "/")
|
|
171
|
+
matcher = _compile_glob(glob) if glob else None
|
|
172
|
+
matches: list[GrepMatch] = []
|
|
173
|
+
# ponytail: scan rendered views; add an index only if literal grep becomes a measured bottleneck.
|
|
174
|
+
for candidate in self._candidate_files(base_path, scope_key=scope_key):
|
|
175
|
+
candidate_path = candidate["path"]
|
|
176
|
+
if matcher is not None and not matcher(_relative_path(candidate_path, base_path)):
|
|
177
|
+
continue
|
|
178
|
+
result = self.read(candidate_path, limit=sys.maxsize)
|
|
179
|
+
if result.error is not None:
|
|
180
|
+
return GrepResult(error=result.error)
|
|
181
|
+
assert result.file_data is not None
|
|
182
|
+
for line_number, line in enumerate(result.file_data["content"].splitlines(), 1):
|
|
183
|
+
if pattern in line:
|
|
184
|
+
if max_count is not None and len(matches) >= max_count:
|
|
185
|
+
if hasattr(GrepResult, "truncated"):
|
|
186
|
+
return GrepResult(matches=matches, truncated=True)
|
|
187
|
+
return GrepResult(error="Grep exceeded max_count. Narrow the search or use Deep Agents 0.7 for truncation metadata.")
|
|
188
|
+
matches.append({"path": _maybe_prefix(candidate_path, had_graph_prefix), "line": line_number, "text": line})
|
|
189
|
+
return GrepResult(matches=matches)
|
|
190
|
+
except (GraphMemoryError, ValueError) as exc:
|
|
191
|
+
return GrepResult(error=str(exc), matches=None)
|
|
192
|
+
|
|
193
|
+
def glob(self, pattern: str, path: str | None = None) -> GlobResult:
|
|
194
|
+
"""Find graph memory virtual files by glob pattern."""
|
|
195
|
+
try:
|
|
196
|
+
scope_key = self._scope_key()
|
|
197
|
+
base_path, had_graph_prefix = normalize_graph_path(path or "/")
|
|
198
|
+
internal_pattern, _pattern_had_graph_prefix = _normalize_pattern(pattern)
|
|
199
|
+
matcher = _compile_glob(internal_pattern)
|
|
200
|
+
recursive = not internal_pattern.startswith("/") or "/" in internal_pattern.strip("/") or "**" in internal_pattern
|
|
201
|
+
matches = []
|
|
202
|
+
for candidate in self._candidate_files(base_path, scope_key=scope_key, recursive=recursive):
|
|
203
|
+
if matcher(_relative_path(candidate["path"], base_path)):
|
|
204
|
+
matches.append(_prefix_file_info(candidate) if had_graph_prefix else candidate)
|
|
205
|
+
matches.sort(key=lambda item: item["path"])
|
|
206
|
+
return GlobResult(matches=matches)
|
|
207
|
+
except (GraphMemoryError, ValueError) as exc:
|
|
208
|
+
return GlobResult(error=str(exc), matches=None)
|
|
209
|
+
|
|
210
|
+
def write(self, file_path: str, content: str) -> WriteResult:
|
|
211
|
+
"""Reject writes to generated graph memory views."""
|
|
212
|
+
del content
|
|
213
|
+
return WriteResult(error=READ_ONLY_ERROR, path=file_path)
|
|
214
|
+
|
|
215
|
+
def edit(self, file_path: str, old_string: str, new_string: str, replace_all: bool = False) -> EditResult:
|
|
216
|
+
"""Reject edits to generated graph memory views."""
|
|
217
|
+
del old_string, new_string, replace_all
|
|
218
|
+
return EditResult(error=READ_ONLY_ERROR, path=file_path)
|
|
219
|
+
|
|
220
|
+
def upload_files(self, files: list[tuple[str, bytes]]) -> list[FileUploadResponse]:
|
|
221
|
+
"""Reject uploads to generated graph memory views."""
|
|
222
|
+
return [FileUploadResponse(path=path, error=READ_ONLY_ERROR) for path, _content in files]
|
|
223
|
+
|
|
224
|
+
def download_files(self, paths: list[str]) -> list[FileDownloadResponse]:
|
|
225
|
+
"""Download generated graph memory views for Deep Agents memory loading."""
|
|
226
|
+
responses = []
|
|
227
|
+
for path in paths:
|
|
228
|
+
result = self.read(path, offset=0, limit=sys.maxsize)
|
|
229
|
+
if result.error is not None or result.file_data is None:
|
|
230
|
+
error = "file_not_found" if result.error and "not found" in result.error.casefold() else result.error
|
|
231
|
+
responses.append(FileDownloadResponse(path=path, error=error))
|
|
232
|
+
continue
|
|
233
|
+
responses.append(FileDownloadResponse(path=path, content=result.file_data["content"].encode("utf-8")))
|
|
234
|
+
return responses
|
|
235
|
+
|
|
236
|
+
def add_graph_node(self, label: str, node_id: str, properties: dict[str, Any] | None = None, **metadata: Any) -> None:
|
|
237
|
+
"""Add or update a graph node through a controlled write path."""
|
|
238
|
+
validate_identifier(label, field="label")
|
|
239
|
+
validate_node_id(node_id)
|
|
240
|
+
scope_key = self._scope_key()
|
|
241
|
+
merged = merge_metadata(properties, scope_key=scope_key, metadata=metadata)
|
|
242
|
+
self.store.add_node(label, node_id, properties=merged, scope_key=scope_key)
|
|
243
|
+
|
|
244
|
+
def add_graph_edge(
|
|
245
|
+
self,
|
|
246
|
+
source_label: str,
|
|
247
|
+
source_id: str,
|
|
248
|
+
relationship: str,
|
|
249
|
+
target_label: str,
|
|
250
|
+
target_id: str,
|
|
251
|
+
properties: dict[str, Any] | None = None,
|
|
252
|
+
**metadata: Any,
|
|
253
|
+
) -> None:
|
|
254
|
+
"""Add or update a graph edge through a controlled write path."""
|
|
255
|
+
validate_identifier(source_label, field="source_label")
|
|
256
|
+
validate_identifier(target_label, field="target_label")
|
|
257
|
+
validate_identifier(relationship, field="relationship")
|
|
258
|
+
validate_node_id(source_id)
|
|
259
|
+
validate_node_id(target_id)
|
|
260
|
+
scope_key = self._scope_key()
|
|
261
|
+
merged = merge_metadata(properties, scope_key=scope_key, metadata=metadata)
|
|
262
|
+
self.store.add_edge(source_label, source_id, relationship, target_label, target_id, properties=merged, scope_key=scope_key)
|
|
263
|
+
|
|
264
|
+
def add_graph_documents(self, documents: Sequence[Any]) -> None:
|
|
265
|
+
"""Add LangChain graph documents through the configured adapter."""
|
|
266
|
+
self.store.add_graph_documents(documents, scope_key=self._scope_key())
|
|
267
|
+
|
|
268
|
+
def record_graph_trace(
|
|
269
|
+
self,
|
|
270
|
+
*,
|
|
271
|
+
situation: str,
|
|
272
|
+
rationale: str,
|
|
273
|
+
action: str,
|
|
274
|
+
outcome: str,
|
|
275
|
+
trace_id: str | None = None,
|
|
276
|
+
operation_id: str | None = None,
|
|
277
|
+
artifacts: Sequence[str] | None = None,
|
|
278
|
+
evidence: Sequence[str] | None = None,
|
|
279
|
+
evidence_refs: list[dict[str, str]] | None = None,
|
|
280
|
+
run_id: str | None = None,
|
|
281
|
+
agent_id: str | None = None,
|
|
282
|
+
subagent_id: str | None = None,
|
|
283
|
+
task_id: str | None = None,
|
|
284
|
+
subject: str | None = None,
|
|
285
|
+
observed_at: str | None = None,
|
|
286
|
+
supersedes: list[str] | None = None,
|
|
287
|
+
resolves: list[str] | None = None,
|
|
288
|
+
depends_on: list[str] | None = None,
|
|
289
|
+
finding_type: Literal["state", "interpretation"] = "interpretation",
|
|
290
|
+
**metadata: Any,
|
|
291
|
+
) -> str:
|
|
292
|
+
"""Record a Situation/Rationale/Action/Outcome reasoning trace.
|
|
293
|
+
|
|
294
|
+
Args:
|
|
295
|
+
situation: What the agent observed.
|
|
296
|
+
rationale: Why the agent chose the action.
|
|
297
|
+
action: What the agent did.
|
|
298
|
+
outcome: What happened after the action.
|
|
299
|
+
trace_id: Optional caller-provided trace id.
|
|
300
|
+
operation_id: Optional stable identity reused only for retrying this exact request.
|
|
301
|
+
artifacts: Optional files or artifacts involved in the action.
|
|
302
|
+
evidence: Optional evidence supporting the rationale or outcome.
|
|
303
|
+
evidence_refs: Optional references to captured evidence sources.
|
|
304
|
+
run_id: Optional run scope id.
|
|
305
|
+
agent_id: Optional agent id.
|
|
306
|
+
subagent_id: Optional subagent id.
|
|
307
|
+
task_id: Optional task id.
|
|
308
|
+
subject: Stable question about one entity and environment within the namespace.
|
|
309
|
+
observed_at: Time of observation, if known, as a timezone-aware ISO 8601 value.
|
|
310
|
+
supersedes: Earlier state trace IDs this evidenced observation replaces.
|
|
311
|
+
resolves: At least two same-subject traces reviewed by an evidenced resolution.
|
|
312
|
+
depends_on: Existing trace IDs whose findings support this conclusion.
|
|
313
|
+
finding_type: Whether the finding reports mutable state or an interpretation.
|
|
314
|
+
**metadata: Additional JSON-serializable metadata written to trace nodes and edges.
|
|
315
|
+
|
|
316
|
+
Returns:
|
|
317
|
+
The trace id used for the recorded graph.
|
|
318
|
+
"""
|
|
319
|
+
if operation_id is not None:
|
|
320
|
+
operation_id = _validate_trace_text(operation_id, field="operation_id")
|
|
321
|
+
if trace_id is not None:
|
|
322
|
+
raise GraphMemoryValidationError("trace_id and operation_id cannot both be supplied.")
|
|
323
|
+
trace_id = _value_id("trace", operation_id)
|
|
324
|
+
trace_id = validate_node_id(_new_trace_id() if trace_id is None else trace_id)
|
|
325
|
+
scope_key = self._scope_key()
|
|
326
|
+
trace_context = _without_none(
|
|
327
|
+
{
|
|
328
|
+
"run_id": run_id,
|
|
329
|
+
"agent_id": agent_id,
|
|
330
|
+
"subagent_id": subagent_id,
|
|
331
|
+
"task_id": task_id,
|
|
332
|
+
}
|
|
333
|
+
)
|
|
334
|
+
situation = _validate_trace_text(situation, field="situation")
|
|
335
|
+
rationale = _validate_trace_text(rationale, field="rationale")
|
|
336
|
+
action = _validate_trace_text(action, field="action")
|
|
337
|
+
outcome = _validate_trace_text(outcome, field="outcome")
|
|
338
|
+
if artifacts is not None and (not isinstance(artifacts, Sequence) or isinstance(artifacts, str | bytes)):
|
|
339
|
+
msg = "artifacts must be a sequence of strings."
|
|
340
|
+
raise GraphMemoryValidationError(msg)
|
|
341
|
+
if evidence is not None and (not isinstance(evidence, Sequence) or isinstance(evidence, str | bytes)):
|
|
342
|
+
msg = "evidence must be a sequence of strings."
|
|
343
|
+
raise GraphMemoryValidationError(msg)
|
|
344
|
+
artifacts = [_validate_trace_text(value, field="artifact") for value in artifacts or []]
|
|
345
|
+
evidence = [_validate_trace_text(value, field="evidence") for value in evidence or []]
|
|
346
|
+
evidence_refs = _normalize_evidence_refs(evidence_refs)
|
|
347
|
+
if subject is not None:
|
|
348
|
+
subject = validate_subject(subject)
|
|
349
|
+
if not isinstance(finding_type, str) or finding_type not in {"state", "interpretation"}:
|
|
350
|
+
raise GraphMemoryValidationError("finding_type must be state or interpretation.")
|
|
351
|
+
observed_at = _normalize_observed_at(observed_at) if observed_at is not None else None
|
|
352
|
+
supersedes_supplied = supersedes is not None
|
|
353
|
+
if supersedes_supplied and (not isinstance(supersedes, list) or any(not isinstance(item, str) for item in supersedes)):
|
|
354
|
+
raise GraphMemoryValidationError("supersedes must be a list of trace IDs.")
|
|
355
|
+
supersedes = list(dict.fromkeys(validate_node_id(item) for item in supersedes or []))
|
|
356
|
+
if supersedes and (subject is None or finding_type != "state" or observed_at is None or not (evidence or evidence_refs)):
|
|
357
|
+
raise GraphMemoryValidationError("supersedes requires a subject, state finding, observed_at, and evidence.")
|
|
358
|
+
resolves_supplied = resolves is not None
|
|
359
|
+
if resolves_supplied and (not isinstance(resolves, list) or any(not isinstance(item, str) for item in resolves)):
|
|
360
|
+
raise GraphMemoryValidationError("resolves must be a list of trace IDs.")
|
|
361
|
+
resolves = list(dict.fromkeys(validate_node_id(item) for item in resolves or []))
|
|
362
|
+
if resolves and (subject is None or not (evidence or evidence_refs) or len(resolves) < 2 or supersedes):
|
|
363
|
+
raise GraphMemoryValidationError("resolves requires a subject, evidence, at least two distinct traces, and no supersedes.")
|
|
364
|
+
if depends_on is not None and (not isinstance(depends_on, list) or any(not isinstance(item, str) for item in depends_on)):
|
|
365
|
+
raise GraphMemoryValidationError("depends_on must be a list of trace IDs.")
|
|
366
|
+
depends_on = list(dict.fromkeys(validate_node_id(item) for item in depends_on or []))
|
|
367
|
+
recorded_at = utc_now()
|
|
368
|
+
node_specs = [
|
|
369
|
+
("Situation", validate_node_id(f"{trace_id}-situation"), situation),
|
|
370
|
+
("Rationale", validate_node_id(f"{trace_id}-rationale"), rationale),
|
|
371
|
+
("Action", validate_node_id(f"{trace_id}-action"), action),
|
|
372
|
+
("Outcome", validate_node_id(f"{trace_id}-outcome"), outcome),
|
|
373
|
+
]
|
|
374
|
+
reserved = {
|
|
375
|
+
"kind": "reasoning_trace",
|
|
376
|
+
"situation": situation,
|
|
377
|
+
"rationale": rationale,
|
|
378
|
+
"action": action,
|
|
379
|
+
"outcome": outcome,
|
|
380
|
+
"trace_id": trace_id,
|
|
381
|
+
**trace_context,
|
|
382
|
+
}
|
|
383
|
+
if any(key in metadata for key in ("operation_id", "request_fingerprint")):
|
|
384
|
+
raise GraphMemoryValidationError("operation_id and request_fingerprint are reserved trace metadata.")
|
|
385
|
+
if any(
|
|
386
|
+
key in metadata
|
|
387
|
+
for key in ("subject", "finding_type", "observed_at", "recorded_at", "supersedes", "resolves", "depends_on", "evidence", "evidence_refs")
|
|
388
|
+
):
|
|
389
|
+
raise GraphMemoryValidationError("subject, finding, time, supersession, and evidence metadata must use their explicit arguments.")
|
|
390
|
+
for key, value in reserved.items():
|
|
391
|
+
if key in metadata and metadata[key] != value:
|
|
392
|
+
msg = f"trace metadata cannot override {key}."
|
|
393
|
+
raise GraphMemoryValidationError(msg)
|
|
394
|
+
if "text" in metadata or any(key in metadata for key in ("run_id", "agent_id", "subagent_id", "task_id") if key not in trace_context):
|
|
395
|
+
msg = "trace text and identity metadata must use their explicit arguments."
|
|
396
|
+
raise GraphMemoryValidationError(msg)
|
|
397
|
+
request_fingerprint = None
|
|
398
|
+
if operation_id is not None:
|
|
399
|
+
request = {
|
|
400
|
+
"situation": situation,
|
|
401
|
+
"rationale": rationale,
|
|
402
|
+
"action": action,
|
|
403
|
+
"outcome": outcome,
|
|
404
|
+
"artifacts": sorted(set(artifacts)),
|
|
405
|
+
"evidence": sorted(set(evidence)),
|
|
406
|
+
"evidence_refs": evidence_refs,
|
|
407
|
+
"run_id": run_id,
|
|
408
|
+
"agent_id": agent_id,
|
|
409
|
+
"subagent_id": subagent_id,
|
|
410
|
+
"task_id": task_id,
|
|
411
|
+
"subject": subject,
|
|
412
|
+
"observed_at": observed_at,
|
|
413
|
+
"supersedes": sorted(supersedes),
|
|
414
|
+
"resolves": sorted(resolves),
|
|
415
|
+
"depends_on": sorted(depends_on),
|
|
416
|
+
"finding_type": finding_type,
|
|
417
|
+
"metadata": validate_properties(metadata),
|
|
418
|
+
}
|
|
419
|
+
request_fingerprint = hashlib.sha256(json.dumps(request, sort_keys=True, ensure_ascii=False, separators=(",", ":")).encode()).hexdigest()
|
|
420
|
+
edge_metadata = {**metadata, **trace_context, "trace_id": trace_id}
|
|
421
|
+
shared_metadata = {key: metadata[key] for key in ("source", "created_by") if key in metadata}
|
|
422
|
+
trace_metadata = merge_metadata(
|
|
423
|
+
{
|
|
424
|
+
"kind": "reasoning_trace",
|
|
425
|
+
"situation": situation,
|
|
426
|
+
"rationale": rationale,
|
|
427
|
+
"action": action,
|
|
428
|
+
"outcome": outcome,
|
|
429
|
+
"recorded_at": recorded_at,
|
|
430
|
+
"evidence": evidence,
|
|
431
|
+
"evidence_refs": evidence_refs,
|
|
432
|
+
**({"operation_id": operation_id, "request_fingerprint": request_fingerprint} if operation_id is not None else {}),
|
|
433
|
+
**({"subject": subject, "finding_type": finding_type} if subject is not None else {}),
|
|
434
|
+
**({"observed_at": observed_at} if observed_at is not None else {}),
|
|
435
|
+
**({"depends_on": depends_on} if depends_on else {}),
|
|
436
|
+
**trace_context,
|
|
437
|
+
},
|
|
438
|
+
scope_key=scope_key,
|
|
439
|
+
metadata={**metadata, "source": metadata.get("source", "graph_trace")},
|
|
440
|
+
)
|
|
441
|
+
with self.store.transaction():
|
|
442
|
+
if operation_id is not None:
|
|
443
|
+
existing = self.store.get_node("Trace", trace_id, scope_key=scope_key)
|
|
444
|
+
if existing is not None:
|
|
445
|
+
if (
|
|
446
|
+
existing.properties.get("operation_id") == operation_id
|
|
447
|
+
and existing.properties.get("request_fingerprint") == request_fingerprint
|
|
448
|
+
):
|
|
449
|
+
return trace_id
|
|
450
|
+
raise GraphMemoryValidationError(f"operation_id {operation_id!r} was already used with a different request.")
|
|
451
|
+
for reviewed_id in resolves:
|
|
452
|
+
reviewed = self.store.get_node("Trace", reviewed_id, scope_key=scope_key)
|
|
453
|
+
if reviewed is None or reviewed.properties.get("subject") != subject:
|
|
454
|
+
raise GraphMemoryValidationError(f"resolved trace {reviewed_id} must exist for the same subject.")
|
|
455
|
+
for premise_id in depends_on:
|
|
456
|
+
if self.store.get_node("Trace", premise_id, scope_key=scope_key) is None:
|
|
457
|
+
raise GraphMemoryValidationError(f"dependency trace {premise_id} must exist in the same namespace.")
|
|
458
|
+
for old_id in supersedes:
|
|
459
|
+
old = self.store.get_node("Trace", old_id, scope_key=scope_key)
|
|
460
|
+
if old is None or old.properties.get("subject") != subject or old.properties.get("finding_type") != "state":
|
|
461
|
+
raise GraphMemoryValidationError(f"superseded trace {old_id} must be an existing state finding for the same subject.")
|
|
462
|
+
old_time = old.properties.get("observed_at")
|
|
463
|
+
if not isinstance(old_time, str) or _normalize_observed_at(old_time) >= observed_at:
|
|
464
|
+
raise GraphMemoryValidationError(f"superseded trace {old_id} must have an earlier observed_at.")
|
|
465
|
+
for label, node_id in [("Trace", trace_id), *((label, node_id) for label, node_id, _text in node_specs)]:
|
|
466
|
+
if self.store.get_node(label, node_id, scope_key=scope_key) is not None:
|
|
467
|
+
msg = f"trace node {label}/{node_id} already exists."
|
|
468
|
+
raise GraphMemoryValidationError(msg)
|
|
469
|
+
self.store.add_node("Trace", trace_id, properties=trace_metadata, scope_key=scope_key)
|
|
470
|
+
if subject is not None:
|
|
471
|
+
subject_id = _value_id("subject", subject)
|
|
472
|
+
existing_subject = self.store.get_node("Subject", subject_id, scope_key=scope_key)
|
|
473
|
+
if existing_subject is not None and existing_subject.properties.get("value") != subject:
|
|
474
|
+
raise GraphMemoryValidationError(f"Subject/{subject_id} has a conflicting value.")
|
|
475
|
+
if existing_subject is None:
|
|
476
|
+
self.store.add_node(
|
|
477
|
+
"Subject", subject_id, properties=merge_metadata({"value": subject}, scope_key=scope_key), scope_key=scope_key
|
|
478
|
+
)
|
|
479
|
+
self._add_trace_edge("Trace", trace_id, "ABOUT", "Subject", subject_id, scope_key=scope_key, metadata=edge_metadata)
|
|
480
|
+
for old_id in supersedes:
|
|
481
|
+
self._add_trace_edge("Trace", trace_id, "SUPERSEDES", "Trace", old_id, scope_key=scope_key, metadata=edge_metadata)
|
|
482
|
+
for reviewed_id in resolves:
|
|
483
|
+
self._add_trace_edge("Trace", trace_id, "RESOLVES", "Trace", reviewed_id, scope_key=scope_key, metadata=edge_metadata)
|
|
484
|
+
for premise_id in depends_on:
|
|
485
|
+
self._add_trace_edge("Trace", trace_id, "BASED_ON", "Trace", premise_id, scope_key=scope_key, metadata=edge_metadata)
|
|
486
|
+
|
|
487
|
+
for label, node_id, text in node_specs:
|
|
488
|
+
self.store.add_node(
|
|
489
|
+
label,
|
|
490
|
+
node_id,
|
|
491
|
+
properties=merge_metadata(
|
|
492
|
+
{
|
|
493
|
+
"text": text,
|
|
494
|
+
"trace_id": trace_id,
|
|
495
|
+
**trace_context,
|
|
496
|
+
},
|
|
497
|
+
scope_key=scope_key,
|
|
498
|
+
metadata=edge_metadata,
|
|
499
|
+
),
|
|
500
|
+
scope_key=scope_key,
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
self._add_trace_edge(
|
|
504
|
+
"Trace", trace_id, "HAS_SITUATION", "Situation", f"{trace_id}-situation", scope_key=scope_key, metadata=edge_metadata
|
|
505
|
+
)
|
|
506
|
+
self._add_trace_edge(
|
|
507
|
+
"Trace", trace_id, "HAS_RATIONALE", "Rationale", f"{trace_id}-rationale", scope_key=scope_key, metadata=edge_metadata
|
|
508
|
+
)
|
|
509
|
+
self._add_trace_edge("Trace", trace_id, "HAS_ACTION", "Action", f"{trace_id}-action", scope_key=scope_key, metadata=edge_metadata)
|
|
510
|
+
self._add_trace_edge("Trace", trace_id, "HAS_OUTCOME", "Outcome", f"{trace_id}-outcome", scope_key=scope_key, metadata=edge_metadata)
|
|
511
|
+
self._add_trace_edge(
|
|
512
|
+
"Situation",
|
|
513
|
+
f"{trace_id}-situation",
|
|
514
|
+
"LED_TO",
|
|
515
|
+
"Rationale",
|
|
516
|
+
f"{trace_id}-rationale",
|
|
517
|
+
scope_key=scope_key,
|
|
518
|
+
metadata=edge_metadata,
|
|
519
|
+
)
|
|
520
|
+
self._add_trace_edge(
|
|
521
|
+
"Rationale",
|
|
522
|
+
f"{trace_id}-rationale",
|
|
523
|
+
"JUSTIFIED",
|
|
524
|
+
"Action",
|
|
525
|
+
f"{trace_id}-action",
|
|
526
|
+
scope_key=scope_key,
|
|
527
|
+
metadata=edge_metadata,
|
|
528
|
+
)
|
|
529
|
+
self._add_trace_edge(
|
|
530
|
+
"Action", f"{trace_id}-action", "PRODUCED", "Outcome", f"{trace_id}-outcome", scope_key=scope_key, metadata=edge_metadata
|
|
531
|
+
)
|
|
532
|
+
|
|
533
|
+
self._link_scope_node("Run", run_id, "HAS_TRACE", trace_id, scope_key=scope_key, metadata=edge_metadata)
|
|
534
|
+
self._link_scope_node("Agent", agent_id, "RECORDED", trace_id, scope_key=scope_key, metadata=edge_metadata)
|
|
535
|
+
self._link_scope_node("Subagent", subagent_id, "RECORDED", trace_id, scope_key=scope_key, metadata=edge_metadata)
|
|
536
|
+
self._link_scope_node("Task", task_id, "HAS_TRACE", trace_id, scope_key=scope_key, metadata=edge_metadata)
|
|
537
|
+
|
|
538
|
+
for artifact_text in artifacts:
|
|
539
|
+
artifact_id = _value_id("artifact", artifact_text)
|
|
540
|
+
existing_artifact = self.store.get_node("Artifact", artifact_id, scope_key=scope_key)
|
|
541
|
+
if existing_artifact is not None and existing_artifact.properties.get("value") != artifact_text:
|
|
542
|
+
msg = f"Artifact/{artifact_id} has a conflicting value."
|
|
543
|
+
raise GraphMemoryValidationError(msg)
|
|
544
|
+
if existing_artifact is None:
|
|
545
|
+
self.store.add_node(
|
|
546
|
+
"Artifact",
|
|
547
|
+
artifact_id,
|
|
548
|
+
properties=merge_metadata({"value": artifact_text}, scope_key=scope_key, metadata=shared_metadata),
|
|
549
|
+
scope_key=scope_key,
|
|
550
|
+
)
|
|
551
|
+
self._add_trace_edge("Trace", trace_id, "INVOLVED", "Artifact", artifact_id, scope_key=scope_key, metadata=edge_metadata)
|
|
552
|
+
self._add_trace_edge("Action", f"{trace_id}-action", "INVOLVED", "Artifact", artifact_id, scope_key=scope_key, metadata=edge_metadata)
|
|
553
|
+
|
|
554
|
+
for evidence_text in evidence:
|
|
555
|
+
evidence_id = _value_id("evidence", evidence_text)
|
|
556
|
+
existing_evidence = self.store.get_node("Evidence", evidence_id, scope_key=scope_key)
|
|
557
|
+
if existing_evidence is not None and existing_evidence.properties.get("value") != evidence_text:
|
|
558
|
+
msg = f"Evidence/{evidence_id} has a conflicting value."
|
|
559
|
+
raise GraphMemoryValidationError(msg)
|
|
560
|
+
if existing_evidence is None:
|
|
561
|
+
self.store.add_node(
|
|
562
|
+
"Evidence",
|
|
563
|
+
evidence_id,
|
|
564
|
+
properties=merge_metadata({"value": evidence_text}, scope_key=scope_key, metadata=shared_metadata),
|
|
565
|
+
scope_key=scope_key,
|
|
566
|
+
)
|
|
567
|
+
self._add_trace_edge(
|
|
568
|
+
"Evidence", evidence_id, "SUPPORTS", "Rationale", f"{trace_id}-rationale", scope_key=scope_key, metadata=edge_metadata
|
|
569
|
+
)
|
|
570
|
+
self._add_trace_edge(
|
|
571
|
+
"Evidence", evidence_id, "SUPPORTS", "Outcome", f"{trace_id}-outcome", scope_key=scope_key, metadata=edge_metadata
|
|
572
|
+
)
|
|
573
|
+
for ref in evidence_refs:
|
|
574
|
+
source_id = _value_id("evidence-source", ref["source_id"])
|
|
575
|
+
identity = {key: ref[key] for key in ("source_id", "locator", "revision", "observed_at") if key in ref}
|
|
576
|
+
source = self.store.get_node("EvidenceSource", source_id, scope_key=scope_key)
|
|
577
|
+
if source is not None and any(
|
|
578
|
+
source.properties.get(key) != identity.get(key) for key in ("source_id", "locator", "revision", "observed_at")
|
|
579
|
+
):
|
|
580
|
+
raise GraphMemoryValidationError(f"EvidenceSource/{source_id} has conflicting identity metadata.")
|
|
581
|
+
if source is None:
|
|
582
|
+
self.store.add_node("EvidenceSource", source_id, properties=merge_metadata(identity, scope_key=scope_key), scope_key=scope_key)
|
|
583
|
+
citation_metadata = {key: value for key, value in edge_metadata.items() if key != "summary"}
|
|
584
|
+
if "summary" in ref:
|
|
585
|
+
citation_metadata["summary"] = ref["summary"]
|
|
586
|
+
self._add_trace_edge("Trace", trace_id, "CITES", "EvidenceSource", source_id, scope_key=scope_key, metadata=citation_metadata)
|
|
587
|
+
|
|
588
|
+
return trace_id
|
|
589
|
+
|
|
590
|
+
def recall_graph_memory(
|
|
591
|
+
self,
|
|
592
|
+
query: str,
|
|
593
|
+
*,
|
|
594
|
+
anchors: Sequence[str] | None = None,
|
|
595
|
+
mode: RecallMode = "auto",
|
|
596
|
+
token_budget: int = 2000,
|
|
597
|
+
max_depth: int = 3,
|
|
598
|
+
max_nodes: int = 50,
|
|
599
|
+
max_edges: int = 100,
|
|
600
|
+
) -> str:
|
|
601
|
+
"""Recall relevant graph memory with adaptive traversal.
|
|
602
|
+
|
|
603
|
+
Args:
|
|
604
|
+
query: Natural-language recall query.
|
|
605
|
+
anchors: Optional concrete starting hints such as file paths, run ids, task ids, or subagent ids.
|
|
606
|
+
mode: Recall expansion mode. `auto` expands while relevant, `local` reads one hop, and `deep` expands to `max_depth`.
|
|
607
|
+
token_budget: Approximate output token budget.
|
|
608
|
+
max_depth: Maximum traversal depth.
|
|
609
|
+
max_nodes: Maximum nodes to include.
|
|
610
|
+
max_edges: Maximum edges to include.
|
|
611
|
+
|
|
612
|
+
Returns:
|
|
613
|
+
Compact markdown with source graph paths.
|
|
614
|
+
"""
|
|
615
|
+
return _recall_graph_memory(
|
|
616
|
+
self.store,
|
|
617
|
+
query,
|
|
618
|
+
scope_key=self._scope_key(),
|
|
619
|
+
anchors=anchors,
|
|
620
|
+
mode=mode,
|
|
621
|
+
token_budget=token_budget,
|
|
622
|
+
max_depth=max_depth,
|
|
623
|
+
max_nodes=max_nodes,
|
|
624
|
+
max_edges=max_edges,
|
|
625
|
+
)
|
|
626
|
+
|
|
627
|
+
def _ls_internal(self, normalized: str, *, scope_key: str | None) -> list[FileInfo]:
|
|
628
|
+
normalized = normalized.rstrip("/") or "/"
|
|
629
|
+
if normalized == "/":
|
|
630
|
+
return [
|
|
631
|
+
{"path": "/index.md", "is_dir": False},
|
|
632
|
+
{"path": "/nodes/", "is_dir": True},
|
|
633
|
+
{"path": "/schema.md", "is_dir": False},
|
|
634
|
+
{"path": "/search/", "is_dir": True},
|
|
635
|
+
]
|
|
636
|
+
if normalized == "/nodes":
|
|
637
|
+
result = self.store.list_labels(scope_key=scope_key, limit=self.max_nodes)
|
|
638
|
+
entries = [{"path": f"/nodes/{label}/", "is_dir": True} for label in result.items]
|
|
639
|
+
elif normalized.startswith("/nodes/") and normalized.count("/") == 2:
|
|
640
|
+
label = validate_identifier(normalized.rsplit("/", 1)[-1], field="label")
|
|
641
|
+
result = self.store.list_node_ids(label, scope_key=scope_key, limit=self.max_nodes)
|
|
642
|
+
entries = [{"path": node_path(label, node_id), "is_dir": False} for node_id in result.items]
|
|
643
|
+
else:
|
|
644
|
+
return []
|
|
645
|
+
if result.truncated:
|
|
646
|
+
raise GraphMemoryValidationError(f"Directory listing exceeds max_nodes={self.max_nodes}. Increase max_nodes or use a known node path.")
|
|
647
|
+
return sorted(entries, key=lambda entry: entry["path"])
|
|
648
|
+
|
|
649
|
+
def _candidate_files(self, path: str, *, scope_key: str | None, recursive: bool = True) -> list[FileInfo]:
|
|
650
|
+
if path.endswith(".md"):
|
|
651
|
+
parsed = parse_graph_path(path)
|
|
652
|
+
if parsed.kind == "node":
|
|
653
|
+
assert parsed.label is not None and parsed.node_id is not None
|
|
654
|
+
if self.store.get_node(parsed.label, parsed.node_id, scope_key=scope_key) is None:
|
|
655
|
+
return []
|
|
656
|
+
return [{"path": parsed.path, "is_dir": False}]
|
|
657
|
+
files: list[FileInfo] = []
|
|
658
|
+
for entry in self._ls_internal(path, scope_key=scope_key):
|
|
659
|
+
if entry.get("is_dir"):
|
|
660
|
+
if recursive:
|
|
661
|
+
files.extend(self._candidate_files(entry["path"], scope_key=scope_key))
|
|
662
|
+
else:
|
|
663
|
+
files.append(entry)
|
|
664
|
+
return sorted(files, key=lambda entry: entry["path"])
|
|
665
|
+
|
|
666
|
+
def _scope_key(self) -> str | None:
|
|
667
|
+
namespace = self.namespace
|
|
668
|
+
if namespace is None:
|
|
669
|
+
return None
|
|
670
|
+
if isinstance(namespace, str):
|
|
671
|
+
return validate_namespace((namespace,))[0]
|
|
672
|
+
if isinstance(namespace, Sequence) and not callable(namespace):
|
|
673
|
+
return "|".join(validate_namespace(tuple(namespace)))
|
|
674
|
+
if not callable(namespace):
|
|
675
|
+
msg = "namespace must be a string, sequence of strings, or factory."
|
|
676
|
+
raise GraphMemoryValidationError(msg)
|
|
677
|
+
runtime = _get_runtime_or_none()
|
|
678
|
+
try:
|
|
679
|
+
resolved = namespace(runtime)
|
|
680
|
+
except Exception as exc:
|
|
681
|
+
msg = f"namespace factory failed: {exc}"
|
|
682
|
+
raise GraphMemoryValidationError(msg) from exc
|
|
683
|
+
if not isinstance(resolved, tuple):
|
|
684
|
+
msg = "namespace factory must return a tuple of strings."
|
|
685
|
+
raise GraphMemoryValidationError(msg)
|
|
686
|
+
return "|".join(validate_namespace(resolved))
|
|
687
|
+
|
|
688
|
+
def _add_trace_edge(
|
|
689
|
+
self,
|
|
690
|
+
source_label: str,
|
|
691
|
+
source_id: str,
|
|
692
|
+
relationship: str,
|
|
693
|
+
target_label: str,
|
|
694
|
+
target_id: str,
|
|
695
|
+
*,
|
|
696
|
+
scope_key: str | None,
|
|
697
|
+
metadata: dict[str, Any],
|
|
698
|
+
) -> None:
|
|
699
|
+
properties = merge_metadata({}, scope_key=scope_key, metadata=metadata)
|
|
700
|
+
self.store.add_edge(source_label, source_id, relationship, target_label, target_id, properties=properties, scope_key=scope_key)
|
|
701
|
+
|
|
702
|
+
def _link_scope_node(
|
|
703
|
+
self,
|
|
704
|
+
label: str,
|
|
705
|
+
node_id: str | None,
|
|
706
|
+
relationship: str,
|
|
707
|
+
trace_id: str,
|
|
708
|
+
*,
|
|
709
|
+
scope_key: str | None,
|
|
710
|
+
metadata: dict[str, Any],
|
|
711
|
+
) -> None:
|
|
712
|
+
if node_id is None:
|
|
713
|
+
return
|
|
714
|
+
validate_node_id(node_id)
|
|
715
|
+
if self.store.get_node(label, node_id, scope_key=scope_key) is None:
|
|
716
|
+
node_metadata = {key: metadata[key] for key in ("source", "created_by") if key in metadata}
|
|
717
|
+
self.store.add_node(label, node_id, properties=merge_metadata({}, scope_key=scope_key, metadata=node_metadata), scope_key=scope_key)
|
|
718
|
+
self._add_trace_edge(label, node_id, relationship, "Trace", trace_id, scope_key=scope_key, metadata=metadata)
|
|
719
|
+
|
|
720
|
+
|
|
721
|
+
def _get_runtime_or_none() -> Any | None:
|
|
722
|
+
try:
|
|
723
|
+
from langgraph.runtime import get_runtime
|
|
724
|
+
except ImportError:
|
|
725
|
+
return None
|
|
726
|
+
try:
|
|
727
|
+
return get_runtime()
|
|
728
|
+
except (RuntimeError, KeyError):
|
|
729
|
+
return None
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
def _prefix_file_info(info: FileInfo) -> FileInfo:
|
|
733
|
+
return {**info, "path": _maybe_prefix(info["path"], True)}
|
|
734
|
+
|
|
735
|
+
|
|
736
|
+
def _maybe_prefix(path: str, use_prefix: bool) -> str:
|
|
737
|
+
if not use_prefix:
|
|
738
|
+
return path
|
|
739
|
+
if path == "/":
|
|
740
|
+
return "/graph/"
|
|
741
|
+
return f"/graph{path}"
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def _relative_path(candidate_path: str, base_path: str) -> str:
|
|
745
|
+
if candidate_path == base_path:
|
|
746
|
+
return candidate_path.rsplit("/", 1)[-1]
|
|
747
|
+
return candidate_path.removeprefix(base_path.rstrip("/") + "/")
|
|
748
|
+
|
|
749
|
+
|
|
750
|
+
def _compile_glob(pattern: str) -> Callable[[str], bool]:
|
|
751
|
+
# Keep validation and POSIX matching consistent across Deep Agents versions and hosts.
|
|
752
|
+
if ".." in pattern.replace("\\", "/").split("/"):
|
|
753
|
+
raise GraphMemoryValidationError("Path traversal not allowed in glob patterns.")
|
|
754
|
+
try:
|
|
755
|
+
compiled = wcglob.compile(pattern.lstrip("/"), flags=wcglob.BRACE | wcglob.GLOBSTAR | wcglob.FORCEUNIX)
|
|
756
|
+
except Exception as exc:
|
|
757
|
+
# wcmatch exposes only private exception types for pattern expansion limits.
|
|
758
|
+
raise GraphMemoryValidationError(f"Invalid glob pattern: {exc}") from exc
|
|
759
|
+
return lambda path: bool(compiled.match(path if "/" in pattern else path.rsplit("/", 1)[-1]))
|
|
760
|
+
|
|
761
|
+
|
|
762
|
+
def _normalize_pattern(pattern: str) -> tuple[str, bool]:
|
|
763
|
+
if pattern.startswith("/graph/"):
|
|
764
|
+
return f"/{pattern.removeprefix('/graph/')}", True
|
|
765
|
+
if pattern == "/graph":
|
|
766
|
+
return "/", True
|
|
767
|
+
return pattern, False
|
|
768
|
+
|
|
769
|
+
|
|
770
|
+
def _new_trace_id() -> str:
|
|
771
|
+
return f"trace-{uuid4().hex}"
|
|
772
|
+
|
|
773
|
+
|
|
774
|
+
def _value_id(prefix: str, value: str) -> str:
|
|
775
|
+
digest = hashlib.sha256(value.encode("utf-8")).hexdigest()
|
|
776
|
+
return f"{prefix}-{digest}"
|
|
777
|
+
|
|
778
|
+
|
|
779
|
+
def _validate_trace_text(value: str, *, field: str) -> str:
|
|
780
|
+
if not isinstance(value, str):
|
|
781
|
+
msg = f"{field} must be a string."
|
|
782
|
+
raise GraphMemoryValidationError(msg)
|
|
783
|
+
normalized = value.strip()
|
|
784
|
+
if not normalized:
|
|
785
|
+
msg = f"{field} must not be empty."
|
|
786
|
+
raise GraphMemoryValidationError(msg)
|
|
787
|
+
if _has_unsafe_control_char(normalized):
|
|
788
|
+
msg = f"{field} must not contain NUL bytes or unsafe control characters."
|
|
789
|
+
raise GraphMemoryValidationError(msg)
|
|
790
|
+
return normalized
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
def _normalize_evidence_refs(refs: list[dict[str, str]] | None) -> list[dict[str, str]]:
|
|
794
|
+
if refs is None:
|
|
795
|
+
return []
|
|
796
|
+
if not isinstance(refs, list) or len(refs) > 50:
|
|
797
|
+
raise GraphMemoryValidationError("evidence_refs must be a list of at most 50 references.")
|
|
798
|
+
limits = {"source_id": 512, "locator": 2048, "revision": 512, "observed_at": 64, "summary": 1000}
|
|
799
|
+
normalized: dict[str, dict[str, str]] = {}
|
|
800
|
+
for ref in refs:
|
|
801
|
+
if not isinstance(ref, dict) or not {"source_id", "locator"} <= ref.keys() or ref.keys() - limits.keys():
|
|
802
|
+
raise GraphMemoryValidationError("evidence_refs require source_id and locator and allow only revision, observed_at, and summary.")
|
|
803
|
+
item: dict[str, str] = {}
|
|
804
|
+
for key, limit in limits.items():
|
|
805
|
+
if key not in ref:
|
|
806
|
+
continue
|
|
807
|
+
value = _validate_trace_text(ref[key], field=f"evidence_refs.{key}")
|
|
808
|
+
if len(value) > limit:
|
|
809
|
+
raise GraphMemoryValidationError(f"evidence_refs.{key} must be at most {limit} characters.")
|
|
810
|
+
item[key] = _normalize_observed_at(value) if key == "observed_at" else value
|
|
811
|
+
previous = normalized.get(item["source_id"])
|
|
812
|
+
if previous is not None and previous != item:
|
|
813
|
+
raise GraphMemoryValidationError(f"evidence_refs has conflicting references for source_id {item['source_id']!r}.")
|
|
814
|
+
normalized[item["source_id"]] = item
|
|
815
|
+
return [normalized[source_id] for source_id in sorted(normalized)]
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def _normalize_observed_at(value: str) -> str:
|
|
819
|
+
if not isinstance(value, str):
|
|
820
|
+
raise GraphMemoryValidationError("observed_at must be a timezone-aware ISO 8601 string.")
|
|
821
|
+
try:
|
|
822
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
823
|
+
except ValueError as exc:
|
|
824
|
+
raise GraphMemoryValidationError("observed_at must be a timezone-aware ISO 8601 string.") from exc
|
|
825
|
+
if parsed.tzinfo is None or parsed.utcoffset() is None:
|
|
826
|
+
raise GraphMemoryValidationError("observed_at must be a timezone-aware ISO 8601 string.")
|
|
827
|
+
try:
|
|
828
|
+
return parsed.astimezone(UTC).isoformat()
|
|
829
|
+
except OverflowError as exc:
|
|
830
|
+
raise GraphMemoryValidationError("observed_at is outside the supported datetime range.") from exc
|
|
831
|
+
|
|
832
|
+
|
|
833
|
+
def _has_unsafe_control_char(value: str) -> bool:
|
|
834
|
+
return any(ord(char) < 32 and char not in TRACE_TEXT_ALLOWED_CONTROL_CHARS for char in value)
|
|
835
|
+
|
|
836
|
+
|
|
837
|
+
def _without_none(values: dict[str, Any]) -> dict[str, Any]:
|
|
838
|
+
return {key: value for key, value in values.items() if value is not None}
|