langgraph-goodmem 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,31 @@
1
+ """LangGraph integration for GoodMem vector-based memory storage and retrieval."""
2
+
3
+ from langgraph_goodmem._client import GoodMemClient
4
+ from langgraph_goodmem.tools import (
5
+ GoodMemCreateMemory,
6
+ GoodMemCreateSpace,
7
+ GoodMemDeleteMemory,
8
+ GoodMemDeleteSpace,
9
+ GoodMemGetMemory,
10
+ GoodMemGetSpace,
11
+ GoodMemListEmbedders,
12
+ GoodMemListMemories,
13
+ GoodMemListSpaces,
14
+ GoodMemRetrieveMemories,
15
+ GoodMemUpdateSpace,
16
+ )
17
+
18
+ __all__ = [
19
+ "GoodMemClient",
20
+ "GoodMemCreateMemory",
21
+ "GoodMemCreateSpace",
22
+ "GoodMemDeleteMemory",
23
+ "GoodMemDeleteSpace",
24
+ "GoodMemGetMemory",
25
+ "GoodMemGetSpace",
26
+ "GoodMemListEmbedders",
27
+ "GoodMemListMemories",
28
+ "GoodMemListSpaces",
29
+ "GoodMemRetrieveMemories",
30
+ "GoodMemUpdateSpace",
31
+ ]
@@ -0,0 +1,665 @@
1
+ """HTTP client for the GoodMem API."""
2
+
3
+ import base64
4
+ import json
5
+ import mimetypes
6
+ import time
7
+ from typing import Any
8
+
9
+ import httpx
10
+
11
+
12
+ class GoodMemClient:
13
+ """Low-level HTTP client for communicating with the GoodMem API.
14
+
15
+ Handles authentication, URL normalization, and common request patterns
16
+ used by all GoodMem tools.
17
+
18
+ Args:
19
+ base_url: The base URL of the GoodMem API server.
20
+ api_key: The API key for authentication via `X-API-Key` header.
21
+ timeout: Request timeout in seconds.
22
+ verify_ssl: Whether to verify SSL certificates.
23
+ """
24
+
25
+ def __init__(
26
+ self,
27
+ *,
28
+ base_url: str,
29
+ api_key: str,
30
+ timeout: float = 30.0,
31
+ verify_ssl: bool = True,
32
+ ) -> None:
33
+ self._base_url = base_url.rstrip("/")
34
+ self._api_key = api_key
35
+ self._timeout = timeout
36
+ self._verify_ssl = verify_ssl
37
+
38
+ @property
39
+ def base_url(self) -> str:
40
+ """The normalized base URL."""
41
+ return self._base_url
42
+
43
+ def _headers(self, *, content_type: str = "application/json") -> dict[str, str]:
44
+ """Build common request headers.
45
+
46
+ Args:
47
+ content_type: The Content-Type header value.
48
+
49
+ Returns:
50
+ Dictionary of HTTP headers.
51
+ """
52
+ return {
53
+ "X-API-Key": self._api_key,
54
+ "Content-Type": content_type,
55
+ "Accept": "application/json",
56
+ }
57
+
58
+ def _ndjson_headers(self) -> dict[str, str]:
59
+ """Build headers for NDJSON streaming requests.
60
+
61
+ Returns:
62
+ Dictionary of HTTP headers with NDJSON accept type.
63
+ """
64
+ return {
65
+ "X-API-Key": self._api_key,
66
+ "Content-Type": "application/json",
67
+ "Accept": "application/x-ndjson",
68
+ }
69
+
70
+ def _url(self, path: str) -> str:
71
+ """Build a full URL from a path.
72
+
73
+ Args:
74
+ path: The API path, e.g. `/v1/spaces`.
75
+
76
+ Returns:
77
+ The full URL.
78
+ """
79
+ return f"{self._base_url}{path}"
80
+
81
+ def _client(self) -> httpx.Client:
82
+ """Create a new synchronous HTTP client.
83
+
84
+ Returns:
85
+ A configured `httpx.Client` instance.
86
+ """
87
+ return httpx.Client(timeout=self._timeout, verify=self._verify_ssl)
88
+
89
+ # -- Space operations --
90
+
91
+ _DEFAULT_CHUNK_SIZE: int = 512
92
+ _DEFAULT_CHUNK_OVERLAP: int = 50
93
+
94
+ def create_space(
95
+ self,
96
+ *,
97
+ name: str,
98
+ embedder_id: str,
99
+ chunking_strategy: str = "recursive",
100
+ chunk_size: int = 512,
101
+ chunk_overlap: int = 50,
102
+ ) -> dict[str, Any]:
103
+ """Create a new space or return an existing one with the same name.
104
+
105
+ First lists existing spaces to check for a name match. If found,
106
+ returns the existing space info. Otherwise creates a new space.
107
+
108
+ Args:
109
+ name: The name of the space.
110
+ embedder_id: The ID of the embedder to associate with the space.
111
+ chunking_strategy: The chunking strategy for the space.
112
+ One of `recursive`, `sentence`, or `none`.
113
+ chunk_size: The maximum chunk size in characters for recursive or
114
+ sentence strategies.
115
+ chunk_overlap: The overlap between consecutive chunks in characters.
116
+
117
+ Returns:
118
+ A dictionary with space info and a `reused` flag.
119
+
120
+ Raises:
121
+ httpx.HTTPStatusError: If the API returns an error status.
122
+ """
123
+ # Check for existing space with the same name
124
+ try:
125
+ spaces = self.list_spaces()
126
+ for space in spaces:
127
+ if space.get("name") == name:
128
+ return {
129
+ "success": True,
130
+ "spaceId": space["spaceId"],
131
+ "name": space["name"],
132
+ "embedderId": embedder_id,
133
+ "message": "Space already exists, reusing existing space",
134
+ "reused": True,
135
+ }
136
+ except httpx.HTTPStatusError:
137
+ pass # If listing fails, proceed to create
138
+
139
+ if chunking_strategy == "none":
140
+ chunking_config: dict[str, Any] = {"none": {}}
141
+ else:
142
+ chunking_config = {
143
+ chunking_strategy: {
144
+ "chunkSize": chunk_size,
145
+ "chunkOverlap": chunk_overlap,
146
+ },
147
+ }
148
+
149
+ with self._client() as client:
150
+ response = client.post(
151
+ self._url("/v1/spaces"),
152
+ headers=self._headers(),
153
+ json={
154
+ "name": name,
155
+ "spaceEmbedders": [{"embedderId": embedder_id}],
156
+ "defaultChunkingConfig": chunking_config,
157
+ },
158
+ )
159
+ response.raise_for_status()
160
+ body = response.json()
161
+
162
+ return {
163
+ "success": True,
164
+ "spaceId": body["spaceId"],
165
+ "name": body["name"],
166
+ "embedderId": embedder_id,
167
+ "message": "Space created successfully",
168
+ "reused": False,
169
+ }
170
+
171
+ def list_spaces(self) -> list[dict[str, Any]]:
172
+ """List all spaces.
173
+
174
+ Returns:
175
+ A list of space dictionaries.
176
+
177
+ Raises:
178
+ httpx.HTTPStatusError: If the API returns an error status.
179
+ """
180
+ with self._client() as client:
181
+ response = client.get(
182
+ self._url("/v1/spaces"),
183
+ headers=self._headers(),
184
+ )
185
+ response.raise_for_status()
186
+ body = response.json()
187
+
188
+ if isinstance(body, list):
189
+ return body
190
+ return body.get("spaces", [])
191
+
192
+ def get_space(self, *, space_id: str) -> dict[str, Any]:
193
+ """Fetch a space by ID.
194
+
195
+ Args:
196
+ space_id: The UUID of the space.
197
+
198
+ Returns:
199
+ The space dictionary.
200
+
201
+ Raises:
202
+ httpx.HTTPStatusError: If the API returns an error status.
203
+ """
204
+ with self._client() as client:
205
+ response = client.get(
206
+ self._url(f"/v1/spaces/{space_id}"),
207
+ headers=self._headers(),
208
+ )
209
+ response.raise_for_status()
210
+ return response.json()
211
+
212
+ def update_space(
213
+ self,
214
+ *,
215
+ space_id: str,
216
+ name: str | None = None,
217
+ public_read: bool | None = None,
218
+ replace_labels: dict[str, str] | None = None,
219
+ merge_labels: dict[str, str] | None = None,
220
+ ) -> dict[str, Any]:
221
+ """Update mutable fields on a space.
222
+
223
+ Only `name`, `publicRead`, and labels are mutable. Fields not provided
224
+ are left unchanged.
225
+
226
+ Args:
227
+ space_id: The UUID of the space to update.
228
+ name: New name for the space.
229
+ public_read: Whether the space should be public-read.
230
+ replace_labels: If provided, replaces the entire label set.
231
+ merge_labels: If provided, merges into the existing label set.
232
+
233
+ Returns:
234
+ The updated space dictionary.
235
+
236
+ Raises:
237
+ httpx.HTTPStatusError: If the API returns an error status.
238
+ """
239
+ body: dict[str, Any] = {}
240
+ if name is not None:
241
+ body["name"] = name
242
+ if public_read is not None:
243
+ body["publicRead"] = public_read
244
+ if replace_labels is not None:
245
+ body["replaceLabels"] = replace_labels
246
+ if merge_labels is not None:
247
+ body["mergeLabels"] = merge_labels
248
+
249
+ with self._client() as client:
250
+ response = client.put(
251
+ self._url(f"/v1/spaces/{space_id}"),
252
+ headers=self._headers(),
253
+ json=body,
254
+ )
255
+ response.raise_for_status()
256
+ return response.json()
257
+
258
+ def delete_space(self, *, space_id: str) -> dict[str, Any]:
259
+ """Delete a space by ID.
260
+
261
+ Args:
262
+ space_id: The UUID of the space to delete.
263
+
264
+ Returns:
265
+ A dictionary confirming the deletion.
266
+
267
+ Raises:
268
+ httpx.HTTPStatusError: If the API returns an error status.
269
+ """
270
+ with self._client() as client:
271
+ response = client.delete(
272
+ self._url(f"/v1/spaces/{space_id}"),
273
+ headers=self._headers(),
274
+ )
275
+ response.raise_for_status()
276
+
277
+ return {
278
+ "success": True,
279
+ "spaceId": space_id,
280
+ "message": "Space deleted successfully",
281
+ }
282
+
283
+ # -- Memory operations --
284
+
285
+ def create_memory(
286
+ self,
287
+ *,
288
+ space_id: str,
289
+ text_content: str | None = None,
290
+ file_path: str | None = None,
291
+ metadata: dict[str, Any] | None = None,
292
+ ) -> dict[str, Any]:
293
+ """Create a new memory in a space from text or a file.
294
+
295
+ If both `file_path` and `text_content` are provided, the file takes
296
+ priority. The file content type is auto-detected from its extension.
297
+
298
+ Args:
299
+ space_id: The UUID of the target space.
300
+ text_content: Plain text content to store.
301
+ file_path: Local file path to upload as memory.
302
+ metadata: Optional key-value metadata as a dictionary.
303
+
304
+ Returns:
305
+ A dictionary with the created memory info.
306
+
307
+ Raises:
308
+ ValueError: If neither `text_content` nor `file_path` is provided.
309
+ httpx.HTTPStatusError: If the API returns an error status.
310
+ """
311
+ request_body: dict[str, Any] = {"spaceId": space_id}
312
+
313
+ if file_path:
314
+ mime_type, _ = mimetypes.guess_type(file_path)
315
+ if mime_type is None:
316
+ mime_type = "application/octet-stream"
317
+
318
+ with open(file_path, "rb") as f:
319
+ file_bytes = f.read()
320
+
321
+ if mime_type.startswith("text/"):
322
+ request_body["contentType"] = mime_type
323
+ request_body["originalContent"] = file_bytes.decode("utf-8")
324
+ else:
325
+ request_body["contentType"] = mime_type
326
+ request_body["originalContentB64"] = base64.b64encode(
327
+ file_bytes
328
+ ).decode("ascii")
329
+ elif text_content is not None:
330
+ request_body["contentType"] = "text/plain"
331
+ request_body["originalContent"] = text_content
332
+ else:
333
+ msg = "No content provided. Provide either text_content or file_path."
334
+ raise ValueError(msg)
335
+
336
+ if metadata:
337
+ request_body["metadata"] = metadata
338
+
339
+ with self._client() as client:
340
+ response = client.post(
341
+ self._url("/v1/memories"),
342
+ headers=self._headers(),
343
+ json=request_body,
344
+ )
345
+ response.raise_for_status()
346
+ body = response.json()
347
+
348
+ return {
349
+ "success": True,
350
+ "memoryId": body["memoryId"],
351
+ "spaceId": body["spaceId"],
352
+ "status": body.get("processingStatus", "PENDING"),
353
+ "contentType": request_body["contentType"],
354
+ "message": "Memory created successfully",
355
+ }
356
+
357
+ _CHAT_POSTPROCESSOR = "com.goodmem.retrieval.postprocess.ChatPostProcessorFactory"
358
+
359
+ def retrieve_memories(
360
+ self,
361
+ *,
362
+ query: str,
363
+ space_ids: str,
364
+ max_results: int = 5,
365
+ include_memory_definition: bool = True,
366
+ wait_for_indexing: bool = True,
367
+ reranker_id: str | None = None,
368
+ llm_id: str | None = None,
369
+ relevance_threshold: float | None = None,
370
+ llm_temperature: float | None = None,
371
+ chronological_resort: bool | None = None,
372
+ ) -> dict[str, Any]:
373
+ """Retrieve memories via semantic similarity search.
374
+
375
+ Supports polling for up to 60 seconds when `wait_for_indexing` is
376
+ enabled and no results are found initially.
377
+
378
+ When any of `reranker_id`, `llm_id`, `relevance_threshold`,
379
+ `llm_temperature`, or `chronological_resort` is provided, a
380
+ `ChatPostProcessor` stage is appended that reranks, filters,
381
+ re-sorts, and/or generates an LLM summary (`abstractReply`).
382
+
383
+ Args:
384
+ query: The natural language search query.
385
+ space_ids: Comma-separated space UUIDs to search across.
386
+ max_results: Maximum number of matching chunks to return. Also
387
+ used as the post-processor `max_results` when one is configured.
388
+ include_memory_definition: Whether to include full memory metadata.
389
+ wait_for_indexing: Whether to retry for up to 60s if no results.
390
+ reranker_id: UUID of a reranker to improve result ordering.
391
+ llm_id: UUID of an LLM that generates a contextual `abstractReply`.
392
+ relevance_threshold: Minimum relevance score (0-1) for inclusion.
393
+ llm_temperature: Creativity setting for the LLM (0-2).
394
+ chronological_resort: Reorder final results by creation time.
395
+
396
+ Returns:
397
+ A dictionary containing matched results and memory definitions.
398
+ Includes `abstractReply` when an LLM summary was generated.
399
+
400
+ Raises:
401
+ ValueError: If no valid space IDs are provided.
402
+ httpx.HTTPStatusError: If the API returns an error status.
403
+ """
404
+ space_keys = [
405
+ {"spaceId": sid.strip()} for sid in space_ids.split(",") if sid.strip()
406
+ ]
407
+ if not space_keys:
408
+ msg = "At least one valid Space ID is required."
409
+ raise ValueError(msg)
410
+
411
+ request_body: dict[str, Any] = {
412
+ "message": query,
413
+ "spaceKeys": space_keys,
414
+ "requestedSize": max_results,
415
+ "fetchMemory": include_memory_definition,
416
+ }
417
+
418
+ post_config: dict[str, Any] = {}
419
+ if reranker_id is not None:
420
+ post_config["reranker_id"] = reranker_id
421
+ if llm_id is not None:
422
+ post_config["llm_id"] = llm_id
423
+ if relevance_threshold is not None:
424
+ post_config["relevance_threshold"] = relevance_threshold
425
+ if llm_temperature is not None:
426
+ post_config["llm_temp"] = llm_temperature
427
+ if chronological_resort is not None:
428
+ post_config["chronological_resort"] = chronological_resort
429
+ if post_config:
430
+ post_config.setdefault("max_results", max_results)
431
+ request_body["postProcessor"] = {
432
+ "name": self._CHAT_POSTPROCESSOR,
433
+ "config": post_config,
434
+ }
435
+
436
+ max_wait = 60.0
437
+ poll_interval = 5.0
438
+ start = time.monotonic()
439
+ last_result: dict[str, Any] | None = None
440
+
441
+ while True:
442
+ with self._client() as client:
443
+ response = client.post(
444
+ self._url("/v1/memories:retrieve"),
445
+ headers=self._ndjson_headers(),
446
+ json=request_body,
447
+ )
448
+ response.raise_for_status()
449
+
450
+ results: list[dict[str, Any]] = []
451
+ memories: list[dict[str, Any]] = []
452
+ result_set_id = ""
453
+ abstract_reply: dict[str, Any] | None = None
454
+
455
+ response_text = response.text
456
+ for line in response_text.strip().split("\n"):
457
+ json_str = line.strip()
458
+ if not json_str:
459
+ continue
460
+ if json_str.startswith("data:"):
461
+ json_str = json_str[5:].strip()
462
+ if json_str.startswith("event:") or not json_str:
463
+ continue
464
+
465
+ try:
466
+ item = json.loads(json_str)
467
+
468
+ if item.get("resultSetBoundary"):
469
+ result_set_id = item["resultSetBoundary"].get("resultSetId", "")
470
+ elif item.get("memoryDefinition"):
471
+ memories.append(item["memoryDefinition"])
472
+ elif item.get("abstractReply"):
473
+ abstract_reply = item["abstractReply"]
474
+ elif item.get("retrievedItem"):
475
+ ri = item["retrievedItem"]
476
+ chunk_data = ri.get("chunk", {})
477
+ chunk = chunk_data.get("chunk", {})
478
+ results.append(
479
+ {
480
+ "chunkId": chunk.get("chunkId"),
481
+ "chunkText": chunk.get("chunkText"),
482
+ "memoryId": chunk.get("memoryId"),
483
+ "relevanceScore": chunk_data.get("relevanceScore"),
484
+ "memoryIndex": chunk_data.get("memoryIndex"),
485
+ }
486
+ )
487
+ except (ValueError, KeyError):
488
+ continue
489
+
490
+ last_result = {
491
+ "success": True,
492
+ "resultSetId": result_set_id,
493
+ "results": results,
494
+ "memories": memories,
495
+ "totalResults": len(results),
496
+ "query": query,
497
+ }
498
+ if abstract_reply is not None:
499
+ last_result["abstractReply"] = abstract_reply
500
+
501
+ if results or not wait_for_indexing:
502
+ return last_result
503
+
504
+ elapsed = time.monotonic() - start
505
+ if elapsed >= max_wait:
506
+ last_result["message"] = (
507
+ "No results found after waiting 60 seconds for indexing. "
508
+ "Memories may still be processing."
509
+ )
510
+ return last_result
511
+
512
+ time.sleep(poll_interval)
513
+
514
+ def get_memory(
515
+ self,
516
+ *,
517
+ memory_id: str,
518
+ include_content: bool = True,
519
+ ) -> dict[str, Any]:
520
+ """Fetch a specific memory by ID.
521
+
522
+ Args:
523
+ memory_id: The UUID of the memory.
524
+ include_content: Whether to also fetch the original document content.
525
+
526
+ Returns:
527
+ A dictionary containing the memory metadata and optionally content.
528
+
529
+ Raises:
530
+ httpx.HTTPStatusError: If the API returns an error status.
531
+ """
532
+ with self._client() as client:
533
+ response = client.get(
534
+ self._url(f"/v1/memories/{memory_id}"),
535
+ headers=self._headers(),
536
+ )
537
+ response.raise_for_status()
538
+ result: dict[str, Any] = {
539
+ "success": True,
540
+ "memory": response.json(),
541
+ }
542
+
543
+ if include_content:
544
+ try:
545
+ with self._client() as client:
546
+ content_response = client.get(
547
+ self._url(f"/v1/memories/{memory_id}/content"),
548
+ headers=self._headers(),
549
+ )
550
+ content_response.raise_for_status()
551
+ content_type = content_response.headers.get("content-type", "")
552
+ if "application/json" in content_type:
553
+ result["content"] = content_response.json()
554
+ else:
555
+ result["content"] = content_response.text
556
+ except (httpx.HTTPStatusError, ValueError) as e:
557
+ result["contentError"] = f"Failed to fetch content: {e}"
558
+
559
+ return result
560
+
561
+ def list_memories(
562
+ self,
563
+ *,
564
+ space_id: str,
565
+ max_results: int | None = None,
566
+ next_token: str | None = None,
567
+ status_filter: str | None = None,
568
+ include_content: bool = False,
569
+ filter_expression: str | None = None,
570
+ ) -> dict[str, Any]:
571
+ """List memories in a space, with optional pagination and filtering.
572
+
573
+ Args:
574
+ space_id: The UUID of the space.
575
+ max_results: Maximum results per page (clamped server-side).
576
+ next_token: Opaque pagination token from a previous response.
577
+ status_filter: Filter by processing status: PENDING, PROCESSING,
578
+ COMPLETED, FAILED.
579
+ include_content: Whether to include the original content in the
580
+ response.
581
+ filter_expression: Metadata filter expression (see GoodMem
582
+ Filter Expressions reference).
583
+
584
+ Returns:
585
+ A dictionary with `memories` list and `nextToken`.
586
+
587
+ Raises:
588
+ httpx.HTTPStatusError: If the API returns an error status.
589
+ """
590
+ params: dict[str, Any] = {}
591
+ if max_results is not None:
592
+ params["maxResults"] = max_results
593
+ if next_token is not None:
594
+ params["nextToken"] = next_token
595
+ if status_filter is not None:
596
+ params["statusFilter"] = status_filter
597
+ if include_content:
598
+ params["includeContent"] = True
599
+ if filter_expression is not None:
600
+ params["filter"] = filter_expression
601
+
602
+ with self._client() as client:
603
+ response = client.get(
604
+ self._url(f"/v1/spaces/{space_id}/memories"),
605
+ headers=self._headers(),
606
+ params=params,
607
+ )
608
+ response.raise_for_status()
609
+ body = response.json()
610
+
611
+ memories = body.get("memories", []) if isinstance(body, dict) else []
612
+ next_token_resp = body.get("nextToken") if isinstance(body, dict) else None
613
+ return {
614
+ "success": True,
615
+ "spaceId": space_id,
616
+ "memories": memories,
617
+ "totalMemories": len(memories),
618
+ "nextToken": next_token_resp,
619
+ }
620
+
621
+ def delete_memory(self, *, memory_id: str) -> dict[str, Any]:
622
+ """Delete a memory by ID.
623
+
624
+ Args:
625
+ memory_id: The UUID of the memory to delete.
626
+
627
+ Returns:
628
+ A dictionary confirming the deletion.
629
+
630
+ Raises:
631
+ httpx.HTTPStatusError: If the API returns an error status.
632
+ """
633
+ with self._client() as client:
634
+ response = client.delete(
635
+ self._url(f"/v1/memories/{memory_id}"),
636
+ headers=self._headers(),
637
+ )
638
+ response.raise_for_status()
639
+
640
+ return {
641
+ "success": True,
642
+ "memoryId": memory_id,
643
+ "message": "Memory deleted successfully",
644
+ }
645
+
646
+ def list_embedders(self) -> list[dict[str, Any]]:
647
+ """List all available embedder models.
648
+
649
+ Returns:
650
+ A list of embedder dictionaries.
651
+
652
+ Raises:
653
+ httpx.HTTPStatusError: If the API returns an error status.
654
+ """
655
+ with self._client() as client:
656
+ response = client.get(
657
+ self._url("/v1/embedders"),
658
+ headers=self._headers(),
659
+ )
660
+ response.raise_for_status()
661
+ body = response.json()
662
+
663
+ if isinstance(body, list):
664
+ return body
665
+ return body.get("embedders", [])
File without changes