hermes-memory-pgvector 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pgvector/__init__.py ADDED
@@ -0,0 +1,806 @@
1
+ """pgvector — Postgres + pgvector memory provider for hermes-agent.
2
+
3
+ Mirrors hermes-agent's built-in `memory` tool entries (MEMORY.md / USER.md
4
+ in tools/memory_tool.py) into a single Postgres table, adds 768-dim
5
+ embeddings for semantic recall, and scopes by `agent_identity` so each
6
+ named agent (marketing / sales / trading / incident / …) has its own
7
+ theme.
8
+
9
+ Design philosophy: this is a STORAGE LAYER for hermes-agent's native
10
+ memory model, not a new memory model. We don't invent facts, entities,
11
+ trust scores, deriver pipelines, or dialectic synthesis. We give the
12
+ built-in `memory` tool a durable Postgres backing + semantic search,
13
+ nothing more. Honcho went heavy and exploded; this stays lean.
14
+
15
+ Config in $HERMES_HOME/config.yaml under plugins.pgvector:
16
+
17
+ plugins:
18
+ pgvector:
19
+ dsn: "dbname=hermes_memory user=hermes host=/var/run/postgresql"
20
+ embed_url: "http://192.168.100.50:11434"
21
+ embed_model: "nomic-embed-text"
22
+ prefetch_limit: 5
23
+ min_similarity: 0.30
24
+ embed_on_write: true
25
+ scope_default: "current" # 'current' | 'all'
26
+
27
+ Tools exposed: `recall_memory` (one explicit search tool). All built-in
28
+ memory writes (add/replace/remove) are mirrored automatically via the
29
+ on_memory_write hook — no agent-facing change.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import json
35
+ import logging
36
+ import re
37
+ from pathlib import Path
38
+ from typing import Any, Dict, List, Optional
39
+
40
+ from agent.memory_provider import MemoryProvider
41
+ from tools.registry import tool_error
42
+ from hermes_cli.config import cfg_get
43
+
44
+ from .embed import embed, EmbeddingError
45
+ from .store import MemoryStore
46
+ from .writer import AsyncWriter, _PendingWrite
47
+
48
+
49
+ # Boilerplate / acknowledgement-only turns that are not worth embedding or
50
+ # storing. Case-insensitive whole-string match after strip. Combined with
51
+ # a length floor (default 40 chars) in _is_noise.
52
+ _NOISE_RE = re.compile(
53
+ r"^("
54
+ r"ok(ay)?|thanks?( you)?|thx|ty|np|"
55
+ r"yes|no|sure|got it|done|cool|nice|great|"
56
+ r"continue|please|exit|cancel|stop|quit|"
57
+ r"yeah|yep|nope|alright"
58
+ r")[\s\.\!\?]*$",
59
+ re.IGNORECASE,
60
+ )
61
+
62
+ logger = logging.getLogger(__name__)
63
+
64
+
65
+ # ---------------------------------------------------------------------------
66
+ # Tool schema — one explicit search over memory_entries
67
+ # ---------------------------------------------------------------------------
68
+
69
+ RECALL_CONVERSATION_SCHEMA = {
70
+ "name": "recall_conversation",
71
+ "description": (
72
+ "Semantic search over past chat turns (every substantive "
73
+ "user/assistant exchange across all sessions). Use this when "
74
+ "the user references something you discussed earlier — last week, "
75
+ "yesterday, in another session — and you need the actual turn "
76
+ "text, not just a durable memory entry. Returns top-K matching "
77
+ "turns with role, content, session_id, and timestamp.\n\n"
78
+ "SCOPES: 'current' (your theme — default), 'session' (current "
79
+ "session only), 'all' (every theme).\n\n"
80
+ "Skip for in-session continuity (already in your context). Skip "
81
+ "for durable facts (use recall_memory instead — that's the "
82
+ "MEMORY.md / USER.md entries the agent decided to remember)."
83
+ ),
84
+ "parameters": {
85
+ "type": "object",
86
+ "properties": {
87
+ "query": {
88
+ "type": "string",
89
+ "description": "Free-text query describing what to recall.",
90
+ },
91
+ "scope": {
92
+ "type": "string",
93
+ "description": "Theme scope: 'current', 'session', 'all', or a named agent.",
94
+ "default": "current",
95
+ },
96
+ "limit": {
97
+ "type": "integer",
98
+ "description": "Max results (1-20, default 5).",
99
+ "default": 5,
100
+ },
101
+ },
102
+ "required": ["query"],
103
+ },
104
+ }
105
+
106
+
107
+ RECALL_MEMORY_SCHEMA = {
108
+ "name": "recall_memory",
109
+ "description": (
110
+ "Semantic search over durable memory entries (the same entries the "
111
+ "built-in `memory` tool writes to MEMORY.md / USER.md, stored "
112
+ "durably in Postgres with embeddings).\n\n"
113
+ "WHEN TO USE: when the answer might be in a past memory entry that "
114
+ "is NOT already in your system prompt's memory block — older "
115
+ "entries, or entries from another named agent. The current scope's "
116
+ "recent entries are already injected ambient; only use this tool "
117
+ "for deeper / cross-scope recall.\n\n"
118
+ "SCOPES:\n"
119
+ " 'current' — your own theme (default; e.g. 'marketing')\n"
120
+ " 'all' — across all agent themes\n"
121
+ " '<name>' — a specific theme: 'marketing', 'sales', 'trading', 'incident', …"
122
+ ),
123
+ "parameters": {
124
+ "type": "object",
125
+ "properties": {
126
+ "query": {
127
+ "type": "string",
128
+ "description": "Free-text query describing what to recall.",
129
+ },
130
+ "scope": {
131
+ "type": "string",
132
+ "description": "Theme scope: 'current', 'all', or a named agent.",
133
+ "default": "current",
134
+ },
135
+ "target": {
136
+ "type": "string",
137
+ "enum": ["memory", "user", "both"],
138
+ "description": "Which store to search. Default 'both'.",
139
+ "default": "both",
140
+ },
141
+ "limit": {
142
+ "type": "integer",
143
+ "description": "Max results (1-20, default 5).",
144
+ "default": 5,
145
+ },
146
+ },
147
+ "required": ["query"],
148
+ },
149
+ }
150
+
151
+
152
+ # ---------------------------------------------------------------------------
153
+ # Config
154
+ # ---------------------------------------------------------------------------
155
+
156
+ DEFAULTS = {
157
+ "dsn": "dbname=hermes_memory user=hermes host=/var/run/postgresql connect_timeout=5",
158
+ "embed_url": "http://192.168.100.50:11434",
159
+ "embed_model": "nomic-embed-text",
160
+ "prefetch_limit": 5,
161
+ "min_similarity": 0.30,
162
+ "embed_on_write": True,
163
+ "scope_default": "current",
164
+ "write_queue_maxsize": 256,
165
+ # v0.1.1 — bulk sync MEMORY.md / USER.md on init
166
+ "bulk_sync_on_init": True,
167
+ # v0.2 — conversation turn capture
168
+ "sync_turns": True,
169
+ "turn_min_chars": 40, # turns shorter than this are noise unless > 200 chars or contain tool refs
170
+ }
171
+
172
+
173
+ def _load_plugin_config() -> dict:
174
+ try:
175
+ from hermes_constants import get_hermes_home
176
+ config_path = get_hermes_home() / "config.yaml"
177
+ if not config_path.exists():
178
+ return {}
179
+ import yaml
180
+ with open(config_path, encoding="utf-8-sig") as fh:
181
+ data = yaml.safe_load(fh) or {}
182
+ return cfg_get(data, "plugins", "pgvector", default={}) or {}
183
+ except Exception: # noqa: BLE001
184
+ return {}
185
+
186
+
187
+ # ---------------------------------------------------------------------------
188
+ # Provider
189
+ # ---------------------------------------------------------------------------
190
+
191
+ class PgvectorMemoryProvider(MemoryProvider):
192
+ """Postgres mirror of built-in memory entries, with semantic recall."""
193
+
194
+ def __init__(self, config: dict | None = None):
195
+ self._config = {**DEFAULTS, **(config or {})}
196
+ self._store: Optional[MemoryStore] = None
197
+ self._writer: Optional[AsyncWriter] = None
198
+ self._agent_identity: str = "default"
199
+ self._session_id: str = ""
200
+ self._healthy: bool = False
201
+ self._embed_warned: bool = False
202
+
203
+ @property
204
+ def name(self) -> str:
205
+ return "pgvector"
206
+
207
+ # -- Lifecycle -----------------------------------------------------------
208
+
209
+ def is_available(self) -> bool:
210
+ try:
211
+ import psycopg # noqa: F401
212
+ return True
213
+ except ImportError:
214
+ return False
215
+
216
+ def initialize(self, session_id: str, **kwargs) -> None:
217
+ self._session_id = session_id
218
+ # Per-agent theme scoping — priority order:
219
+ # 1. gateway_session_key — from the `X-Hermes-Session-Key` header on
220
+ # API requests. This is the EXPLICIT minion-scope signal sent by
221
+ # systemd-run callers (marketing-daily, sales-daily, intraday
222
+ # workers, …) and takes precedence over the profile fallback
223
+ # because the gateway always sets agent_identity='default' for
224
+ # API traffic — without prioritising the header, every minion
225
+ # collapses to one shared 'default' scope.
226
+ # 2. agent_identity ≠ 'default' — explicit profile name from CLI
227
+ # (`hermes --profile marketing`). Skipped when it's the
228
+ # auto-default sentinel to allow header (#1) to win.
229
+ # 3. agent_workspace — shared workspace name from some platforms.
230
+ # 4. agent_identity == 'default' — accept it now (no other source).
231
+ # 5. 'default' — last-resort bucket for unscoped traffic.
232
+ explicit_identity = kwargs.get("agent_identity")
233
+ if explicit_identity == "default":
234
+ explicit_identity = None # sentinel — let header take over
235
+ self._agent_identity = (
236
+ kwargs.get("gateway_session_key")
237
+ or explicit_identity
238
+ or kwargs.get("agent_workspace")
239
+ or kwargs.get("agent_identity") # accept 'default' if nothing else set
240
+ or "default"
241
+ )
242
+ self._store = MemoryStore(self._config["dsn"])
243
+ try:
244
+ # Schema is verify-only at runtime — admin applies the
245
+ # migration out-of-band (see plugin README install step).
246
+ self._store.ensure_schema()
247
+ health = self._store.health()
248
+ self._healthy = bool(health.get("ok"))
249
+ if not self._healthy:
250
+ logger.warning("pgvector unhealthy on init: %s", health.get("error"))
251
+ except MemoryStore.SchemaNotApplied as exc:
252
+ logger.error("pgvector schema not applied — %s", exc)
253
+ self._healthy = False
254
+ except Exception as exc: # noqa: BLE001
255
+ logger.warning("pgvector init failed: %s", exc)
256
+ self._healthy = False
257
+
258
+ # Background writer — bounded queue, lazy thread start. Decouples
259
+ # on_memory_write + sync_turn from the (potentially slow) embed +
260
+ # DB write so the agent loop never blocks on a stalled embed
261
+ # endpoint.
262
+ self._writer = AsyncWriter(
263
+ self._worker,
264
+ maxsize=int(self._config.get("write_queue_maxsize", 256)),
265
+ )
266
+
267
+ # v0.1.1: bulk import existing MEMORY.md / USER.md content so the
268
+ # plugin sees pre-plugin entries + direct file edits, not just the
269
+ # new writes captured via on_memory_write.
270
+ if self._healthy and self._config.get("bulk_sync_on_init", True):
271
+ self._bulk_sync_from_disk(kwargs.get("hermes_home"))
272
+
273
+ def shutdown(self) -> None:
274
+ # Drain the in-flight writes first so we don't drop work...
275
+ if self._writer:
276
+ self._writer.shutdown(timeout=5.0)
277
+ self._writer = None
278
+ # ...then close the pool the writer was draining into.
279
+ if self._store:
280
+ self._store.close()
281
+ self._store = None
282
+ self._healthy = False
283
+
284
+ def on_session_switch(self, new_session_id: str, **kwargs) -> None:
285
+ self._session_id = new_session_id
286
+
287
+ # -- System prompt + ambient recall --------------------------------------
288
+
289
+ def system_prompt_block(self) -> str:
290
+ if not self._healthy or not self._store:
291
+ return ""
292
+ try:
293
+ count_scoped = self._store.count(agent_identity=self._agent_identity)
294
+ count_all = self._store.count()
295
+ except Exception: # noqa: BLE001
296
+ count_scoped = count_all = 0
297
+ if count_all == 0:
298
+ return (
299
+ "# pgvector memory\n"
300
+ "Active. Empty store. Use the built-in `memory` tool to save "
301
+ "durable notes — entries are mirrored to Postgres with "
302
+ "embeddings for semantic recall across sessions."
303
+ )
304
+ return (
305
+ "# pgvector memory\n"
306
+ f"Active. {count_scoped} entries for '{self._agent_identity}', "
307
+ f"{count_all} total across all themes. "
308
+ "Use `recall_memory(query, scope='all'|'<theme>')` for deeper / "
309
+ "cross-theme recall beyond what's in the built-in memory block."
310
+ )
311
+
312
+ def prefetch(self, query: str, *, session_id: str = "") -> str:
313
+ if not self._healthy or not self._store or not query:
314
+ return ""
315
+ try:
316
+ vec = embed(
317
+ query,
318
+ base_url=self._config["embed_url"],
319
+ model=self._config["embed_model"],
320
+ )
321
+ except EmbeddingError as exc:
322
+ logger.debug("pgvector prefetch embed failed: %s", exc)
323
+ return ""
324
+
325
+ # Ambient prefetch is scoped to the current agent_identity by
326
+ # default — keeps marketing turns from polluting trading recall.
327
+ try:
328
+ rows = self._store.search(
329
+ query_embedding=vec,
330
+ agent_identity=self._agent_identity,
331
+ limit=int(self._config.get("prefetch_limit", 5)),
332
+ min_similarity=float(self._config.get("min_similarity", 0.30)),
333
+ )
334
+ except Exception as exc: # noqa: BLE001
335
+ logger.debug("pgvector prefetch query failed: %s", exc)
336
+ return ""
337
+ if not rows:
338
+ return ""
339
+
340
+ lines = [f"## Recall (pgvector, {self._agent_identity})"]
341
+ for r in rows:
342
+ score = r.get("score") or 0.0
343
+ tgt = r.get("target") or "?"
344
+ content = (r.get("content") or "").strip().replace("\n", " ")
345
+ if len(content) > 280:
346
+ content = content[:280] + "…"
347
+ lines.append(f"- [{score:.2f}] ({tgt}) {content}")
348
+ return "\n".join(lines)
349
+
350
+ # -- Turn capture (v0.2) -------------------------------------------------
351
+
352
+ def sync_turn(
353
+ self,
354
+ user_content: str,
355
+ assistant_content: str,
356
+ *,
357
+ session_id: str = "",
358
+ ) -> None:
359
+ """Persist a (user, assistant) turn pair to the conversations table.
360
+
361
+ Non-blocking — enqueues writes; the async writer drains, embeds,
362
+ and INSERTs. Boilerplate / very short turns are filtered out so
363
+ the recall table stays high-signal.
364
+ """
365
+ if not self._healthy or not self._writer:
366
+ return
367
+ if not self._config.get("sync_turns", True):
368
+ return
369
+
370
+ sid = session_id or self._session_id or "default"
371
+ min_chars = int(self._config.get("turn_min_chars", 40))
372
+
373
+ for role, content in (("user", user_content), ("assistant", assistant_content)):
374
+ if not content:
375
+ continue
376
+ if self._is_noise(content, min_chars=min_chars):
377
+ continue
378
+ self._writer.enqueue(
379
+ action="turn",
380
+ agent_identity=self._agent_identity,
381
+ target="conversations", # synthetic; worker dispatches on action
382
+ content=content,
383
+ extra={"role": role, "session_id": sid},
384
+ metadata={"session_id": sid},
385
+ )
386
+
387
+ @staticmethod
388
+ def _is_noise(content: str, *, min_chars: int) -> bool:
389
+ """True for short / boilerplate content we don't want in recall."""
390
+ stripped = (content or "").strip()
391
+ if not stripped:
392
+ return True
393
+ if len(stripped) < min_chars:
394
+ return True
395
+ if _NOISE_RE.match(stripped):
396
+ return True
397
+ return False
398
+
399
+ # -- Built-in memory mirror (THE main integration point) ----------------
400
+
401
+ def on_memory_write(
402
+ self,
403
+ action: str,
404
+ target: str,
405
+ content: str,
406
+ metadata: Optional[Dict[str, Any]] = None,
407
+ ) -> None:
408
+ """Mirror built-in `memory` tool writes to Postgres (non-blocking).
409
+
410
+ Built-in tool fires this on every add/replace/remove. We enqueue
411
+ the write; the background thread drains, embeds, and INSERTs.
412
+ Returns instantly so the agent loop never blocks on the embed
413
+ endpoint or the DB.
414
+ """
415
+ if not self._healthy or not self._writer:
416
+ return
417
+ if target not in ("memory", "user"):
418
+ logger.debug("pgvector ignoring unsupported target: %r", target)
419
+ return
420
+ if action not in ("add", "replace", "remove"):
421
+ logger.debug("pgvector ignoring unknown action: %r", action)
422
+ return
423
+
424
+ meta = dict(metadata or {})
425
+ meta.setdefault("session_id", self._session_id)
426
+ old_text = meta.get("old_text") or meta.get("replaces")
427
+
428
+ self._writer.enqueue(
429
+ action=action,
430
+ agent_identity=self._agent_identity,
431
+ target=target,
432
+ content=content,
433
+ extra={"old_text": str(old_text)} if old_text else {},
434
+ metadata=meta,
435
+ )
436
+
437
+ def _worker(self, item: "_PendingWrite") -> None:
438
+ """Drain-thread worker: embed + DB write for a single queued item.
439
+
440
+ Must NOT raise — the AsyncWriter logs + survives if we do, but
441
+ we still want failures to degrade gracefully (drop the write,
442
+ keep the queue moving).
443
+ """
444
+ if not self._store:
445
+ return
446
+ try:
447
+ if item.action == "add":
448
+ vec = self._maybe_embed(item.content)
449
+ self._store.add(
450
+ agent_identity=item.agent_identity,
451
+ target=item.target,
452
+ content=item.content,
453
+ embedding=vec,
454
+ metadata=item.metadata,
455
+ )
456
+ elif item.action == "replace":
457
+ old_text = item.extra.get("old_text")
458
+ vec = self._maybe_embed(item.content)
459
+ if old_text:
460
+ n = self._store.replace(
461
+ agent_identity=item.agent_identity,
462
+ target=item.target,
463
+ old_text=old_text,
464
+ new_content=item.content,
465
+ new_embedding=vec,
466
+ )
467
+ if n == 0:
468
+ # Nothing matched — degrade to add (built-in wrote
469
+ # the new entry to disk; mirror it).
470
+ self._store.add(
471
+ agent_identity=item.agent_identity,
472
+ target=item.target,
473
+ content=item.content,
474
+ embedding=vec,
475
+ metadata=item.metadata,
476
+ )
477
+ else:
478
+ # No old_text in metadata → can't locate prior row;
479
+ # add the new content so we don't lose it.
480
+ self._store.add(
481
+ agent_identity=item.agent_identity,
482
+ target=item.target,
483
+ content=item.content,
484
+ embedding=vec,
485
+ metadata=item.metadata,
486
+ )
487
+ elif item.action == "remove":
488
+ self._store.remove(
489
+ agent_identity=item.agent_identity,
490
+ target=item.target,
491
+ old_text=item.content,
492
+ )
493
+ elif item.action == "turn":
494
+ role = item.extra.get("role") or "user"
495
+ sid = item.extra.get("session_id") or "default"
496
+ vec = self._maybe_embed(item.content)
497
+ self._store.append_turn(
498
+ session_id=sid,
499
+ agent_identity=item.agent_identity,
500
+ role=role,
501
+ content=item.content,
502
+ embedding=vec,
503
+ metadata=item.metadata,
504
+ )
505
+ except Exception as exc: # noqa: BLE001
506
+ logger.debug(
507
+ "pgvector worker (%s/%s/%s) failed: %s",
508
+ item.action,
509
+ item.agent_identity,
510
+ item.target,
511
+ str(exc)[:200],
512
+ )
513
+
514
+ # -- Bulk sync (v0.1.1) --------------------------------------------------
515
+
516
+ def _bulk_sync_from_disk(self, hermes_home: Optional[str]) -> None:
517
+ """Import MEMORY.md + USER.md entries from disk into memory_entries.
518
+
519
+ Called by initialize(). Runs synchronously (not via async writer)
520
+ so the table is warm before the first turn's prefetch. Cheap on
521
+ re-init: an existence pre-check skips already-imported entries
522
+ without re-embedding.
523
+ """
524
+ if not self._store:
525
+ return
526
+ if not hermes_home:
527
+ # Fall back to hermes_constants if the runtime didn't pass it.
528
+ try:
529
+ from hermes_constants import get_hermes_home
530
+ hermes_home = str(get_hermes_home())
531
+ except Exception: # noqa: BLE001
532
+ return
533
+
534
+ memories_dir = Path(hermes_home) / "memories"
535
+ embed_fn = self._make_embed_fn()
536
+
537
+ for target, fname in (("memory", "MEMORY.md"), ("user", "USER.md")):
538
+ try:
539
+ result = self._store.bulk_upsert_md(
540
+ agent_identity=self._agent_identity,
541
+ target=target,
542
+ file_path=memories_dir / fname,
543
+ embed_fn=embed_fn,
544
+ )
545
+ if result.get("inserted"):
546
+ logger.info(
547
+ "pgvector bulk-sync %s: parsed=%d inserted=%d skipped=%d",
548
+ fname,
549
+ result.get("parsed", 0),
550
+ result.get("inserted", 0),
551
+ result.get("skipped", 0),
552
+ )
553
+ except Exception as exc: # noqa: BLE001
554
+ logger.warning("pgvector bulk-sync %s failed: %s", fname, exc)
555
+
556
+ def _make_embed_fn(self):
557
+ """Return a closure over the configured embed endpoint, or None."""
558
+ if not self._config.get("embed_on_write", True):
559
+ return None
560
+ base_url = self._config["embed_url"]
561
+ model = self._config["embed_model"]
562
+ def _fn(text: str):
563
+ return embed(text, base_url=base_url, model=model)
564
+ return _fn
565
+
566
+ # -- Tool surface --------------------------------------------------------
567
+
568
+ def get_tool_schemas(self) -> List[Dict[str, Any]]:
569
+ return [RECALL_MEMORY_SCHEMA, RECALL_CONVERSATION_SCHEMA]
570
+
571
+ def handle_tool_call(self, tool_name: str, args: Dict[str, Any], **kwargs) -> str:
572
+ if tool_name == "recall_conversation":
573
+ return self._handle_recall_conversation(args)
574
+ if tool_name != "recall_memory":
575
+ return tool_error(f"Unknown tool: {tool_name}")
576
+ if not self._healthy or not self._store:
577
+ return json.dumps({"results": [], "count": 0, "error": "pgvector unavailable"})
578
+
579
+ query = (args.get("query") or "").strip()
580
+ if not query:
581
+ return tool_error("Missing required arg: query")
582
+
583
+ try:
584
+ limit = max(1, min(int(args.get("limit", 5)), 20))
585
+ except (TypeError, ValueError):
586
+ limit = 5
587
+
588
+ # Scope resolution: 'current' → my agent_identity; 'all' → no filter;
589
+ # anything else → treat as explicit theme name.
590
+ scope = (args.get("scope") or self._config.get("scope_default") or "current").strip()
591
+ if scope == "current":
592
+ agent_filter: Optional[str] = self._agent_identity
593
+ elif scope == "all":
594
+ agent_filter = None
595
+ else:
596
+ agent_filter = scope
597
+
598
+ # Target resolution: 'memory'/'user'/'both'.
599
+ target_arg = (args.get("target") or "both").strip()
600
+ target_filter: Optional[str] = None if target_arg == "both" else target_arg
601
+ if target_filter not in (None, "memory", "user"):
602
+ return tool_error(f"Invalid target: {target_arg!r}")
603
+
604
+ try:
605
+ vec = embed(
606
+ query,
607
+ base_url=self._config["embed_url"],
608
+ model=self._config["embed_model"],
609
+ )
610
+ except EmbeddingError as exc:
611
+ return json.dumps({"results": [], "count": 0, "error": f"embed: {exc}"})
612
+
613
+ try:
614
+ rows = self._store.search(
615
+ query_embedding=vec,
616
+ agent_identity=agent_filter,
617
+ target=target_filter,
618
+ limit=limit,
619
+ )
620
+ except Exception as exc: # noqa: BLE001
621
+ return json.dumps({"results": [], "count": 0, "error": f"db: {exc}"})
622
+
623
+ results = []
624
+ for r in rows:
625
+ ts = r.get("updated_at") or r.get("created_at")
626
+ results.append(
627
+ {
628
+ "id": r.get("id"),
629
+ "agent_identity": r.get("agent_identity"),
630
+ "target": r.get("target"),
631
+ "ts": ts.isoformat() if ts else None,
632
+ "score": round(float(r.get("score") or 0.0), 4),
633
+ "content": (r.get("content") or "")[:2000],
634
+ }
635
+ )
636
+ return json.dumps({"results": results, "count": len(results)})
637
+
638
+ def _handle_recall_conversation(self, args: Dict[str, Any]) -> str:
639
+ """Tool handler for recall_conversation over the conversations table."""
640
+ if not self._healthy or not self._store:
641
+ return json.dumps({"results": [], "count": 0, "error": "pgvector unavailable"})
642
+
643
+ query = (args.get("query") or "").strip()
644
+ if not query:
645
+ return tool_error("Missing required arg: query")
646
+ try:
647
+ limit = max(1, min(int(args.get("limit", 5)), 20))
648
+ except (TypeError, ValueError):
649
+ limit = 5
650
+
651
+ scope = (args.get("scope") or "current").strip()
652
+ agent_filter: Optional[str] = None
653
+ session_filter: Optional[str] = None
654
+ if scope == "current":
655
+ agent_filter = self._agent_identity
656
+ elif scope == "session":
657
+ session_filter = self._session_id or None
658
+ elif scope == "all":
659
+ pass # no filters
660
+ else:
661
+ agent_filter = scope # treat as a specific theme name
662
+
663
+ try:
664
+ vec = embed(
665
+ query,
666
+ base_url=self._config["embed_url"],
667
+ model=self._config["embed_model"],
668
+ )
669
+ except EmbeddingError as exc:
670
+ return json.dumps({"results": [], "count": 0, "error": f"embed: {exc}"})
671
+
672
+ try:
673
+ rows = self._store.search_turns(
674
+ query_embedding=vec,
675
+ agent_identity=agent_filter,
676
+ session_id=session_filter,
677
+ limit=limit,
678
+ )
679
+ except Exception as exc: # noqa: BLE001
680
+ return json.dumps({"results": [], "count": 0, "error": f"db: {exc}"})
681
+
682
+ results = []
683
+ for r in rows:
684
+ ts = r.get("ts")
685
+ results.append(
686
+ {
687
+ "id": r.get("id"),
688
+ "agent_identity": r.get("agent_identity"),
689
+ "session_id": r.get("session_id"),
690
+ "role": r.get("role"),
691
+ "ts": ts.isoformat() if ts else None,
692
+ "score": round(float(r.get("score") or 0.0), 4),
693
+ "content": (r.get("content") or "")[:2000],
694
+ }
695
+ )
696
+ return json.dumps({"results": results, "count": len(results)})
697
+
698
+ # -- Setup hooks ---------------------------------------------------------
699
+
700
+ def get_config_schema(self) -> List[Dict[str, Any]]:
701
+ return [
702
+ {
703
+ "key": "dsn",
704
+ "description": "Postgres DSN (psycopg connection string)",
705
+ "default": DEFAULTS["dsn"],
706
+ "required": True,
707
+ },
708
+ {
709
+ "key": "embed_url",
710
+ "description": "Embedding endpoint base URL (OpenAI-compatible or Ollama native)",
711
+ "default": DEFAULTS["embed_url"],
712
+ "required": True,
713
+ },
714
+ {
715
+ "key": "embed_model",
716
+ "description": "Embedding model name (must return 768-dim vectors)",
717
+ "default": DEFAULTS["embed_model"],
718
+ },
719
+ {
720
+ "key": "prefetch_limit",
721
+ "description": "Max ambient recall results injected per turn",
722
+ "default": str(DEFAULTS["prefetch_limit"]),
723
+ },
724
+ {
725
+ "key": "min_similarity",
726
+ "description": "Cosine similarity cutoff for ambient prefetch (0.0–1.0)",
727
+ "default": str(DEFAULTS["min_similarity"]),
728
+ },
729
+ {
730
+ "key": "embed_on_write",
731
+ "description": "Compute embedding on each write; turn off for text-only mode",
732
+ "default": "true",
733
+ "choices": ["true", "false"],
734
+ },
735
+ {
736
+ "key": "scope_default",
737
+ "description": "Default scope for recall_memory when caller omits it",
738
+ "default": DEFAULTS["scope_default"],
739
+ "choices": ["current", "all"],
740
+ },
741
+ {
742
+ "key": "write_queue_maxsize",
743
+ "description": "Bounded async-writer queue size; full = oldest writes drop with a warning",
744
+ "default": str(DEFAULTS["write_queue_maxsize"]),
745
+ },
746
+ {
747
+ "key": "bulk_sync_on_init",
748
+ "description": "Import MEMORY.md / USER.md content from disk on agent init (v0.1.1)",
749
+ "default": "true",
750
+ "choices": ["true", "false"],
751
+ },
752
+ {
753
+ "key": "sync_turns",
754
+ "description": "Capture every substantive (user, assistant) turn pair into the conversations table",
755
+ "default": "true",
756
+ "choices": ["true", "false"],
757
+ },
758
+ {
759
+ "key": "turn_min_chars",
760
+ "description": "Turns shorter than this (after strip) are treated as boilerplate and skipped",
761
+ "default": str(DEFAULTS["turn_min_chars"]),
762
+ },
763
+ ]
764
+
765
+ def save_config(self, values: Dict[str, Any], hermes_home: str) -> None:
766
+ from pathlib import Path
767
+ config_path = Path(hermes_home) / "config.yaml"
768
+ try:
769
+ import yaml
770
+ existing: Dict[str, Any] = {}
771
+ if config_path.exists():
772
+ with open(config_path, encoding="utf-8-sig") as fh:
773
+ existing = yaml.safe_load(fh) or {}
774
+ existing.setdefault("plugins", {})
775
+ existing["plugins"]["pgvector"] = values
776
+ with open(config_path, "w", encoding="utf-8") as fh:
777
+ yaml.dump(existing, fh, default_flow_style=False)
778
+ except Exception as exc: # noqa: BLE001
779
+ logger.warning("pgvector save_config failed: %s", exc)
780
+
781
+ # -- Helpers -------------------------------------------------------------
782
+
783
+ def _maybe_embed(self, content: str) -> Optional[List[float]]:
784
+ if not self._config.get("embed_on_write", True):
785
+ return None
786
+ try:
787
+ return embed(
788
+ content,
789
+ base_url=self._config["embed_url"],
790
+ model=self._config["embed_model"],
791
+ )
792
+ except EmbeddingError as exc:
793
+ if not self._embed_warned:
794
+ logger.warning("pgvector embed failed (degrading to text-only): %s", exc)
795
+ self._embed_warned = True
796
+ return None
797
+
798
+
799
+ # ---------------------------------------------------------------------------
800
+ # Plugin entry point
801
+ # ---------------------------------------------------------------------------
802
+
803
+ def register(ctx) -> None:
804
+ """Register the pgvector memory provider with the plugin system."""
805
+ provider = PgvectorMemoryProvider(config=_load_plugin_config())
806
+ ctx.register_memory_provider(provider)