hermes-memory-pgvector 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hermes_memory_pgvector-0.3.0.dist-info/METADATA +260 -0
- hermes_memory_pgvector-0.3.0.dist-info/RECORD +11 -0
- hermes_memory_pgvector-0.3.0.dist-info/WHEEL +5 -0
- hermes_memory_pgvector-0.3.0.dist-info/licenses/LICENSE +28 -0
- hermes_memory_pgvector-0.3.0.dist-info/top_level.txt +1 -0
- pgvector/__init__.py +806 -0
- pgvector/embed.py +103 -0
- pgvector/migrations/001_schema.sql +95 -0
- pgvector/plugin.yaml +9 -0
- pgvector/store.py +507 -0
- pgvector/writer.py +170 -0
pgvector/__init__.py
ADDED
|
@@ -0,0 +1,806 @@
|
|
|
1
|
+
"""pgvector — Postgres + pgvector memory provider for hermes-agent.
|
|
2
|
+
|
|
3
|
+
Mirrors hermes-agent's built-in `memory` tool entries (MEMORY.md / USER.md
|
|
4
|
+
in tools/memory_tool.py) into a single Postgres table, adds 768-dim
|
|
5
|
+
embeddings for semantic recall, and scopes by `agent_identity` so each
|
|
6
|
+
named agent (marketing / sales / trading / incident / …) has its own
|
|
7
|
+
theme.
|
|
8
|
+
|
|
9
|
+
Design philosophy: this is a STORAGE LAYER for hermes-agent's native
|
|
10
|
+
memory model, not a new memory model. We don't invent facts, entities,
|
|
11
|
+
trust scores, deriver pipelines, or dialectic synthesis. We give the
|
|
12
|
+
built-in `memory` tool a durable Postgres backing + semantic search,
|
|
13
|
+
nothing more. Honcho went heavy and exploded; this stays lean.
|
|
14
|
+
|
|
15
|
+
Config in $HERMES_HOME/config.yaml under plugins.pgvector:
|
|
16
|
+
|
|
17
|
+
plugins:
|
|
18
|
+
pgvector:
|
|
19
|
+
dsn: "dbname=hermes_memory user=hermes host=/var/run/postgresql"
|
|
20
|
+
embed_url: "http://192.168.100.50:11434"
|
|
21
|
+
embed_model: "nomic-embed-text"
|
|
22
|
+
prefetch_limit: 5
|
|
23
|
+
min_similarity: 0.30
|
|
24
|
+
embed_on_write: true
|
|
25
|
+
scope_default: "current" # 'current' | 'all'
|
|
26
|
+
|
|
27
|
+
Tools exposed: `recall_memory` (one explicit search tool). All built-in
|
|
28
|
+
memory writes (add/replace/remove) are mirrored automatically via the
|
|
29
|
+
on_memory_write hook — no agent-facing change.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import json
|
|
35
|
+
import logging
|
|
36
|
+
import re
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Any, Dict, List, Optional
|
|
39
|
+
|
|
40
|
+
from agent.memory_provider import MemoryProvider
|
|
41
|
+
from tools.registry import tool_error
|
|
42
|
+
from hermes_cli.config import cfg_get
|
|
43
|
+
|
|
44
|
+
from .embed import embed, EmbeddingError
|
|
45
|
+
from .store import MemoryStore
|
|
46
|
+
from .writer import AsyncWriter, _PendingWrite
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# Boilerplate / acknowledgement-only turns that are not worth embedding or
|
|
50
|
+
# storing. Case-insensitive whole-string match after strip. Combined with
|
|
51
|
+
# a length floor (default 40 chars) in _is_noise.
|
|
52
|
+
_NOISE_RE = re.compile(
|
|
53
|
+
r"^("
|
|
54
|
+
r"ok(ay)?|thanks?( you)?|thx|ty|np|"
|
|
55
|
+
r"yes|no|sure|got it|done|cool|nice|great|"
|
|
56
|
+
r"continue|please|exit|cancel|stop|quit|"
|
|
57
|
+
r"yeah|yep|nope|alright"
|
|
58
|
+
r")[\s\.\!\?]*$",
|
|
59
|
+
re.IGNORECASE,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
logger = logging.getLogger(__name__)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# ---------------------------------------------------------------------------
|
|
66
|
+
# Tool schema — one explicit search over memory_entries
|
|
67
|
+
# ---------------------------------------------------------------------------
|
|
68
|
+
|
|
69
|
+
RECALL_CONVERSATION_SCHEMA = {
|
|
70
|
+
"name": "recall_conversation",
|
|
71
|
+
"description": (
|
|
72
|
+
"Semantic search over past chat turns (every substantive "
|
|
73
|
+
"user/assistant exchange across all sessions). Use this when "
|
|
74
|
+
"the user references something you discussed earlier — last week, "
|
|
75
|
+
"yesterday, in another session — and you need the actual turn "
|
|
76
|
+
"text, not just a durable memory entry. Returns top-K matching "
|
|
77
|
+
"turns with role, content, session_id, and timestamp.\n\n"
|
|
78
|
+
"SCOPES: 'current' (your theme — default), 'session' (current "
|
|
79
|
+
"session only), 'all' (every theme).\n\n"
|
|
80
|
+
"Skip for in-session continuity (already in your context). Skip "
|
|
81
|
+
"for durable facts (use recall_memory instead — that's the "
|
|
82
|
+
"MEMORY.md / USER.md entries the agent decided to remember)."
|
|
83
|
+
),
|
|
84
|
+
"parameters": {
|
|
85
|
+
"type": "object",
|
|
86
|
+
"properties": {
|
|
87
|
+
"query": {
|
|
88
|
+
"type": "string",
|
|
89
|
+
"description": "Free-text query describing what to recall.",
|
|
90
|
+
},
|
|
91
|
+
"scope": {
|
|
92
|
+
"type": "string",
|
|
93
|
+
"description": "Theme scope: 'current', 'session', 'all', or a named agent.",
|
|
94
|
+
"default": "current",
|
|
95
|
+
},
|
|
96
|
+
"limit": {
|
|
97
|
+
"type": "integer",
|
|
98
|
+
"description": "Max results (1-20, default 5).",
|
|
99
|
+
"default": 5,
|
|
100
|
+
},
|
|
101
|
+
},
|
|
102
|
+
"required": ["query"],
|
|
103
|
+
},
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
RECALL_MEMORY_SCHEMA = {
|
|
108
|
+
"name": "recall_memory",
|
|
109
|
+
"description": (
|
|
110
|
+
"Semantic search over durable memory entries (the same entries the "
|
|
111
|
+
"built-in `memory` tool writes to MEMORY.md / USER.md, stored "
|
|
112
|
+
"durably in Postgres with embeddings).\n\n"
|
|
113
|
+
"WHEN TO USE: when the answer might be in a past memory entry that "
|
|
114
|
+
"is NOT already in your system prompt's memory block — older "
|
|
115
|
+
"entries, or entries from another named agent. The current scope's "
|
|
116
|
+
"recent entries are already injected ambient; only use this tool "
|
|
117
|
+
"for deeper / cross-scope recall.\n\n"
|
|
118
|
+
"SCOPES:\n"
|
|
119
|
+
" 'current' — your own theme (default; e.g. 'marketing')\n"
|
|
120
|
+
" 'all' — across all agent themes\n"
|
|
121
|
+
" '<name>' — a specific theme: 'marketing', 'sales', 'trading', 'incident', …"
|
|
122
|
+
),
|
|
123
|
+
"parameters": {
|
|
124
|
+
"type": "object",
|
|
125
|
+
"properties": {
|
|
126
|
+
"query": {
|
|
127
|
+
"type": "string",
|
|
128
|
+
"description": "Free-text query describing what to recall.",
|
|
129
|
+
},
|
|
130
|
+
"scope": {
|
|
131
|
+
"type": "string",
|
|
132
|
+
"description": "Theme scope: 'current', 'all', or a named agent.",
|
|
133
|
+
"default": "current",
|
|
134
|
+
},
|
|
135
|
+
"target": {
|
|
136
|
+
"type": "string",
|
|
137
|
+
"enum": ["memory", "user", "both"],
|
|
138
|
+
"description": "Which store to search. Default 'both'.",
|
|
139
|
+
"default": "both",
|
|
140
|
+
},
|
|
141
|
+
"limit": {
|
|
142
|
+
"type": "integer",
|
|
143
|
+
"description": "Max results (1-20, default 5).",
|
|
144
|
+
"default": 5,
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
"required": ["query"],
|
|
148
|
+
},
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
# ---------------------------------------------------------------------------
|
|
153
|
+
# Config
|
|
154
|
+
# ---------------------------------------------------------------------------
|
|
155
|
+
|
|
156
|
+
DEFAULTS = {
|
|
157
|
+
"dsn": "dbname=hermes_memory user=hermes host=/var/run/postgresql connect_timeout=5",
|
|
158
|
+
"embed_url": "http://192.168.100.50:11434",
|
|
159
|
+
"embed_model": "nomic-embed-text",
|
|
160
|
+
"prefetch_limit": 5,
|
|
161
|
+
"min_similarity": 0.30,
|
|
162
|
+
"embed_on_write": True,
|
|
163
|
+
"scope_default": "current",
|
|
164
|
+
"write_queue_maxsize": 256,
|
|
165
|
+
# v0.1.1 — bulk sync MEMORY.md / USER.md on init
|
|
166
|
+
"bulk_sync_on_init": True,
|
|
167
|
+
# v0.2 — conversation turn capture
|
|
168
|
+
"sync_turns": True,
|
|
169
|
+
"turn_min_chars": 40, # turns shorter than this are noise unless > 200 chars or contain tool refs
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def _load_plugin_config() -> dict:
|
|
174
|
+
try:
|
|
175
|
+
from hermes_constants import get_hermes_home
|
|
176
|
+
config_path = get_hermes_home() / "config.yaml"
|
|
177
|
+
if not config_path.exists():
|
|
178
|
+
return {}
|
|
179
|
+
import yaml
|
|
180
|
+
with open(config_path, encoding="utf-8-sig") as fh:
|
|
181
|
+
data = yaml.safe_load(fh) or {}
|
|
182
|
+
return cfg_get(data, "plugins", "pgvector", default={}) or {}
|
|
183
|
+
except Exception: # noqa: BLE001
|
|
184
|
+
return {}
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# ---------------------------------------------------------------------------
|
|
188
|
+
# Provider
|
|
189
|
+
# ---------------------------------------------------------------------------
|
|
190
|
+
|
|
191
|
+
class PgvectorMemoryProvider(MemoryProvider):
|
|
192
|
+
"""Postgres mirror of built-in memory entries, with semantic recall."""
|
|
193
|
+
|
|
194
|
+
def __init__(self, config: dict | None = None):
|
|
195
|
+
self._config = {**DEFAULTS, **(config or {})}
|
|
196
|
+
self._store: Optional[MemoryStore] = None
|
|
197
|
+
self._writer: Optional[AsyncWriter] = None
|
|
198
|
+
self._agent_identity: str = "default"
|
|
199
|
+
self._session_id: str = ""
|
|
200
|
+
self._healthy: bool = False
|
|
201
|
+
self._embed_warned: bool = False
|
|
202
|
+
|
|
203
|
+
@property
|
|
204
|
+
def name(self) -> str:
|
|
205
|
+
return "pgvector"
|
|
206
|
+
|
|
207
|
+
# -- Lifecycle -----------------------------------------------------------
|
|
208
|
+
|
|
209
|
+
def is_available(self) -> bool:
|
|
210
|
+
try:
|
|
211
|
+
import psycopg # noqa: F401
|
|
212
|
+
return True
|
|
213
|
+
except ImportError:
|
|
214
|
+
return False
|
|
215
|
+
|
|
216
|
+
def initialize(self, session_id: str, **kwargs) -> None:
|
|
217
|
+
self._session_id = session_id
|
|
218
|
+
# Per-agent theme scoping — priority order:
|
|
219
|
+
# 1. gateway_session_key — from the `X-Hermes-Session-Key` header on
|
|
220
|
+
# API requests. This is the EXPLICIT minion-scope signal sent by
|
|
221
|
+
# systemd-run callers (marketing-daily, sales-daily, intraday
|
|
222
|
+
# workers, …) and takes precedence over the profile fallback
|
|
223
|
+
# because the gateway always sets agent_identity='default' for
|
|
224
|
+
# API traffic — without prioritising the header, every minion
|
|
225
|
+
# collapses to one shared 'default' scope.
|
|
226
|
+
# 2. agent_identity ≠ 'default' — explicit profile name from CLI
|
|
227
|
+
# (`hermes --profile marketing`). Skipped when it's the
|
|
228
|
+
# auto-default sentinel to allow header (#1) to win.
|
|
229
|
+
# 3. agent_workspace — shared workspace name from some platforms.
|
|
230
|
+
# 4. agent_identity == 'default' — accept it now (no other source).
|
|
231
|
+
# 5. 'default' — last-resort bucket for unscoped traffic.
|
|
232
|
+
explicit_identity = kwargs.get("agent_identity")
|
|
233
|
+
if explicit_identity == "default":
|
|
234
|
+
explicit_identity = None # sentinel — let header take over
|
|
235
|
+
self._agent_identity = (
|
|
236
|
+
kwargs.get("gateway_session_key")
|
|
237
|
+
or explicit_identity
|
|
238
|
+
or kwargs.get("agent_workspace")
|
|
239
|
+
or kwargs.get("agent_identity") # accept 'default' if nothing else set
|
|
240
|
+
or "default"
|
|
241
|
+
)
|
|
242
|
+
self._store = MemoryStore(self._config["dsn"])
|
|
243
|
+
try:
|
|
244
|
+
# Schema is verify-only at runtime — admin applies the
|
|
245
|
+
# migration out-of-band (see plugin README install step).
|
|
246
|
+
self._store.ensure_schema()
|
|
247
|
+
health = self._store.health()
|
|
248
|
+
self._healthy = bool(health.get("ok"))
|
|
249
|
+
if not self._healthy:
|
|
250
|
+
logger.warning("pgvector unhealthy on init: %s", health.get("error"))
|
|
251
|
+
except MemoryStore.SchemaNotApplied as exc:
|
|
252
|
+
logger.error("pgvector schema not applied — %s", exc)
|
|
253
|
+
self._healthy = False
|
|
254
|
+
except Exception as exc: # noqa: BLE001
|
|
255
|
+
logger.warning("pgvector init failed: %s", exc)
|
|
256
|
+
self._healthy = False
|
|
257
|
+
|
|
258
|
+
# Background writer — bounded queue, lazy thread start. Decouples
|
|
259
|
+
# on_memory_write + sync_turn from the (potentially slow) embed +
|
|
260
|
+
# DB write so the agent loop never blocks on a stalled embed
|
|
261
|
+
# endpoint.
|
|
262
|
+
self._writer = AsyncWriter(
|
|
263
|
+
self._worker,
|
|
264
|
+
maxsize=int(self._config.get("write_queue_maxsize", 256)),
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
# v0.1.1: bulk import existing MEMORY.md / USER.md content so the
|
|
268
|
+
# plugin sees pre-plugin entries + direct file edits, not just the
|
|
269
|
+
# new writes captured via on_memory_write.
|
|
270
|
+
if self._healthy and self._config.get("bulk_sync_on_init", True):
|
|
271
|
+
self._bulk_sync_from_disk(kwargs.get("hermes_home"))
|
|
272
|
+
|
|
273
|
+
def shutdown(self) -> None:
|
|
274
|
+
# Drain the in-flight writes first so we don't drop work...
|
|
275
|
+
if self._writer:
|
|
276
|
+
self._writer.shutdown(timeout=5.0)
|
|
277
|
+
self._writer = None
|
|
278
|
+
# ...then close the pool the writer was draining into.
|
|
279
|
+
if self._store:
|
|
280
|
+
self._store.close()
|
|
281
|
+
self._store = None
|
|
282
|
+
self._healthy = False
|
|
283
|
+
|
|
284
|
+
def on_session_switch(self, new_session_id: str, **kwargs) -> None:
|
|
285
|
+
self._session_id = new_session_id
|
|
286
|
+
|
|
287
|
+
# -- System prompt + ambient recall --------------------------------------
|
|
288
|
+
|
|
289
|
+
def system_prompt_block(self) -> str:
|
|
290
|
+
if not self._healthy or not self._store:
|
|
291
|
+
return ""
|
|
292
|
+
try:
|
|
293
|
+
count_scoped = self._store.count(agent_identity=self._agent_identity)
|
|
294
|
+
count_all = self._store.count()
|
|
295
|
+
except Exception: # noqa: BLE001
|
|
296
|
+
count_scoped = count_all = 0
|
|
297
|
+
if count_all == 0:
|
|
298
|
+
return (
|
|
299
|
+
"# pgvector memory\n"
|
|
300
|
+
"Active. Empty store. Use the built-in `memory` tool to save "
|
|
301
|
+
"durable notes — entries are mirrored to Postgres with "
|
|
302
|
+
"embeddings for semantic recall across sessions."
|
|
303
|
+
)
|
|
304
|
+
return (
|
|
305
|
+
"# pgvector memory\n"
|
|
306
|
+
f"Active. {count_scoped} entries for '{self._agent_identity}', "
|
|
307
|
+
f"{count_all} total across all themes. "
|
|
308
|
+
"Use `recall_memory(query, scope='all'|'<theme>')` for deeper / "
|
|
309
|
+
"cross-theme recall beyond what's in the built-in memory block."
|
|
310
|
+
)
|
|
311
|
+
|
|
312
|
+
def prefetch(self, query: str, *, session_id: str = "") -> str:
|
|
313
|
+
if not self._healthy or not self._store or not query:
|
|
314
|
+
return ""
|
|
315
|
+
try:
|
|
316
|
+
vec = embed(
|
|
317
|
+
query,
|
|
318
|
+
base_url=self._config["embed_url"],
|
|
319
|
+
model=self._config["embed_model"],
|
|
320
|
+
)
|
|
321
|
+
except EmbeddingError as exc:
|
|
322
|
+
logger.debug("pgvector prefetch embed failed: %s", exc)
|
|
323
|
+
return ""
|
|
324
|
+
|
|
325
|
+
# Ambient prefetch is scoped to the current agent_identity by
|
|
326
|
+
# default — keeps marketing turns from polluting trading recall.
|
|
327
|
+
try:
|
|
328
|
+
rows = self._store.search(
|
|
329
|
+
query_embedding=vec,
|
|
330
|
+
agent_identity=self._agent_identity,
|
|
331
|
+
limit=int(self._config.get("prefetch_limit", 5)),
|
|
332
|
+
min_similarity=float(self._config.get("min_similarity", 0.30)),
|
|
333
|
+
)
|
|
334
|
+
except Exception as exc: # noqa: BLE001
|
|
335
|
+
logger.debug("pgvector prefetch query failed: %s", exc)
|
|
336
|
+
return ""
|
|
337
|
+
if not rows:
|
|
338
|
+
return ""
|
|
339
|
+
|
|
340
|
+
lines = [f"## Recall (pgvector, {self._agent_identity})"]
|
|
341
|
+
for r in rows:
|
|
342
|
+
score = r.get("score") or 0.0
|
|
343
|
+
tgt = r.get("target") or "?"
|
|
344
|
+
content = (r.get("content") or "").strip().replace("\n", " ")
|
|
345
|
+
if len(content) > 280:
|
|
346
|
+
content = content[:280] + "…"
|
|
347
|
+
lines.append(f"- [{score:.2f}] ({tgt}) {content}")
|
|
348
|
+
return "\n".join(lines)
|
|
349
|
+
|
|
350
|
+
# -- Turn capture (v0.2) -------------------------------------------------
|
|
351
|
+
|
|
352
|
+
def sync_turn(
|
|
353
|
+
self,
|
|
354
|
+
user_content: str,
|
|
355
|
+
assistant_content: str,
|
|
356
|
+
*,
|
|
357
|
+
session_id: str = "",
|
|
358
|
+
) -> None:
|
|
359
|
+
"""Persist a (user, assistant) turn pair to the conversations table.
|
|
360
|
+
|
|
361
|
+
Non-blocking — enqueues writes; the async writer drains, embeds,
|
|
362
|
+
and INSERTs. Boilerplate / very short turns are filtered out so
|
|
363
|
+
the recall table stays high-signal.
|
|
364
|
+
"""
|
|
365
|
+
if not self._healthy or not self._writer:
|
|
366
|
+
return
|
|
367
|
+
if not self._config.get("sync_turns", True):
|
|
368
|
+
return
|
|
369
|
+
|
|
370
|
+
sid = session_id or self._session_id or "default"
|
|
371
|
+
min_chars = int(self._config.get("turn_min_chars", 40))
|
|
372
|
+
|
|
373
|
+
for role, content in (("user", user_content), ("assistant", assistant_content)):
|
|
374
|
+
if not content:
|
|
375
|
+
continue
|
|
376
|
+
if self._is_noise(content, min_chars=min_chars):
|
|
377
|
+
continue
|
|
378
|
+
self._writer.enqueue(
|
|
379
|
+
action="turn",
|
|
380
|
+
agent_identity=self._agent_identity,
|
|
381
|
+
target="conversations", # synthetic; worker dispatches on action
|
|
382
|
+
content=content,
|
|
383
|
+
extra={"role": role, "session_id": sid},
|
|
384
|
+
metadata={"session_id": sid},
|
|
385
|
+
)
|
|
386
|
+
|
|
387
|
+
@staticmethod
|
|
388
|
+
def _is_noise(content: str, *, min_chars: int) -> bool:
|
|
389
|
+
"""True for short / boilerplate content we don't want in recall."""
|
|
390
|
+
stripped = (content or "").strip()
|
|
391
|
+
if not stripped:
|
|
392
|
+
return True
|
|
393
|
+
if len(stripped) < min_chars:
|
|
394
|
+
return True
|
|
395
|
+
if _NOISE_RE.match(stripped):
|
|
396
|
+
return True
|
|
397
|
+
return False
|
|
398
|
+
|
|
399
|
+
# -- Built-in memory mirror (THE main integration point) ----------------
|
|
400
|
+
|
|
401
|
+
def on_memory_write(
|
|
402
|
+
self,
|
|
403
|
+
action: str,
|
|
404
|
+
target: str,
|
|
405
|
+
content: str,
|
|
406
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
407
|
+
) -> None:
|
|
408
|
+
"""Mirror built-in `memory` tool writes to Postgres (non-blocking).
|
|
409
|
+
|
|
410
|
+
Built-in tool fires this on every add/replace/remove. We enqueue
|
|
411
|
+
the write; the background thread drains, embeds, and INSERTs.
|
|
412
|
+
Returns instantly so the agent loop never blocks on the embed
|
|
413
|
+
endpoint or the DB.
|
|
414
|
+
"""
|
|
415
|
+
if not self._healthy or not self._writer:
|
|
416
|
+
return
|
|
417
|
+
if target not in ("memory", "user"):
|
|
418
|
+
logger.debug("pgvector ignoring unsupported target: %r", target)
|
|
419
|
+
return
|
|
420
|
+
if action not in ("add", "replace", "remove"):
|
|
421
|
+
logger.debug("pgvector ignoring unknown action: %r", action)
|
|
422
|
+
return
|
|
423
|
+
|
|
424
|
+
meta = dict(metadata or {})
|
|
425
|
+
meta.setdefault("session_id", self._session_id)
|
|
426
|
+
old_text = meta.get("old_text") or meta.get("replaces")
|
|
427
|
+
|
|
428
|
+
self._writer.enqueue(
|
|
429
|
+
action=action,
|
|
430
|
+
agent_identity=self._agent_identity,
|
|
431
|
+
target=target,
|
|
432
|
+
content=content,
|
|
433
|
+
extra={"old_text": str(old_text)} if old_text else {},
|
|
434
|
+
metadata=meta,
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
def _worker(self, item: "_PendingWrite") -> None:
|
|
438
|
+
"""Drain-thread worker: embed + DB write for a single queued item.
|
|
439
|
+
|
|
440
|
+
Must NOT raise — the AsyncWriter logs + survives if we do, but
|
|
441
|
+
we still want failures to degrade gracefully (drop the write,
|
|
442
|
+
keep the queue moving).
|
|
443
|
+
"""
|
|
444
|
+
if not self._store:
|
|
445
|
+
return
|
|
446
|
+
try:
|
|
447
|
+
if item.action == "add":
|
|
448
|
+
vec = self._maybe_embed(item.content)
|
|
449
|
+
self._store.add(
|
|
450
|
+
agent_identity=item.agent_identity,
|
|
451
|
+
target=item.target,
|
|
452
|
+
content=item.content,
|
|
453
|
+
embedding=vec,
|
|
454
|
+
metadata=item.metadata,
|
|
455
|
+
)
|
|
456
|
+
elif item.action == "replace":
|
|
457
|
+
old_text = item.extra.get("old_text")
|
|
458
|
+
vec = self._maybe_embed(item.content)
|
|
459
|
+
if old_text:
|
|
460
|
+
n = self._store.replace(
|
|
461
|
+
agent_identity=item.agent_identity,
|
|
462
|
+
target=item.target,
|
|
463
|
+
old_text=old_text,
|
|
464
|
+
new_content=item.content,
|
|
465
|
+
new_embedding=vec,
|
|
466
|
+
)
|
|
467
|
+
if n == 0:
|
|
468
|
+
# Nothing matched — degrade to add (built-in wrote
|
|
469
|
+
# the new entry to disk; mirror it).
|
|
470
|
+
self._store.add(
|
|
471
|
+
agent_identity=item.agent_identity,
|
|
472
|
+
target=item.target,
|
|
473
|
+
content=item.content,
|
|
474
|
+
embedding=vec,
|
|
475
|
+
metadata=item.metadata,
|
|
476
|
+
)
|
|
477
|
+
else:
|
|
478
|
+
# No old_text in metadata → can't locate prior row;
|
|
479
|
+
# add the new content so we don't lose it.
|
|
480
|
+
self._store.add(
|
|
481
|
+
agent_identity=item.agent_identity,
|
|
482
|
+
target=item.target,
|
|
483
|
+
content=item.content,
|
|
484
|
+
embedding=vec,
|
|
485
|
+
metadata=item.metadata,
|
|
486
|
+
)
|
|
487
|
+
elif item.action == "remove":
|
|
488
|
+
self._store.remove(
|
|
489
|
+
agent_identity=item.agent_identity,
|
|
490
|
+
target=item.target,
|
|
491
|
+
old_text=item.content,
|
|
492
|
+
)
|
|
493
|
+
elif item.action == "turn":
|
|
494
|
+
role = item.extra.get("role") or "user"
|
|
495
|
+
sid = item.extra.get("session_id") or "default"
|
|
496
|
+
vec = self._maybe_embed(item.content)
|
|
497
|
+
self._store.append_turn(
|
|
498
|
+
session_id=sid,
|
|
499
|
+
agent_identity=item.agent_identity,
|
|
500
|
+
role=role,
|
|
501
|
+
content=item.content,
|
|
502
|
+
embedding=vec,
|
|
503
|
+
metadata=item.metadata,
|
|
504
|
+
)
|
|
505
|
+
except Exception as exc: # noqa: BLE001
|
|
506
|
+
logger.debug(
|
|
507
|
+
"pgvector worker (%s/%s/%s) failed: %s",
|
|
508
|
+
item.action,
|
|
509
|
+
item.agent_identity,
|
|
510
|
+
item.target,
|
|
511
|
+
str(exc)[:200],
|
|
512
|
+
)
|
|
513
|
+
|
|
514
|
+
# -- Bulk sync (v0.1.1) --------------------------------------------------
|
|
515
|
+
|
|
516
|
+
def _bulk_sync_from_disk(self, hermes_home: Optional[str]) -> None:
|
|
517
|
+
"""Import MEMORY.md + USER.md entries from disk into memory_entries.
|
|
518
|
+
|
|
519
|
+
Called by initialize(). Runs synchronously (not via async writer)
|
|
520
|
+
so the table is warm before the first turn's prefetch. Cheap on
|
|
521
|
+
re-init: an existence pre-check skips already-imported entries
|
|
522
|
+
without re-embedding.
|
|
523
|
+
"""
|
|
524
|
+
if not self._store:
|
|
525
|
+
return
|
|
526
|
+
if not hermes_home:
|
|
527
|
+
# Fall back to hermes_constants if the runtime didn't pass it.
|
|
528
|
+
try:
|
|
529
|
+
from hermes_constants import get_hermes_home
|
|
530
|
+
hermes_home = str(get_hermes_home())
|
|
531
|
+
except Exception: # noqa: BLE001
|
|
532
|
+
return
|
|
533
|
+
|
|
534
|
+
memories_dir = Path(hermes_home) / "memories"
|
|
535
|
+
embed_fn = self._make_embed_fn()
|
|
536
|
+
|
|
537
|
+
for target, fname in (("memory", "MEMORY.md"), ("user", "USER.md")):
|
|
538
|
+
try:
|
|
539
|
+
result = self._store.bulk_upsert_md(
|
|
540
|
+
agent_identity=self._agent_identity,
|
|
541
|
+
target=target,
|
|
542
|
+
file_path=memories_dir / fname,
|
|
543
|
+
embed_fn=embed_fn,
|
|
544
|
+
)
|
|
545
|
+
if result.get("inserted"):
|
|
546
|
+
logger.info(
|
|
547
|
+
"pgvector bulk-sync %s: parsed=%d inserted=%d skipped=%d",
|
|
548
|
+
fname,
|
|
549
|
+
result.get("parsed", 0),
|
|
550
|
+
result.get("inserted", 0),
|
|
551
|
+
result.get("skipped", 0),
|
|
552
|
+
)
|
|
553
|
+
except Exception as exc: # noqa: BLE001
|
|
554
|
+
logger.warning("pgvector bulk-sync %s failed: %s", fname, exc)
|
|
555
|
+
|
|
556
|
+
def _make_embed_fn(self):
|
|
557
|
+
"""Return a closure over the configured embed endpoint, or None."""
|
|
558
|
+
if not self._config.get("embed_on_write", True):
|
|
559
|
+
return None
|
|
560
|
+
base_url = self._config["embed_url"]
|
|
561
|
+
model = self._config["embed_model"]
|
|
562
|
+
def _fn(text: str):
|
|
563
|
+
return embed(text, base_url=base_url, model=model)
|
|
564
|
+
return _fn
|
|
565
|
+
|
|
566
|
+
# -- Tool surface --------------------------------------------------------
|
|
567
|
+
|
|
568
|
+
def get_tool_schemas(self) -> List[Dict[str, Any]]:
|
|
569
|
+
return [RECALL_MEMORY_SCHEMA, RECALL_CONVERSATION_SCHEMA]
|
|
570
|
+
|
|
571
|
+
def handle_tool_call(self, tool_name: str, args: Dict[str, Any], **kwargs) -> str:
|
|
572
|
+
if tool_name == "recall_conversation":
|
|
573
|
+
return self._handle_recall_conversation(args)
|
|
574
|
+
if tool_name != "recall_memory":
|
|
575
|
+
return tool_error(f"Unknown tool: {tool_name}")
|
|
576
|
+
if not self._healthy or not self._store:
|
|
577
|
+
return json.dumps({"results": [], "count": 0, "error": "pgvector unavailable"})
|
|
578
|
+
|
|
579
|
+
query = (args.get("query") or "").strip()
|
|
580
|
+
if not query:
|
|
581
|
+
return tool_error("Missing required arg: query")
|
|
582
|
+
|
|
583
|
+
try:
|
|
584
|
+
limit = max(1, min(int(args.get("limit", 5)), 20))
|
|
585
|
+
except (TypeError, ValueError):
|
|
586
|
+
limit = 5
|
|
587
|
+
|
|
588
|
+
# Scope resolution: 'current' → my agent_identity; 'all' → no filter;
|
|
589
|
+
# anything else → treat as explicit theme name.
|
|
590
|
+
scope = (args.get("scope") or self._config.get("scope_default") or "current").strip()
|
|
591
|
+
if scope == "current":
|
|
592
|
+
agent_filter: Optional[str] = self._agent_identity
|
|
593
|
+
elif scope == "all":
|
|
594
|
+
agent_filter = None
|
|
595
|
+
else:
|
|
596
|
+
agent_filter = scope
|
|
597
|
+
|
|
598
|
+
# Target resolution: 'memory'/'user'/'both'.
|
|
599
|
+
target_arg = (args.get("target") or "both").strip()
|
|
600
|
+
target_filter: Optional[str] = None if target_arg == "both" else target_arg
|
|
601
|
+
if target_filter not in (None, "memory", "user"):
|
|
602
|
+
return tool_error(f"Invalid target: {target_arg!r}")
|
|
603
|
+
|
|
604
|
+
try:
|
|
605
|
+
vec = embed(
|
|
606
|
+
query,
|
|
607
|
+
base_url=self._config["embed_url"],
|
|
608
|
+
model=self._config["embed_model"],
|
|
609
|
+
)
|
|
610
|
+
except EmbeddingError as exc:
|
|
611
|
+
return json.dumps({"results": [], "count": 0, "error": f"embed: {exc}"})
|
|
612
|
+
|
|
613
|
+
try:
|
|
614
|
+
rows = self._store.search(
|
|
615
|
+
query_embedding=vec,
|
|
616
|
+
agent_identity=agent_filter,
|
|
617
|
+
target=target_filter,
|
|
618
|
+
limit=limit,
|
|
619
|
+
)
|
|
620
|
+
except Exception as exc: # noqa: BLE001
|
|
621
|
+
return json.dumps({"results": [], "count": 0, "error": f"db: {exc}"})
|
|
622
|
+
|
|
623
|
+
results = []
|
|
624
|
+
for r in rows:
|
|
625
|
+
ts = r.get("updated_at") or r.get("created_at")
|
|
626
|
+
results.append(
|
|
627
|
+
{
|
|
628
|
+
"id": r.get("id"),
|
|
629
|
+
"agent_identity": r.get("agent_identity"),
|
|
630
|
+
"target": r.get("target"),
|
|
631
|
+
"ts": ts.isoformat() if ts else None,
|
|
632
|
+
"score": round(float(r.get("score") or 0.0), 4),
|
|
633
|
+
"content": (r.get("content") or "")[:2000],
|
|
634
|
+
}
|
|
635
|
+
)
|
|
636
|
+
return json.dumps({"results": results, "count": len(results)})
|
|
637
|
+
|
|
638
|
+
def _handle_recall_conversation(self, args: Dict[str, Any]) -> str:
|
|
639
|
+
"""Tool handler for recall_conversation over the conversations table."""
|
|
640
|
+
if not self._healthy or not self._store:
|
|
641
|
+
return json.dumps({"results": [], "count": 0, "error": "pgvector unavailable"})
|
|
642
|
+
|
|
643
|
+
query = (args.get("query") or "").strip()
|
|
644
|
+
if not query:
|
|
645
|
+
return tool_error("Missing required arg: query")
|
|
646
|
+
try:
|
|
647
|
+
limit = max(1, min(int(args.get("limit", 5)), 20))
|
|
648
|
+
except (TypeError, ValueError):
|
|
649
|
+
limit = 5
|
|
650
|
+
|
|
651
|
+
scope = (args.get("scope") or "current").strip()
|
|
652
|
+
agent_filter: Optional[str] = None
|
|
653
|
+
session_filter: Optional[str] = None
|
|
654
|
+
if scope == "current":
|
|
655
|
+
agent_filter = self._agent_identity
|
|
656
|
+
elif scope == "session":
|
|
657
|
+
session_filter = self._session_id or None
|
|
658
|
+
elif scope == "all":
|
|
659
|
+
pass # no filters
|
|
660
|
+
else:
|
|
661
|
+
agent_filter = scope # treat as a specific theme name
|
|
662
|
+
|
|
663
|
+
try:
|
|
664
|
+
vec = embed(
|
|
665
|
+
query,
|
|
666
|
+
base_url=self._config["embed_url"],
|
|
667
|
+
model=self._config["embed_model"],
|
|
668
|
+
)
|
|
669
|
+
except EmbeddingError as exc:
|
|
670
|
+
return json.dumps({"results": [], "count": 0, "error": f"embed: {exc}"})
|
|
671
|
+
|
|
672
|
+
try:
|
|
673
|
+
rows = self._store.search_turns(
|
|
674
|
+
query_embedding=vec,
|
|
675
|
+
agent_identity=agent_filter,
|
|
676
|
+
session_id=session_filter,
|
|
677
|
+
limit=limit,
|
|
678
|
+
)
|
|
679
|
+
except Exception as exc: # noqa: BLE001
|
|
680
|
+
return json.dumps({"results": [], "count": 0, "error": f"db: {exc}"})
|
|
681
|
+
|
|
682
|
+
results = []
|
|
683
|
+
for r in rows:
|
|
684
|
+
ts = r.get("ts")
|
|
685
|
+
results.append(
|
|
686
|
+
{
|
|
687
|
+
"id": r.get("id"),
|
|
688
|
+
"agent_identity": r.get("agent_identity"),
|
|
689
|
+
"session_id": r.get("session_id"),
|
|
690
|
+
"role": r.get("role"),
|
|
691
|
+
"ts": ts.isoformat() if ts else None,
|
|
692
|
+
"score": round(float(r.get("score") or 0.0), 4),
|
|
693
|
+
"content": (r.get("content") or "")[:2000],
|
|
694
|
+
}
|
|
695
|
+
)
|
|
696
|
+
return json.dumps({"results": results, "count": len(results)})
|
|
697
|
+
|
|
698
|
+
# -- Setup hooks ---------------------------------------------------------
|
|
699
|
+
|
|
700
|
+
def get_config_schema(self) -> List[Dict[str, Any]]:
|
|
701
|
+
return [
|
|
702
|
+
{
|
|
703
|
+
"key": "dsn",
|
|
704
|
+
"description": "Postgres DSN (psycopg connection string)",
|
|
705
|
+
"default": DEFAULTS["dsn"],
|
|
706
|
+
"required": True,
|
|
707
|
+
},
|
|
708
|
+
{
|
|
709
|
+
"key": "embed_url",
|
|
710
|
+
"description": "Embedding endpoint base URL (OpenAI-compatible or Ollama native)",
|
|
711
|
+
"default": DEFAULTS["embed_url"],
|
|
712
|
+
"required": True,
|
|
713
|
+
},
|
|
714
|
+
{
|
|
715
|
+
"key": "embed_model",
|
|
716
|
+
"description": "Embedding model name (must return 768-dim vectors)",
|
|
717
|
+
"default": DEFAULTS["embed_model"],
|
|
718
|
+
},
|
|
719
|
+
{
|
|
720
|
+
"key": "prefetch_limit",
|
|
721
|
+
"description": "Max ambient recall results injected per turn",
|
|
722
|
+
"default": str(DEFAULTS["prefetch_limit"]),
|
|
723
|
+
},
|
|
724
|
+
{
|
|
725
|
+
"key": "min_similarity",
|
|
726
|
+
"description": "Cosine similarity cutoff for ambient prefetch (0.0–1.0)",
|
|
727
|
+
"default": str(DEFAULTS["min_similarity"]),
|
|
728
|
+
},
|
|
729
|
+
{
|
|
730
|
+
"key": "embed_on_write",
|
|
731
|
+
"description": "Compute embedding on each write; turn off for text-only mode",
|
|
732
|
+
"default": "true",
|
|
733
|
+
"choices": ["true", "false"],
|
|
734
|
+
},
|
|
735
|
+
{
|
|
736
|
+
"key": "scope_default",
|
|
737
|
+
"description": "Default scope for recall_memory when caller omits it",
|
|
738
|
+
"default": DEFAULTS["scope_default"],
|
|
739
|
+
"choices": ["current", "all"],
|
|
740
|
+
},
|
|
741
|
+
{
|
|
742
|
+
"key": "write_queue_maxsize",
|
|
743
|
+
"description": "Bounded async-writer queue size; full = oldest writes drop with a warning",
|
|
744
|
+
"default": str(DEFAULTS["write_queue_maxsize"]),
|
|
745
|
+
},
|
|
746
|
+
{
|
|
747
|
+
"key": "bulk_sync_on_init",
|
|
748
|
+
"description": "Import MEMORY.md / USER.md content from disk on agent init (v0.1.1)",
|
|
749
|
+
"default": "true",
|
|
750
|
+
"choices": ["true", "false"],
|
|
751
|
+
},
|
|
752
|
+
{
|
|
753
|
+
"key": "sync_turns",
|
|
754
|
+
"description": "Capture every substantive (user, assistant) turn pair into the conversations table",
|
|
755
|
+
"default": "true",
|
|
756
|
+
"choices": ["true", "false"],
|
|
757
|
+
},
|
|
758
|
+
{
|
|
759
|
+
"key": "turn_min_chars",
|
|
760
|
+
"description": "Turns shorter than this (after strip) are treated as boilerplate and skipped",
|
|
761
|
+
"default": str(DEFAULTS["turn_min_chars"]),
|
|
762
|
+
},
|
|
763
|
+
]
|
|
764
|
+
|
|
765
|
+
def save_config(self, values: Dict[str, Any], hermes_home: str) -> None:
|
|
766
|
+
from pathlib import Path
|
|
767
|
+
config_path = Path(hermes_home) / "config.yaml"
|
|
768
|
+
try:
|
|
769
|
+
import yaml
|
|
770
|
+
existing: Dict[str, Any] = {}
|
|
771
|
+
if config_path.exists():
|
|
772
|
+
with open(config_path, encoding="utf-8-sig") as fh:
|
|
773
|
+
existing = yaml.safe_load(fh) or {}
|
|
774
|
+
existing.setdefault("plugins", {})
|
|
775
|
+
existing["plugins"]["pgvector"] = values
|
|
776
|
+
with open(config_path, "w", encoding="utf-8") as fh:
|
|
777
|
+
yaml.dump(existing, fh, default_flow_style=False)
|
|
778
|
+
except Exception as exc: # noqa: BLE001
|
|
779
|
+
logger.warning("pgvector save_config failed: %s", exc)
|
|
780
|
+
|
|
781
|
+
# -- Helpers -------------------------------------------------------------
|
|
782
|
+
|
|
783
|
+
def _maybe_embed(self, content: str) -> Optional[List[float]]:
|
|
784
|
+
if not self._config.get("embed_on_write", True):
|
|
785
|
+
return None
|
|
786
|
+
try:
|
|
787
|
+
return embed(
|
|
788
|
+
content,
|
|
789
|
+
base_url=self._config["embed_url"],
|
|
790
|
+
model=self._config["embed_model"],
|
|
791
|
+
)
|
|
792
|
+
except EmbeddingError as exc:
|
|
793
|
+
if not self._embed_warned:
|
|
794
|
+
logger.warning("pgvector embed failed (degrading to text-only): %s", exc)
|
|
795
|
+
self._embed_warned = True
|
|
796
|
+
return None
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
# ---------------------------------------------------------------------------
|
|
800
|
+
# Plugin entry point
|
|
801
|
+
# ---------------------------------------------------------------------------
|
|
802
|
+
|
|
803
|
+
def register(ctx) -> None:
|
|
804
|
+
"""Register the pgvector memory provider with the plugin system."""
|
|
805
|
+
provider = PgvectorMemoryProvider(config=_load_plugin_config())
|
|
806
|
+
ctx.register_memory_provider(provider)
|