hermes-memory-pgvector 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hermes_memory_pgvector-0.3.0.dist-info/METADATA +260 -0
- hermes_memory_pgvector-0.3.0.dist-info/RECORD +11 -0
- hermes_memory_pgvector-0.3.0.dist-info/WHEEL +5 -0
- hermes_memory_pgvector-0.3.0.dist-info/licenses/LICENSE +28 -0
- hermes_memory_pgvector-0.3.0.dist-info/top_level.txt +1 -0
- pgvector/__init__.py +806 -0
- pgvector/embed.py +103 -0
- pgvector/migrations/001_schema.sql +95 -0
- pgvector/plugin.yaml +9 -0
- pgvector/store.py +507 -0
- pgvector/writer.py +170 -0
pgvector/embed.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""embed.py — minimal embedding client for the pgvector memory plugin.
|
|
2
|
+
|
|
3
|
+
Posts to an OpenAI-compatible /v1/embeddings or Ollama native /api/embed
|
|
4
|
+
endpoint and returns a list of floats. No retries beyond a single attempt
|
|
5
|
+
— callers decide what to do with failures (we want fail-soft, not retry
|
|
6
|
+
storms — that was Honcho's mistake).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import logging
|
|
13
|
+
import urllib.error
|
|
14
|
+
import urllib.request
|
|
15
|
+
from typing import List, Optional
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class EmbeddingError(Exception):
|
|
21
|
+
"""Raised when the embedding endpoint fails to return a usable vector."""
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def embed(
|
|
25
|
+
text: str,
|
|
26
|
+
*,
|
|
27
|
+
base_url: str,
|
|
28
|
+
model: str = "nomic-embed-text",
|
|
29
|
+
timeout: float = 10.0,
|
|
30
|
+
) -> List[float]:
|
|
31
|
+
"""Return a 768-dim embedding for `text`.
|
|
32
|
+
|
|
33
|
+
Tries the OpenAI-compatible `/v1/embeddings` path first; falls back to
|
|
34
|
+
Ollama's native `/api/embed`. Raises EmbeddingError on any failure.
|
|
35
|
+
Single attempt — no retries.
|
|
36
|
+
"""
|
|
37
|
+
if not text or not text.strip():
|
|
38
|
+
raise EmbeddingError("empty input")
|
|
39
|
+
|
|
40
|
+
# Trim to avoid the n_ctx_train=2048 cliff on nomic — at ~4 chars/token
|
|
41
|
+
# that's ~8000 chars. Keep a safety margin.
|
|
42
|
+
if len(text) > 6000:
|
|
43
|
+
text = text[:6000]
|
|
44
|
+
|
|
45
|
+
base_url = base_url.rstrip("/")
|
|
46
|
+
|
|
47
|
+
# Path A: OpenAI-compatible
|
|
48
|
+
try:
|
|
49
|
+
return _post(
|
|
50
|
+
f"{base_url}/v1/embeddings",
|
|
51
|
+
{"model": model, "input": text},
|
|
52
|
+
timeout=timeout,
|
|
53
|
+
extract=lambda d: d["data"][0]["embedding"],
|
|
54
|
+
)
|
|
55
|
+
except EmbeddingError as exc:
|
|
56
|
+
logger.debug("OpenAI-compat embed failed (%s); trying native", exc)
|
|
57
|
+
|
|
58
|
+
# Path B: Ollama native
|
|
59
|
+
return _post(
|
|
60
|
+
f"{base_url}/api/embed",
|
|
61
|
+
{"model": model, "input": text},
|
|
62
|
+
timeout=timeout,
|
|
63
|
+
extract=lambda d: (d.get("embeddings") or [d.get("embedding")])[0],
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _post(url: str, body: dict, *, timeout: float, extract) -> List[float]:
|
|
68
|
+
data = json.dumps(body).encode("utf-8")
|
|
69
|
+
req = urllib.request.Request(
|
|
70
|
+
url,
|
|
71
|
+
data=data,
|
|
72
|
+
headers={"Content-Type": "application/json"},
|
|
73
|
+
method="POST",
|
|
74
|
+
)
|
|
75
|
+
try:
|
|
76
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
77
|
+
payload = json.loads(resp.read())
|
|
78
|
+
except urllib.error.HTTPError as exc:
|
|
79
|
+
raise EmbeddingError(f"HTTP {exc.code}: {exc.reason}") from exc
|
|
80
|
+
except urllib.error.URLError as exc:
|
|
81
|
+
raise EmbeddingError(f"connection failed: {exc.reason}") from exc
|
|
82
|
+
except (json.JSONDecodeError, ValueError) as exc:
|
|
83
|
+
raise EmbeddingError(f"invalid JSON response: {exc}") from exc
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
vec = extract(payload)
|
|
87
|
+
except (KeyError, IndexError, TypeError) as exc:
|
|
88
|
+
raise EmbeddingError(f"unexpected response shape: {exc}") from exc
|
|
89
|
+
|
|
90
|
+
if not isinstance(vec, list) or not vec:
|
|
91
|
+
raise EmbeddingError("response had no embedding array")
|
|
92
|
+
if len(vec) != 768:
|
|
93
|
+
raise EmbeddingError(f"expected 768 dims, got {len(vec)}")
|
|
94
|
+
return vec
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def to_pgvector_literal(vec: List[float]) -> str:
|
|
98
|
+
"""Render a Python list of floats as a pgvector input literal.
|
|
99
|
+
|
|
100
|
+
psycopg can also handle this via type adapters, but the literal form
|
|
101
|
+
keeps the plugin dependency-light.
|
|
102
|
+
"""
|
|
103
|
+
return "[" + ",".join(f"{x:.6g}" for x in vec) + "]"
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
-- 001_schema.sql — pgvector memory plugin schema.
|
|
2
|
+
--
|
|
3
|
+
-- TWO tables, both in the existing hermes_memory database:
|
|
4
|
+
-- memory_entries → mirrors hermes-agent's built-in `memory` tool
|
|
5
|
+
-- (MEMORY.md / USER.md from tools/memory_tool.py)
|
|
6
|
+
-- conversations → every (user, assistant) chat turn, semantic-searchable
|
|
7
|
+
-- for cross-session recall of "what did we talk about"
|
|
8
|
+
--
|
|
9
|
+
-- Both scoped per agent_identity (marketing / sales / trading / incident / …),
|
|
10
|
+
-- both with 768-dim embeddings, both with HNSW indexes tuned the same way as
|
|
11
|
+
-- hermes_memory.events.
|
|
12
|
+
--
|
|
13
|
+
-- Apply once:
|
|
14
|
+
-- sudo -u postgres psql -d hermes_memory -f 001_schema.sql
|
|
15
|
+
-- Idempotent (CREATE IF NOT EXISTS everywhere); safe to re-run.
|
|
16
|
+
|
|
17
|
+
CREATE EXTENSION IF NOT EXISTS vector;
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
-- ---------------------------------------------------------------------------
|
|
21
|
+
-- memory_entries
|
|
22
|
+
-- One row per add/replace from hermes-agent's built-in `memory` tool,
|
|
23
|
+
-- mirrored via the on_memory_write hook. Hard-deleted on `remove`.
|
|
24
|
+
-- ---------------------------------------------------------------------------
|
|
25
|
+
|
|
26
|
+
CREATE TABLE IF NOT EXISTS memory_entries (
|
|
27
|
+
id BIGSERIAL PRIMARY KEY,
|
|
28
|
+
|
|
29
|
+
-- Per-agent theme. Maps to hermes-agent's `agent_identity` profile
|
|
30
|
+
-- (passed to MemoryProvider.initialize via kwargs).
|
|
31
|
+
-- Examples: 'marketing', 'sales', 'trading', 'incident', 'default'.
|
|
32
|
+
agent_identity TEXT NOT NULL DEFAULT 'default',
|
|
33
|
+
|
|
34
|
+
-- Mirrors hermes-agent's two built-in stores:
|
|
35
|
+
-- 'memory' → MEMORY.md (env facts, project conventions, tool quirks)
|
|
36
|
+
-- 'user' → USER.md (about the user: name, role, preferences)
|
|
37
|
+
target TEXT NOT NULL CHECK (target IN ('memory', 'user')),
|
|
38
|
+
|
|
39
|
+
content TEXT NOT NULL,
|
|
40
|
+
embedding vector(768),
|
|
41
|
+
|
|
42
|
+
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
43
|
+
updated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
44
|
+
|
|
45
|
+
-- Provenance from on_memory_write kwargs: session_id, platform,
|
|
46
|
+
-- write_origin, tool_name, parent_session_id, etc.
|
|
47
|
+
metadata JSONB NOT NULL DEFAULT '{}'::jsonb,
|
|
48
|
+
|
|
49
|
+
-- Built-in tool dedupes on (target, exact content match). We replicate
|
|
50
|
+
-- that within a single agent's scope so add-of-existing is a no-op.
|
|
51
|
+
CONSTRAINT memory_entries_unique
|
|
52
|
+
UNIQUE (agent_identity, target, content)
|
|
53
|
+
);
|
|
54
|
+
|
|
55
|
+
-- Per-agent + target listing (the most common scan).
|
|
56
|
+
CREATE INDEX IF NOT EXISTS ix_memory_entries_agent_target
|
|
57
|
+
ON memory_entries (agent_identity, target, updated_at DESC);
|
|
58
|
+
|
|
59
|
+
-- Semantic recall: SELECT … ORDER BY embedding <=> $1 LIMIT $2
|
|
60
|
+
-- Cosine distance via HNSW — same tuning as hermes_memory.events.
|
|
61
|
+
CREATE INDEX IF NOT EXISTS ix_memory_entries_embedding_hnsw
|
|
62
|
+
ON memory_entries USING hnsw (embedding vector_cosine_ops)
|
|
63
|
+
WITH (m = 16, ef_construction = 64);
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
-- ---------------------------------------------------------------------------
|
|
67
|
+
-- conversations
|
|
68
|
+
-- One row per substantive chat turn, captured via sync_turn(). Short /
|
|
69
|
+
-- boilerplate turns ("ok", "thanks", < 40 chars) are filtered out at
|
|
70
|
+
-- the plugin layer before INSERT; what lands here is recall-worthy.
|
|
71
|
+
-- ---------------------------------------------------------------------------
|
|
72
|
+
|
|
73
|
+
CREATE TABLE IF NOT EXISTS conversations (
|
|
74
|
+
id BIGSERIAL PRIMARY KEY,
|
|
75
|
+
session_id TEXT NOT NULL,
|
|
76
|
+
agent_identity TEXT NOT NULL DEFAULT 'default',
|
|
77
|
+
role TEXT NOT NULL CHECK (role IN ('user', 'assistant', 'system', 'tool')),
|
|
78
|
+
content TEXT NOT NULL,
|
|
79
|
+
ts TIMESTAMPTZ NOT NULL DEFAULT now(),
|
|
80
|
+
embedding vector(768),
|
|
81
|
+
metadata JSONB NOT NULL DEFAULT '{}'::jsonb
|
|
82
|
+
);
|
|
83
|
+
|
|
84
|
+
-- Per-session timeline (the "recent turns in this conversation" path).
|
|
85
|
+
CREATE INDEX IF NOT EXISTS ix_conversations_session_ts
|
|
86
|
+
ON conversations (session_id, ts DESC);
|
|
87
|
+
|
|
88
|
+
-- Per-agent timeline (the "what's marketing been doing lately" path).
|
|
89
|
+
CREATE INDEX IF NOT EXISTS ix_conversations_agent_ts
|
|
90
|
+
ON conversations (agent_identity, ts DESC);
|
|
91
|
+
|
|
92
|
+
-- Semantic recall across all turns. Same HNSW tuning as memory_entries.
|
|
93
|
+
CREATE INDEX IF NOT EXISTS ix_conversations_embedding_hnsw
|
|
94
|
+
ON conversations USING hnsw (embedding vector_cosine_ops)
|
|
95
|
+
WITH (m = 16, ef_construction = 64);
|
pgvector/plugin.yaml
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
name: pgvector
|
|
2
|
+
version: 0.3.0
|
|
3
|
+
description: "Postgres + pgvector storage layer for hermes-agent. Mirrors built-in MEMORY.md / USER.md + captures every substantive (user, assistant) chat turn into a conversations table. Per-agent themes (marketing/sales/trading/incident/morning-report/…) via X-Hermes-Session-Key header — every systemd-run minion gets its own scope. 768-dim semantic recall. Async writer + connection pool. No LLM mediation."
|
|
4
|
+
pip_dependencies:
|
|
5
|
+
- "psycopg[binary]>=3.3.4"
|
|
6
|
+
- "psycopg-pool>=3.3.1"
|
|
7
|
+
requires_env: []
|
|
8
|
+
hooks:
|
|
9
|
+
- on_session_end
|