hermes-memory-pgvector 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pgvector/embed.py ADDED
@@ -0,0 +1,103 @@
1
+ """embed.py — minimal embedding client for the pgvector memory plugin.
2
+
3
+ Posts to an OpenAI-compatible /v1/embeddings or Ollama native /api/embed
4
+ endpoint and returns a list of floats. No retries beyond a single attempt
5
+ — callers decide what to do with failures (we want fail-soft, not retry
6
+ storms — that was Honcho's mistake).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import logging
13
+ import urllib.error
14
+ import urllib.request
15
+ from typing import List, Optional
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+
20
+ class EmbeddingError(Exception):
21
+ """Raised when the embedding endpoint fails to return a usable vector."""
22
+
23
+
24
+ def embed(
25
+ text: str,
26
+ *,
27
+ base_url: str,
28
+ model: str = "nomic-embed-text",
29
+ timeout: float = 10.0,
30
+ ) -> List[float]:
31
+ """Return a 768-dim embedding for `text`.
32
+
33
+ Tries the OpenAI-compatible `/v1/embeddings` path first; falls back to
34
+ Ollama's native `/api/embed`. Raises EmbeddingError on any failure.
35
+ Single attempt — no retries.
36
+ """
37
+ if not text or not text.strip():
38
+ raise EmbeddingError("empty input")
39
+
40
+ # Trim to avoid the n_ctx_train=2048 cliff on nomic — at ~4 chars/token
41
+ # that's ~8000 chars. Keep a safety margin.
42
+ if len(text) > 6000:
43
+ text = text[:6000]
44
+
45
+ base_url = base_url.rstrip("/")
46
+
47
+ # Path A: OpenAI-compatible
48
+ try:
49
+ return _post(
50
+ f"{base_url}/v1/embeddings",
51
+ {"model": model, "input": text},
52
+ timeout=timeout,
53
+ extract=lambda d: d["data"][0]["embedding"],
54
+ )
55
+ except EmbeddingError as exc:
56
+ logger.debug("OpenAI-compat embed failed (%s); trying native", exc)
57
+
58
+ # Path B: Ollama native
59
+ return _post(
60
+ f"{base_url}/api/embed",
61
+ {"model": model, "input": text},
62
+ timeout=timeout,
63
+ extract=lambda d: (d.get("embeddings") or [d.get("embedding")])[0],
64
+ )
65
+
66
+
67
+ def _post(url: str, body: dict, *, timeout: float, extract) -> List[float]:
68
+ data = json.dumps(body).encode("utf-8")
69
+ req = urllib.request.Request(
70
+ url,
71
+ data=data,
72
+ headers={"Content-Type": "application/json"},
73
+ method="POST",
74
+ )
75
+ try:
76
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
77
+ payload = json.loads(resp.read())
78
+ except urllib.error.HTTPError as exc:
79
+ raise EmbeddingError(f"HTTP {exc.code}: {exc.reason}") from exc
80
+ except urllib.error.URLError as exc:
81
+ raise EmbeddingError(f"connection failed: {exc.reason}") from exc
82
+ except (json.JSONDecodeError, ValueError) as exc:
83
+ raise EmbeddingError(f"invalid JSON response: {exc}") from exc
84
+
85
+ try:
86
+ vec = extract(payload)
87
+ except (KeyError, IndexError, TypeError) as exc:
88
+ raise EmbeddingError(f"unexpected response shape: {exc}") from exc
89
+
90
+ if not isinstance(vec, list) or not vec:
91
+ raise EmbeddingError("response had no embedding array")
92
+ if len(vec) != 768:
93
+ raise EmbeddingError(f"expected 768 dims, got {len(vec)}")
94
+ return vec
95
+
96
+
97
+ def to_pgvector_literal(vec: List[float]) -> str:
98
+ """Render a Python list of floats as a pgvector input literal.
99
+
100
+ psycopg can also handle this via type adapters, but the literal form
101
+ keeps the plugin dependency-light.
102
+ """
103
+ return "[" + ",".join(f"{x:.6g}" for x in vec) + "]"
@@ -0,0 +1,95 @@
1
+ -- 001_schema.sql — pgvector memory plugin schema.
2
+ --
3
+ -- TWO tables, both in the existing hermes_memory database:
4
+ -- memory_entries → mirrors hermes-agent's built-in `memory` tool
5
+ -- (MEMORY.md / USER.md from tools/memory_tool.py)
6
+ -- conversations → every (user, assistant) chat turn, semantic-searchable
7
+ -- for cross-session recall of "what did we talk about"
8
+ --
9
+ -- Both scoped per agent_identity (marketing / sales / trading / incident / …),
10
+ -- both with 768-dim embeddings, both with HNSW indexes tuned the same way as
11
+ -- hermes_memory.events.
12
+ --
13
+ -- Apply once:
14
+ -- sudo -u postgres psql -d hermes_memory -f 001_schema.sql
15
+ -- Idempotent (CREATE IF NOT EXISTS everywhere); safe to re-run.
16
+
17
+ CREATE EXTENSION IF NOT EXISTS vector;
18
+
19
+
20
+ -- ---------------------------------------------------------------------------
21
+ -- memory_entries
22
+ -- One row per add/replace from hermes-agent's built-in `memory` tool,
23
+ -- mirrored via the on_memory_write hook. Hard-deleted on `remove`.
24
+ -- ---------------------------------------------------------------------------
25
+
26
+ CREATE TABLE IF NOT EXISTS memory_entries (
27
+ id BIGSERIAL PRIMARY KEY,
28
+
29
+ -- Per-agent theme. Maps to hermes-agent's `agent_identity` profile
30
+ -- (passed to MemoryProvider.initialize via kwargs).
31
+ -- Examples: 'marketing', 'sales', 'trading', 'incident', 'default'.
32
+ agent_identity TEXT NOT NULL DEFAULT 'default',
33
+
34
+ -- Mirrors hermes-agent's two built-in stores:
35
+ -- 'memory' → MEMORY.md (env facts, project conventions, tool quirks)
36
+ -- 'user' → USER.md (about the user: name, role, preferences)
37
+ target TEXT NOT NULL CHECK (target IN ('memory', 'user')),
38
+
39
+ content TEXT NOT NULL,
40
+ embedding vector(768),
41
+
42
+ created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
43
+ updated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
44
+
45
+ -- Provenance from on_memory_write kwargs: session_id, platform,
46
+ -- write_origin, tool_name, parent_session_id, etc.
47
+ metadata JSONB NOT NULL DEFAULT '{}'::jsonb,
48
+
49
+ -- Built-in tool dedupes on (target, exact content match). We replicate
50
+ -- that within a single agent's scope so add-of-existing is a no-op.
51
+ CONSTRAINT memory_entries_unique
52
+ UNIQUE (agent_identity, target, content)
53
+ );
54
+
55
+ -- Per-agent + target listing (the most common scan).
56
+ CREATE INDEX IF NOT EXISTS ix_memory_entries_agent_target
57
+ ON memory_entries (agent_identity, target, updated_at DESC);
58
+
59
+ -- Semantic recall: SELECT … ORDER BY embedding <=> $1 LIMIT $2
60
+ -- Cosine distance via HNSW — same tuning as hermes_memory.events.
61
+ CREATE INDEX IF NOT EXISTS ix_memory_entries_embedding_hnsw
62
+ ON memory_entries USING hnsw (embedding vector_cosine_ops)
63
+ WITH (m = 16, ef_construction = 64);
64
+
65
+
66
+ -- ---------------------------------------------------------------------------
67
+ -- conversations
68
+ -- One row per substantive chat turn, captured via sync_turn(). Short /
69
+ -- boilerplate turns ("ok", "thanks", < 40 chars) are filtered out at
70
+ -- the plugin layer before INSERT; what lands here is recall-worthy.
71
+ -- ---------------------------------------------------------------------------
72
+
73
+ CREATE TABLE IF NOT EXISTS conversations (
74
+ id BIGSERIAL PRIMARY KEY,
75
+ session_id TEXT NOT NULL,
76
+ agent_identity TEXT NOT NULL DEFAULT 'default',
77
+ role TEXT NOT NULL CHECK (role IN ('user', 'assistant', 'system', 'tool')),
78
+ content TEXT NOT NULL,
79
+ ts TIMESTAMPTZ NOT NULL DEFAULT now(),
80
+ embedding vector(768),
81
+ metadata JSONB NOT NULL DEFAULT '{}'::jsonb
82
+ );
83
+
84
+ -- Per-session timeline (the "recent turns in this conversation" path).
85
+ CREATE INDEX IF NOT EXISTS ix_conversations_session_ts
86
+ ON conversations (session_id, ts DESC);
87
+
88
+ -- Per-agent timeline (the "what's marketing been doing lately" path).
89
+ CREATE INDEX IF NOT EXISTS ix_conversations_agent_ts
90
+ ON conversations (agent_identity, ts DESC);
91
+
92
+ -- Semantic recall across all turns. Same HNSW tuning as memory_entries.
93
+ CREATE INDEX IF NOT EXISTS ix_conversations_embedding_hnsw
94
+ ON conversations USING hnsw (embedding vector_cosine_ops)
95
+ WITH (m = 16, ef_construction = 64);
pgvector/plugin.yaml ADDED
@@ -0,0 +1,9 @@
1
+ name: pgvector
2
+ version: 0.3.0
3
+ description: "Postgres + pgvector storage layer for hermes-agent. Mirrors built-in MEMORY.md / USER.md + captures every substantive (user, assistant) chat turn into a conversations table. Per-agent themes (marketing/sales/trading/incident/morning-report/…) via X-Hermes-Session-Key header — every systemd-run minion gets its own scope. 768-dim semantic recall. Async writer + connection pool. No LLM mediation."
4
+ pip_dependencies:
5
+ - "psycopg[binary]>=3.3.4"
6
+ - "psycopg-pool>=3.3.1"
7
+ requires_env: []
8
+ hooks:
9
+ - on_session_end