hermes-memory-pgvector 0.5.2__tar.gz → 0.5.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/PKG-INFO +77 -6
  2. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/README.md +74 -3
  3. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_memory_pgvector.egg-info/PKG-INFO +77 -6
  4. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_memory_pgvector.egg-info/SOURCES.txt +2 -0
  5. hermes_memory_pgvector-0.5.4/hermes_memory_pgvector.egg-info/requires.txt +6 -0
  6. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/__init__.py +101 -30
  7. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/__main__.py +30 -9
  8. hermes_memory_pgvector-0.5.4/hermes_pgvector/embed.py +311 -0
  9. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/plugin.yaml +4 -4
  10. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/store.py +11 -7
  11. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/pyproject.toml +3 -3
  12. hermes_memory_pgvector-0.5.4/tests/test_embed_config.py +610 -0
  13. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_embed_timeouts.py +249 -249
  14. hermes_memory_pgvector-0.5.4/tests/test_loader_embed_clobber.py +188 -0
  15. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_read_side_gate.py +187 -187
  16. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_tool_args_hardening.py +128 -128
  17. hermes_memory_pgvector-0.5.2/hermes_memory_pgvector.egg-info/requires.txt +0 -6
  18. hermes_memory_pgvector-0.5.2/hermes_pgvector/embed.py +0 -156
  19. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/LICENSE +0 -0
  20. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_memory_pgvector.egg-info/dependency_links.txt +0 -0
  21. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_memory_pgvector.egg-info/entry_points.txt +0 -0
  22. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_memory_pgvector.egg-info/top_level.txt +0 -0
  23. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/identity.py +0 -0
  24. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/migrations/001_schema.sql +0 -0
  25. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/migrations/002_agent_attribution.sql +0 -0
  26. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/migrations/003_hybrid_search_fts.sql +0 -0
  27. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/migrations/004_runtime_grants.sql +0 -0
  28. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/hermes_pgvector/writer.py +0 -0
  29. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/setup.cfg +0 -0
  30. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_async_writer.py +0 -0
  31. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_config_coercion.py +0 -0
  32. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_empty_content.py +0 -0
  33. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_exclude_identities_live.py +0 -0
  34. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_hybrid_search.py +0 -0
  35. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_identity.py +0 -0
  36. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_install_shim.py +0 -0
  37. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_save_config_merge.py +0 -0
  38. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_session_switch_contract.py +0 -0
  39. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_smoke.py +0 -0
  40. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_store_v04.py +0 -0
  41. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_system_prompt_block.py +0 -0
  42. {hermes_memory_pgvector-0.5.2 → hermes_memory_pgvector-0.5.4}/tests/test_turn_dedup.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hermes-memory-pgvector
3
- Version: 0.5.2
3
+ Version: 0.5.4
4
4
  Summary: Postgres + pgvector memory provider plugin for hermes-agent. Multi-agent storage layer with per-minion themes, identity governance, agent attribution + delegation provenance, async writer, no LLM in the memory hot path.
5
5
  Author: Andrea Borghi
6
6
  License: BSD-3-Clause
@@ -24,8 +24,8 @@ Classifier: Topic :: System :: Distributed Computing
24
24
  Requires-Python: >=3.11
25
25
  Description-Content-Type: text/markdown
26
26
  License-File: LICENSE
27
- Requires-Dist: psycopg[binary]<4,>=3.3.5
28
- Requires-Dist: psycopg-pool<4,>=3.3.1
27
+ Requires-Dist: psycopg[binary]<4,>=3.3.6
28
+ Requires-Dist: psycopg-pool<4,>=3.3.2
29
29
  Requires-Dist: PyYAML<7,>=6.0
30
30
  Provides-Extra: test
31
31
  Requires-Dist: pytest<9,>=7.4; extra == "test"
@@ -202,7 +202,7 @@ Found in production: one `memory_entries` row sat with a NULL embedding and **ze
202
202
 
203
203
  ## New in v0.5.2 — documentation only
204
204
 
205
- **No code changes.** `git diff v0.5.1..v0.5.2` touches only `README.md` and one test file; nothing under `hermes_pgvector/` differs, so the installed behaviour is byte-for-byte identical to v0.5.1. There is no reason to redeploy for this release.
205
+ **No behaviour changes.** `git diff v0.5.1..v0.5.2` shows four files: `README.md`, one test file, and the two version strings (`pyproject.toml` and `hermes_pgvector/plugin.yaml`). The only packaged file that differs is `plugin.yaml`, and only its `version:` line — no logic changed anywhere, so an installed 0.5.2 behaves identically to 0.5.1. There is no reason to redeploy for this release.
206
206
 
207
207
  It exists because PyPI renders a project's README **frozen at upload time**: two fixes that landed after v0.5.1 shipped were visible on GitHub but not on the package page.
208
208
 
@@ -211,6 +211,21 @@ It exists because PyPI renders a project's README **frozen at upload time**: two
211
211
 
212
212
  If you are on v0.5.1 you already have every fix in this release. If you are on **v0.5.0 or earlier, upgrade** — v0.5.1 fixed a data-loss bug where a single `memory remove` deleted a whole theme's mirrored memory.
213
213
 
214
+ ## New in v0.5.3 - configurable embedding model (dimension, auth, protocol)
215
+
216
+ **Defaults are unchanged:** 768 dimensions, no `Authorization` header, `auto` protocol. No schema changes and no new migrations; a deployment that sets none of the new keys behaves as before, apart from the two fixes at the end of this list.
217
+
218
+ - **The embedding dimension is configuration, not code.** `embed_dim` (default `768`) replaces the literal 768 in the response check, in the backfill dimension guard (`backfill_null_embeddings(expected_dim=...)`), and in the `stats` dry-run. The check was moved, not relaxed: a mismatch still fails fast with `expected N dims (embed_dim), got M`. Changing the value on a database that already holds vectors needs a column migration; see [Changing the embedding dimension](#changing-the-embedding-dimension).
219
+ - **Bearer auth for hosted endpoints.** `embed_api_key_env` holds the *name* of an environment variable (for example `OPENROUTER_API_KEY`). When that variable is set and non-empty, the plugin sends `Authorization: Bearer <value>`. The value is read at call time and is never logged, stored in config, or included in exception messages. Unset or empty means no header, as before.
220
+ - **Explicit protocol selection.** `embed_protocol: openai` uses only `/v1/embeddings`, so a 401 or an unknown-model error from a hosted endpoint is reported as-is instead of being replaced by a 404 from the Ollama-native fallback. `ollama` uses only `/api/embed`. `auto` keeps the old try-OpenAI-then-Ollama behaviour. Unknown values fall back to `auto` with a warning.
221
+ - **One embed path.** Prefetch, both recall tools, the init-time bulk import, the writer drain and `hermes-pgvector backfill` all resolve the endpoint through one helper, so the settings apply the same way everywhere (`stats` reads the same `embed_dim`). Timeouts and retries are unchanged: one attempt on the agent thread, bounded retries on the writer. `backfill` gains `--embed-dim`, `--embed-api-key-env` and `--embed-protocol` (CLI flag > `--config` file > default).
222
+ - **Fixed: embeds broke under hermes-agent's plugin loader.** After it runs the package, `plugins/plugin_loader.py:load_plugin_module` binds every sibling module back onto it, including `setattr(pkg, "embed", <the embed submodule>)`. That replaced the `embed` function the provider called, so every embed raised `TypeError: 'module' object is not callable`. That is not an `EmbeddingError`, so nothing degraded gracefully: prefetch and the recall tools raised out of the hook, and the writer dropped each mirrored write and captured turn outright instead of storing it text-only. Call sites now use a private alias the loader never touches. An external patch that re-binds `embed` inside `register()` is no longer needed, and does no harm if it is still present. `from hermes_pgvector import embed` still works.
223
+ - **Fixed: a read timeout escaped as a bare `TimeoutError`.** urllib wraps errors raised while *sending* a request, but a server that accepts the connection and answers slower than the timeout raises `TimeoutError` from the response read. That slipped past every `except EmbeddingError`: on the agent thread it raised out of prefetch and the recall tools, `auto` never tried its fallback, and on the writer the retries never ran and the write was dropped instead of being stored text-only. It is now an `EmbeddingError`, like every other endpoint failure.
224
+
225
+ ## New in v0.5.4 - psycopg 3.3.6 / psycopg-pool 3.3.2 floor
226
+
227
+ - **Dependency floor raised**: `psycopg[binary]>=3.3.6`, `psycopg-pool>=3.3.2` (upstream patch releases, 2026-09-18). psycopg 3.3.6: Python 3.15 support; a cancelled query no longer waits forever on an unresponsive server (needs libpq 17+); cancels the running query on `SystemExit`; interval `Column.precision` now reports `None` instead of `65535`; fixes dumping nested list subclasses as arrays; discards prepared statements on `DEALLOCATE ALL`; better guards dumping a large int to binary numeric; faster async waits. psycopg-pool 3.3.2: propagates cancellation and other base exceptions raised during a connection check -- relevant here since this package opens one shared `ConnectionPool` across the agent and async-writer threads. No code changes.
228
+
214
229
  ## Multi-agent / per-minion themes
215
230
 
216
231
  Each systemd-run minion sets one header on its OpenAI client; everything else flows automatically:
@@ -274,7 +289,7 @@ That:
274
289
 
275
290
  ```bash
276
291
  # Python deps
277
- pip install 'psycopg[binary]>=3.3.5,<4' 'psycopg-pool>=3.3.1,<4' 'PyYAML>=6.0,<7'
292
+ pip install 'psycopg[binary]>=3.3.6,<4' 'psycopg-pool>=3.3.2,<4' 'PyYAML>=6.0,<7'
278
293
 
279
294
  # Plugin module
280
295
  mkdir -p ~/.hermes/plugins
@@ -323,6 +338,9 @@ plugins:
323
338
  dsn: "dbname=hermes_memory user=hermes host=/var/run/postgresql"
324
339
  embed_url: "http://your-embed-endpoint:11434"
325
340
  embed_model: "nomic-embed-text"
341
+ embed_dim: 768 # v0.5.3: must match the model AND the vector(N) columns
342
+ embed_api_key_env: "" # v0.5.3: NAME of an env var holding a bearer token
343
+ embed_protocol: "auto" # v0.5.3: auto | openai | ollama
326
344
  prefetch_limit: 5
327
345
  min_similarity: 0.30
328
346
  embed_on_write: true
@@ -339,7 +357,58 @@ plugins:
339
357
  embed_write_retries: 2 # writer-path only; hot path stays single-attempt
340
358
  ```
341
359
 
342
- The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL that returns **768-dim vectors** (the schema is hard-coded to `vector(768)` to match `nomic-embed-text`). Use a different model only if it produces 768-dim output, or edit the migration before applying it.
360
+ The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL. Its vectors must be exactly `embed_dim` long, and `embed_dim` must match the database's `vector(N)` columns. Migration 001 creates `vector(768)` to match `nomic-embed-text`, which is why 768 is the default.
361
+
362
+ ### Embedding endpoint keys (v0.5.3)
363
+
364
+ | Key | Default | Meaning |
365
+ |---|---|---|
366
+ | `embed_dim` | `768` | Vector length the model returns. Every embedding is checked against it, and so is the `hermes-pgvector backfill` probe. It must equal the `vector(N)` column size: changing it on an existing database is a migration, see [below](#changing-the-embedding-dimension). |
367
+ | `embed_api_key_env` | unset | **Name** of an environment variable holding a bearer token, e.g. `OPENROUTER_API_KEY`. Never put the token itself in config. The variable is read on every request; when it is set and non-empty the plugin sends `Authorization: Bearer <value>`, otherwise no header. |
368
+ | `embed_protocol` | `auto` | `auto`: try `/v1/embeddings`, then fall back to `/api/embed`. `openai`: `/v1/embeddings` only, so auth and model errors surface as-is (use this for hosted OpenAI-compatible APIs). `ollama`: `/api/embed` only. Unknown values fall back to `auto` with a warning. |
369
+
370
+ Example: OpenAI `text-embedding-3-small` (1536 dimensions) through OpenRouter:
371
+
372
+ ```yaml
373
+ plugins:
374
+ pgvector:
375
+ embed_url: "https://openrouter.ai/api" # the plugin appends /v1/embeddings
376
+ embed_model: "openai/text-embedding-3-small"
377
+ embed_dim: 1536
378
+ embed_api_key_env: "OPENROUTER_API_KEY" # the variable's NAME, not the key
379
+ embed_protocol: "openai"
380
+ ```
381
+
382
+ The variable has to be in the environment of every process that loads the provider (each hermes service) and of any `hermes-pgvector backfill` job. To call OpenAI directly instead, use `embed_url: "https://api.openai.com"`, `embed_model: "text-embedding-3-small"` and a variable holding an OpenAI key.
383
+
384
+ ### Changing the embedding dimension
385
+
386
+ `embed_dim` has to agree with the columns, so switching to a model with a different output size is a migration, not a config edit. Vectors from two different models are not comparable anyway, so every row must be re-embedded. Until config and columns agree, Postgres rejects each write whose vector has the wrong length (`expected 1536 dimensions, not 768`), and the whole row is lost, not stored text-only. Stop the services first.
387
+
388
+ The shipped migration files are not meant to be edited for this. As the table owner:
389
+
390
+ ```sql
391
+ -- 1. With every hermes service that loads the provider stopped:
392
+ BEGIN;
393
+ DROP INDEX IF EXISTS ix_memory_entries_embedding_hnsw;
394
+ DROP INDEX IF EXISTS ix_conversations_embedding_hnsw;
395
+ ALTER TABLE memory_entries ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
396
+ ALTER TABLE conversations ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
397
+ COMMIT;
398
+ ```
399
+
400
+ 2. Set `embed_model` and `embed_dim` (plus `embed_url`, `embed_api_key_env` and `embed_protocol` as needed), then start the services. New writes are embedded with the new model.
401
+ 3. Re-embed the existing rows, which are all NULL now: `hermes-pgvector backfill --config $HERMES_HOME/config.yaml`. Repeat until every table reports `remaining: 0`. If the endpoint does not return `embed_dim`-length vectors, the run aborts on its first probe, before touching any row, and the logged warning names both sizes.
402
+ 4. Rebuild the HNSW indexes with the shipped tuning. Building them after the backfill is faster than maintaining them during it:
403
+
404
+ ```sql
405
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_memory_entries_embedding_hnsw
406
+ ON memory_entries USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
407
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_conversations_embedding_hnsw
408
+ ON conversations USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
409
+ ```
410
+
411
+ Until step 3 completes, rows without a vector are reachable only through full-text recall (`hybrid_search: true`). pgvector's HNSW index supports `vector` columns of up to 2,000 dimensions. The plugin maintains only `memory_entries` and `conversations`; any other embedding columns in the same database need the same change from whatever writes them.
343
412
 
344
413
  ## Schema
345
414
 
@@ -370,6 +439,8 @@ CREATE TABLE conversations (
370
439
 
371
440
  Indexes: HNSW on each `embedding` column (m=16, ef_construction=64) plus per-agent + per-session btree timelines. Full DDL in [`hermes_pgvector/migrations/001_schema.sql`](hermes_pgvector/migrations/001_schema.sql).
372
441
 
442
+ `vector(768)` is the size migration 001 creates. A deployment on a model with a different output size changes both columns and sets `embed_dim` to match; see [Changing the embedding dimension](#changing-the-embedding-dimension).
443
+
373
444
  ## Tests
374
445
 
375
446
  ```bash
@@ -169,7 +169,7 @@ Found in production: one `memory_entries` row sat with a NULL embedding and **ze
169
169
 
170
170
  ## New in v0.5.2 — documentation only
171
171
 
172
- **No code changes.** `git diff v0.5.1..v0.5.2` touches only `README.md` and one test file; nothing under `hermes_pgvector/` differs, so the installed behaviour is byte-for-byte identical to v0.5.1. There is no reason to redeploy for this release.
172
+ **No behaviour changes.** `git diff v0.5.1..v0.5.2` shows four files: `README.md`, one test file, and the two version strings (`pyproject.toml` and `hermes_pgvector/plugin.yaml`). The only packaged file that differs is `plugin.yaml`, and only its `version:` line — no logic changed anywhere, so an installed 0.5.2 behaves identically to 0.5.1. There is no reason to redeploy for this release.
173
173
 
174
174
  It exists because PyPI renders a project's README **frozen at upload time**: two fixes that landed after v0.5.1 shipped were visible on GitHub but not on the package page.
175
175
 
@@ -178,6 +178,21 @@ It exists because PyPI renders a project's README **frozen at upload time**: two
178
178
 
179
179
  If you are on v0.5.1 you already have every fix in this release. If you are on **v0.5.0 or earlier, upgrade** — v0.5.1 fixed a data-loss bug where a single `memory remove` deleted a whole theme's mirrored memory.
180
180
 
181
+ ## New in v0.5.3 - configurable embedding model (dimension, auth, protocol)
182
+
183
+ **Defaults are unchanged:** 768 dimensions, no `Authorization` header, `auto` protocol. No schema changes and no new migrations; a deployment that sets none of the new keys behaves as before, apart from the two fixes at the end of this list.
184
+
185
+ - **The embedding dimension is configuration, not code.** `embed_dim` (default `768`) replaces the literal 768 in the response check, in the backfill dimension guard (`backfill_null_embeddings(expected_dim=...)`), and in the `stats` dry-run. The check was moved, not relaxed: a mismatch still fails fast with `expected N dims (embed_dim), got M`. Changing the value on a database that already holds vectors needs a column migration; see [Changing the embedding dimension](#changing-the-embedding-dimension).
186
+ - **Bearer auth for hosted endpoints.** `embed_api_key_env` holds the *name* of an environment variable (for example `OPENROUTER_API_KEY`). When that variable is set and non-empty, the plugin sends `Authorization: Bearer <value>`. The value is read at call time and is never logged, stored in config, or included in exception messages. Unset or empty means no header, as before.
187
+ - **Explicit protocol selection.** `embed_protocol: openai` uses only `/v1/embeddings`, so a 401 or an unknown-model error from a hosted endpoint is reported as-is instead of being replaced by a 404 from the Ollama-native fallback. `ollama` uses only `/api/embed`. `auto` keeps the old try-OpenAI-then-Ollama behaviour. Unknown values fall back to `auto` with a warning.
188
+ - **One embed path.** Prefetch, both recall tools, the init-time bulk import, the writer drain and `hermes-pgvector backfill` all resolve the endpoint through one helper, so the settings apply the same way everywhere (`stats` reads the same `embed_dim`). Timeouts and retries are unchanged: one attempt on the agent thread, bounded retries on the writer. `backfill` gains `--embed-dim`, `--embed-api-key-env` and `--embed-protocol` (CLI flag > `--config` file > default).
189
+ - **Fixed: embeds broke under hermes-agent's plugin loader.** After it runs the package, `plugins/plugin_loader.py:load_plugin_module` binds every sibling module back onto it, including `setattr(pkg, "embed", <the embed submodule>)`. That replaced the `embed` function the provider called, so every embed raised `TypeError: 'module' object is not callable`. That is not an `EmbeddingError`, so nothing degraded gracefully: prefetch and the recall tools raised out of the hook, and the writer dropped each mirrored write and captured turn outright instead of storing it text-only. Call sites now use a private alias the loader never touches. An external patch that re-binds `embed` inside `register()` is no longer needed, and does no harm if it is still present. `from hermes_pgvector import embed` still works.
190
+ - **Fixed: a read timeout escaped as a bare `TimeoutError`.** urllib wraps errors raised while *sending* a request, but a server that accepts the connection and answers slower than the timeout raises `TimeoutError` from the response read. That slipped past every `except EmbeddingError`: on the agent thread it raised out of prefetch and the recall tools, `auto` never tried its fallback, and on the writer the retries never ran and the write was dropped instead of being stored text-only. It is now an `EmbeddingError`, like every other endpoint failure.
191
+
192
+ ## New in v0.5.4 - psycopg 3.3.6 / psycopg-pool 3.3.2 floor
193
+
194
+ - **Dependency floor raised**: `psycopg[binary]>=3.3.6`, `psycopg-pool>=3.3.2` (upstream patch releases, 2026-09-18). psycopg 3.3.6: Python 3.15 support; a cancelled query no longer waits forever on an unresponsive server (needs libpq 17+); cancels the running query on `SystemExit`; interval `Column.precision` now reports `None` instead of `65535`; fixes dumping nested list subclasses as arrays; discards prepared statements on `DEALLOCATE ALL`; better guards dumping a large int to binary numeric; faster async waits. psycopg-pool 3.3.2: propagates cancellation and other base exceptions raised during a connection check -- relevant here since this package opens one shared `ConnectionPool` across the agent and async-writer threads. No code changes.
195
+
181
196
  ## Multi-agent / per-minion themes
182
197
 
183
198
  Each systemd-run minion sets one header on its OpenAI client; everything else flows automatically:
@@ -241,7 +256,7 @@ That:
241
256
 
242
257
  ```bash
243
258
  # Python deps
244
- pip install 'psycopg[binary]>=3.3.5,<4' 'psycopg-pool>=3.3.1,<4' 'PyYAML>=6.0,<7'
259
+ pip install 'psycopg[binary]>=3.3.6,<4' 'psycopg-pool>=3.3.2,<4' 'PyYAML>=6.0,<7'
245
260
 
246
261
  # Plugin module
247
262
  mkdir -p ~/.hermes/plugins
@@ -290,6 +305,9 @@ plugins:
290
305
  dsn: "dbname=hermes_memory user=hermes host=/var/run/postgresql"
291
306
  embed_url: "http://your-embed-endpoint:11434"
292
307
  embed_model: "nomic-embed-text"
308
+ embed_dim: 768 # v0.5.3: must match the model AND the vector(N) columns
309
+ embed_api_key_env: "" # v0.5.3: NAME of an env var holding a bearer token
310
+ embed_protocol: "auto" # v0.5.3: auto | openai | ollama
293
311
  prefetch_limit: 5
294
312
  min_similarity: 0.30
295
313
  embed_on_write: true
@@ -306,7 +324,58 @@ plugins:
306
324
  embed_write_retries: 2 # writer-path only; hot path stays single-attempt
307
325
  ```
308
326
 
309
- The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL that returns **768-dim vectors** (the schema is hard-coded to `vector(768)` to match `nomic-embed-text`). Use a different model only if it produces 768-dim output, or edit the migration before applying it.
327
+ The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL. Its vectors must be exactly `embed_dim` long, and `embed_dim` must match the database's `vector(N)` columns. Migration 001 creates `vector(768)` to match `nomic-embed-text`, which is why 768 is the default.
328
+
329
+ ### Embedding endpoint keys (v0.5.3)
330
+
331
+ | Key | Default | Meaning |
332
+ |---|---|---|
333
+ | `embed_dim` | `768` | Vector length the model returns. Every embedding is checked against it, and so is the `hermes-pgvector backfill` probe. It must equal the `vector(N)` column size: changing it on an existing database is a migration, see [below](#changing-the-embedding-dimension). |
334
+ | `embed_api_key_env` | unset | **Name** of an environment variable holding a bearer token, e.g. `OPENROUTER_API_KEY`. Never put the token itself in config. The variable is read on every request; when it is set and non-empty the plugin sends `Authorization: Bearer <value>`, otherwise no header. |
335
+ | `embed_protocol` | `auto` | `auto`: try `/v1/embeddings`, then fall back to `/api/embed`. `openai`: `/v1/embeddings` only, so auth and model errors surface as-is (use this for hosted OpenAI-compatible APIs). `ollama`: `/api/embed` only. Unknown values fall back to `auto` with a warning. |
336
+
337
+ Example: OpenAI `text-embedding-3-small` (1536 dimensions) through OpenRouter:
338
+
339
+ ```yaml
340
+ plugins:
341
+ pgvector:
342
+ embed_url: "https://openrouter.ai/api" # the plugin appends /v1/embeddings
343
+ embed_model: "openai/text-embedding-3-small"
344
+ embed_dim: 1536
345
+ embed_api_key_env: "OPENROUTER_API_KEY" # the variable's NAME, not the key
346
+ embed_protocol: "openai"
347
+ ```
348
+
349
+ The variable has to be in the environment of every process that loads the provider (each hermes service) and of any `hermes-pgvector backfill` job. To call OpenAI directly instead, use `embed_url: "https://api.openai.com"`, `embed_model: "text-embedding-3-small"` and a variable holding an OpenAI key.
350
+
351
+ ### Changing the embedding dimension
352
+
353
+ `embed_dim` has to agree with the columns, so switching to a model with a different output size is a migration, not a config edit. Vectors from two different models are not comparable anyway, so every row must be re-embedded. Until config and columns agree, Postgres rejects each write whose vector has the wrong length (`expected 1536 dimensions, not 768`), and the whole row is lost, not stored text-only. Stop the services first.
354
+
355
+ The shipped migration files are not meant to be edited for this. As the table owner:
356
+
357
+ ```sql
358
+ -- 1. With every hermes service that loads the provider stopped:
359
+ BEGIN;
360
+ DROP INDEX IF EXISTS ix_memory_entries_embedding_hnsw;
361
+ DROP INDEX IF EXISTS ix_conversations_embedding_hnsw;
362
+ ALTER TABLE memory_entries ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
363
+ ALTER TABLE conversations ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
364
+ COMMIT;
365
+ ```
366
+
367
+ 2. Set `embed_model` and `embed_dim` (plus `embed_url`, `embed_api_key_env` and `embed_protocol` as needed), then start the services. New writes are embedded with the new model.
368
+ 3. Re-embed the existing rows, which are all NULL now: `hermes-pgvector backfill --config $HERMES_HOME/config.yaml`. Repeat until every table reports `remaining: 0`. If the endpoint does not return `embed_dim`-length vectors, the run aborts on its first probe, before touching any row, and the logged warning names both sizes.
369
+ 4. Rebuild the HNSW indexes with the shipped tuning. Building them after the backfill is faster than maintaining them during it:
370
+
371
+ ```sql
372
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_memory_entries_embedding_hnsw
373
+ ON memory_entries USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
374
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_conversations_embedding_hnsw
375
+ ON conversations USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
376
+ ```
377
+
378
+ Until step 3 completes, rows without a vector are reachable only through full-text recall (`hybrid_search: true`). pgvector's HNSW index supports `vector` columns of up to 2,000 dimensions. The plugin maintains only `memory_entries` and `conversations`; any other embedding columns in the same database need the same change from whatever writes them.
310
379
 
311
380
  ## Schema
312
381
 
@@ -337,6 +406,8 @@ CREATE TABLE conversations (
337
406
 
338
407
  Indexes: HNSW on each `embedding` column (m=16, ef_construction=64) plus per-agent + per-session btree timelines. Full DDL in [`hermes_pgvector/migrations/001_schema.sql`](hermes_pgvector/migrations/001_schema.sql).
339
408
 
409
+ `vector(768)` is the size migration 001 creates. A deployment on a model with a different output size changes both columns and sets `embed_dim` to match; see [Changing the embedding dimension](#changing-the-embedding-dimension).
410
+
340
411
  ## Tests
341
412
 
342
413
  ```bash
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: hermes-memory-pgvector
3
- Version: 0.5.2
3
+ Version: 0.5.4
4
4
  Summary: Postgres + pgvector memory provider plugin for hermes-agent. Multi-agent storage layer with per-minion themes, identity governance, agent attribution + delegation provenance, async writer, no LLM in the memory hot path.
5
5
  Author: Andrea Borghi
6
6
  License: BSD-3-Clause
@@ -24,8 +24,8 @@ Classifier: Topic :: System :: Distributed Computing
24
24
  Requires-Python: >=3.11
25
25
  Description-Content-Type: text/markdown
26
26
  License-File: LICENSE
27
- Requires-Dist: psycopg[binary]<4,>=3.3.5
28
- Requires-Dist: psycopg-pool<4,>=3.3.1
27
+ Requires-Dist: psycopg[binary]<4,>=3.3.6
28
+ Requires-Dist: psycopg-pool<4,>=3.3.2
29
29
  Requires-Dist: PyYAML<7,>=6.0
30
30
  Provides-Extra: test
31
31
  Requires-Dist: pytest<9,>=7.4; extra == "test"
@@ -202,7 +202,7 @@ Found in production: one `memory_entries` row sat with a NULL embedding and **ze
202
202
 
203
203
  ## New in v0.5.2 — documentation only
204
204
 
205
- **No code changes.** `git diff v0.5.1..v0.5.2` touches only `README.md` and one test file; nothing under `hermes_pgvector/` differs, so the installed behaviour is byte-for-byte identical to v0.5.1. There is no reason to redeploy for this release.
205
+ **No behaviour changes.** `git diff v0.5.1..v0.5.2` shows four files: `README.md`, one test file, and the two version strings (`pyproject.toml` and `hermes_pgvector/plugin.yaml`). The only packaged file that differs is `plugin.yaml`, and only its `version:` line — no logic changed anywhere, so an installed 0.5.2 behaves identically to 0.5.1. There is no reason to redeploy for this release.
206
206
 
207
207
  It exists because PyPI renders a project's README **frozen at upload time**: two fixes that landed after v0.5.1 shipped were visible on GitHub but not on the package page.
208
208
 
@@ -211,6 +211,21 @@ It exists because PyPI renders a project's README **frozen at upload time**: two
211
211
 
212
212
  If you are on v0.5.1 you already have every fix in this release. If you are on **v0.5.0 or earlier, upgrade** — v0.5.1 fixed a data-loss bug where a single `memory remove` deleted a whole theme's mirrored memory.
213
213
 
214
+ ## New in v0.5.3 - configurable embedding model (dimension, auth, protocol)
215
+
216
+ **Defaults are unchanged:** 768 dimensions, no `Authorization` header, `auto` protocol. No schema changes and no new migrations; a deployment that sets none of the new keys behaves as before, apart from the two fixes at the end of this list.
217
+
218
+ - **The embedding dimension is configuration, not code.** `embed_dim` (default `768`) replaces the literal 768 in the response check, in the backfill dimension guard (`backfill_null_embeddings(expected_dim=...)`), and in the `stats` dry-run. The check was moved, not relaxed: a mismatch still fails fast with `expected N dims (embed_dim), got M`. Changing the value on a database that already holds vectors needs a column migration; see [Changing the embedding dimension](#changing-the-embedding-dimension).
219
+ - **Bearer auth for hosted endpoints.** `embed_api_key_env` holds the *name* of an environment variable (for example `OPENROUTER_API_KEY`). When that variable is set and non-empty, the plugin sends `Authorization: Bearer <value>`. The value is read at call time and is never logged, stored in config, or included in exception messages. Unset or empty means no header, as before.
220
+ - **Explicit protocol selection.** `embed_protocol: openai` uses only `/v1/embeddings`, so a 401 or an unknown-model error from a hosted endpoint is reported as-is instead of being replaced by a 404 from the Ollama-native fallback. `ollama` uses only `/api/embed`. `auto` keeps the old try-OpenAI-then-Ollama behaviour. Unknown values fall back to `auto` with a warning.
221
+ - **One embed path.** Prefetch, both recall tools, the init-time bulk import, the writer drain and `hermes-pgvector backfill` all resolve the endpoint through one helper, so the settings apply the same way everywhere (`stats` reads the same `embed_dim`). Timeouts and retries are unchanged: one attempt on the agent thread, bounded retries on the writer. `backfill` gains `--embed-dim`, `--embed-api-key-env` and `--embed-protocol` (CLI flag > `--config` file > default).
222
+ - **Fixed: embeds broke under hermes-agent's plugin loader.** After it runs the package, `plugins/plugin_loader.py:load_plugin_module` binds every sibling module back onto it, including `setattr(pkg, "embed", <the embed submodule>)`. That replaced the `embed` function the provider called, so every embed raised `TypeError: 'module' object is not callable`. That is not an `EmbeddingError`, so nothing degraded gracefully: prefetch and the recall tools raised out of the hook, and the writer dropped each mirrored write and captured turn outright instead of storing it text-only. Call sites now use a private alias the loader never touches. An external patch that re-binds `embed` inside `register()` is no longer needed, and does no harm if it is still present. `from hermes_pgvector import embed` still works.
223
+ - **Fixed: a read timeout escaped as a bare `TimeoutError`.** urllib wraps errors raised while *sending* a request, but a server that accepts the connection and answers slower than the timeout raises `TimeoutError` from the response read. That slipped past every `except EmbeddingError`: on the agent thread it raised out of prefetch and the recall tools, `auto` never tried its fallback, and on the writer the retries never ran and the write was dropped instead of being stored text-only. It is now an `EmbeddingError`, like every other endpoint failure.
224
+
225
+ ## New in v0.5.4 - psycopg 3.3.6 / psycopg-pool 3.3.2 floor
226
+
227
+ - **Dependency floor raised**: `psycopg[binary]>=3.3.6`, `psycopg-pool>=3.3.2` (upstream patch releases, 2026-09-18). psycopg 3.3.6: Python 3.15 support; a cancelled query no longer waits forever on an unresponsive server (needs libpq 17+); cancels the running query on `SystemExit`; interval `Column.precision` now reports `None` instead of `65535`; fixes dumping nested list subclasses as arrays; discards prepared statements on `DEALLOCATE ALL`; better guards dumping a large int to binary numeric; faster async waits. psycopg-pool 3.3.2: propagates cancellation and other base exceptions raised during a connection check -- relevant here since this package opens one shared `ConnectionPool` across the agent and async-writer threads. No code changes.
228
+
214
229
  ## Multi-agent / per-minion themes
215
230
 
216
231
  Each systemd-run minion sets one header on its OpenAI client; everything else flows automatically:
@@ -274,7 +289,7 @@ That:
274
289
 
275
290
  ```bash
276
291
  # Python deps
277
- pip install 'psycopg[binary]>=3.3.5,<4' 'psycopg-pool>=3.3.1,<4' 'PyYAML>=6.0,<7'
292
+ pip install 'psycopg[binary]>=3.3.6,<4' 'psycopg-pool>=3.3.2,<4' 'PyYAML>=6.0,<7'
278
293
 
279
294
  # Plugin module
280
295
  mkdir -p ~/.hermes/plugins
@@ -323,6 +338,9 @@ plugins:
323
338
  dsn: "dbname=hermes_memory user=hermes host=/var/run/postgresql"
324
339
  embed_url: "http://your-embed-endpoint:11434"
325
340
  embed_model: "nomic-embed-text"
341
+ embed_dim: 768 # v0.5.3: must match the model AND the vector(N) columns
342
+ embed_api_key_env: "" # v0.5.3: NAME of an env var holding a bearer token
343
+ embed_protocol: "auto" # v0.5.3: auto | openai | ollama
326
344
  prefetch_limit: 5
327
345
  min_similarity: 0.30
328
346
  embed_on_write: true
@@ -339,7 +357,58 @@ plugins:
339
357
  embed_write_retries: 2 # writer-path only; hot path stays single-attempt
340
358
  ```
341
359
 
342
- The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL that returns **768-dim vectors** (the schema is hard-coded to `vector(768)` to match `nomic-embed-text`). Use a different model only if it produces 768-dim output, or edit the migration before applying it.
360
+ The embed endpoint can be any OpenAI-compatible `/v1/embeddings` or Ollama-native `/api/embed` URL. Its vectors must be exactly `embed_dim` long, and `embed_dim` must match the database's `vector(N)` columns. Migration 001 creates `vector(768)` to match `nomic-embed-text`, which is why 768 is the default.
361
+
362
+ ### Embedding endpoint keys (v0.5.3)
363
+
364
+ | Key | Default | Meaning |
365
+ |---|---|---|
366
+ | `embed_dim` | `768` | Vector length the model returns. Every embedding is checked against it, and so is the `hermes-pgvector backfill` probe. It must equal the `vector(N)` column size: changing it on an existing database is a migration, see [below](#changing-the-embedding-dimension). |
367
+ | `embed_api_key_env` | unset | **Name** of an environment variable holding a bearer token, e.g. `OPENROUTER_API_KEY`. Never put the token itself in config. The variable is read on every request; when it is set and non-empty the plugin sends `Authorization: Bearer <value>`, otherwise no header. |
368
+ | `embed_protocol` | `auto` | `auto`: try `/v1/embeddings`, then fall back to `/api/embed`. `openai`: `/v1/embeddings` only, so auth and model errors surface as-is (use this for hosted OpenAI-compatible APIs). `ollama`: `/api/embed` only. Unknown values fall back to `auto` with a warning. |
369
+
370
+ Example: OpenAI `text-embedding-3-small` (1536 dimensions) through OpenRouter:
371
+
372
+ ```yaml
373
+ plugins:
374
+ pgvector:
375
+ embed_url: "https://openrouter.ai/api" # the plugin appends /v1/embeddings
376
+ embed_model: "openai/text-embedding-3-small"
377
+ embed_dim: 1536
378
+ embed_api_key_env: "OPENROUTER_API_KEY" # the variable's NAME, not the key
379
+ embed_protocol: "openai"
380
+ ```
381
+
382
+ The variable has to be in the environment of every process that loads the provider (each hermes service) and of any `hermes-pgvector backfill` job. To call OpenAI directly instead, use `embed_url: "https://api.openai.com"`, `embed_model: "text-embedding-3-small"` and a variable holding an OpenAI key.
383
+
384
+ ### Changing the embedding dimension
385
+
386
+ `embed_dim` has to agree with the columns, so switching to a model with a different output size is a migration, not a config edit. Vectors from two different models are not comparable anyway, so every row must be re-embedded. Until config and columns agree, Postgres rejects each write whose vector has the wrong length (`expected 1536 dimensions, not 768`), and the whole row is lost, not stored text-only. Stop the services first.
387
+
388
+ The shipped migration files are not meant to be edited for this. As the table owner:
389
+
390
+ ```sql
391
+ -- 1. With every hermes service that loads the provider stopped:
392
+ BEGIN;
393
+ DROP INDEX IF EXISTS ix_memory_entries_embedding_hnsw;
394
+ DROP INDEX IF EXISTS ix_conversations_embedding_hnsw;
395
+ ALTER TABLE memory_entries ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
396
+ ALTER TABLE conversations ALTER COLUMN embedding TYPE vector(1536) USING NULL::vector(1536);
397
+ COMMIT;
398
+ ```
399
+
400
+ 2. Set `embed_model` and `embed_dim` (plus `embed_url`, `embed_api_key_env` and `embed_protocol` as needed), then start the services. New writes are embedded with the new model.
401
+ 3. Re-embed the existing rows, which are all NULL now: `hermes-pgvector backfill --config $HERMES_HOME/config.yaml`. Repeat until every table reports `remaining: 0`. If the endpoint does not return `embed_dim`-length vectors, the run aborts on its first probe, before touching any row, and the logged warning names both sizes.
402
+ 4. Rebuild the HNSW indexes with the shipped tuning. Building them after the backfill is faster than maintaining them during it:
403
+
404
+ ```sql
405
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_memory_entries_embedding_hnsw
406
+ ON memory_entries USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
407
+ CREATE INDEX CONCURRENTLY IF NOT EXISTS ix_conversations_embedding_hnsw
408
+ ON conversations USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 64);
409
+ ```
410
+
411
+ Until step 3 completes, rows without a vector are reachable only through full-text recall (`hybrid_search: true`). pgvector's HNSW index supports `vector` columns of up to 2,000 dimensions. The plugin maintains only `memory_entries` and `conversations`; any other embedding columns in the same database need the same change from whatever writes them.
343
412
 
344
413
  ## Schema
345
414
 
@@ -370,6 +439,8 @@ CREATE TABLE conversations (
370
439
 
371
440
  Indexes: HNSW on each `embedding` column (m=16, ef_construction=64) plus per-agent + per-session btree timelines. Full DDL in [`hermes_pgvector/migrations/001_schema.sql`](hermes_pgvector/migrations/001_schema.sql).
372
441
 
442
+ `vector(768)` is the size migration 001 creates. A deployment on a model with a different output size changes both columns and sets `embed_dim` to match; see [Changing the embedding dimension](#changing-the-embedding-dimension).
443
+
373
444
  ## Tests
374
445
 
375
446
  ```bash
@@ -20,12 +20,14 @@ hermes_pgvector/migrations/003_hybrid_search_fts.sql
20
20
  hermes_pgvector/migrations/004_runtime_grants.sql
21
21
  tests/test_async_writer.py
22
22
  tests/test_config_coercion.py
23
+ tests/test_embed_config.py
23
24
  tests/test_embed_timeouts.py
24
25
  tests/test_empty_content.py
25
26
  tests/test_exclude_identities_live.py
26
27
  tests/test_hybrid_search.py
27
28
  tests/test_identity.py
28
29
  tests/test_install_shim.py
30
+ tests/test_loader_embed_clobber.py
29
31
  tests/test_read_side_gate.py
30
32
  tests/test_save_config_merge.py
31
33
  tests/test_session_switch_contract.py
@@ -0,0 +1,6 @@
1
+ psycopg[binary]<4,>=3.3.6
2
+ psycopg-pool<4,>=3.3.2
3
+ PyYAML<7,>=6.0
4
+
5
+ [test]
6
+ pytest<9,>=7.4