compartment 3.5.2__tar.gz → 4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {compartment-3.5.2 → compartment-4}/PKG-INFO +212 -13
  2. {compartment-3.5.2 → compartment-4}/README.md +211 -12
  3. {compartment-3.5.2 → compartment-4}/pyproject.toml +1 -1
  4. {compartment-3.5.2 → compartment-4}/src/compartment/__init__.py +1 -1
  5. compartment-4/src/compartment/agent_skill.py +143 -0
  6. {compartment-3.5.2 → compartment-4}/src/compartment/cli.py +61 -3
  7. {compartment-3.5.2 → compartment-4}/src/compartment/dash.py +8 -22
  8. compartment-4/src/compartment/data/agent-skill/SKILL.md +49 -0
  9. {compartment-3.5.2 → compartment-4}/src/compartment/data/starter.mpack +0 -0
  10. {compartment-3.5.2 → compartment-4}/src/compartment/embed.py +60 -1
  11. {compartment-3.5.2 → compartment-4}/src/compartment/longmemeval.py +29 -15
  12. {compartment-3.5.2 → compartment-4}/src/compartment/menubar.py +11 -2
  13. compartment-4/src/compartment/ranking.py +186 -0
  14. {compartment-3.5.2 → compartment-4}/src/compartment/server.py +2 -2
  15. {compartment-3.5.2 → compartment-4}/src/compartment/store.py +145 -13
  16. {compartment-3.5.2 → compartment-4}/src/compartment/vault.py +136 -50
  17. {compartment-3.5.2 → compartment-4}/src/compartment/vindex.py +9 -1
  18. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/PKG-INFO +212 -13
  19. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/SOURCES.txt +6 -0
  20. compartment-4/tests/test_agent_skill.py +159 -0
  21. compartment-4/tests/test_long_memory_recall.py +156 -0
  22. compartment-4/tests/test_ranking.py +174 -0
  23. {compartment-3.5.2 → compartment-4}/tests/test_starter_data_quality.py +9 -1
  24. {compartment-3.5.2 → compartment-4}/setup.cfg +0 -0
  25. {compartment-3.5.2 → compartment-4}/src/compartment/acl.py +0 -0
  26. {compartment-3.5.2 → compartment-4}/src/compartment/audit.py +0 -0
  27. {compartment-3.5.2 → compartment-4}/src/compartment/bench.py +0 -0
  28. {compartment-3.5.2 → compartment-4}/src/compartment/claude_desktop.py +0 -0
  29. {compartment-3.5.2 → compartment-4}/src/compartment/claude_hooks.py +0 -0
  30. {compartment-3.5.2 → compartment-4}/src/compartment/claude_memory.py +0 -0
  31. {compartment-3.5.2 → compartment-4}/src/compartment/crypto.py +0 -0
  32. {compartment-3.5.2 → compartment-4}/src/compartment/data/app.icns +0 -0
  33. {compartment-3.5.2 → compartment-4}/src/compartment/data/app.png +0 -0
  34. {compartment-3.5.2 → compartment-4}/src/compartment/data/hermes-plugin/__init__.py +0 -0
  35. {compartment-3.5.2 → compartment-4}/src/compartment/data/hermes-plugin/plugin.yaml +0 -0
  36. {compartment-3.5.2 → compartment-4}/src/compartment/data/menubar.png +0 -0
  37. {compartment-3.5.2 → compartment-4}/src/compartment/data/menubar@2x.png +0 -0
  38. {compartment-3.5.2 → compartment-4}/src/compartment/data/tray.ico +0 -0
  39. {compartment-3.5.2 → compartment-4}/src/compartment/home.py +0 -0
  40. {compartment-3.5.2 → compartment-4}/src/compartment/models/bge-small-en-v1.5-int8/model_quantized.onnx +0 -0
  41. {compartment-3.5.2 → compartment-4}/src/compartment/models/bge-small-en-v1.5-int8/tokenizer.json +0 -0
  42. {compartment-3.5.2 → compartment-4}/src/compartment/offline_guard.py +0 -0
  43. {compartment-3.5.2 → compartment-4}/src/compartment/packs.py +0 -0
  44. {compartment-3.5.2 → compartment-4}/src/compartment/platforms.py +0 -0
  45. {compartment-3.5.2 → compartment-4}/src/compartment/salience.py +0 -0
  46. {compartment-3.5.2 → compartment-4}/src/compartment/selftest.py +0 -0
  47. {compartment-3.5.2 → compartment-4}/src/compartment/session.py +0 -0
  48. {compartment-3.5.2 → compartment-4}/src/compartment/systray.py +0 -0
  49. {compartment-3.5.2 → compartment-4}/src/compartment/vaultfile.py +0 -0
  50. {compartment-3.5.2 → compartment-4}/src/compartment/wire.py +0 -0
  51. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/dependency_links.txt +0 -0
  52. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/entry_points.txt +0 -0
  53. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/requires.txt +0 -0
  54. {compartment-3.5.2 → compartment-4}/src/compartment.egg-info/top_level.txt +0 -0
  55. {compartment-3.5.2 → compartment-4}/tests/test_acl_audit_packs.py +0 -0
  56. {compartment-3.5.2 → compartment-4}/tests/test_acl_precedence.py +0 -0
  57. {compartment-3.5.2 → compartment-4}/tests/test_attempt_pacing.py +0 -0
  58. {compartment-3.5.2 → compartment-4}/tests/test_audit_chain.py +0 -0
  59. {compartment-3.5.2 → compartment-4}/tests/test_autoconnect.py +0 -0
  60. {compartment-3.5.2 → compartment-4}/tests/test_claude_desktop.py +0 -0
  61. {compartment-3.5.2 → compartment-4}/tests/test_claude_hooks_and_integrate.py +0 -0
  62. {compartment-3.5.2 → compartment-4}/tests/test_claude_memory_and_recent.py +0 -0
  63. {compartment-3.5.2 → compartment-4}/tests/test_connect_buttons.py +0 -0
  64. {compartment-3.5.2 → compartment-4}/tests/test_connect_environment.py +0 -0
  65. {compartment-3.5.2 → compartment-4}/tests/test_crypto.py +0 -0
  66. {compartment-3.5.2 → compartment-4}/tests/test_dash.py +0 -0
  67. {compartment-3.5.2 → compartment-4}/tests/test_instructions.py +0 -0
  68. {compartment-3.5.2 → compartment-4}/tests/test_journal_framing.py +0 -0
  69. {compartment-3.5.2 → compartment-4}/tests/test_linux_panel.py +0 -0
  70. {compartment-3.5.2 → compartment-4}/tests/test_login_item_identity.py +0 -0
  71. {compartment-3.5.2 → compartment-4}/tests/test_macos_bundle.py +0 -0
  72. {compartment-3.5.2 → compartment-4}/tests/test_menubar.py +0 -0
  73. {compartment-3.5.2 → compartment-4}/tests/test_offline_guard_paths.py +0 -0
  74. {compartment-3.5.2 → compartment-4}/tests/test_offline_selftest_bench.py +0 -0
  75. {compartment-3.5.2 → compartment-4}/tests/test_ondisk_format.py +0 -0
  76. {compartment-3.5.2 → compartment-4}/tests/test_pack_trust.py +0 -0
  77. {compartment-3.5.2 → compartment-4}/tests/test_platforms.py +0 -0
  78. {compartment-3.5.2 → compartment-4}/tests/test_relations.py +0 -0
  79. {compartment-3.5.2 → compartment-4}/tests/test_salience.py +0 -0
  80. {compartment-3.5.2 → compartment-4}/tests/test_session_entropy.py +0 -0
  81. {compartment-3.5.2 → compartment-4}/tests/test_session_lock.py +0 -0
  82. {compartment-3.5.2 → compartment-4}/tests/test_single_instance.py +0 -0
  83. {compartment-3.5.2 → compartment-4}/tests/test_starter_packs.py +0 -0
  84. {compartment-3.5.2 → compartment-4}/tests/test_starter_visibility.py +0 -0
  85. {compartment-3.5.2 → compartment-4}/tests/test_status_bar_install.py +0 -0
  86. {compartment-3.5.2 → compartment-4}/tests/test_systray.py +0 -0
  87. {compartment-3.5.2 → compartment-4}/tests/test_tamper.py +0 -0
  88. {compartment-3.5.2 → compartment-4}/tests/test_twofa.py +0 -0
  89. {compartment-3.5.2 → compartment-4}/tests/test_vault_ops.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: compartment
3
- Version: 3.5.2
3
+ Version: 4
4
4
  Summary: Compartment - high-security, fully offline, encrypted vector memory for AI agents (MCP)
5
5
  Project-URL: Homepage, https://github.com/MaxFreedomPollard/Compartment
6
6
  Project-URL: Source, https://github.com/MaxFreedomPollard/Compartment
@@ -55,7 +55,7 @@ on your own computer. Hermes, Claude, OpenClaw and other AI Agents can
55
55
  install in one command. One fully-transferable memory store is shared
56
56
  simultaneously by all agents on the computer. 100% offline: no network, no
57
57
  API key, no cloud account, no telemetry. The embedding model ships inside
58
- the package, and a full search returns in under 9 ms, beating the round-trip
58
+ the package, and a full search returns in about 12 ms, beating the round-trip
59
59
  a hosted memory charges you for. Every byte at rest is AEAD-encrypted, the
60
60
  embedding vectors included, and only your passphrase opens it.
61
61
 
@@ -102,7 +102,18 @@ Then connect it to the agent you use:
102
102
  compartment integrate claude
103
103
  ```
104
104
 
105
- `claude`, `hermes` and `openclaw` are the three auto-connect targets.
105
+ `claude`, `hermes` and `openclaw` are the three auto-connect targets. Each one
106
+ also gets the **`/compartmentalize`** skill installed into its own skills
107
+ directory.
108
+
109
+ **`/compartmentalize` saves the conversation before it is thrown away.** Every
110
+ agent eventually compacts or summarizes a long session, and the summary is
111
+ written by a pass that has no tools, so nothing can be stored from inside it:
112
+ whatever the model did not think to save is simply gone. Type
113
+ `/compartmentalize` and the whole conversation is swept into the vault first -
114
+ people and contacts, credentials and where they live, URLs and hosts, decisions
115
+ and the reasoning behind them, and a narrative of the session itself. Then
116
+ compact, and nothing is lost. It works on its own at any point too.
106
117
 
107
118
  **One click install (for people not good with command line).** Download
108
119
  **Compartment.pkg** from the [latest release](https://github.com/MaxFreedomPollard/Compartment/releases/latest)
@@ -129,6 +140,8 @@ of the box. Every option is in [Configuration](#configuration).
129
140
  bar, the Windows notification area, a window on Linux.
130
141
  - Every feature toggles in that panel instead of a config file:
131
142
  model-independent capture, starter facts in search, auto-lock.
143
+ - `/compartmentalize` is installed into every agent it connects, so one command
144
+ banks a whole conversation before compaction throws it away.
132
145
  - Your vault ships full. The 6,718 seeded facts are ordinary memories,
133
146
  editable and forgettable, and one switch keeps them out of search.
134
147
  - Runs under what you already use: Hermes ("no setup needed"), Claude Code
@@ -151,7 +164,7 @@ of the box. Every option is in [Configuration](#configuration).
151
164
 
152
165
  **Search that beats a network call**
153
166
 
154
- - 0.68 ms vector search. 8.8 ms for the full hybrid pipeline. A cloud memory
167
+ - 0.68 ms vector search. About 12 ms for the full hybrid pipeline. A cloud memory
155
168
  spends longer than that saying hello.
156
169
  - Exact below 20k records: recall = 1.0 by construction, not an
157
170
  approximation.
@@ -199,11 +212,12 @@ user say to email the client?"* later retrieves exactly that record.
199
212
  **Deterministic importance tiers rank recall**: decisions/consent 0.90,
200
213
  personal facts and preferences 0.80, the user's machine and configuration
201
214
  0.75, other substantive statements 0.55, pleasantries 0.20 (kept, ranked
202
- last). The fused score is
203
- `RRF(vector) + RRF(keyword) + 0.02·cosine + 0.006·importance`: cosine
204
- magnitude keeps the genuinely best match on top, importance settles
205
- near-ties in favor of what matters. The agent learns the user and the
206
- computer first, the world second, and forgets nothing.
215
+ last). Importance multiplies a match rather than adding to it, so it settles
216
+ near-ties in favour of what matters and can never surface a memory for a
217
+ question it has nothing to do with. The whole scoring model, and the numbers
218
+ it was chosen against, are in [The mathematics](#the-mathematics). The agent
219
+ learns the user and the computer first, the world second, and forgets
220
+ nothing.
207
221
 
208
222
  **One memory, not two.** Agent hosts increasingly ship a memory of their
209
223
  own - Claude Code keeps per-project Markdown files with an auto-loaded
@@ -268,6 +282,173 @@ offline guarantee absolute and every decision reproducible. Pair Compartment
268
282
  with an offline LLM and the whole agent stack can run usefully with no
269
283
  network at all.
270
284
 
285
+ ## The mathematics
286
+
287
+ Everything below lives in one file, [`src/compartment/ranking.py`](src/compartment/ranking.py),
288
+ which the vault, the dashboard and the benchmark all import. A benchmark score
289
+ is therefore a measurement of the product and not of a copy of it that has
290
+ drifted.
291
+
292
+ ### Storage: a memory is embedded in windows, not truncated
293
+
294
+ The encoder reads 512 tokens. Text past that is not weighted less, it is not
295
+ seen at all, so a long memory used to be searchable only by its opening. On a
296
+ real 6,705-memory vault, 40% of records ran past the window and **57.6% of the
297
+ whole corpus was invisible to semantic search**.
298
+
299
+ So a record is embedded as overlapping windows of `W = 448` tokens at a stride
300
+ of `S = 384`, giving 64 tokens of overlap so no fact is cut in half by a
301
+ boundary, and the record is scored by its best window:
302
+
303
+ ```
304
+ windows(d) = ceil( max(0, tokens(d) - W) / S ) + 1 capped at 64
305
+
306
+ s_vec(d) = max over windows w of d : cos(q, w)
307
+ ```
308
+
309
+ Max-pooling, not averaging: a memory is relevant if **any** part of it is, and
310
+ an average would punish a long memory for the parts that are about something
311
+ else. With one window per record it reduces exactly to the old behaviour, so it
312
+ can never be worse for a short memory. The cost is small because most memories
313
+ are short: on that vault, 6,705 records produced 6,785 windows.
314
+
315
+ Windows are measured in model tokens, never characters. A character budget is
316
+ wrong by a factor of three between prose and a hex digest, and being wrong here
317
+ means silently dropping the end of a memory.
318
+
319
+ ### Recall: two channels, combined as evidence rather than added
320
+
321
+ Two indexes look for a memory and they answer different questions. The vector
322
+ index answers *what does this mean*. The keyword index answers *what does this
323
+ say*. Their scores are not denominated in the same thing, and combining them is
324
+ the entire difficulty.
325
+
326
+ The obvious move, and what Compartment shipped until now, is to add them.
327
+ Adding is the wrong operation: it lets a merely-good semantic match outvote
328
+ conclusive literal evidence. Searching a real vault for a commit sha occurring
329
+ in exactly one memory out of 6,705 returned that memory **below ten paraphrases
330
+ of it** - the keyword index had ranked it first and the sum buried it.
331
+
332
+ The two channels are not addends, they are **alternatives**: either one alone
333
+ can establish relevance. That is a soft OR over independent evidence,
334
+
335
+ ```
336
+ P(relevant) = 1 - (1 - p_vec)(1 - p_lex)
337
+ ```
338
+
339
+ and the score is its logarithm, which ranks identically while continuing to
340
+ spread results apart near the top instead of saturating at 1:
341
+
342
+ ```
343
+ score(d) = - w_vec · log(1 - p_vec(d)) - w_lex · log(1 - p_lex(d))
344
+
345
+ w_vec = 0.75 w_lex = 0.25
346
+ ```
347
+
348
+ Either channel approaching certainty carries the memory on its own, and neither
349
+ can veto the other.
350
+
351
+ **Reading a cosine as a probability.** An L2-normalized encoder gives cosines
352
+ that are comparable *across* queries, so they map through fixed bounds.
353
+ Per-query min-max normalization is the obvious alternative and it is a trap: it
354
+ rescales the best hit of a hopeless query up to 1.0 and throws that calibration
355
+ away.
356
+
357
+ ```
358
+ p_vec(d) = clamp( (cos(q, d) - 0.25) / (0.85 - 0.25), 0, 0.88 )
359
+ ```
360
+
361
+ That ceiling of 0.88 is doing real work. A cosine is a similarity, never an
362
+ identity: an encoder can say *this is about the same thing*, but it can never
363
+ say *this is the record you named*. A literal match on a string unique to one
364
+ memory can say exactly that. So the semantic channel is capped below the
365
+ certainty the literal channel may reach, and the bound is forced rather than
366
+ chosen - the literal channel tops out at `0.25 · -log(1 - 0.999) = 1.727`, so
367
+ the cap must satisfy `0.75 · -log(1 - cap) < 1.727`, giving `cap < 0.90`.
368
+
369
+ **Reading a keyword hit as a probability, and deliberately not with BM25.**
370
+ BM25 answers *how well does this match*, which is not what settles a contest
371
+ against a semantic hit. What settles it is how unlikely the match was by
372
+ chance. So each query term carries its self-information over the vault, and a
373
+ memory scores the **fraction of the query's information it accounts for**:
374
+
375
+ ```
376
+ I(t) = log( N / (1 + df(t)) ) N = records in the vault
377
+
378
+ p_lex(d) = ( Σ I(t) for query terms t present in d ) / ( Σ I(t) for all t )
379
+ ```
380
+
381
+ A term unique to one memory is near-conclusive evidence. A term appearing in a
382
+ tenth of the vault is nearly none, whatever its BM25 happens to be. This is the
383
+ piece that makes a literal hit and a semantic hit comparable at all.
384
+
385
+ The keyword index is queried as AND first, since an exact phrase match is the
386
+ strongest signal available. FTS5's implicit AND means a nine-word question has
387
+ to appear word for word, so when AND finds nothing it falls back to OR over
388
+ only the terms carrying information - anything appearing in more than 10% of
389
+ records is dropped. That ceiling is measured from the vault rather than taken
390
+ from an English stopword list, so it behaves the same for a vault full of code,
391
+ of names, or of another language.
392
+
393
+ A small rank-agreement residue is added, the one thing reciprocal-rank fusion is
394
+ genuinely good at, sized to break ties rather than decide them:
395
+
396
+ ```
397
+ + w_rrf · k · [ 1/(k + rank_vec) + 1/(k + rank_lex) ] w_rrf = 0.10, k = 20
398
+ ```
399
+
400
+ ### Importance ranking: priors multiply, they never add
401
+
402
+ ```
403
+ final(d) = score(d) · ( 1 + w_imp · (2·importance(d) - 1)
404
+ + w_rec · 2^( -age_days(d) / 180 ) )
405
+
406
+ w_imp = 0.15 w_rec = 0.10
407
+ ```
408
+
409
+ **Multiplicative, so a prior can only reorder a memory that already matched.**
410
+ An additive prior lets a very important memory surface for a question it has
411
+ nothing to do with, which is how a memory system starts feeling haunted. A
412
+ memory that matched nothing scores zero, and nothing can lift it off zero.
413
+
414
+ **Centred on the 0.5 default**, which is why `2·importance - 1` appears rather
415
+ than `importance`. Every unweighted memory carries 0.5, including the thousands
416
+ of starting facts a vault ships with. Uncentred, they all collect the same
417
+ silent boost, which is another way of saying importance did nothing at all.
418
+ Centred, an unweighted memory is exactly neutral and a deliberate weight is the
419
+ only thing that moves.
420
+
421
+ The tiers the capture path writes: decisions and consent 0.90, personal facts
422
+ and preferences 0.80, the user's machine and configuration 0.75, other
423
+ substantive statements 0.55, pleasantries 0.20. Recency halves every 180 days.
424
+
425
+ ### Retrieval order, and why the pool is wide
426
+
427
+ Namespace, tag, date and starter-fact filters run *after* ranking, so a
428
+ candidate pool sized to the number of results requested can be emptied by them
429
+ while matching memories sit just past the cut. The pool starts at 200 per
430
+ channel and widens up to three times when filtering leaves too few.
431
+
432
+ ### Measured
433
+
434
+ Against the previous scorer, end to end through `Vault.search`, on a real
435
+ 6,705-memory vault with 44 queries in four families:
436
+
437
+ | | before | after |
438
+ |---|---|---|
439
+ | Recall@1 | 0.523 | **0.773** |
440
+ | Recall@5 | 0.705 | **0.977** |
441
+ | MRR@10 | 0.601 | **0.845** |
442
+ | nDCG@10 | 0.627 | **0.878** |
443
+ | exact identifiers found in top 5 | 4/10 | **10/10** |
444
+ | facts past the encoder window | 0/6 | **5/6** |
445
+ | paraphrases | 16/16 | 16/16 |
446
+ | median search latency | 4.4 ms | 11.6 ms |
447
+
448
+ Nothing regressed in any family. The weights were chosen from a sensitivity
449
+ sweep and are deliberately round: the result is flat around them, because a
450
+ ranker that only works at `w_lex = 0.37` is a ranker that does not work.
451
+
271
452
  ## Wiring each agent
272
453
 
273
454
  One command per platform. Each installs the package, creates your
@@ -359,11 +540,11 @@ and `compartment bench`.
359
540
  |---|---|
360
541
  | Fresh install → open vault, offline | seconds, zero network |
361
542
  | Vector search, 20k records (HNSW) | p95 0.68 ms |
362
- | Full hybrid search (embed + vector + BM25 + fuse) | p95 8.8 ms |
543
+ | Full hybrid search (embed + windows + BM25 + evidence fusion) | median 11.6 ms, p95 14.7 ms |
363
544
  | Peak RSS, model + vault + index resident | 319 MB |
364
545
  | Store one memory (embed + encrypt + fsync journal) | ~40 ms |
365
546
  | Wheel size, model included | ~30 MB |
366
- | Test suite (crypto, tamper, crash, offline, concurrency, 2FA, graph, dash) | 249 tests, ~60 s |
547
+ | Test suite (crypto, tamper, crash, offline, concurrency, 2FA, graph, dash, ranking) | 566 tests, ~110 s |
367
548
 
368
549
  A single network round-trip to a cloud memory API costs more than this
369
550
  entire pipeline. The property that makes Compartment secure (no plaintext
@@ -501,7 +682,7 @@ which on Linux is the applications menu entry.
501
682
  | `search` / `recent` | find things. `--namespace`, `--tag`, `--top-k`, `--limit`, `--all`, `--json` |
502
683
  | `link` / `relations` / `unlink` | the relation graph, with validity windows (`--from`, `--to`, `--as-of`) |
503
684
  | `panel` (`menubar`, `tray`) | the app. `--show`, `--self-check`, `--render`, `--login` |
504
- | `integrate <agent>` | wire claude, hermes or openclaw. `--no-import`, `--no-hooks` |
685
+ | `integrate <agent>` | wire claude, hermes or openclaw, and install `/compartmentalize` for it. `--no-import`, `--no-hooks` |
505
686
  | `hook` | capture hook: `install --pin-vault`, `uninstall`, `status`, `capture` |
506
687
  | `serve` | the MCP server, over stdio |
507
688
  | `dash` | read the vault in a browser: 127.0.0.1, one-time token, GET only |
@@ -511,7 +692,7 @@ which on Linux is the applications menu entry.
511
692
  | `2fa` | `enable`, `disable`, `status` - a keyfile as a second factor |
512
693
  | `audit` | `verify`, `repair` the hash-chained history |
513
694
  | `pack` | `build`, `install`, `remove`, `list`, `export` signed memory packs (`--trusted-key`) |
514
- | `reindex` | `--int8`, `--f32`, `--re-embed`, `--model` |
695
+ | `reindex` | rebuild the index, and give long records the embedding windows they are missing. `--int8`, `--f32`, `--re-embed`, `--model` |
515
696
  | `bench` | `--records`, `--longmemeval`, `--variant`, `--limit` |
516
697
  | `setup` | `download-model`, `download-longmemeval`, `airgap-bundle` |
517
698
  | `update` | upgrade in place. `--source` takes GitHub main, `--no-app` skips the restart |
@@ -520,6 +701,24 @@ which on Linux is the applications menu entry.
520
701
  Global flags, before the command: `--vault PATH`, `--caller NAME`,
521
702
  `--keyfile PATH`, `--assert-offline`, `--version`.
522
703
 
704
+ ### The /compartmentalize skill
705
+
706
+ `compartment integrate <agent>` writes one file into that agent's own skills
707
+ directory, and `compartment uninstall` takes it back:
708
+
709
+ | Agent | Path |
710
+ |---|---|
711
+ | Claude Code | `~/.claude/skills/compartmentalize/SKILL.md` |
712
+ | Hermes | `$HERMES_HOME` or `~/.hermes/skills/compartmentalize/SKILL.md` |
713
+ | OpenClaw | `$OPENCLAW_HOME` or `~/.openclaw/skills/compartmentalize/SKILL.md` |
714
+
715
+ All three read the same Agent Skills layout, so it is one packaged file. It is
716
+ user-invoked only: no agent runs it on its own guess. Edit your copy freely -
717
+ a later install backs up anything that differs rather than overwriting it, and
718
+ leaves the backup behind when the skill is removed. Invoking it makes the agent
719
+ sweep the conversation and write to the vault, so expect a burst of
720
+ `memory_store` calls; that is the point of it.
721
+
523
722
  ### Settings file
524
723
 
525
724
  `<vault>.config.json`, beside the vault, holding grants per caller and:
@@ -9,7 +9,7 @@ on your own computer. Hermes, Claude, OpenClaw and other AI Agents can
9
9
  install in one command. One fully-transferable memory store is shared
10
10
  simultaneously by all agents on the computer. 100% offline: no network, no
11
11
  API key, no cloud account, no telemetry. The embedding model ships inside
12
- the package, and a full search returns in under 9 ms, beating the round-trip
12
+ the package, and a full search returns in about 12 ms, beating the round-trip
13
13
  a hosted memory charges you for. Every byte at rest is AEAD-encrypted, the
14
14
  embedding vectors included, and only your passphrase opens it.
15
15
 
@@ -56,7 +56,18 @@ Then connect it to the agent you use:
56
56
  compartment integrate claude
57
57
  ```
58
58
 
59
- `claude`, `hermes` and `openclaw` are the three auto-connect targets.
59
+ `claude`, `hermes` and `openclaw` are the three auto-connect targets. Each one
60
+ also gets the **`/compartmentalize`** skill installed into its own skills
61
+ directory.
62
+
63
+ **`/compartmentalize` saves the conversation before it is thrown away.** Every
64
+ agent eventually compacts or summarizes a long session, and the summary is
65
+ written by a pass that has no tools, so nothing can be stored from inside it:
66
+ whatever the model did not think to save is simply gone. Type
67
+ `/compartmentalize` and the whole conversation is swept into the vault first -
68
+ people and contacts, credentials and where they live, URLs and hosts, decisions
69
+ and the reasoning behind them, and a narrative of the session itself. Then
70
+ compact, and nothing is lost. It works on its own at any point too.
60
71
 
61
72
  **One click install (for people not good with command line).** Download
62
73
  **Compartment.pkg** from the [latest release](https://github.com/MaxFreedomPollard/Compartment/releases/latest)
@@ -83,6 +94,8 @@ of the box. Every option is in [Configuration](#configuration).
83
94
  bar, the Windows notification area, a window on Linux.
84
95
  - Every feature toggles in that panel instead of a config file:
85
96
  model-independent capture, starter facts in search, auto-lock.
97
+ - `/compartmentalize` is installed into every agent it connects, so one command
98
+ banks a whole conversation before compaction throws it away.
86
99
  - Your vault ships full. The 6,718 seeded facts are ordinary memories,
87
100
  editable and forgettable, and one switch keeps them out of search.
88
101
  - Runs under what you already use: Hermes ("no setup needed"), Claude Code
@@ -105,7 +118,7 @@ of the box. Every option is in [Configuration](#configuration).
105
118
 
106
119
  **Search that beats a network call**
107
120
 
108
- - 0.68 ms vector search. 8.8 ms for the full hybrid pipeline. A cloud memory
121
+ - 0.68 ms vector search. About 12 ms for the full hybrid pipeline. A cloud memory
109
122
  spends longer than that saying hello.
110
123
  - Exact below 20k records: recall = 1.0 by construction, not an
111
124
  approximation.
@@ -153,11 +166,12 @@ user say to email the client?"* later retrieves exactly that record.
153
166
  **Deterministic importance tiers rank recall**: decisions/consent 0.90,
154
167
  personal facts and preferences 0.80, the user's machine and configuration
155
168
  0.75, other substantive statements 0.55, pleasantries 0.20 (kept, ranked
156
- last). The fused score is
157
- `RRF(vector) + RRF(keyword) + 0.02·cosine + 0.006·importance`: cosine
158
- magnitude keeps the genuinely best match on top, importance settles
159
- near-ties in favor of what matters. The agent learns the user and the
160
- computer first, the world second, and forgets nothing.
169
+ last). Importance multiplies a match rather than adding to it, so it settles
170
+ near-ties in favour of what matters and can never surface a memory for a
171
+ question it has nothing to do with. The whole scoring model, and the numbers
172
+ it was chosen against, are in [The mathematics](#the-mathematics). The agent
173
+ learns the user and the computer first, the world second, and forgets
174
+ nothing.
161
175
 
162
176
  **One memory, not two.** Agent hosts increasingly ship a memory of their
163
177
  own - Claude Code keeps per-project Markdown files with an auto-loaded
@@ -222,6 +236,173 @@ offline guarantee absolute and every decision reproducible. Pair Compartment
222
236
  with an offline LLM and the whole agent stack can run usefully with no
223
237
  network at all.
224
238
 
239
+ ## The mathematics
240
+
241
+ Everything below lives in one file, [`src/compartment/ranking.py`](src/compartment/ranking.py),
242
+ which the vault, the dashboard and the benchmark all import. A benchmark score
243
+ is therefore a measurement of the product and not of a copy of it that has
244
+ drifted.
245
+
246
+ ### Storage: a memory is embedded in windows, not truncated
247
+
248
+ The encoder reads 512 tokens. Text past that is not weighted less, it is not
249
+ seen at all, so a long memory used to be searchable only by its opening. On a
250
+ real 6,705-memory vault, 40% of records ran past the window and **57.6% of the
251
+ whole corpus was invisible to semantic search**.
252
+
253
+ So a record is embedded as overlapping windows of `W = 448` tokens at a stride
254
+ of `S = 384`, giving 64 tokens of overlap so no fact is cut in half by a
255
+ boundary, and the record is scored by its best window:
256
+
257
+ ```
258
+ windows(d) = ceil( max(0, tokens(d) - W) / S ) + 1 capped at 64
259
+
260
+ s_vec(d) = max over windows w of d : cos(q, w)
261
+ ```
262
+
263
+ Max-pooling, not averaging: a memory is relevant if **any** part of it is, and
264
+ an average would punish a long memory for the parts that are about something
265
+ else. With one window per record it reduces exactly to the old behaviour, so it
266
+ can never be worse for a short memory. The cost is small because most memories
267
+ are short: on that vault, 6,705 records produced 6,785 windows.
268
+
269
+ Windows are measured in model tokens, never characters. A character budget is
270
+ wrong by a factor of three between prose and a hex digest, and being wrong here
271
+ means silently dropping the end of a memory.
272
+
273
+ ### Recall: two channels, combined as evidence rather than added
274
+
275
+ Two indexes look for a memory and they answer different questions. The vector
276
+ index answers *what does this mean*. The keyword index answers *what does this
277
+ say*. Their scores are not denominated in the same thing, and combining them is
278
+ the entire difficulty.
279
+
280
+ The obvious move, and what Compartment shipped until now, is to add them.
281
+ Adding is the wrong operation: it lets a merely-good semantic match outvote
282
+ conclusive literal evidence. Searching a real vault for a commit sha occurring
283
+ in exactly one memory out of 6,705 returned that memory **below ten paraphrases
284
+ of it** - the keyword index had ranked it first and the sum buried it.
285
+
286
+ The two channels are not addends, they are **alternatives**: either one alone
287
+ can establish relevance. That is a soft OR over independent evidence,
288
+
289
+ ```
290
+ P(relevant) = 1 - (1 - p_vec)(1 - p_lex)
291
+ ```
292
+
293
+ and the score is its logarithm, which ranks identically while continuing to
294
+ spread results apart near the top instead of saturating at 1:
295
+
296
+ ```
297
+ score(d) = - w_vec · log(1 - p_vec(d)) - w_lex · log(1 - p_lex(d))
298
+
299
+ w_vec = 0.75 w_lex = 0.25
300
+ ```
301
+
302
+ Either channel approaching certainty carries the memory on its own, and neither
303
+ can veto the other.
304
+
305
+ **Reading a cosine as a probability.** An L2-normalized encoder gives cosines
306
+ that are comparable *across* queries, so they map through fixed bounds.
307
+ Per-query min-max normalization is the obvious alternative and it is a trap: it
308
+ rescales the best hit of a hopeless query up to 1.0 and throws that calibration
309
+ away.
310
+
311
+ ```
312
+ p_vec(d) = clamp( (cos(q, d) - 0.25) / (0.85 - 0.25), 0, 0.88 )
313
+ ```
314
+
315
+ That ceiling of 0.88 is doing real work. A cosine is a similarity, never an
316
+ identity: an encoder can say *this is about the same thing*, but it can never
317
+ say *this is the record you named*. A literal match on a string unique to one
318
+ memory can say exactly that. So the semantic channel is capped below the
319
+ certainty the literal channel may reach, and the bound is forced rather than
320
+ chosen - the literal channel tops out at `0.25 · -log(1 - 0.999) = 1.727`, so
321
+ the cap must satisfy `0.75 · -log(1 - cap) < 1.727`, giving `cap < 0.90`.
322
+
323
+ **Reading a keyword hit as a probability, and deliberately not with BM25.**
324
+ BM25 answers *how well does this match*, which is not what settles a contest
325
+ against a semantic hit. What settles it is how unlikely the match was by
326
+ chance. So each query term carries its self-information over the vault, and a
327
+ memory scores the **fraction of the query's information it accounts for**:
328
+
329
+ ```
330
+ I(t) = log( N / (1 + df(t)) ) N = records in the vault
331
+
332
+ p_lex(d) = ( Σ I(t) for query terms t present in d ) / ( Σ I(t) for all t )
333
+ ```
334
+
335
+ A term unique to one memory is near-conclusive evidence. A term appearing in a
336
+ tenth of the vault is nearly none, whatever its BM25 happens to be. This is the
337
+ piece that makes a literal hit and a semantic hit comparable at all.
338
+
339
+ The keyword index is queried as AND first, since an exact phrase match is the
340
+ strongest signal available. FTS5's implicit AND means a nine-word question has
341
+ to appear word for word, so when AND finds nothing it falls back to OR over
342
+ only the terms carrying information - anything appearing in more than 10% of
343
+ records is dropped. That ceiling is measured from the vault rather than taken
344
+ from an English stopword list, so it behaves the same for a vault full of code,
345
+ of names, or of another language.
346
+
347
+ A small rank-agreement residue is added, the one thing reciprocal-rank fusion is
348
+ genuinely good at, sized to break ties rather than decide them:
349
+
350
+ ```
351
+ + w_rrf · k · [ 1/(k + rank_vec) + 1/(k + rank_lex) ] w_rrf = 0.10, k = 20
352
+ ```
353
+
354
+ ### Importance ranking: priors multiply, they never add
355
+
356
+ ```
357
+ final(d) = score(d) · ( 1 + w_imp · (2·importance(d) - 1)
358
+ + w_rec · 2^( -age_days(d) / 180 ) )
359
+
360
+ w_imp = 0.15 w_rec = 0.10
361
+ ```
362
+
363
+ **Multiplicative, so a prior can only reorder a memory that already matched.**
364
+ An additive prior lets a very important memory surface for a question it has
365
+ nothing to do with, which is how a memory system starts feeling haunted. A
366
+ memory that matched nothing scores zero, and nothing can lift it off zero.
367
+
368
+ **Centred on the 0.5 default**, which is why `2·importance - 1` appears rather
369
+ than `importance`. Every unweighted memory carries 0.5, including the thousands
370
+ of starting facts a vault ships with. Uncentred, they all collect the same
371
+ silent boost, which is another way of saying importance did nothing at all.
372
+ Centred, an unweighted memory is exactly neutral and a deliberate weight is the
373
+ only thing that moves.
374
+
375
+ The tiers the capture path writes: decisions and consent 0.90, personal facts
376
+ and preferences 0.80, the user's machine and configuration 0.75, other
377
+ substantive statements 0.55, pleasantries 0.20. Recency halves every 180 days.
378
+
379
+ ### Retrieval order, and why the pool is wide
380
+
381
+ Namespace, tag, date and starter-fact filters run *after* ranking, so a
382
+ candidate pool sized to the number of results requested can be emptied by them
383
+ while matching memories sit just past the cut. The pool starts at 200 per
384
+ channel and widens up to three times when filtering leaves too few.
385
+
386
+ ### Measured
387
+
388
+ Against the previous scorer, end to end through `Vault.search`, on a real
389
+ 6,705-memory vault with 44 queries in four families:
390
+
391
+ | | before | after |
392
+ |---|---|---|
393
+ | Recall@1 | 0.523 | **0.773** |
394
+ | Recall@5 | 0.705 | **0.977** |
395
+ | MRR@10 | 0.601 | **0.845** |
396
+ | nDCG@10 | 0.627 | **0.878** |
397
+ | exact identifiers found in top 5 | 4/10 | **10/10** |
398
+ | facts past the encoder window | 0/6 | **5/6** |
399
+ | paraphrases | 16/16 | 16/16 |
400
+ | median search latency | 4.4 ms | 11.6 ms |
401
+
402
+ Nothing regressed in any family. The weights were chosen from a sensitivity
403
+ sweep and are deliberately round: the result is flat around them, because a
404
+ ranker that only works at `w_lex = 0.37` is a ranker that does not work.
405
+
225
406
  ## Wiring each agent
226
407
 
227
408
  One command per platform. Each installs the package, creates your
@@ -313,11 +494,11 @@ and `compartment bench`.
313
494
  |---|---|
314
495
  | Fresh install → open vault, offline | seconds, zero network |
315
496
  | Vector search, 20k records (HNSW) | p95 0.68 ms |
316
- | Full hybrid search (embed + vector + BM25 + fuse) | p95 8.8 ms |
497
+ | Full hybrid search (embed + windows + BM25 + evidence fusion) | median 11.6 ms, p95 14.7 ms |
317
498
  | Peak RSS, model + vault + index resident | 319 MB |
318
499
  | Store one memory (embed + encrypt + fsync journal) | ~40 ms |
319
500
  | Wheel size, model included | ~30 MB |
320
- | Test suite (crypto, tamper, crash, offline, concurrency, 2FA, graph, dash) | 249 tests, ~60 s |
501
+ | Test suite (crypto, tamper, crash, offline, concurrency, 2FA, graph, dash, ranking) | 566 tests, ~110 s |
321
502
 
322
503
  A single network round-trip to a cloud memory API costs more than this
323
504
  entire pipeline. The property that makes Compartment secure (no plaintext
@@ -455,7 +636,7 @@ which on Linux is the applications menu entry.
455
636
  | `search` / `recent` | find things. `--namespace`, `--tag`, `--top-k`, `--limit`, `--all`, `--json` |
456
637
  | `link` / `relations` / `unlink` | the relation graph, with validity windows (`--from`, `--to`, `--as-of`) |
457
638
  | `panel` (`menubar`, `tray`) | the app. `--show`, `--self-check`, `--render`, `--login` |
458
- | `integrate <agent>` | wire claude, hermes or openclaw. `--no-import`, `--no-hooks` |
639
+ | `integrate <agent>` | wire claude, hermes or openclaw, and install `/compartmentalize` for it. `--no-import`, `--no-hooks` |
459
640
  | `hook` | capture hook: `install --pin-vault`, `uninstall`, `status`, `capture` |
460
641
  | `serve` | the MCP server, over stdio |
461
642
  | `dash` | read the vault in a browser: 127.0.0.1, one-time token, GET only |
@@ -465,7 +646,7 @@ which on Linux is the applications menu entry.
465
646
  | `2fa` | `enable`, `disable`, `status` - a keyfile as a second factor |
466
647
  | `audit` | `verify`, `repair` the hash-chained history |
467
648
  | `pack` | `build`, `install`, `remove`, `list`, `export` signed memory packs (`--trusted-key`) |
468
- | `reindex` | `--int8`, `--f32`, `--re-embed`, `--model` |
649
+ | `reindex` | rebuild the index, and give long records the embedding windows they are missing. `--int8`, `--f32`, `--re-embed`, `--model` |
469
650
  | `bench` | `--records`, `--longmemeval`, `--variant`, `--limit` |
470
651
  | `setup` | `download-model`, `download-longmemeval`, `airgap-bundle` |
471
652
  | `update` | upgrade in place. `--source` takes GitHub main, `--no-app` skips the restart |
@@ -474,6 +655,24 @@ which on Linux is the applications menu entry.
474
655
  Global flags, before the command: `--vault PATH`, `--caller NAME`,
475
656
  `--keyfile PATH`, `--assert-offline`, `--version`.
476
657
 
658
+ ### The /compartmentalize skill
659
+
660
+ `compartment integrate <agent>` writes one file into that agent's own skills
661
+ directory, and `compartment uninstall` takes it back:
662
+
663
+ | Agent | Path |
664
+ |---|---|
665
+ | Claude Code | `~/.claude/skills/compartmentalize/SKILL.md` |
666
+ | Hermes | `$HERMES_HOME` or `~/.hermes/skills/compartmentalize/SKILL.md` |
667
+ | OpenClaw | `$OPENCLAW_HOME` or `~/.openclaw/skills/compartmentalize/SKILL.md` |
668
+
669
+ All three read the same Agent Skills layout, so it is one packaged file. It is
670
+ user-invoked only: no agent runs it on its own guess. Edit your copy freely -
671
+ a later install backs up anything that differs rather than overwriting it, and
672
+ leaves the backup behind when the skill is removed. Invoking it makes the agent
673
+ sweep the conversation and write to the vault, so expect a burst of
674
+ `memory_store` calls; that is the point of it.
675
+
477
676
  ### Settings file
478
677
 
479
678
  `<vault>.config.json`, beside the vault, holding grants per caller and:
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "compartment"
7
- version = "3.5.2"
7
+ version = "4"
8
8
  description = "Compartment - high-security, fully offline, encrypted vector memory for AI agents (MCP)"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.11"
@@ -1,6 +1,6 @@
1
1
  """Compartment - high-security, fully offline, encrypted vector memory for AI agents."""
2
2
 
3
- __version__ = "3.5.2"
3
+ __version__ = "4"
4
4
 
5
5
  from . import offline_guard as _og
6
6