vexicon 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. vexicon-0.5.0/PKG-INFO +100 -0
  2. vexicon-0.5.0/README.md +81 -0
  3. vexicon-0.5.0/pyproject.toml +126 -0
  4. vexicon-0.5.0/pyproject.toml.orig +127 -0
  5. vexicon-0.5.0/src/vexicon/__init__.py +0 -0
  6. vexicon-0.5.0/src/vexicon/client/__init__.py +0 -0
  7. vexicon-0.5.0/src/vexicon/client/hybrid_client.py +417 -0
  8. vexicon-0.5.0/src/vexicon/client/repo.py +94 -0
  9. vexicon-0.5.0/src/vexicon/db.py +42 -0
  10. vexicon-0.5.0/src/vexicon/deps.py +66 -0
  11. vexicon-0.5.0/src/vexicon/embedding.py +129 -0
  12. vexicon-0.5.0/src/vexicon/idle_proxy.py +87 -0
  13. vexicon-0.5.0/src/vexicon/migrations/__init__.py +0 -0
  14. vexicon-0.5.0/src/vexicon/migrations/env.py +36 -0
  15. vexicon-0.5.0/src/vexicon/migrations/script.py.mako +28 -0
  16. vexicon-0.5.0/src/vexicon/migrations/versions/0001_create_collections_documents_and_fts.py +79 -0
  17. vexicon-0.5.0/src/vexicon/migrations/versions/__init__.py +0 -0
  18. vexicon-0.5.0/src/vexicon/models/__init__.py +0 -0
  19. vexicon-0.5.0/src/vexicon/models/base.py +52 -0
  20. vexicon-0.5.0/src/vexicon/models/embeddings.py +23 -0
  21. vexicon-0.5.0/src/vexicon/models/entries.py +67 -0
  22. vexicon-0.5.0/src/vexicon/models/spaces.py +51 -0
  23. vexicon-0.5.0/src/vexicon/orm.py +54 -0
  24. vexicon-0.5.0/src/vexicon/queries/__init__.py +0 -0
  25. vexicon-0.5.0/src/vexicon/queries/filtering.py +134 -0
  26. vexicon-0.5.0/src/vexicon/queries/fusion.py +61 -0
  27. vexicon-0.5.0/src/vexicon/queries/matching.py +9 -0
  28. vexicon-0.5.0/src/vexicon/queries/selects.py +62 -0
  29. vexicon-0.5.0/src/vexicon/resources.py +60 -0
  30. vexicon-0.5.0/src/vexicon/server.py +42 -0
  31. vexicon-0.5.0/src/vexicon/settings.py +64 -0
  32. vexicon-0.5.0/src/vexicon/text_tools.py +148 -0
  33. vexicon-0.5.0/src/vexicon/tools/__init__.py +5 -0
  34. vexicon-0.5.0/src/vexicon/tools/embeddings.py +45 -0
  35. vexicon-0.5.0/src/vexicon/tools/entries.py +120 -0
  36. vexicon-0.5.0/src/vexicon/tools/spaces.py +112 -0
vexicon-0.5.0/PKG-INFO ADDED
@@ -0,0 +1,100 @@
1
+ Metadata-Version: 2.3
2
+ Name: vexicon
3
+ Version: 0.5.0
4
+ Summary: Hybrid vector + keyword MCP server powered by chroma and sqlite
5
+ Author: estasney
6
+ Author-email: estasney <estasney@users.noreply.github.com>
7
+ Requires-Dist: aiosqlite>=0.22.1,<0.23.0
8
+ Requires-Dist: alembic>=1.19.1,<2.0.0
9
+ Requires-Dist: chromadb>=1.5.9,<2.0.0
10
+ Requires-Dist: fastmcp>=4.0.10,<5.0.0
11
+ Requires-Dist: huggingface-hub>=1.28.0,<2.0.0
12
+ Requires-Dist: numpy>=2.5.2,<3.0.0
13
+ Requires-Dist: pydantic>=2.13.4,<3.0.0
14
+ Requires-Dist: pydantic-settings>=2.15.0,<3.0.0
15
+ Requires-Dist: sentence-transformers>=6.0.0,<7.0.0
16
+ Requires-Dist: sqlalchemy>=2.0.52,<3.0.0
17
+ Requires-Python: >=3.12
18
+ Description-Content-Type: text/markdown
19
+
20
+ # vexicon
21
+
22
+ ## About
23
+
24
+ Knowledge store MCP Server.
25
+
26
+ Provides an RRF implementation using SQLite. Tunable via settings.
27
+
28
+ Introduces some opinionated defaults
29
+
30
+
31
+ Notes and reference material live in
32
+ named spaces backed by Chroma collections with a SQLite FTS5 keyword index;
33
+ searches fuse the vector and keyword rankings by reciprocal rank fusion.
34
+
35
+ Run the server over stdio with `uv run vexicon`.
36
+
37
+ Settings are read from `VEXICON_*` environment variables; see
38
+ `src/vexicon/settings.py`. Data is stored under `~/.vexicon` unless
39
+ `VEXICON_PERSISTENT_PATH` and `VEXICON_INDEX_DB_PATH` say otherwise.
40
+
41
+ ## Settings
42
+
43
+ | Variable | Default | Meaning |
44
+ |---------------------------|------------------------|---------------------------------------------------------------------------|
45
+ | `VEXICON_PERSISTENT_PATH` | `~/.vexicon/chroma` | Directory where Chroma stores its database. |
46
+ | `VEXICON_INDEX_DB_PATH` | `~/.vexicon/hybrid.db` | SQLite file holding the keyword index. |
47
+ | `VEXICON_VECTOR_WEIGHT` | `1.0` | Weight of the vector ranking in fusion. |
48
+ | `VEXICON_KEYWORD_WEIGHT` | `1.0` | Weight of the keyword ranking in fusion. |
49
+ | `VEXICON_RRF_RANK_OFFSET` | `60` | Rank offset in reciprocal rank fusion. |
50
+ | `VEXICON_DEVICE` | `auto` | Device that runs embedding models: `auto`, `cpu`, or `cuda`. |
51
+ | `VEXICON_IDLE_SECONDS` | `300` | Seconds without activity before Chroma and embedding models are unloaded. |
52
+
53
+ ## Compared with chroma-mcp
54
+
55
+ Compared against chroma-mcp 0.2.6 and chromadb 1.5.9.
56
+
57
+ ### Hybrid search without Chroma Cloud
58
+
59
+ - vexicon combines vector and BM25 keyword rankings with reciprocal rank
60
+ fusion, using a SQLite FTS5 index beside each local Chroma collection.
61
+
62
+ ### Token usage
63
+
64
+ - vexicon's tool definitions take about 30% fewer tokens than chroma-mcp's.
65
+ - `search` returns at most `limit` entries in total, while chroma-mcp returns
66
+ `n_results` per query, so its response grows with every query a model adds.
67
+ - Results always leave out embedding vectors, which add tokens without giving
68
+ a model anything it can use.
69
+
70
+ ### Embedding models
71
+
72
+ - Each space can use any sentence-transformers model, and tools let a model
73
+ find and download one from Hugging Face.
74
+ - Cached models load without network calls.
75
+ - Models trained with separate query and document prompts get the matching
76
+ prompt for searches and for stored entries.
77
+ - Spaces always have their embedding model's max token size available in
78
+ metadata for sizing entries before they are added.
79
+
80
+ ### Memory
81
+
82
+ - Embedding models unload from memory after `VEXICON_IDLE_SECONDS` without
83
+ use.
84
+
85
+ ### Defaults
86
+
87
+ - Storage persists under `~/.vexicon` by default.
88
+ - Entry IDs are kebab-case mnemonics.
89
+ - Duplicate entry IDs raise an error.
90
+ - Spaces are published as MCP resources.
91
+ - Chroma telemetry is off.
92
+
93
+ vexicon keeps its own fields in Chroma metadata:
94
+
95
+ - Each entry's metadata holds `created_at` in epoch seconds for recency
96
+ filters in `where`.
97
+ - Each space's metadata holds its `readme`, `embedding_repo_id`, and
98
+ `embedding_max_tokens`.
99
+ - Callers cannot set those three keys through a space's `metadata` argument.
100
+ - Tool results show these fields apart from the caller's own metadata.
@@ -0,0 +1,81 @@
1
+ # vexicon
2
+
3
+ ## About
4
+
5
+ Knowledge store MCP Server.
6
+
7
+ Provides an RRF implementation using SQLite. Tunable via settings.
8
+
9
+ Introduces some opinionated defaults
10
+
11
+
12
+ Notes and reference material live in
13
+ named spaces backed by Chroma collections with a SQLite FTS5 keyword index;
14
+ searches fuse the vector and keyword rankings by reciprocal rank fusion.
15
+
16
+ Run the server over stdio with `uv run vexicon`.
17
+
18
+ Settings are read from `VEXICON_*` environment variables; see
19
+ `src/vexicon/settings.py`. Data is stored under `~/.vexicon` unless
20
+ `VEXICON_PERSISTENT_PATH` and `VEXICON_INDEX_DB_PATH` say otherwise.
21
+
22
+ ## Settings
23
+
24
+ | Variable | Default | Meaning |
25
+ |---------------------------|------------------------|---------------------------------------------------------------------------|
26
+ | `VEXICON_PERSISTENT_PATH` | `~/.vexicon/chroma` | Directory where Chroma stores its database. |
27
+ | `VEXICON_INDEX_DB_PATH` | `~/.vexicon/hybrid.db` | SQLite file holding the keyword index. |
28
+ | `VEXICON_VECTOR_WEIGHT` | `1.0` | Weight of the vector ranking in fusion. |
29
+ | `VEXICON_KEYWORD_WEIGHT` | `1.0` | Weight of the keyword ranking in fusion. |
30
+ | `VEXICON_RRF_RANK_OFFSET` | `60` | Rank offset in reciprocal rank fusion. |
31
+ | `VEXICON_DEVICE` | `auto` | Device that runs embedding models: `auto`, `cpu`, or `cuda`. |
32
+ | `VEXICON_IDLE_SECONDS` | `300` | Seconds without activity before Chroma and embedding models are unloaded. |
33
+
34
+ ## Compared with chroma-mcp
35
+
36
+ Compared against chroma-mcp 0.2.6 and chromadb 1.5.9.
37
+
38
+ ### Hybrid search without Chroma Cloud
39
+
40
+ - vexicon combines vector and BM25 keyword rankings with reciprocal rank
41
+ fusion, using a SQLite FTS5 index beside each local Chroma collection.
42
+
43
+ ### Token usage
44
+
45
+ - vexicon's tool definitions take about 30% fewer tokens than chroma-mcp's.
46
+ - `search` returns at most `limit` entries in total, while chroma-mcp returns
47
+ `n_results` per query, so its response grows with every query a model adds.
48
+ - Results always leave out embedding vectors, which add tokens without giving
49
+ a model anything it can use.
50
+
51
+ ### Embedding models
52
+
53
+ - Each space can use any sentence-transformers model, and tools let a model
54
+ find and download one from Hugging Face.
55
+ - Cached models load without network calls.
56
+ - Models trained with separate query and document prompts get the matching
57
+ prompt for searches and for stored entries.
58
+ - Spaces always have their embedding model's max token size available in
59
+ metadata for sizing entries before they are added.
60
+
61
+ ### Memory
62
+
63
+ - Embedding models unload from memory after `VEXICON_IDLE_SECONDS` without
64
+ use.
65
+
66
+ ### Defaults
67
+
68
+ - Storage persists under `~/.vexicon` by default.
69
+ - Entry IDs are kebab-case mnemonics.
70
+ - Duplicate entry IDs raise an error.
71
+ - Spaces are published as MCP resources.
72
+ - Chroma telemetry is off.
73
+
74
+ vexicon keeps its own fields in Chroma metadata:
75
+
76
+ - Each entry's metadata holds `created_at` in epoch seconds for recency
77
+ filters in `where`.
78
+ - Each space's metadata holds its `readme`, `embedding_repo_id`, and
79
+ `embedding_max_tokens`.
80
+ - Callers cannot set those three keys through a space's `metadata` argument.
81
+ - Tool results show these fields apart from the caller's own metadata.
@@ -0,0 +1,126 @@
1
+ [project]
2
+ name = "vexicon"
3
+ version = "0.5.0"
4
+ description = "Hybrid vector + keyword MCP server powered by chroma and sqlite"
5
+ readme = "README.md"
6
+ requires-python = ">=3.12"
7
+ dependencies = [
8
+ "aiosqlite>=0.22.1,<0.23.0",
9
+ "alembic>=1.19.1,<2.0.0",
10
+ "chromadb>=1.5.9,<2.0.0",
11
+ "fastmcp>=4.0.10,<5.0.0",
12
+ "huggingface-hub>=1.28.0,<2.0.0",
13
+ "numpy>=2.5.2,<3.0.0",
14
+ "pydantic>=2.13.4,<3.0.0",
15
+ "pydantic-settings>=2.15.0,<3.0.0",
16
+ "sentence-transformers>=6.0.0,<7.0.0",
17
+ "sqlalchemy>=2.0.52,<3.0.0",
18
+ ]
19
+
20
+ [[project.authors]]
21
+ name = "estasney"
22
+ email = "estasney@users.noreply.github.com"
23
+
24
+ [project.scripts]
25
+ vexicon = "vexicon.server:main"
26
+
27
+ [tool.uv]
28
+ add-bounds = "major"
29
+
30
+ [tool.basedpyright]
31
+ include = ["src"]
32
+ extraPaths = ["src"]
33
+ pythonVersion = "3.12"
34
+ reportAny = false
35
+ reportExplicitAny = "hint"
36
+ reportConstantRedefinition = false
37
+ reportImplicitOverride = false
38
+ reportImplicitStringConcatenation = false
39
+ reportImportCycles = false
40
+ reportMissingModuleSource = false
41
+ reportMissingSuperCall = false
42
+ reportUnannotatedClassAttribute = false
43
+ reportUnknownLambdaType = false
44
+ reportUnknownMemberType = false
45
+ reportUnnecessaryComparison = false
46
+ reportUnsafeMultipleInheritance = false
47
+ reportUnusedCallResult = false
48
+ reportUnusedImport = false
49
+ reportUnusedParameter = false
50
+ strictDictionaryInference = true
51
+ strictListInference = true
52
+ strictSetInference = true
53
+
54
+ [tool.ruff.lint]
55
+ select = [
56
+ "A",
57
+ "ANN",
58
+ "ASYNC",
59
+ "B",
60
+ "BLE",
61
+ "C4",
62
+ "DTZ",
63
+ "ERA",
64
+ "F",
65
+ "FA",
66
+ "FAST",
67
+ "FBT",
68
+ "FLY",
69
+ "FURB",
70
+ "I",
71
+ "INP",
72
+ "N",
73
+ "NPY",
74
+ "PD",
75
+ "PERF",
76
+ "PIE",
77
+ "PLC0105",
78
+ "PLC0131",
79
+ "PLC0132",
80
+ "PLC0207",
81
+ "PLC0208",
82
+ "PLE",
83
+ "PLR",
84
+ "PLW",
85
+ "PYI",
86
+ "RET",
87
+ "RUF",
88
+ "SIM",
89
+ "TRY",
90
+ "UP",
91
+ "YTT",
92
+ ]
93
+ ignore = [
94
+ "A002",
95
+ "ANN002",
96
+ "ANN003",
97
+ "ANN202",
98
+ "ANN401",
99
+ "ASYNC119",
100
+ "N818",
101
+ "PLR0904",
102
+ "PLR0911",
103
+ "PLR0912",
104
+ "PLR0913",
105
+ "PLR0915",
106
+ "PLR0917",
107
+ "PLR2004",
108
+ "PLW0603",
109
+ "RET504",
110
+ "RUF001",
111
+ "RUF105",
112
+ "RUF106",
113
+ "RUF201",
114
+ "TRY003",
115
+ "UP051",
116
+ ]
117
+
118
+ [tool.ruff.lint.per-file-ignores]
119
+ "**/__init__.py" = ["I"]
120
+
121
+ [build-system]
122
+ requires = ["uv_build>=0.12.13,<0.13.0"]
123
+ build-backend = "uv_build"
124
+
125
+ [dependency-groups]
126
+ dev = ["pytest>=9.1.1,<10.0.0"]
@@ -0,0 +1,127 @@
1
+ [project]
2
+ name = "vexicon"
3
+ version = "0.5.0"
4
+ description = "Hybrid vector + keyword MCP server powered by chroma and sqlite"
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "estasney", email = "estasney@users.noreply.github.com" }
8
+ ]
9
+ requires-python = ">=3.12"
10
+ dependencies = [
11
+ "aiosqlite>=0.22.1,<0.23.0",
12
+ "alembic>=1.19.1,<2.0.0",
13
+ "chromadb>=1.5.9,<2.0.0",
14
+ "fastmcp>=4.0.10,<5.0.0",
15
+ "huggingface-hub>=1.28.0,<2.0.0",
16
+ "numpy>=2.5.2,<3.0.0",
17
+ "pydantic>=2.13.4,<3.0.0",
18
+ "pydantic-settings>=2.15.0,<3.0.0",
19
+ "sentence-transformers>=6.0.0,<7.0.0",
20
+ "sqlalchemy>=2.0.52,<3.0.0",
21
+ ]
22
+
23
+ [tool.uv]
24
+ add-bounds = "major"
25
+
26
+ [project.scripts]
27
+ vexicon = "vexicon.server:main"
28
+
29
+ [build-system]
30
+ requires = ["uv_build>=0.12.13,<0.13.0"]
31
+ build-backend = "uv_build"
32
+
33
+ [dependency-groups]
34
+ dev = [
35
+ "pytest>=9.1.1,<10.0.0",
36
+ ]
37
+
38
+ [tool.basedpyright]
39
+ include = ["src"]
40
+ extraPaths = ["src"]
41
+ pythonVersion = "3.12"
42
+ reportAny = false
43
+ reportExplicitAny = "hint"
44
+ reportConstantRedefinition = false
45
+ reportImplicitOverride = false
46
+ reportImplicitStringConcatenation = false
47
+ reportImportCycles = false
48
+ reportMissingModuleSource = false
49
+ reportMissingSuperCall = false
50
+ reportUnannotatedClassAttribute = false
51
+ reportUnknownLambdaType = false
52
+ reportUnknownMemberType = false
53
+ reportUnnecessaryComparison = false
54
+ reportUnsafeMultipleInheritance = false
55
+ reportUnusedCallResult = false
56
+ reportUnusedImport = false
57
+ reportUnusedParameter = false
58
+ strictDictionaryInference = true
59
+ strictListInference = true
60
+ strictSetInference = true
61
+
62
+ [tool.ruff.lint]
63
+ select = [
64
+ "A",
65
+ "ANN",
66
+ "ASYNC",
67
+ "B",
68
+ "BLE",
69
+ "C4",
70
+ "DTZ",
71
+ "ERA",
72
+ "F",
73
+ "FA",
74
+ "FAST",
75
+ "FBT",
76
+ "FLY",
77
+ "FURB",
78
+ "I",
79
+ "INP",
80
+ "N",
81
+ "NPY",
82
+ "PD",
83
+ "PERF",
84
+ "PIE",
85
+ "PLC0105",
86
+ "PLC0131",
87
+ "PLC0132",
88
+ "PLC0207",
89
+ "PLC0208",
90
+ "PLE",
91
+ "PLR",
92
+ "PLW",
93
+ "PYI",
94
+ "RET",
95
+ "RUF",
96
+ "SIM",
97
+ "TRY",
98
+ "UP",
99
+ "YTT",
100
+ ]
101
+ ignore = [
102
+ "A002",
103
+ "ANN002",
104
+ "ANN003",
105
+ "ANN202",
106
+ "ANN401",
107
+ "ASYNC119",
108
+ "N818",
109
+ "PLR0904",
110
+ "PLR0911",
111
+ "PLR0912",
112
+ "PLR0913",
113
+ "PLR0915",
114
+ "PLR0917",
115
+ "PLR2004",
116
+ "PLW0603",
117
+ "RET504",
118
+ "RUF001",
119
+ "RUF105",
120
+ "RUF106",
121
+ "RUF201",
122
+ "TRY003",
123
+ "UP051",
124
+ ]
125
+
126
+ [tool.ruff.lint.per-file-ignores]
127
+ "**/__init__.py" = ["I"]
File without changes
File without changes