msgsearch 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {msgsearch-0.2.1 → msgsearch-0.2.2}/ARCHITECTURE.md +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/CHANGELOG.md +16 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/PKG-INFO +27 -16
- {msgsearch-0.2.1 → msgsearch-0.2.2}/README.md +26 -15
- {msgsearch-0.2.1 → msgsearch-0.2.2}/pyproject.toml +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/__init__.py +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/cli.py +58 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/config.py +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/doctor.py +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/embedder.py +1 -1
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_cli.py +23 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/agents/corpus-explorer.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/agents/retrieval-critic.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/commands/bench.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/commands/go.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/commands/recon.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/hooks/verify-retrieval.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.claude/settings.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.github/workflows/ci.yml +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.github/workflows/release.yml +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/.gitignore +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/AGENTS.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/CLAUDE.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/CONTRIBUTING.md +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/LICENSE +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/__init__.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/bench.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/gold.jsonl +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/label.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/results/baseline.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/results/retriever.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/results/retriever_bm25.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/results/retriever_dense.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/results/retriever_norerank.json +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/retriever.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/retriever_bm25.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/retriever_dense.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/eval/retriever_norerank.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/attributed_body.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/chunk.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/contacts.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/explore.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/extract.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/index.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/py.typed +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/search.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/sync.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/src/msgsearch/tagging.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_chunk.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_contacts.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_index.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_search.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_sync.py +0 -0
- {msgsearch-0.2.1 → msgsearch-0.2.2}/tests/test_tagging.py +0 -0
|
@@ -125,7 +125,7 @@ Both run locally on Metal via MPS.
|
|
|
125
125
|
|
|
126
126
|
| role | model | notes |
|
|
127
127
|
|---|---|---|
|
|
128
|
-
| embedding | `google/embeddinggemma-300m` | 768-dim. **Gated** — needs Gemma licence acceptance and `
|
|
128
|
+
| embedding | `google/embeddinggemma-300m` | 768-dim. **Gated** — needs Gemma licence acceptance and `msgsearch login`. Matryoshka truncation to 512/256/128 available if the array grows. |
|
|
129
129
|
| rerank | `Qwen/Qwen3-Reranker-0.6B` | matches qmd's choice |
|
|
130
130
|
|
|
131
131
|
Rerankers measured over 50 candidates on the one query available:
|
|
@@ -3,6 +3,22 @@
|
|
|
3
3
|
Notable changes to msgsearch. Retrieval changes carry the measurement that
|
|
4
4
|
justified them; see `ARCHITECTURE.md` for the full evaluation.
|
|
5
5
|
|
|
6
|
+
## [0.2.2] — 2026-09-06
|
|
7
|
+
|
|
8
|
+
### Added
|
|
9
|
+
- `msgsearch login`, which authenticates with HuggingFace. The instructions
|
|
10
|
+
previously said to run `hf auth login`, but installing msgsearch does not put
|
|
11
|
+
`hf` on your PATH -- pipx exposes only the entry points a package declares, so
|
|
12
|
+
that step failed with "command not found" for everyone who installed normally.
|
|
13
|
+
It also verifies afterwards that the gated model is actually reachable, since
|
|
14
|
+
being logged in and having accepted the model licence are different things
|
|
15
|
+
whose failures look identical.
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
- Full Disk Access instructions are now step by step, including that macOS grants
|
|
19
|
+
it to the application rather than the shell and that the application must be
|
|
20
|
+
restarted before it takes effect.
|
|
21
|
+
|
|
6
22
|
## [0.2.1] — 2026-09-06
|
|
7
23
|
|
|
8
24
|
Documentation only. PyPI freezes a project's description at publish time, so
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: msgsearch
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Search your iMessage history by meaning, entirely on your own machine.
|
|
5
5
|
Project-URL: Homepage, https://github.com/dhruv1707/msgsearch
|
|
6
6
|
Project-URL: Documentation, https://github.com/dhruv1707/msgsearch/blob/main/README.md
|
|
@@ -69,9 +69,8 @@ what lands on your disk is unencrypted.
|
|
|
69
69
|
# 1. install (needs an arm64 Python — see Install if this errors)
|
|
70
70
|
pipx install msgsearch
|
|
71
71
|
|
|
72
|
-
# 2. get the embedding model
|
|
73
|
-
|
|
74
|
-
hf auth login
|
|
72
|
+
# 2. get the embedding model (accept the licence in a browser first, then:)
|
|
73
|
+
msgsearch login
|
|
75
74
|
|
|
76
75
|
# 3. snapshot your messages [needs Full Disk Access]
|
|
77
76
|
msgsearch sync
|
|
@@ -93,12 +92,21 @@ so it takes seconds:
|
|
|
93
92
|
msgsearch sync --index
|
|
94
93
|
```
|
|
95
94
|
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
95
|
+
### Granting Full Disk Access
|
|
96
|
+
|
|
97
|
+
Step 3 fails without it. macOS grants this to the **application**, not to your
|
|
98
|
+
shell, so the grant goes to whatever program you type commands into.
|
|
99
|
+
|
|
100
|
+
1. Open **System Settings → Privacy & Security → Full Disk Access**
|
|
101
|
+
2. Click **+**
|
|
102
|
+
3. Press **⌘⇧G** and paste `/Applications/Utilities/Terminal.app`, then Open
|
|
103
|
+
(if you use iTerm, VS Code or another terminal, choose that instead)
|
|
104
|
+
4. Make sure its toggle is **on**
|
|
105
|
+
5. **Quit and reopen that application** — the permission is only read at launch
|
|
106
|
+
|
|
107
|
+
Resolving contact names needs a *separate* permission, **Contacts**, granted the
|
|
108
|
+
same way in the same place. It is optional; without it speakers appear as phone
|
|
109
|
+
numbers. Full Disk Access does not include it.
|
|
102
110
|
|
|
103
111
|
## Requirements
|
|
104
112
|
|
|
@@ -160,12 +168,15 @@ arch -arm64 python3 -m venv .venv # must be an arm64 interpreter
|
|
|
160
168
|
|
|
161
169
|
## Setup
|
|
162
170
|
|
|
163
|
-
Get access to the embedding model. `google/embeddinggemma-300m` is gated, so
|
|
164
|
-
accept the licence at <https://huggingface.co/google/embeddinggemma-300m>, then:
|
|
171
|
+
Get access to the embedding model. `google/embeddinggemma-300m` is gated, so:
|
|
165
172
|
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
173
|
+
1. Accept the licence at <https://huggingface.co/google/embeddinggemma-300m>
|
|
174
|
+
2. Create a read token at <https://huggingface.co/settings/tokens>
|
|
175
|
+
3. Run `msgsearch login` and paste it
|
|
176
|
+
|
|
177
|
+
`msgsearch login` checks afterwards that you can actually reach the model, since
|
|
178
|
+
being logged in and having accepted the licence are different things — and the
|
|
179
|
+
failure for the second looks identical to the first.
|
|
169
180
|
|
|
170
181
|
Any sentence-transformers model works instead, for example
|
|
171
182
|
`MSGSEARCH_EMBED_MODEL=BAAI/bge-small-en-v1.5`.
|
|
@@ -193,7 +204,7 @@ accept the Gemma licence at
|
|
|
193
204
|
<https://huggingface.co/google/embeddinggemma-300m> and authenticate:
|
|
194
205
|
|
|
195
206
|
```bash
|
|
196
|
-
./.venv/bin/
|
|
207
|
+
./.venv/bin/msgsearch login
|
|
197
208
|
```
|
|
198
209
|
|
|
199
210
|
Any sentence-transformers model works instead if you would rather not, for example
|
|
@@ -23,9 +23,8 @@ what lands on your disk is unencrypted.
|
|
|
23
23
|
# 1. install (needs an arm64 Python — see Install if this errors)
|
|
24
24
|
pipx install msgsearch
|
|
25
25
|
|
|
26
|
-
# 2. get the embedding model
|
|
27
|
-
|
|
28
|
-
hf auth login
|
|
26
|
+
# 2. get the embedding model (accept the licence in a browser first, then:)
|
|
27
|
+
msgsearch login
|
|
29
28
|
|
|
30
29
|
# 3. snapshot your messages [needs Full Disk Access]
|
|
31
30
|
msgsearch sync
|
|
@@ -47,12 +46,21 @@ so it takes seconds:
|
|
|
47
46
|
msgsearch sync --index
|
|
48
47
|
```
|
|
49
48
|
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
49
|
+
### Granting Full Disk Access
|
|
50
|
+
|
|
51
|
+
Step 3 fails without it. macOS grants this to the **application**, not to your
|
|
52
|
+
shell, so the grant goes to whatever program you type commands into.
|
|
53
|
+
|
|
54
|
+
1. Open **System Settings → Privacy & Security → Full Disk Access**
|
|
55
|
+
2. Click **+**
|
|
56
|
+
3. Press **⌘⇧G** and paste `/Applications/Utilities/Terminal.app`, then Open
|
|
57
|
+
(if you use iTerm, VS Code or another terminal, choose that instead)
|
|
58
|
+
4. Make sure its toggle is **on**
|
|
59
|
+
5. **Quit and reopen that application** — the permission is only read at launch
|
|
60
|
+
|
|
61
|
+
Resolving contact names needs a *separate* permission, **Contacts**, granted the
|
|
62
|
+
same way in the same place. It is optional; without it speakers appear as phone
|
|
63
|
+
numbers. Full Disk Access does not include it.
|
|
56
64
|
|
|
57
65
|
## Requirements
|
|
58
66
|
|
|
@@ -114,12 +122,15 @@ arch -arm64 python3 -m venv .venv # must be an arm64 interpreter
|
|
|
114
122
|
|
|
115
123
|
## Setup
|
|
116
124
|
|
|
117
|
-
Get access to the embedding model. `google/embeddinggemma-300m` is gated, so
|
|
118
|
-
accept the licence at <https://huggingface.co/google/embeddinggemma-300m>, then:
|
|
125
|
+
Get access to the embedding model. `google/embeddinggemma-300m` is gated, so:
|
|
119
126
|
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
127
|
+
1. Accept the licence at <https://huggingface.co/google/embeddinggemma-300m>
|
|
128
|
+
2. Create a read token at <https://huggingface.co/settings/tokens>
|
|
129
|
+
3. Run `msgsearch login` and paste it
|
|
130
|
+
|
|
131
|
+
`msgsearch login` checks afterwards that you can actually reach the model, since
|
|
132
|
+
being logged in and having accepted the licence are different things — and the
|
|
133
|
+
failure for the second looks identical to the first.
|
|
123
134
|
|
|
124
135
|
Any sentence-transformers model works instead, for example
|
|
125
136
|
`MSGSEARCH_EMBED_MODEL=BAAI/bge-small-en-v1.5`.
|
|
@@ -147,7 +158,7 @@ accept the Gemma licence at
|
|
|
147
158
|
<https://huggingface.co/google/embeddinggemma-300m> and authenticate:
|
|
148
159
|
|
|
149
160
|
```bash
|
|
150
|
-
./.venv/bin/
|
|
161
|
+
./.venv/bin/msgsearch login
|
|
151
162
|
```
|
|
152
163
|
|
|
153
164
|
Any sentence-transformers model works instead if you would rather not, for example
|
|
@@ -156,6 +156,63 @@ def _run_search(args) -> int:
|
|
|
156
156
|
return 0
|
|
157
157
|
|
|
158
158
|
|
|
159
|
+
def _add_login_command(subparsers) -> None:
|
|
160
|
+
parser = subparsers.add_parser(
|
|
161
|
+
"login",
|
|
162
|
+
help="authenticate with HuggingFace to download the embedding model",
|
|
163
|
+
description="The default embedding model is gated, so it needs a "
|
|
164
|
+
"HuggingFace account that has accepted its licence. This wraps the same "
|
|
165
|
+
"login the huggingface_hub library provides -- msgsearch owns the "
|
|
166
|
+
"command because installing msgsearch does not put `hf` on your PATH.",
|
|
167
|
+
)
|
|
168
|
+
parser.add_argument(
|
|
169
|
+
"--token",
|
|
170
|
+
metavar="TOKEN",
|
|
171
|
+
help="read token from https://huggingface.co/settings/tokens "
|
|
172
|
+
"(prompted for if omitted)",
|
|
173
|
+
)
|
|
174
|
+
parser.set_defaults(handler=_run_login)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _run_login(args) -> int:
|
|
178
|
+
from getpass import getpass
|
|
179
|
+
|
|
180
|
+
from huggingface_hub import login
|
|
181
|
+
|
|
182
|
+
from . import config
|
|
183
|
+
|
|
184
|
+
token = args.token
|
|
185
|
+
if not token:
|
|
186
|
+
print(f"A read token is needed to download {config.EMBED_MODEL}.")
|
|
187
|
+
print("Create one at https://huggingface.co/settings/tokens")
|
|
188
|
+
print(f"and accept the licence at https://huggingface.co/{config.EMBED_MODEL}")
|
|
189
|
+
try:
|
|
190
|
+
token = getpass("\nToken (input hidden): ").strip()
|
|
191
|
+
except (EOFError, KeyboardInterrupt):
|
|
192
|
+
print()
|
|
193
|
+
return 130
|
|
194
|
+
if not token:
|
|
195
|
+
print("No token given.", file=sys.stderr)
|
|
196
|
+
return 1
|
|
197
|
+
|
|
198
|
+
try:
|
|
199
|
+
login(token=token)
|
|
200
|
+
except Exception as error:
|
|
201
|
+
print(f"Login failed: {error}", file=sys.stderr)
|
|
202
|
+
return 1
|
|
203
|
+
|
|
204
|
+
# Logging in successfully is not the same as having access to a gated model,
|
|
205
|
+
# and that difference is exactly what produces a baffling 401 later.
|
|
206
|
+
from .doctor import check_model
|
|
207
|
+
|
|
208
|
+
result = check_model()
|
|
209
|
+
print(f"\nlogged in. model: {result.detail}")
|
|
210
|
+
if result.fix:
|
|
211
|
+
print(result.fix)
|
|
212
|
+
return 1
|
|
213
|
+
return 0
|
|
214
|
+
|
|
215
|
+
|
|
159
216
|
def _add_contacts_command(subparsers) -> None:
|
|
160
217
|
parser = subparsers.add_parser(
|
|
161
218
|
"contacts",
|
|
@@ -325,6 +382,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
325
382
|
_add_doctor_command(subparsers)
|
|
326
383
|
_add_explore_command(subparsers)
|
|
327
384
|
_add_index_command(subparsers)
|
|
385
|
+
_add_login_command(subparsers)
|
|
328
386
|
_add_search_command(subparsers)
|
|
329
387
|
_add_sync_command(subparsers)
|
|
330
388
|
return parser
|
|
@@ -76,7 +76,7 @@ PASSAGE_STRIDE = 1
|
|
|
76
76
|
|
|
77
77
|
# --- models ------------------------------------------------------------------
|
|
78
78
|
|
|
79
|
-
# Gated on HuggingFace: accept the model terms and `
|
|
79
|
+
# Gated on HuggingFace: accept the model terms and run `msgsearch login` once,
|
|
80
80
|
# 401s. Must match the model the index was built with -- search.py checks.
|
|
81
81
|
EMBED_MODEL = os.environ.get("MSGSEARCH_EMBED_MODEL", "google/embeddinggemma-300m")
|
|
82
82
|
|
|
@@ -124,7 +124,7 @@ def check_model() -> Check:
|
|
|
124
124
|
FAIL,
|
|
125
125
|
f"{name} is gated and this machine is not authorised",
|
|
126
126
|
f"Accept the licence at https://huggingface.co/{name}, then run "
|
|
127
|
-
"'
|
|
127
|
+
"'msgsearch login'. An expired token gives this same error, so log in "
|
|
128
128
|
"again even if you have before. Or set "
|
|
129
129
|
"MSGSEARCH_EMBED_MODEL=BAAI/bge-small-en-v1.5 to use an ungated model.",
|
|
130
130
|
)
|
|
@@ -37,7 +37,7 @@ def _load_failure_hint(model_name: str, error: Exception) -> str:
|
|
|
37
37
|
f"is not authorised.\n\n"
|
|
38
38
|
f" 1. Accept the licence at https://huggingface.co/{model_name}\n"
|
|
39
39
|
f" 2. Create a read token at https://huggingface.co/settings/tokens\n"
|
|
40
|
-
f" 3. Run:
|
|
40
|
+
f" 3. Run: msgsearch login\n\n"
|
|
41
41
|
f"Note that an *expired* token produces this same error, so re-run step 3 "
|
|
42
42
|
f"even if you have logged in before. To use an ungated model instead:\n\n"
|
|
43
43
|
f" MSGSEARCH_EMBED_MODEL=BAAI/bge-small-en-v1.5\n\n"
|
|
@@ -75,3 +75,26 @@ class TestEntryPoint(unittest.TestCase):
|
|
|
75
75
|
|
|
76
76
|
if __name__ == "__main__":
|
|
77
77
|
unittest.main()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class TestLoginCommand(unittest.TestCase):
|
|
81
|
+
"""Authentication is a msgsearch command because `hf` is not on the PATH.
|
|
82
|
+
|
|
83
|
+
Installing msgsearch exposes only the entry points msgsearch declares, so
|
|
84
|
+
telling people to run `hf auth login` failed with "command not found" for
|
|
85
|
+
everyone who installed it normally rather than from a checkout.
|
|
86
|
+
"""
|
|
87
|
+
|
|
88
|
+
def setUp(self):
|
|
89
|
+
self.parser = cli.build_parser()
|
|
90
|
+
|
|
91
|
+
def test_login_is_a_command(self):
|
|
92
|
+
args = self.parser.parse_args(["login"])
|
|
93
|
+
self.assertTrue(callable(args.handler))
|
|
94
|
+
|
|
95
|
+
def test_token_can_be_passed_non_interactively(self):
|
|
96
|
+
args = self.parser.parse_args(["login", "--token", "hf_example"])
|
|
97
|
+
self.assertEqual(args.token, "hf_example")
|
|
98
|
+
|
|
99
|
+
def test_token_is_optional_so_it_can_prompt(self):
|
|
100
|
+
self.assertIsNone(self.parser.parse_args(["login"]).token)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|