interactive-semantic-hunt 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. interactive_semantic_hunt-0.1.0.dist-info/METADATA +316 -0
  2. interactive_semantic_hunt-0.1.0.dist-info/RECORD +54 -0
  3. interactive_semantic_hunt-0.1.0.dist-info/WHEEL +5 -0
  4. interactive_semantic_hunt-0.1.0.dist-info/entry_points.txt +4 -0
  5. interactive_semantic_hunt-0.1.0.dist-info/licenses/LICENSE +21 -0
  6. interactive_semantic_hunt-0.1.0.dist-info/top_level.txt +1 -0
  7. ish/__init__.py +6 -0
  8. ish/adapters/__init__.py +0 -0
  9. ish/adapters/embedder/__init__.py +0 -0
  10. ish/adapters/embedder/llama_cpp.py +50 -0
  11. ish/adapters/embedder/ollama.py +112 -0
  12. ish/adapters/embedder/prefixes.py +87 -0
  13. ish/adapters/embedder/sentence_transformer.py +47 -0
  14. ish/adapters/parser/__init__.py +0 -0
  15. ish/adapters/parser/limits.py +90 -0
  16. ish/adapters/parser/markup.py +93 -0
  17. ish/adapters/parser/plugins.py +135 -0
  18. ish/adapters/parser/python.py +127 -0
  19. ish/adapters/parser/structured.py +234 -0
  20. ish/adapters/parser/tree_sitter.py +249 -0
  21. ish/adapters/vcs/__init__.py +0 -0
  22. ish/adapters/vcs/git.py +90 -0
  23. ish/adapters/vector_store/__init__.py +0 -0
  24. ish/adapters/vector_store/federated.py +117 -0
  25. ish/adapters/vector_store/pure_python.py +157 -0
  26. ish/adapters/vector_store/sqlite.py +538 -0
  27. ish/application/__init__.py +0 -0
  28. ish/application/index.py +227 -0
  29. ish/application/ports/__init__.py +0 -0
  30. ish/application/ports/embedder.py +31 -0
  31. ish/application/ports/parser.py +39 -0
  32. ish/application/ports/vector_store.py +165 -0
  33. ish/application/preview.py +39 -0
  34. ish/application/scan.py +179 -0
  35. ish/application/search.py +367 -0
  36. ish/bootstrap.py +412 -0
  37. ish/domain/__init__.py +0 -0
  38. ish/domain/chunk.py +30 -0
  39. ish/interfaces/__init__.py +0 -0
  40. ish/interfaces/cli/__init__.py +0 -0
  41. ish/interfaces/cli/args.py +165 -0
  42. ish/interfaces/cli/complete.py +48 -0
  43. ish/interfaces/cli/log.py +141 -0
  44. ish/interfaces/cli/main.py +168 -0
  45. ish/interfaces/complete.py +115 -0
  46. ish/interfaces/format.py +50 -0
  47. ish/interfaces/mcp/__init__.py +0 -0
  48. ish/interfaces/mcp/protocol.py +171 -0
  49. ish/interfaces/mcp/server.py +384 -0
  50. ish/interfaces/python/__init__.py +0 -0
  51. ish/interfaces/python/api.py +195 -0
  52. ish/interfaces/tui/__init__.py +0 -0
  53. ish/interfaces/tui/app.py +393 -0
  54. ish/settings.py +387 -0
@@ -0,0 +1,316 @@
1
+ Metadata-Version: 2.4
2
+ Name: interactive-semantic-hunt
3
+ Version: 0.1.0
4
+ Summary: Interactive semantic search for code, inspired by fzf
5
+ Author-email: David Kristiansen <david@kristiansen.tech>
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/davetothek/interactive-semantic-hunt
8
+ Project-URL: Repository, https://github.com/davetothek/interactive-semantic-hunt
9
+ Project-URL: Issues, https://github.com/davetothek/interactive-semantic-hunt/issues
10
+ Keywords: search,semantic,embeddings,code,fzf,tui,mcp
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Programming Language :: Python :: 3.14
18
+ Classifier: Topic :: Software Development
19
+ Classifier: Topic :: Text Processing :: Indexing
20
+ Classifier: Typing :: Typed
21
+ Requires-Python: >=3.12
22
+ Description-Content-Type: text/markdown
23
+ License-File: LICENSE
24
+ Requires-Dist: numpy>=2.5.2
25
+ Requires-Dist: pyyaml>=6.0
26
+ Requires-Dist: textual>=0.70.0
27
+ Requires-Dist: tree-sitter>=0.26.0
28
+ Requires-Dist: tree-sitter-c>=0.24.2
29
+ Requires-Dist: tree-sitter-cpp>=0.23.4
30
+ Provides-Extra: llama
31
+ Requires-Dist: llama-cpp-python>=0.2.75; extra == "llama"
32
+ Requires-Dist: huggingface-hub>=0.23.0; extra == "llama"
33
+ Provides-Extra: st
34
+ Requires-Dist: sentence-transformers; extra == "st"
35
+ Dynamic: license-file
36
+
37
+ # ish — Interactive Semantic Hunt
38
+
39
+ Semantic search for code, inspired by `fzf`. Point `ish` at a directory. It parses
40
+ each source file into named chunks, embeds them with a local model, and ranks them
41
+ against your query. Everything runs on your machine.
42
+
43
+ Languages: Python, C and C++, Markdown, and AsciiDoc. Documentation is indexed
44
+ beside the code it describes, so one query searches both.
45
+
46
+ ## Install
47
+
48
+ ```sh
49
+ pip install interactive-semantic-hunt
50
+ ```
51
+
52
+ Nothing in that install compiles: the default backend reaches Ollama over
53
+ HTTP with the standard library. A backend that runs the model in this
54
+ process is an extra — `[llama]` for llama.cpp, `[st]` for
55
+ sentence-transformers.
56
+
57
+ To work on ish itself, clone it and run `uv sync`.
58
+
59
+ The default embedding backend is Ollama, which keeps the model resident so no
60
+ run pays a model load. Start it once and pull the embedding model:
61
+
62
+ ```sh
63
+ ollama serve
64
+ ollama pull nomic-embed-text
65
+ ```
66
+
67
+ Set `OLLAMA_HOST` to reach a daemon elsewhere.
68
+
69
+ Two other backends need no daemon:
70
+
71
+ - `--embedder llama.cpp` downloads a GGUF model and loads it per run. Slower per
72
+ query, faster for a first index of a large tree. Needs the `[llama]` extra.
73
+ - `--embedder st` uses sentence-transformers. Needs the `[st]` extra.
74
+
75
+ ## Use
76
+
77
+ List every chunk under a path:
78
+
79
+ ```sh
80
+ ish src/
81
+ ```
82
+
83
+ ```
84
+ src/ish/domain/chunk.py:7-28 class Chunk
85
+ ```
86
+
87
+ Search for a query:
88
+
89
+ ```sh
90
+ ish "parse a python file" src/
91
+ ```
92
+
93
+ ```
94
+ [0.71] src/ish/adapters/parser/python.py:14-31 method PythonParser.parse
95
+ ```
96
+
97
+ Run the interactive picker and open the selection in your editor. Type to
98
+ search, `up`/`down` or `ctrl+p`/`ctrl+n` to move, `enter` to choose, `escape`
99
+ to quit. Narrow without leaving the query line:
100
+
101
+ ```text
102
+ state machine transitions every language
103
+ lang:cpp state machine transitions the implementation
104
+ lang:yaml under:/10.System/ state the tests that cover it
105
+ type:doc how do I configure this the prose, not the code
106
+ type:test,doc retry backoff the tests and what they document
107
+ ```
108
+
109
+ Press Tab to finish a filter word. `ty` becomes `type:`, `lang:cp` becomes
110
+ `lang:cpp`, and `under:/s` becomes `under:/src/`; a word with several answers
111
+ grows as far as they agree and names the rest. `ish-complete` does the work, so
112
+ any picker can call it.
113
+
114
+ `lang:`, `under:`, and `type:` work in the query line of every interface —
115
+ the command line, the picker, Neovim, and MCP. The words are taken out
116
+ before the query is embedded, so the model sees the question rather than
117
+ how it was narrowed.
118
+
119
+ A language may be named however it comes to mind. `c`, `c++`, `cxx`, `h`, and
120
+ `hpp` all mean `cpp`, because one parser reads them all; `adoc` means
121
+ `asciidoc`, `md` means `markdown`, `py` means `python`, and `yml` means `yaml`.
122
+
123
+ `type:` sorts every chunk into exactly one kind. A path decides before a
124
+ language does, so a YAML fixture counts as a test rather than as config:
125
+
126
+ | kind | what it holds |
127
+ |---|---|
128
+ | `code` | Python, C, C++, and anything a plugin parser adds |
129
+ | `doc` | Markdown and AsciiDoc |
130
+ | `test` | anything under `tests/`, `spec/`, `fixtures/`, plus `test_*` and `conftest.py` |
131
+ | `config` | YAML, JSON, and TOML outside a test path |
132
+
133
+ A repository that names its trees its own way can say so, in
134
+ `.ish/config.toml`:
135
+
136
+ ```toml
137
+ type_patterns = [
138
+ "test:/[0-9.]*(Tests|Verification)/",
139
+ "doc:/[0-9.]*Specification/",
140
+ ]
141
+ ```
142
+
143
+ The first match wins; anything unmatched keeps the reading above.
144
+
145
+ A config beside a subtree adds to the one above it, so a tree settles only
146
+ what it names and inherits the rest. Searching inside an already-indexed tree
147
+ reads that tree's index and narrows the answers to the path, rather than
148
+ starting a second index of the same files.
149
+
150
+ ```sh
151
+ nvim $(ish -i src/)
152
+ ```
153
+
154
+ ### Options
155
+
156
+ | Flag | Purpose |
157
+ |---|---|
158
+ | `-i`, `--interactive` | Run the TUI picker |
159
+ | `--embedder {llama.cpp,ollama,st}` | Select the embedding backend (default: ollama) |
160
+ | `-v`, `-vv` | Increase log detail |
161
+ | `--color {auto,always,never}` | Control log color |
162
+ | `--limit N` | Maximum search results |
163
+ | `--ignore DIR ...` | Directory names to skip (default `.git .venv venv __pycache__`) |
164
+ | `--include REGEX ...` | Index only paths matching these patterns |
165
+ | `--exclude REGEX ...` | Never index paths matching these patterns |
166
+ | `--git`, `--no-git` | Skip files git ignores (default: on) |
167
+ | `--lang LANG ...` | Show results only from these languages |
168
+ | `--under REGEX` | Show results only from matching paths |
169
+ | `--type TYPE ...` | Show results only of these kinds: `code`, `doc`, `test`, `config` |
170
+ | `--type-patterns TYPE:REGEX ...` | Say what a path holds, overriding the built-in reading |
171
+ | `--model NAME` | Override the backend model |
172
+ | `--refresh` | Bring every stored index at or below the path up to date first |
173
+ | `--reindex` | Discard the stored index and build it again |
174
+ | `--no-cache` | Index in memory only, leaving nothing on disk |
175
+
176
+ Logs go to stderr, so you can pipe stdout safely. A file that cannot be read
177
+ or parsed is counted in one line; `-v` names them.
178
+
179
+ ## Use from Neovim
180
+
181
+ `contrib/nvim/ish.lua` is an fzf-lua picker. Copy it to `lua/utils/ish.lua` and
182
+ bind it:
183
+
184
+ ```lua
185
+ map('n', '<leader>fi', function() require('utils.ish').search() end,
186
+ { desc = 'Semantic search (ish)' })
187
+ ```
188
+
189
+ It reads `--format grep`, so the built-in previewer opens each result at its
190
+ line, and prints the rank in the leftmost column. `search_lang({'cpp'})`,
191
+ `search_type({'doc'})`, and `search_here()` narrow it, as does a `lang:`,
192
+ `type:`, or `under:` word typed into the query.
193
+
194
+ While the index refreshes, `require('utils.ish').statusline()` renders a bar
195
+ for a statusline — `ish ███░░░░░ 38%` — and an empty string when idle. It reads `vim.g.ish_index_status`, which the picker keeps up to date, and
196
+ shows `ish ✓` briefly when a refresh finishes. Nothing is reported through
197
+ `vim.notify`: with `cmdheight = 0` there is no command line to put a message
198
+ in, so nvim draws one over the last screen row — the statusline itself.
199
+
200
+ The picker never blocks the editor: results are written as they arrive, so
201
+ typing stays smooth however long a search takes.
202
+
203
+ `contrib/nvim/ish_server.lua` keeps one `ish-mcp` process per session. It
204
+ starts on the first search and is reused after that, which cuts a keystroke
205
+ from about 500 ms to about 150 ms. Copy it beside the picker.
206
+
207
+ ## Use from Python
208
+
209
+ ```python
210
+ from ish.interfaces.python.api import Ish
211
+
212
+ with Ish("src/") as ish:
213
+ for chunk, score in ish.search("type:doc how to configure", limit=5):
214
+ print(score, chunk.path, chunk.symbol)
215
+ print(ish.status())
216
+ ```
217
+
218
+ `Ish` holds the index open, so a second query costs a search rather than a
219
+ process start. It offers `search()`, `chunks()`, `index()`, `refresh_all()`,
220
+ and `status()`, and reads `lang:`, `type:`, and `under:` out of the query
221
+ exactly as the other interfaces do.
222
+
223
+ ## Use from an agent
224
+
225
+ `ish-mcp` serves the same search over the Model Context Protocol, so an agent
226
+ can query the index directly. Add it to a project with `.mcp.json`:
227
+
228
+ ```json
229
+ {
230
+ "mcpServers": {
231
+ "ish": { "command": "uv", "args": ["run", "ish-mcp"] }
232
+ }
233
+ }
234
+ ```
235
+
236
+ It offers `search_code`, `list_chunks`, `index_status`, and `refresh_index`. The server stays
237
+ resident, so a query costs about 58 ms rather than a process start.
238
+
239
+ A call may narrow one search with `lang`, `under`, `type`, and `limit`, or
240
+ write the same filters into the query text. It cannot change
241
+ what is indexed — those settings come from `ish.toml` only, so no single call can
242
+ shrink an index that another call depends on.
243
+
244
+ ## Index
245
+
246
+ The index persists in SQLite under `$XDG_DATA_HOME/ish/`, one file per scanned
247
+ tree. **A search of a parent reads the indexes below it and refreshes none** —
248
+ choosing one of them to write to would be wrong — so it warns and offers
249
+ `--refresh`, which visits each tree in turn. A repeated query reuses it, so only changed files are parsed and only new
250
+ text is embedded. A renamed file re-embeds nothing.
251
+
252
+ Each index records the tree it was built from, so searching a directory also
253
+ searches every index below it. Index the parts of a large project separately and
254
+ search the whole from its root:
255
+
256
+ ```sh
257
+ ish "warm" project/docs # index one part
258
+ ish "warm" project/firmware # and another
259
+ ish "how is exposure set" project # searches both
260
+ ```
261
+
262
+ Searching a parent never rewrites an index below it. Pass `--no-federate` to use
263
+ only the index of the exact path.
264
+
265
+ The index records where each chunk is — its path, line range, kind, and name —
266
+ together with the embedding vector. It does not store the source, so it is not a
267
+ second readable copy of your code. Previews are read from the file, which also
268
+ means they always show the current content.
269
+
270
+ ## Configure
271
+
272
+ Every command-line option is also a key in `ish.toml`, under the same name.
273
+ Put project settings in `ish.toml` at the root of your repository:
274
+
275
+ ```toml
276
+ embedder = "ollama"
277
+ model = "mxbai-embed-large"
278
+ limit = 10
279
+ ignore = [".git", ".venv", "build", "node_modules"]
280
+
281
+ # Regular expressions, searched against the path.
282
+ exclude = ["/vendor/", "_pb2\\.py$", "(_test|_spec)\\.py$"]
283
+ ```
284
+
285
+ `include` and `exclude` take regular expressions rather than globs, so `/vendor/`
286
+ matches at any depth and alternation works. `exclude` wins over `include`.
287
+
288
+ `--git` is on by default, so anything a `.gitignore` covers stays out of the
289
+ index. Pass `--no-git` to index it anyway.
290
+
291
+ `--lang` and `--under` narrow what a search *returns*. They never change what is
292
+ indexed, so a narrowed query cannot shrink the index:
293
+
294
+ ```sh
295
+ ish "how is ranking done" --lang python
296
+ ish "installation steps" --lang markdown asciidoc
297
+ ish "parse a header" --under '/include/'
298
+ ```
299
+
300
+ User-level defaults go in `~/.config/ish/ish.toml`. Later sources win:
301
+
302
+ ```
303
+ defaults < ~/.config/ish/ish.toml < ./ish.toml < ISH_* environment < command line
304
+ ```
305
+
306
+ Set any option from the environment with the `ISH_` prefix, for example
307
+ `ISH_LIMIT=20` or `ISH_IGNORE=build,dist`.
308
+
309
+ ## Develop
310
+
311
+ ```sh
312
+ uv run poe check # lint, typecheck, test
313
+ ```
314
+
315
+ The architecture is ports and adapters. `spec.md` holds the requirements, and
316
+ `.claude/CLAUDE.md` describes the layers and the composition root.
@@ -0,0 +1,54 @@
1
+ interactive_semantic_hunt-0.1.0.dist-info/licenses/LICENSE,sha256=_tIgsjXEpm2J-zkIN9m7SkL-K2LkgHWfBcS_Y7aFzqY,1074
2
+ ish/__init__.py,sha256=W9wJYhAv0PwnXHkhf6caJfYO5OXH73Y-g_PoehyPLWU,294
3
+ ish/bootstrap.py,sha256=4IiaxP6UaRhSts30Nu4jWWpIX69-pKce0xFMJbWEJhw,13770
4
+ ish/settings.py,sha256=2LhHGDWyhZXTbNq-MlW1sClJjEz5a0Fxs-W1qWHaWmM,12521
5
+ ish/adapters/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
6
+ ish/adapters/embedder/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ ish/adapters/embedder/llama_cpp.py,sha256=_GnRmYcTiBZ-reMFFMnRuSvWdhIMJTUb33pOXTD224I,1746
8
+ ish/adapters/embedder/ollama.py,sha256=cnLcTgR6mkNf5I4LLcVxXqkpPcBn3gLtBL70SB7BxyM,4209
9
+ ish/adapters/embedder/prefixes.py,sha256=M7ThGfjDL8sV_eVZPiCmXR5IR5CT8d9iT8bEOvm3vKk,3076
10
+ ish/adapters/embedder/sentence_transformer.py,sha256=z9z5CiF1WRZ-_bNVYXBm2jWp58nxw0FbuTd4KYejFKM,1796
11
+ ish/adapters/parser/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
12
+ ish/adapters/parser/limits.py,sha256=Vgv9gzsZRpJMcORKgNeJJND6kAfLSN79xzI-6_GQb3g,3161
13
+ ish/adapters/parser/markup.py,sha256=QCwK5ExyqXTkQkmuEfJ3h-7j9OG1mJE4vuuyiZlzQdU,3312
14
+ ish/adapters/parser/plugins.py,sha256=8Ob6J6kPHtp5qVEU9a2Q7AqoIg8ZWvR8EIPNkk9uOZI,4166
15
+ ish/adapters/parser/python.py,sha256=2Oguh249O1EHeuk0Gq1PthSqDFnHFf48xBQ_DzVbU2s,4293
16
+ ish/adapters/parser/structured.py,sha256=HyuKUCVdaz2sJJNWXCcLy1FWob_KywLOJ5pUh4qytRg,8124
17
+ ish/adapters/parser/tree_sitter.py,sha256=KHnWKdMnNke2Y4l3MyeJJDpImtOdZVoyxGFmNUPDjOY,8673
18
+ ish/adapters/vcs/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
19
+ ish/adapters/vcs/git.py,sha256=nzXiLJHwHTYDaHoAjsEFpNYua-CdRE_MqnEkA7FEW9k,2986
20
+ ish/adapters/vector_store/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
21
+ ish/adapters/vector_store/federated.py,sha256=ZmVDXE3LLe225uGNuYCqe-2Drfd-Ef-w5C_0yV-TWzs,4328
22
+ ish/adapters/vector_store/pure_python.py,sha256=fZBmbFlb7zLAEI5wpAHDt3wh9wDpHZBoe4x4ojyt0So,5672
23
+ ish/adapters/vector_store/sqlite.py,sha256=VrnC-Q5rKL0J7Prd7JqVoaOtA89jvhUsLFgpo4mjzzw,20171
24
+ ish/application/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
25
+ ish/application/index.py,sha256=AyE8ka8SLBlpS6HPhzvZqp4by2k-lzEUTSB3bFYwFwo,8022
26
+ ish/application/preview.py,sha256=sSIW4vfqpZW6b4Q9CSqdrsl7qyVRqshqFeKQwStLgtU,1320
27
+ ish/application/scan.py,sha256=WVvkownLTP3jBt1VsAaVzw5N3tUxtKDg0zghs6ATwnI,6589
28
+ ish/application/search.py,sha256=0VpiJwUUfVvTWeODy5kXSX5tSEuT15pKdQnqejZnqTM,12265
29
+ ish/application/ports/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
30
+ ish/application/ports/embedder.py,sha256=ZBiQ-8gz31XcPRRKUvbICSrTo2Ya8xecogKhimsSRWs,987
31
+ ish/application/ports/parser.py,sha256=KjCaB0YDdKna0O9mdab5B5c5MvntWJMTmzWxXiamapg,1059
32
+ ish/application/ports/vector_store.py,sha256=GlbQc_b0ZEwJaKGegzkDIK7G224BKb3uu-CLemMEFAI,5494
33
+ ish/domain/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
34
+ ish/domain/chunk.py,sha256=O5xo3KAaumkScYgGVtX7EAv1oICGKLb1eyjmBAbo66c,1058
35
+ ish/interfaces/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
36
+ ish/interfaces/complete.py,sha256=77f0ZK7ozvEFvST6rQtF4n8YqVO58IueQouSjh69_mg,3947
37
+ ish/interfaces/format.py,sha256=EupJWKg65lSAHqvfB6hu-TzlX9Qu5IP9DYfyVIpg_2U,1584
38
+ ish/interfaces/cli/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
39
+ ish/interfaces/cli/args.py,sha256=3m_aD8BlmG4steJInPxdkpxeHDwtj1OW6T_ua9HjUOU,5406
40
+ ish/interfaces/cli/complete.py,sha256=4STpIDg17JykuJQvYLDSVYOFM6iyTSP7R3IzrjV1l64,1663
41
+ ish/interfaces/cli/log.py,sha256=UCLJslCcQa-zbBwnYde5KtjmHRZEbtJFzLgEbHSy9j0,3988
42
+ ish/interfaces/cli/main.py,sha256=TCYnKgEmnjc-3xCsYUpX5tR5__GP2KS-VGikPzKvx9A,5188
43
+ ish/interfaces/mcp/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
44
+ ish/interfaces/mcp/protocol.py,sha256=b8iPjd7u78HR6ss066l2Iew85chjWa0LORwHBLdCpmM,6042
45
+ ish/interfaces/mcp/server.py,sha256=Td-2FA7ZknXcw_-_-xGkX56bihJclbAHhVGEU_qopRg,14628
46
+ ish/interfaces/python/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
47
+ ish/interfaces/python/api.py,sha256=PA69SyJ6N2k6ZBVOwzCneCu24D9lSDp69CKElqmOSe4,6687
48
+ ish/interfaces/tui/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
49
+ ish/interfaces/tui/app.py,sha256=aFwwbxFR4iku8H3UxecxTowL_nWrLz51dHbgthcD5VY,14840
50
+ interactive_semantic_hunt-0.1.0.dist-info/METADATA,sha256=M9JZEOFrQJgQ2-28tksaswATOwuvc--jJ6sqGlsOPAc,11636
51
+ interactive_semantic_hunt-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
52
+ interactive_semantic_hunt-0.1.0.dist-info/entry_points.txt,sha256=Gw9kcf_Ijny8kgKQ1gDZTdtKrTkp3P1ahGkH7_qRALA,142
53
+ interactive_semantic_hunt-0.1.0.dist-info/top_level.txt,sha256=iEzghqES9lDeQgVwT6pm7BwRfGDSnYFvuQ6jmqbyM50,4
54
+ interactive_semantic_hunt-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,4 @@
1
+ [console_scripts]
2
+ ish = ish.interfaces.cli.main:main
3
+ ish-complete = ish.interfaces.cli.complete:main
4
+ ish-mcp = ish.interfaces.mcp.server:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 David Kristiansen
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
ish/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ """ish — Interactive Semantic Hunt. A semantic search tool for code.
2
+
3
+ Keep this module free of layer imports. The package root must not pull
4
+ in interfaces, application, or adapter code — import from the layer
5
+ modules directly, for example ``from ish.interfaces.python.api import Ish``.
6
+ """
File without changes
File without changes
@@ -0,0 +1,50 @@
1
+ """llama.cpp adapter for the Embedder protocol."""
2
+
3
+ from collections.abc import Sequence
4
+
5
+ from ish.adapters.embedder.prefixes import PrefixingEmbedder
6
+
7
+
8
+ class LlamaCppEmbedder(PrefixingEmbedder):
9
+ """Generate embeddings using a local GGUF model via llama.cpp.
10
+
11
+ Automatically downloads `nomic-embed-text` from Hugging Face if not present.
12
+ """
13
+
14
+ def __init__(
15
+ self,
16
+ repo_id: str = "nomic-ai/nomic-embed-text-v1.5-GGUF",
17
+ filename: str = "nomic-embed-text-v1.5.Q4_K_M.gguf",
18
+ ) -> None:
19
+ # Expose the model identity for cache keying.
20
+ self.model_name = f"{repo_id}/{filename}"
21
+
22
+ import os
23
+
24
+ from huggingface_hub.utils.logging import set_verbosity_error
25
+
26
+ # Suppress huggingface_hub network warnings
27
+ os.environ["HF_HUB_DISABLE_TELEMETRY"] = "1"
28
+ os.environ["HF_HUB_DISABLE_SYMLINKS_WARNING"] = "1"
29
+ set_verbosity_error()
30
+
31
+ from huggingface_hub import hf_hub_download
32
+ from llama_cpp import Llama
33
+
34
+ # 1. Download/find the model on disk
35
+ model_path = hf_hub_download(repo_id=repo_id, filename=filename)
36
+
37
+ # 2. Instantiate the engine. verbose=False hides the massive C++ startup logs.
38
+ self._model = Llama(model_path=model_path, embedding=True, verbose=False)
39
+
40
+ def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
41
+ """Encode texts into vectors via llama.cpp."""
42
+ text_list = list(texts)
43
+
44
+ # create_embedding accepts a single string or a list of strings
45
+ result = self._model.create_embedding(text_list)
46
+
47
+ from typing import cast
48
+
49
+ embeddings = [item["embedding"] for item in result["data"]]
50
+ return cast("Sequence[Sequence[float]]", embeddings)
@@ -0,0 +1,112 @@
1
+ """Ollama adapter for the Embedder protocol.
2
+
3
+ Call the local Ollama daemon over HTTP with the standard library. The
4
+ daemon holds the model resident, so no process pays a model load, and
5
+ the adapter needs no third-party package.
6
+ """
7
+
8
+ import json
9
+ import logging
10
+ import os
11
+ import urllib.error
12
+ import urllib.request
13
+ from collections.abc import Sequence
14
+
15
+ from ish.adapters.embedder.prefixes import PrefixingEmbedder
16
+
17
+ log = logging.getLogger(__name__)
18
+
19
+ DEFAULT_HOST = "http://localhost:11434"
20
+ DEFAULT_MODEL = "nomic-embed-text"
21
+
22
+ # Send this many texts per request. One request for a whole repository
23
+ # would hold the daemon for minutes and risk the timeout.
24
+ DEFAULT_BATCH_SIZE = 64
25
+ # A batch of large definitions can take minutes on a busy daemon.
26
+ TIMEOUT_SECONDS = 600
27
+
28
+
29
+ def _normalize_host(host: str) -> str:
30
+ """Accept a bare ``host:port`` as well as a full URL."""
31
+ host = host.rstrip("/")
32
+ if not host.startswith(("http://", "https://")):
33
+ return f"http://{host}"
34
+ return host
35
+
36
+
37
+ class OllamaEmbedder(PrefixingEmbedder):
38
+ """Generate embeddings through a running Ollama daemon.
39
+
40
+ Read ``OLLAMA_HOST`` when no host is given, matching the Ollama
41
+ command-line tools.
42
+ """
43
+
44
+ def __init__(
45
+ self,
46
+ model_name: str = DEFAULT_MODEL,
47
+ *,
48
+ host: str | None = None,
49
+ batch_size: int = DEFAULT_BATCH_SIZE,
50
+ ) -> None:
51
+ self.model_name = model_name
52
+ chosen = host or os.environ.get("OLLAMA_HOST") or DEFAULT_HOST
53
+ self.host = _normalize_host(chosen)
54
+ self._batch_size = max(1, batch_size)
55
+
56
+ def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
57
+ """Encode texts into vectors, one batch of requests at a time."""
58
+ items = list(texts)
59
+ vectors: list[Sequence[float]] = []
60
+ for start in range(0, len(items), self._batch_size):
61
+ vectors.extend(self._embed_batch(items[start : start + self._batch_size]))
62
+ return vectors
63
+
64
+ # ------------------------------------------------------------------
65
+ # Internal helpers
66
+ # ------------------------------------------------------------------
67
+
68
+ def _embed_batch(self, batch: list[str]) -> list[Sequence[float]]:
69
+ """Send one batch and return its vectors."""
70
+ payload = json.dumps({"model": self.model_name, "input": batch}).encode("utf-8")
71
+ request = urllib.request.Request(
72
+ f"{self.host}/api/embed",
73
+ data=payload,
74
+ headers={"Content-Type": "application/json"},
75
+ )
76
+
77
+ try:
78
+ with urllib.request.urlopen(request, timeout=TIMEOUT_SECONDS) as response:
79
+ body = json.loads(response.read())
80
+ except urllib.error.HTTPError as exc:
81
+ detail = exc.read().decode("utf-8", "replace").strip()
82
+ if exc.code == 404:
83
+ hint = f"Pull it with 'ollama pull {self.model_name}'."
84
+ else:
85
+ hint = (
86
+ f"Confirm that {self.model_name!r} is an embedding model, "
87
+ f"not a generation model."
88
+ )
89
+ raise RuntimeError(
90
+ f"Ollama refused the request ({exc.code}): {detail}. {hint}"
91
+ ) from exc
92
+ except urllib.error.URLError as exc:
93
+ raise RuntimeError(
94
+ f"Cannot reach Ollama at {self.host}: {exc.reason}. "
95
+ f"Start it with 'ollama serve', or install another backend "
96
+ f"and select it, as in "
97
+ f"'pip install interactive-semantic-hunt[llama]' "
98
+ f"then '--embedder llama.cpp'."
99
+ ) from exc
100
+ except json.JSONDecodeError as exc:
101
+ raise RuntimeError(
102
+ f"Ollama at {self.host} returned a reply that is not JSON."
103
+ ) from exc
104
+
105
+ embeddings = body.get("embeddings")
106
+ if not isinstance(embeddings, list) or len(embeddings) != len(batch):
107
+ got = len(embeddings) if isinstance(embeddings, list) else 0
108
+ raise RuntimeError(
109
+ f"Ollama returned {got} vectors for {len(batch)} texts. "
110
+ f"Confirm that {self.model_name!r} is an embedding model."
111
+ )
112
+ return embeddings
@@ -0,0 +1,87 @@
1
+ """Apply the task prefixes an embedding model was trained with.
2
+
3
+ Several retrieval models expect the caller to say whether a text is a
4
+ stored document or a search query, and they lose accuracy without it.
5
+ The convention belongs to the model, not to the backend serving it, so
6
+ both the llama.cpp and the Ollama adapter read the same table.
7
+
8
+ Measured on this repository with nomic-embed-text, 16 queries: the
9
+ prefixes moved top-1 accuracy from 62% to 75%.
10
+ """
11
+
12
+ from collections import OrderedDict
13
+ from collections.abc import Sequence
14
+
15
+ # Model name prefix -> (document prefix, query prefix).
16
+ # Match on the start of the name, so a tag such as ":latest" still hits.
17
+ _CONVENTIONS: dict[str, tuple[str, str]] = {
18
+ "nomic-embed-text": ("search_document: ", "search_query: "),
19
+ "nomic-ai/nomic-embed-text": ("search_document: ", "search_query: "),
20
+ "mxbai-embed-large": (
21
+ "",
22
+ "Represent this sentence for searching relevant passages: ",
23
+ ),
24
+ }
25
+
26
+
27
+ def prefixes_for(model_name: str) -> tuple[str, str]:
28
+ """Return the (document, query) prefixes for *model_name*.
29
+
30
+ Return empty strings for a model with no known convention, which
31
+ leaves the text untouched.
32
+ """
33
+ name = model_name.lower()
34
+ for known, pair in _CONVENTIONS.items():
35
+ if known in name:
36
+ return pair
37
+ return ("", "")
38
+
39
+
40
+ # How many recent queries to keep. Typing walks over the same text as
41
+ # characters are added and removed, so a small cache turns a backspace
42
+ # from a request into a lookup.
43
+ QUERY_CACHE_SIZE = 128
44
+
45
+
46
+ class PrefixingEmbedder:
47
+ """Add task prefixes, then hand the text to the concrete backend.
48
+
49
+ Subclasses set ``model_name`` and implement ``_embed``.
50
+ """
51
+
52
+ model_name: str
53
+
54
+ def embed_documents(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
55
+ """Embed texts that will be stored and searched over."""
56
+ if not texts:
57
+ return []
58
+ prefix, _ = prefixes_for(self.model_name)
59
+ prepared = [f"{prefix}{text}" for text in texts] if prefix else list(texts)
60
+ return self._embed(prepared)
61
+
62
+ def embed_query(self, text: str) -> Sequence[float]:
63
+ """Embed one search query, reusing a recent answer.
64
+
65
+ An interactive search embeds a query on every keystroke, and
66
+ deleting a character asks for text already seen.
67
+ """
68
+ cache = getattr(self, "_query_cache", None)
69
+ if cache is None:
70
+ cache = self._query_cache = OrderedDict()
71
+ held = cache.get(text)
72
+ if held is not None:
73
+ cache.move_to_end(text)
74
+ return held
75
+
76
+ _, prefix = prefixes_for(self.model_name)
77
+ vectors = self._embed([f"{prefix}{text}" if prefix else text])
78
+ vector = vectors[0] if vectors else []
79
+
80
+ cache[text] = vector
81
+ if len(cache) > QUERY_CACHE_SIZE:
82
+ cache.popitem(last=False)
83
+ return vector
84
+
85
+ def _embed(self, texts: Sequence[str]) -> Sequence[Sequence[float]]:
86
+ """Encode already-prepared texts. Implemented by each adapter."""
87
+ raise NotImplementedError