glossdex 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
glossdex/__init__.py ADDED
@@ -0,0 +1,19 @@
1
+ """glossdex: glossed locally, indexed by meaning, gated by evidence."""
2
+
3
+ from .describers import Describer, OllamaDescriber, TextOnlyDescriber
4
+ from .embedders import Embedder, OllamaEmbedder, SentenceTransformersEmbedder
5
+ from .engine import Context, Glossdex, Hit, IndexReport, SearchResult
6
+ from .phrases import parse_phrases
7
+ from .profiles import EMBEDDINGGEMMA, EMBEDDINGGEMMA_PASSAGES, Profile
8
+ from .scoring import ScoringIndex, Trace, Unit, select
9
+ from .sources import FolderSource, Item
10
+
11
+ __version__ = "0.1.0"
12
+
13
+ __all__ = [
14
+ "Glossdex", "SearchResult", "Hit", "Context", "IndexReport",
15
+ "Describer", "OllamaDescriber", "TextOnlyDescriber",
16
+ "Embedder", "OllamaEmbedder", "SentenceTransformersEmbedder",
17
+ "FolderSource", "Item", "Profile", "EMBEDDINGGEMMA", "EMBEDDINGGEMMA_PASSAGES",
18
+ "ScoringIndex", "Unit", "Trace", "select", "parse_phrases",
19
+ ]
glossdex/cli.py ADDED
@@ -0,0 +1,183 @@
1
+ """Command line: ``glossdex index|search|explain|serve|doctor``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import os
8
+ import sys
9
+ import urllib.request
10
+
11
+ from . import __version__
12
+ from .describers import DEFAULT_OLLAMA, OllamaDescriber, TextOnlyDescriber
13
+ from .embedders import OllamaEmbedder
14
+ from .engine import Glossdex
15
+
16
+
17
+ def _engine(args) -> Glossdex:
18
+ if args.describer == "text-only":
19
+ describer = TextOnlyDescriber()
20
+ else:
21
+ describer = OllamaDescriber(model=args.vision_model, text_model=args.text_model,
22
+ host=args.ollama)
23
+ if args.embedder == "st":
24
+ from .embedders import SentenceTransformersEmbedder
25
+ embedder = SentenceTransformersEmbedder()
26
+ else:
27
+ embedder = OllamaEmbedder(model=args.embed_model, host=args.ollama)
28
+ return Glossdex.for_folder(args.folder, describer=describer, embedder=embedder)
29
+
30
+
31
+ def _common(p: argparse.ArgumentParser) -> None:
32
+ p.add_argument("folder", help="the folder to index or search")
33
+ p.add_argument("--ollama", default=os.environ.get("OLLAMA_HOST", DEFAULT_OLLAMA))
34
+ p.add_argument("--vision-model", default="gemma4:e4b",
35
+ help="Ollama model that describes images (default: gemma4:e4b)")
36
+ p.add_argument("--text-model", default=None,
37
+ help="Ollama model that lists facts of documents (default: the vision model)")
38
+ p.add_argument("--embed-model", default="embeddinggemma")
39
+ p.add_argument("--describer", choices=["ollama", "text-only"], default="ollama",
40
+ help="text-only: no language model; a document's sentences are its phrases")
41
+ p.add_argument("--embedder", choices=["ollama", "st"], default="ollama",
42
+ help="st: sentence-transformers (pip install 'glossdex[st]')")
43
+
44
+
45
+ def cmd_index(args) -> int:
46
+ gx = _engine(args)
47
+
48
+ def on_item(item, err):
49
+ p = gx.progress
50
+ status = "FAILED " + err if err else "ok"
51
+ print(f"[{p['done']}/{p['total']}] {item.title} {status}", flush=True)
52
+
53
+ rep = gx.index(rebuild=args.rebuild, on_item=on_item)
54
+ print(f"\nIndexed {rep.added}, unchanged {rep.unchanged}, removed {rep.removed}, "
55
+ f"failed {len(rep.failed)} in {rep.seconds:.1f}s. Index: {gx.store.path}")
56
+ return 1 if rep.failed and not rep.added else 0
57
+
58
+
59
+ def _one_line(text: str, width: int = 100) -> str:
60
+ s = " ".join(text.split())
61
+ return s if len(s) <= width else s[: width - 1] + "…"
62
+
63
+
64
+ def _print_result(res, show_trace: bool) -> None:
65
+ if not res.found:
66
+ print("Nothing found.")
67
+ for i, h in enumerate(res.hits, 1):
68
+ extra = " (near miss)" if h.reason == "pad" else ""
69
+ words = f" words {h.coverage:.0%}" if h.coverage > 0 else ""
70
+ print(f"{i:2d}. match {h.semantic:.3f}{words} {h.title}{extra}\n"
71
+ f" matched: {_one_line(h.matched)}")
72
+ if show_trace:
73
+ print("\n" + res.trace.verdict)
74
+
75
+
76
+ def cmd_search(args) -> int:
77
+ gx = _engine(args)
78
+ res = gx.search(args.query, strictness=args.strictness, compare_top_k=args.top_k)
79
+ if args.json:
80
+ print(json.dumps(res.to_dict(), indent=2))
81
+ else:
82
+ _print_result(res, not args.quiet)
83
+ return 0
84
+
85
+
86
+ def cmd_explain(args) -> int:
87
+ gx = _engine(args)
88
+ res = gx.search(args.query, strictness=args.strictness)
89
+ shown = {h.id for h in res.hits}
90
+ print(res.trace.verdict + "\n\nFull ranking before the cutoff:")
91
+ for i, h in enumerate(gx.explain(args.query, args.n), 1):
92
+ mark = "SHOWN" if h.id in shown else " "
93
+ print(f"{i:2d}. {mark} score {h.score:.4f} semantic {h.semantic:.4f} "
94
+ f"coverage {h.coverage:.3f} {h.title}\n matched: {_one_line(h.matched)}")
95
+ print(f"\nProfile: {res.trace.profile}")
96
+ return 0
97
+
98
+
99
+ def cmd_serve(args) -> int:
100
+ from .server import serve
101
+ serve(_engine(args), host=args.host, port=args.port, open_browser=not args.no_browser)
102
+ return 0
103
+
104
+
105
+ def cmd_doctor(args) -> int:
106
+ ok = True
107
+ print(f"glossdex {__version__}, Python {sys.version.split()[0]}")
108
+ try:
109
+ with urllib.request.urlopen(f"{args.ollama}/api/tags", timeout=5) as r:
110
+ models = {m["name"] for m in json.loads(r.read())["models"]}
111
+ print(f"Ollama at {args.ollama}: running")
112
+ for m in (args.vision_model, args.embed_model):
113
+ have = any(x == m or x.split(":")[0] == m for x in models) or f"{m}:latest" in models
114
+ print(f" {m}: {'installed' if have else 'MISSING, run: ollama pull ' + m}")
115
+ ok &= have
116
+ except Exception as e:
117
+ print(f"Ollama at {args.ollama}: NOT reachable ({e}). Install from https://ollama.com")
118
+ ok = False
119
+ try:
120
+ import PIL # noqa: F401
121
+ print("Pillow: installed")
122
+ except ImportError:
123
+ print("Pillow: missing (images are sent at full size)")
124
+ try:
125
+ import pypdf # noqa: F401
126
+ print("pypdf: installed (PDF files will be indexed)")
127
+ except ImportError:
128
+ print("pypdf: not installed (PDFs skipped; pip install 'glossdex[pdf]')")
129
+ print("Ready." if ok else "Not ready: fix the lines above.")
130
+ return 0 if ok else 1
131
+
132
+
133
+ def main(argv=None) -> int:
134
+ ap = argparse.ArgumentParser(prog="glossdex", description=(
135
+ "Glossed locally, indexed by meaning, gated by evidence. Private semantic search over "
136
+ "a folder of images and documents that shows only what the evidence supports."))
137
+ ap.add_argument("--version", action="version", version=f"glossdex {__version__}")
138
+ sub = ap.add_subparsers(dest="cmd", required=True)
139
+
140
+ p = sub.add_parser("index", help="describe and embed new or changed files")
141
+ _common(p)
142
+ p.add_argument("--rebuild", action="store_true", help="discard the index and start over")
143
+ p.set_defaults(fn=cmd_index)
144
+
145
+ for name, fn, helptext in (("search", cmd_search, "search the index"),
146
+ ("explain", cmd_explain, "show the ranking and every decision")):
147
+ p = sub.add_parser(name, help=helptext)
148
+ _common(p)
149
+ p.add_argument("query")
150
+ p.add_argument("--strictness", type=float, default=None,
151
+ help="0..1; lower shows more (default 0.70)")
152
+ if name == "search":
153
+ p.add_argument("--top-k", type=int, default=0,
154
+ help="show a fixed top-k instead, for comparison")
155
+ p.add_argument("--json", action="store_true")
156
+ p.add_argument("--quiet", action="store_true")
157
+ else:
158
+ p.add_argument("-n", type=int, default=15)
159
+ p.set_defaults(fn=fn)
160
+
161
+ p = sub.add_parser("serve", help="open the local web UI")
162
+ _common(p)
163
+ p.add_argument("--host", default="127.0.0.1")
164
+ p.add_argument("--port", type=int, default=8484)
165
+ p.add_argument("--no-browser", action="store_true")
166
+ p.set_defaults(fn=cmd_serve)
167
+
168
+ p = sub.add_parser("doctor", help="check that Ollama and the models are ready")
169
+ p.add_argument("--ollama", default=os.environ.get("OLLAMA_HOST", DEFAULT_OLLAMA))
170
+ p.add_argument("--vision-model", default="gemma4:e4b")
171
+ p.add_argument("--embed-model", default="embeddinggemma")
172
+ p.set_defaults(fn=cmd_doctor)
173
+
174
+ args = ap.parse_args(argv)
175
+ try:
176
+ return args.fn(args)
177
+ except RuntimeError as e:
178
+ print(f"error: {e}", file=sys.stderr)
179
+ return 2
180
+
181
+
182
+ if __name__ == "__main__":
183
+ sys.exit(main())
glossdex/describers.py ADDED
@@ -0,0 +1,155 @@
1
+ """Description-generator adapters: content item -> a gloss (a list of independent phrases).
2
+
3
+ Any local model works. The built-in adapters talk to Ollama over its local HTTP API; nothing
4
+ leaves the machine. Write your own by implementing :class:`Describer`.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import base64
10
+ import io
11
+ import json
12
+ import time
13
+ import urllib.error
14
+ import urllib.request
15
+ from dataclasses import dataclass
16
+ from typing import List, Optional, Protocol
17
+
18
+ from .phrases import parse_phrases
19
+ from .sources import Item
20
+
21
+ DEFAULT_OLLAMA = "http://localhost:11434"
22
+
23
+ IMAGE_PROMPT = (
24
+ "Describe this image for a search index as up to {n} short phrases.\n"
25
+ "Put each phrase on its own line and end it with a period.\n"
26
+ "One element per phrase: the main subject, what people or animals are doing, objects, "
27
+ "setting, colors, mood. Do not join two elements into one phrase.\n"
28
+ "If the image is a document, receipt, bill, sign or screen: name its type and copy its key "
29
+ "text (names, organizations, dates, amounts, identifiers).\n"
30
+ "Only state what is visible. No introduction, no conclusion, no markdown."
31
+ )
32
+
33
+ DOCUMENT_PROMPT = (
34
+ "List the key facts of this document as up to {n} concise, independent phrases, one fact "
35
+ "per line, each ending with a period: document type, persons or parties named, "
36
+ "organization, dates, amounts, identifiers, subject, findings or decisions, and requested "
37
+ "actions. Use the document's own words for names, numbers and dates. Do not add "
38
+ "information that is not in the document. No intro or outro.\n\nDocument:\n{text}"
39
+ )
40
+
41
+
42
+ @dataclass
43
+ class Description:
44
+ text: str # the raw model output (the whole description)
45
+ phrases: List[str]
46
+ model: str
47
+ seconds: float
48
+
49
+
50
+ class Describer(Protocol):
51
+ name: str
52
+
53
+ def describe(self, item: Item) -> Description: ...
54
+
55
+
56
+ class DescriberError(RuntimeError):
57
+ pass
58
+
59
+
60
+ def _post(url: str, payload: dict, timeout: float) -> dict:
61
+ req = urllib.request.Request(url, data=json.dumps(payload).encode("utf-8"),
62
+ headers={"Content-Type": "application/json"})
63
+ try:
64
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
65
+ return json.loads(resp.read().decode("utf-8"))
66
+ except urllib.error.HTTPError as e:
67
+ body = e.read().decode("utf-8", "replace")[:300]
68
+ raise DescriberError(f"Ollama returned {e.code}: {body}") from e
69
+ except urllib.error.URLError as e:
70
+ raise DescriberError(
71
+ f"Cannot reach Ollama at {url}. Is it running? (`ollama serve`)") from e
72
+
73
+
74
+ def _image_b64(path: str, max_side: int) -> str:
75
+ try:
76
+ from PIL import Image, ImageOps
77
+ except ImportError: # pragma: no cover
78
+ with open(path, "rb") as f:
79
+ return base64.b64encode(f.read()).decode("ascii")
80
+ with Image.open(path) as im:
81
+ im = ImageOps.exif_transpose(im).convert("RGB")
82
+ im.thumbnail((max_side, max_side))
83
+ buf = io.BytesIO()
84
+ im.save(buf, format="JPEG", quality=85)
85
+ return base64.b64encode(buf.getvalue()).decode("ascii")
86
+
87
+
88
+ class OllamaDescriber:
89
+ """Describes images with a local vision-language model and text with the same (or another) model.
90
+
91
+ Defaults to Gemma 4 E4B, which reads both images and text.
92
+ """
93
+
94
+ def __init__(self, model: str = "gemma4:e4b", text_model: Optional[str] = None,
95
+ host: str = DEFAULT_OLLAMA, max_phrases: int = 7, max_side: int = 1024,
96
+ timeout: float = 600.0, image_prompt: str = IMAGE_PROMPT,
97
+ document_prompt: str = DOCUMENT_PROMPT):
98
+ self.model = model
99
+ self.text_model = text_model or model
100
+ self.host = host.rstrip("/")
101
+ self.max_phrases = max_phrases
102
+ self.max_side = max_side
103
+ self.timeout = timeout
104
+ self.image_prompt = image_prompt
105
+ self.document_prompt = document_prompt
106
+ self.name = f"ollama:{self.model}"
107
+
108
+ def _generate(self, model: str, prompt: str, images: Optional[List[str]] = None) -> dict:
109
+ payload = {
110
+ "model": model,
111
+ "prompt": prompt,
112
+ "stream": False,
113
+ "think": False,
114
+ "options": {"temperature": 0, "num_predict": 60 * self.max_phrases + 120},
115
+ }
116
+ if images:
117
+ payload["images"] = images
118
+ try:
119
+ return _post(f"{self.host}/api/generate", payload, self.timeout)
120
+ except DescriberError as e:
121
+ if "think" in str(e): # models without a thinking switch reject the field
122
+ payload.pop("think")
123
+ return _post(f"{self.host}/api/generate", payload, self.timeout)
124
+ raise
125
+
126
+ def describe(self, item: Item) -> Description:
127
+ start = time.time()
128
+ if item.kind == "image":
129
+ res = self._generate(self.model, self.image_prompt.format(n=self.max_phrases),
130
+ [_image_b64(item.path, self.max_side)])
131
+ model = self.model
132
+ else:
133
+ res = self._generate(self.text_model, self.document_prompt.format(
134
+ n=self.max_phrases, text=item.text))
135
+ model = self.text_model
136
+ text = (res.get("response") or "").strip()
137
+ truncated = res.get("done_reason") == "length"
138
+ return Description(text, parse_phrases(text, truncated), model, time.time() - start)
139
+
140
+
141
+ class TextOnlyDescriber:
142
+ """No model at all: an item's own sentences are its phrases. Text and documents only.
143
+
144
+ Useful to try glossdex on documents in seconds, and for text that is already terse
145
+ (notes, tickets, logs). Images are skipped.
146
+ """
147
+
148
+ name = "text-only"
149
+
150
+ def describe(self, item: Item) -> Description:
151
+ if item.kind == "image":
152
+ raise DescriberError("text-only describer cannot describe images")
153
+ start = time.time()
154
+ text = item.text.strip()
155
+ return Description(text, parse_phrases(text), self.name, time.time() - start)
glossdex/embedders.py ADDED
@@ -0,0 +1,104 @@
1
+ """Embedding adapters: text -> vector, in a document mode and a query mode.
2
+
3
+ ``profile_key`` names the model whose score units the vectors are in; it selects the
4
+ thresholds in :mod:`glossdex.profiles`. Two adapters that run the same model (for example
5
+ EmbeddingGemma through Ollama or through sentence-transformers) share a key.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import urllib.error
12
+ import urllib.request
13
+ from typing import List, Protocol, Sequence
14
+
15
+ import numpy as np
16
+
17
+ from .describers import DEFAULT_OLLAMA
18
+
19
+ #: EmbeddingGemma's task prompts for retrieval (from its model card).
20
+ GEMMA_QUERY_PREFIX = "task: search result | query: "
21
+ GEMMA_DOCUMENT_PREFIX = "title: none | text: "
22
+
23
+
24
+ class Embedder(Protocol):
25
+ name: str
26
+ profile_key: str
27
+
28
+ def embed_documents(self, texts: Sequence[str]) -> np.ndarray: ...
29
+
30
+ def embed_query(self, text: str) -> np.ndarray: ...
31
+
32
+
33
+ class EmbedderError(RuntimeError):
34
+ pass
35
+
36
+
37
+ class OllamaEmbedder:
38
+ """EmbeddingGemma (or any embedding model) served by a local Ollama."""
39
+
40
+ def __init__(self, model: str = "embeddinggemma", host: str = DEFAULT_OLLAMA,
41
+ profile_key: str = None, query_prefix: str = None, document_prefix: str = None,
42
+ timeout: float = 120.0, batch: int = 32):
43
+ self.model = model
44
+ self.host = host.rstrip("/")
45
+ self.timeout = timeout
46
+ self.batch = batch
47
+ is_gemma = model.split(":")[0] == "embeddinggemma"
48
+ self.profile_key = profile_key or ("embeddinggemma-300m" if is_gemma else model)
49
+ self.query_prefix = query_prefix if query_prefix is not None else (
50
+ GEMMA_QUERY_PREFIX if is_gemma else "")
51
+ self.document_prefix = document_prefix if document_prefix is not None else (
52
+ GEMMA_DOCUMENT_PREFIX if is_gemma else "")
53
+ self.name = f"ollama:{model}"
54
+
55
+ def _embed(self, texts: List[str]) -> np.ndarray:
56
+ out = []
57
+ for i in range(0, len(texts), self.batch):
58
+ payload = {"model": self.model, "input": texts[i:i + self.batch]}
59
+ req = urllib.request.Request(f"{self.host}/api/embed",
60
+ data=json.dumps(payload).encode("utf-8"),
61
+ headers={"Content-Type": "application/json"})
62
+ try:
63
+ with urllib.request.urlopen(req, timeout=self.timeout) as resp:
64
+ data = json.loads(resp.read().decode("utf-8"))
65
+ except urllib.error.HTTPError as e:
66
+ raise EmbedderError(f"Ollama returned {e.code}: "
67
+ f"{e.read().decode('utf-8', 'replace')[:300]}") from e
68
+ except urllib.error.URLError as e:
69
+ raise EmbedderError(f"Cannot reach Ollama at {self.host}. Is it running?") from e
70
+ out.extend(data["embeddings"])
71
+ return np.asarray(out, dtype=np.float32)
72
+
73
+ def embed_documents(self, texts: Sequence[str]) -> np.ndarray:
74
+ return self._embed([self.document_prefix + t for t in texts])
75
+
76
+ def embed_query(self, text: str) -> np.ndarray:
77
+ return self._embed([self.query_prefix + text])[0]
78
+
79
+
80
+ class SentenceTransformersEmbedder:
81
+ """EmbeddingGemma through sentence-transformers (``pip install glossdex[st]``).
82
+
83
+ ``google/embeddinggemma-300m`` is gated on Hugging Face: accept its license on the model
84
+ page and run ``hf auth login`` once.
85
+ """
86
+
87
+ def __init__(self, model: str = "google/embeddinggemma-300m", device: str = None,
88
+ profile_key: str = None):
89
+ try:
90
+ from sentence_transformers import SentenceTransformer
91
+ except ImportError as e: # pragma: no cover
92
+ raise EmbedderError("pip install 'glossdex[st]' to use sentence-transformers") from e
93
+ self._model = SentenceTransformer(model, device=device)
94
+ self.profile_key = profile_key or (
95
+ "embeddinggemma-300m" if "embeddinggemma" in model else model)
96
+ self.name = f"st:{model}"
97
+
98
+ def embed_documents(self, texts: Sequence[str]) -> np.ndarray:
99
+ encode = getattr(self._model, "encode_document", self._model.encode)
100
+ return np.asarray(encode(list(texts)), dtype=np.float32)
101
+
102
+ def embed_query(self, text: str) -> np.ndarray:
103
+ encode = getattr(self._model, "encode_query", self._model.encode)
104
+ return np.asarray(encode([text])[0], dtype=np.float32)