loom-notes 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- loom_notes/__init__.py +8 -0
- loom_notes/cli.py +229 -0
- loom_notes/embed/__init__.py +3 -0
- loom_notes/embed/base.py +52 -0
- loom_notes/embed/bge_m3.py +66 -0
- loom_notes/embed/fake.py +69 -0
- loom_notes/embed/reranker.py +44 -0
- loom_notes/evals.py +108 -0
- loom_notes/ids.py +13 -0
- loom_notes/ingest/__init__.py +12 -0
- loom_notes/ingest/chunk.py +142 -0
- loom_notes/ingest/dedup.py +16 -0
- loom_notes/ingest/extract.py +207 -0
- loom_notes/ingest/paths.py +53 -0
- loom_notes/models.py +90 -0
- loom_notes/retrieval.py +81 -0
- loom_notes/server.py +216 -0
- loom_notes/service.py +253 -0
- loom_notes/settings.py +88 -0
- loom_notes/store/__init__.py +8 -0
- loom_notes/store/qdrant.py +300 -0
- loom_notes-1.1.0.dist-info/METADATA +339 -0
- loom_notes-1.1.0.dist-info/RECORD +26 -0
- loom_notes-1.1.0.dist-info/WHEEL +4 -0
- loom_notes-1.1.0.dist-info/entry_points.txt +3 -0
- loom_notes-1.1.0.dist-info/licenses/LICENSE +201 -0
loom_notes/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""loom-notes — mémoire locale pour Claude Desktop et les agents IA."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
__version__ = version("loom-notes") # source unique : pyproject.toml
|
|
7
|
+
except PackageNotFoundError: # package non installé (exécution depuis les sources)
|
|
8
|
+
__version__ = "0.0.0"
|
loom_notes/cli.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""CLI d'administration : ingestion, recherche, sauvegarde, réindexation."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
from collections.abc import Awaitable, Callable
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Annotated
|
|
9
|
+
|
|
10
|
+
import typer
|
|
11
|
+
from pydantic import BaseModel
|
|
12
|
+
|
|
13
|
+
from loom_notes.ingest.extract import ExtractionError
|
|
14
|
+
from loom_notes.ingest.paths import PathDeniedError
|
|
15
|
+
from loom_notes.service import MemoryService, NotFoundError, build_service
|
|
16
|
+
from loom_notes.settings import Settings
|
|
17
|
+
from loom_notes.store import ModelMismatchError, StoreUnavailableError
|
|
18
|
+
|
|
19
|
+
app = typer.Typer(no_args_is_help=True, add_completion=False, help="loom-notes — mémoire locale.")
|
|
20
|
+
|
|
21
|
+
_fake = False
|
|
22
|
+
_data_dir: Path | None = None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@app.callback()
|
|
26
|
+
def _root(
|
|
27
|
+
fake: Annotated[
|
|
28
|
+
bool, typer.Option("--fake", help="Modèles factices (tests, sans GPU).")
|
|
29
|
+
] = False,
|
|
30
|
+
data_dir: Annotated[
|
|
31
|
+
Path | None, typer.Option("--data-dir", help="Répertoire des données.")
|
|
32
|
+
] = None,
|
|
33
|
+
) -> None:
|
|
34
|
+
global _fake, _data_dir
|
|
35
|
+
_fake, _data_dir = fake, data_dir
|
|
36
|
+
os.environ.setdefault("TQDM_DISABLE", "1") # barres de progression de FlagEmbedding
|
|
37
|
+
os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _settings() -> Settings:
|
|
41
|
+
return Settings(data_dir=_data_dir) if _data_dir else Settings()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _run[T](fn: Callable[[MemoryService], Awaitable[T]], *, check_model: bool = True) -> T:
|
|
45
|
+
async def go() -> T:
|
|
46
|
+
svc = build_service(_settings(), fake=True if _fake else None)
|
|
47
|
+
try:
|
|
48
|
+
await svc.start(check_model=check_model)
|
|
49
|
+
return await fn(svc)
|
|
50
|
+
finally:
|
|
51
|
+
await svc.close()
|
|
52
|
+
|
|
53
|
+
try:
|
|
54
|
+
return asyncio.run(go())
|
|
55
|
+
except (
|
|
56
|
+
NotFoundError,
|
|
57
|
+
ValueError,
|
|
58
|
+
ModelMismatchError,
|
|
59
|
+
StoreUnavailableError,
|
|
60
|
+
ExtractionError,
|
|
61
|
+
PathDeniedError,
|
|
62
|
+
) as exc:
|
|
63
|
+
typer.secho(f"erreur : {exc}", fg=typer.colors.RED, err=True)
|
|
64
|
+
raise typer.Exit(1) from exc
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _dump(obj: BaseModel | list[BaseModel] | dict[str, int]) -> None:
|
|
68
|
+
if isinstance(obj, BaseModel):
|
|
69
|
+
typer.echo(obj.model_dump_json(indent=2))
|
|
70
|
+
elif isinstance(obj, dict):
|
|
71
|
+
typer.echo(json.dumps(obj, indent=2, ensure_ascii=False))
|
|
72
|
+
else:
|
|
73
|
+
typer.echo(json.dumps([o.model_dump() for o in obj], indent=2, ensure_ascii=False))
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
Tags = Annotated[list[str] | None, typer.Option("--tag", "-t", help="Tag (répétable).")]
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@app.command("add-text")
|
|
80
|
+
def add_text(
|
|
81
|
+
project: str,
|
|
82
|
+
title: str,
|
|
83
|
+
text: Annotated[str | None, typer.Argument(help="Texte ; lu sur stdin si absent.")] = None,
|
|
84
|
+
tags: Tags = None,
|
|
85
|
+
) -> None:
|
|
86
|
+
"""Ajoute un texte brut."""
|
|
87
|
+
content = text if text is not None else typer.get_text_stream("stdin").read()
|
|
88
|
+
_dump(_run(lambda s: s.add_text(content, title, project, tags)))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@app.command("add-url")
|
|
92
|
+
def add_url(project: str, url: str, tags: Tags = None) -> None:
|
|
93
|
+
"""Télécharge une page web et l'ajoute."""
|
|
94
|
+
_dump(_run(lambda s: s.add_url(url, project, tags)))
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@app.command("add-file")
|
|
98
|
+
def add_file(project: str, path: Path, tags: Tags = None) -> None:
|
|
99
|
+
"""Ajoute un fichier markdown local."""
|
|
100
|
+
_dump(_run(lambda s: s.add_file(path, project, tags)))
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@app.command()
|
|
104
|
+
def search(
|
|
105
|
+
query: str,
|
|
106
|
+
project: Annotated[str | None, typer.Option("--project", "-p")] = None,
|
|
107
|
+
tags: Tags = None,
|
|
108
|
+
k: Annotated[int, typer.Option("--k", "-k", min=1, max=10)] = 5,
|
|
109
|
+
) -> None:
|
|
110
|
+
"""Recherche hybride + reranking."""
|
|
111
|
+
hits = _run(lambda s: s.search(query, project, tags, k))
|
|
112
|
+
if not hits:
|
|
113
|
+
typer.echo("aucun résultat")
|
|
114
|
+
return
|
|
115
|
+
for h in hits:
|
|
116
|
+
typer.secho(f"{h.score:.3f} {h.title} [{h.project}] {h.doc_id}", bold=True)
|
|
117
|
+
typer.echo(f" {h.heading_path}{' (extrait)' if h.truncated else ''}")
|
|
118
|
+
typer.echo(" " + h.snippet.replace("\n", "\n ") + "\n")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@app.command()
|
|
122
|
+
def get(doc_id: str) -> None:
|
|
123
|
+
"""Affiche un document complet."""
|
|
124
|
+
_dump(_run(lambda s: s.get(doc_id)))
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@app.command("list")
|
|
128
|
+
def list_docs(
|
|
129
|
+
project: Annotated[str | None, typer.Option("--project", "-p")] = None,
|
|
130
|
+
n: Annotated[int, typer.Option("--n", "-n", min=1, max=500)] = 20,
|
|
131
|
+
) -> None:
|
|
132
|
+
"""Derniers documents ajoutés."""
|
|
133
|
+
for d in _run(lambda s: s.list_docs(project, n)):
|
|
134
|
+
when = (d.updated_at or d.added_at)[:19]
|
|
135
|
+
typer.echo(f"{when} {d.project:<12} {d.title} ({d.chars} car.) {d.doc_id}")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@app.command()
|
|
139
|
+
def projects() -> None:
|
|
140
|
+
"""Projets et nombre de documents."""
|
|
141
|
+
_dump(_run(lambda s: s.projects()))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@app.command()
|
|
145
|
+
def delete(doc_id: str) -> None:
|
|
146
|
+
"""Supprime un document et ses chunks."""
|
|
147
|
+
_dump(_run(lambda s: s.delete(doc_id)))
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
@app.command()
|
|
151
|
+
def export(path: Annotated[Path | None, typer.Argument()] = None) -> None:
|
|
152
|
+
"""Exporte tous les documents en JSONL (sauvegarde de référence)."""
|
|
153
|
+
target = path or _settings().export_path
|
|
154
|
+
n = _run(lambda s: s.export(target))
|
|
155
|
+
typer.echo(f"{n} document(s) exporté(s) vers {target}")
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@app.command("import")
|
|
159
|
+
def import_(path: Annotated[Path | None, typer.Argument()] = None) -> None:
|
|
160
|
+
"""Réimporte un export JSONL (les doc_id existants sont ignorés)."""
|
|
161
|
+
source = path or _settings().export_path
|
|
162
|
+
imported, skipped = _run(lambda s: s.import_(source))
|
|
163
|
+
typer.echo(f"{imported} importé(s), {skipped} ignoré(s)")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@app.command("eval")
|
|
167
|
+
def eval_(
|
|
168
|
+
path: Annotated[
|
|
169
|
+
Path | None, typer.Argument(help="Jeu doré JSONL ; défaut data/golden.jsonl")
|
|
170
|
+
] = None,
|
|
171
|
+
k: Annotated[int, typer.Option("--k", min=1, max=10)] = 10,
|
|
172
|
+
as_json: Annotated[bool, typer.Option("--json", help="Rapport complet en JSON.")] = False,
|
|
173
|
+
) -> None:
|
|
174
|
+
"""Évalue le retrieval sur le jeu doré : recall@1, recall@5, MRR, seuil suggéré."""
|
|
175
|
+
from loom_notes.evals import load_golden, run_eval
|
|
176
|
+
|
|
177
|
+
golden = path or _settings().golden_path
|
|
178
|
+
cases = load_golden(golden)
|
|
179
|
+
if not cases:
|
|
180
|
+
typer.secho(f"aucun cas dans {golden} — ajoute-en avec `loom-notes eval-add`", err=True)
|
|
181
|
+
raise typer.Exit(1)
|
|
182
|
+
report = _run(lambda s: run_eval(s, cases, k))
|
|
183
|
+
if as_json:
|
|
184
|
+
typer.echo(report.model_dump_json(indent=2))
|
|
185
|
+
return
|
|
186
|
+
for r in report.results:
|
|
187
|
+
mark = "ok " if r.rank == 1 else ("~ " if r.rank else "KO ")
|
|
188
|
+
rank = f"#{r.rank}" if r.rank else "absent"
|
|
189
|
+
score = f"{r.expected_score:.3f}" if r.expected_score is not None else " - "
|
|
190
|
+
typer.echo(f"{mark} {rank:>7} {score} {r.query}")
|
|
191
|
+
typer.echo(
|
|
192
|
+
f"\n{report.cases} cas — recall@1 {report.recall_at_1:.2f} "
|
|
193
|
+
f"recall@5 {report.recall_at_5:.2f} MRR {report.mrr:.2f}"
|
|
194
|
+
)
|
|
195
|
+
if report.suggested_min_score is not None:
|
|
196
|
+
noise = report.noise_removed_at_suggested
|
|
197
|
+
typer.echo(
|
|
198
|
+
f"min_score suggéré : {report.suggested_min_score} "
|
|
199
|
+
f"(aucun cas perdu ; {noise:.0%} du bruit filtré)"
|
|
200
|
+
if noise is not None
|
|
201
|
+
else f"min_score suggéré : {report.suggested_min_score}"
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
@app.command("eval-add")
|
|
206
|
+
def eval_add(
|
|
207
|
+
query: str,
|
|
208
|
+
doc_id: str,
|
|
209
|
+
project: Annotated[str | None, typer.Option("--project", "-p")] = None,
|
|
210
|
+
path: Annotated[Path | None, typer.Option("--path")] = None,
|
|
211
|
+
) -> None:
|
|
212
|
+
"""Ajoute un cas au jeu doré (le titre est relu depuis la base pour lisibilité)."""
|
|
213
|
+
from loom_notes.evals import GoldenCase, append_golden
|
|
214
|
+
|
|
215
|
+
doc = _run(lambda s: s.get(doc_id))
|
|
216
|
+
case = GoldenCase(query=query, doc_id=doc.doc_id, title=doc.title, project=project)
|
|
217
|
+
append_golden(path or _settings().golden_path, case)
|
|
218
|
+
typer.echo(f"ajouté : « {query} » → {doc.title}")
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@app.command()
|
|
222
|
+
def reindex() -> None:
|
|
223
|
+
"""Reconstruit les chunks depuis les documents avec le modèle courant."""
|
|
224
|
+
n = _run(lambda s: s.reindex(), check_model=False)
|
|
225
|
+
typer.echo(f"{n} document(s) réindexé(s)")
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
if __name__ == "__main__":
|
|
229
|
+
app()
|
loom_notes/embed/base.py
ADDED
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Interfaces des modèles. Synchrones : le service les exécute dans un thread."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from typing import Any, Protocol
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True, slots=True)
|
|
10
|
+
class SparseVec:
|
|
11
|
+
indices: list[int]
|
|
12
|
+
values: list[float]
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True, slots=True)
|
|
16
|
+
class Embedding:
|
|
17
|
+
dense: list[float]
|
|
18
|
+
sparse: SparseVec
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class Embedder(Protocol):
|
|
22
|
+
@property
|
|
23
|
+
def model_name(self) -> str: ...
|
|
24
|
+
|
|
25
|
+
@property
|
|
26
|
+
def dim(self) -> int: ...
|
|
27
|
+
|
|
28
|
+
def embed_documents(self, texts: Sequence[str]) -> list[Embedding]: ...
|
|
29
|
+
|
|
30
|
+
def embed_query(self, text: str) -> Embedding: ...
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Reranker(Protocol):
|
|
34
|
+
@property
|
|
35
|
+
def model_name(self) -> str: ...
|
|
36
|
+
|
|
37
|
+
def score(self, query: str, passages: Sequence[str]) -> list[float]: ...
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def resolve_device(device: str) -> str:
|
|
41
|
+
"""Résout `auto` en cuda, sinon mps, sinon cpu ; toute autre valeur est rendue telle quelle.
|
|
42
|
+
|
|
43
|
+
torch n'est importé que pour `auto`, au chargement des vrais modèles.
|
|
44
|
+
"""
|
|
45
|
+
if device != "auto":
|
|
46
|
+
return device
|
|
47
|
+
torch: Any = importlib.import_module("torch")
|
|
48
|
+
if torch.cuda.is_available():
|
|
49
|
+
return "cuda"
|
|
50
|
+
if torch.backends.mps.is_available():
|
|
51
|
+
return "mps"
|
|
52
|
+
return "cpu"
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
"""BGE-M3 via FlagEmbedding : dense + sparse en un seul passage, chargé paresseusement."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from loom_notes.embed.base import Embedding, SparseVec, resolve_device
|
|
8
|
+
from loom_notes.settings import Settings
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BgeM3Embedder:
|
|
12
|
+
def __init__(self, settings: Settings) -> None:
|
|
13
|
+
self._settings = settings
|
|
14
|
+
self._model: Any = None
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def model_name(self) -> str:
|
|
18
|
+
return self._settings.dense_model
|
|
19
|
+
|
|
20
|
+
@property
|
|
21
|
+
def dim(self) -> int:
|
|
22
|
+
return self._settings.dense_dim
|
|
23
|
+
|
|
24
|
+
def _load(self) -> Any:
|
|
25
|
+
if self._model is None:
|
|
26
|
+
try:
|
|
27
|
+
flag: Any = importlib.import_module("FlagEmbedding")
|
|
28
|
+
except ImportError as exc:
|
|
29
|
+
raise RuntimeError(
|
|
30
|
+
'FlagEmbedding absent : pip install "loom-notes[models]"'
|
|
31
|
+
) from exc
|
|
32
|
+
device = resolve_device(self._settings.device)
|
|
33
|
+
self._model = flag.BGEM3FlagModel(
|
|
34
|
+
self._settings.dense_model,
|
|
35
|
+
use_fp16=self._settings.use_fp16 and device.startswith("cuda"),
|
|
36
|
+
devices=device,
|
|
37
|
+
)
|
|
38
|
+
return self._model
|
|
39
|
+
|
|
40
|
+
def _encode(self, texts: Sequence[str]) -> list[Embedding]:
|
|
41
|
+
out: Any = self._load().encode(
|
|
42
|
+
list(texts),
|
|
43
|
+
batch_size=16,
|
|
44
|
+
max_length=1024,
|
|
45
|
+
return_dense=True,
|
|
46
|
+
return_sparse=True,
|
|
47
|
+
return_colbert_vecs=False,
|
|
48
|
+
)
|
|
49
|
+
dense_vecs: Any = out["dense_vecs"]
|
|
50
|
+
lexical: Any = out["lexical_weights"]
|
|
51
|
+
result: list[Embedding] = []
|
|
52
|
+
for dense, weights in zip(dense_vecs, lexical, strict=True):
|
|
53
|
+
items = sorted((int(k), float(v)) for k, v in weights.items() if float(v) > 0)
|
|
54
|
+
result.append(
|
|
55
|
+
Embedding(
|
|
56
|
+
dense=[float(x) for x in dense],
|
|
57
|
+
sparse=SparseVec(indices=[i for i, _ in items], values=[v for _, v in items]),
|
|
58
|
+
)
|
|
59
|
+
)
|
|
60
|
+
return result
|
|
61
|
+
|
|
62
|
+
def embed_documents(self, texts: Sequence[str]) -> list[Embedding]:
|
|
63
|
+
return self._encode(texts) if texts else []
|
|
64
|
+
|
|
65
|
+
def embed_query(self, text: str) -> Embedding:
|
|
66
|
+
return self._encode([text])[0]
|
loom_notes/embed/fake.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""Modèles factices et déterministes pour les tests : pas de GPU, pas de téléchargement.
|
|
2
|
+
Le dense est un sac de mots haché, le sparse compte les mots — assez pour que la recherche
|
|
3
|
+
lexicale se comporte de façon plausible."""
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import math
|
|
7
|
+
import re
|
|
8
|
+
from collections import Counter
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
|
|
11
|
+
from loom_notes.embed.base import Embedding, SparseVec
|
|
12
|
+
|
|
13
|
+
_TOKEN = re.compile(r"\w+", re.UNICODE)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _tokens(text: str) -> list[str]:
|
|
17
|
+
return [t.casefold() for t in _TOKEN.findall(text) if len(t) > 1]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _bucket(token: str, mod: int) -> int:
|
|
21
|
+
return int.from_bytes(hashlib.blake2b(token.encode(), digest_size=4).digest(), "big") % mod
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class FakeEmbedder:
|
|
25
|
+
def __init__(self, dim: int = 64) -> None:
|
|
26
|
+
self._dim = dim
|
|
27
|
+
|
|
28
|
+
@property
|
|
29
|
+
def model_name(self) -> str:
|
|
30
|
+
return f"fake-{self._dim}"
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def dim(self) -> int:
|
|
34
|
+
return self._dim
|
|
35
|
+
|
|
36
|
+
def _one(self, text: str) -> Embedding:
|
|
37
|
+
counts = Counter(_tokens(text))
|
|
38
|
+
dense = [0.0] * self._dim
|
|
39
|
+
for tok, n in counts.items():
|
|
40
|
+
dense[_bucket(tok, self._dim)] += float(n)
|
|
41
|
+
norm = math.sqrt(sum(x * x for x in dense)) or 1.0
|
|
42
|
+
dense = [x / norm for x in dense]
|
|
43
|
+
sparse_counts: dict[int, float] = {}
|
|
44
|
+
for tok, n in counts.items():
|
|
45
|
+
idx = _bucket(tok, 1 << 20)
|
|
46
|
+
sparse_counts[idx] = sparse_counts.get(idx, 0.0) + float(n)
|
|
47
|
+
items = sorted(sparse_counts.items())
|
|
48
|
+
return Embedding(
|
|
49
|
+
dense=dense,
|
|
50
|
+
sparse=SparseVec(indices=[i for i, _ in items], values=[v for _, v in items]),
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
def embed_documents(self, texts: Sequence[str]) -> list[Embedding]:
|
|
54
|
+
return [self._one(t) for t in texts]
|
|
55
|
+
|
|
56
|
+
def embed_query(self, text: str) -> Embedding:
|
|
57
|
+
return self._one(text)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class FakeReranker:
|
|
61
|
+
@property
|
|
62
|
+
def model_name(self) -> str:
|
|
63
|
+
return "fake-reranker"
|
|
64
|
+
|
|
65
|
+
def score(self, query: str, passages: Sequence[str]) -> list[float]:
|
|
66
|
+
q = set(_tokens(query))
|
|
67
|
+
if not q:
|
|
68
|
+
return [0.0] * len(passages)
|
|
69
|
+
return [len(q & set(_tokens(p))) / len(q) for p in passages]
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""Cross-encoder bge-reranker-v2-m3 via FlagEmbedding, chargé paresseusement."""
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from loom_notes.embed.base import resolve_device
|
|
8
|
+
from loom_notes.settings import Settings
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BgeReranker:
|
|
12
|
+
def __init__(self, settings: Settings) -> None:
|
|
13
|
+
self._settings = settings
|
|
14
|
+
self._model: Any = None
|
|
15
|
+
|
|
16
|
+
@property
|
|
17
|
+
def model_name(self) -> str:
|
|
18
|
+
return self._settings.reranker_model
|
|
19
|
+
|
|
20
|
+
def _load(self) -> Any:
|
|
21
|
+
if self._model is None:
|
|
22
|
+
try:
|
|
23
|
+
flag: Any = importlib.import_module("FlagEmbedding")
|
|
24
|
+
except ImportError as exc:
|
|
25
|
+
raise RuntimeError(
|
|
26
|
+
'FlagEmbedding absent : pip install "loom-notes[models]"'
|
|
27
|
+
) from exc
|
|
28
|
+
device = resolve_device(self._settings.device)
|
|
29
|
+
self._model = flag.FlagReranker(
|
|
30
|
+
self._settings.reranker_model,
|
|
31
|
+
use_fp16=self._settings.use_fp16 and device.startswith("cuda"),
|
|
32
|
+
devices=device,
|
|
33
|
+
)
|
|
34
|
+
return self._model
|
|
35
|
+
|
|
36
|
+
def score(self, query: str, passages: Sequence[str]) -> list[float]:
|
|
37
|
+
if not passages:
|
|
38
|
+
return []
|
|
39
|
+
scores: Any = self._load().compute_score(
|
|
40
|
+
[[query, p] for p in passages], normalize=True, batch_size=16
|
|
41
|
+
)
|
|
42
|
+
if isinstance(scores, int | float):
|
|
43
|
+
return [float(scores)]
|
|
44
|
+
return [float(s) for s in scores]
|
loom_notes/evals.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Évaluation du retrieval sur un jeu doré : requête → document attendu.
|
|
2
|
+
|
|
3
|
+
Le jeu est un JSONL (`data/golden.jsonl`), une ligne par cas. Les métriques sont celles qu'on
|
|
4
|
+
compare d'une version à l'autre : recall@1, recall@5, MRR. L'analyse de seuil sert à régler
|
|
5
|
+
`min_score` sans perdre de rappel.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, Field
|
|
12
|
+
|
|
13
|
+
from loom_notes.models import Hit
|
|
14
|
+
from loom_notes.service import MemoryService
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class GoldenCase(BaseModel):
|
|
18
|
+
query: str
|
|
19
|
+
doc_id: str
|
|
20
|
+
title: str | None = None # informatif, pour relire le fichier
|
|
21
|
+
project: str | None = None # filtre appliqué à la recherche, comme le ferait Claude
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class CaseResult(BaseModel):
|
|
25
|
+
query: str
|
|
26
|
+
doc_id: str
|
|
27
|
+
rank: int | None = Field(description="Rang du document attendu (1 = premier), None si absent")
|
|
28
|
+
expected_score: float | None
|
|
29
|
+
top_doc_id: str | None
|
|
30
|
+
top_score: float | None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class EvalReport(BaseModel):
|
|
34
|
+
cases: int
|
|
35
|
+
recall_at_1: float
|
|
36
|
+
recall_at_5: float
|
|
37
|
+
mrr: float
|
|
38
|
+
suggested_min_score: float | None = Field(
|
|
39
|
+
description="Plus haut seuil qui ne fait perdre aucun cas trouvé (score min des bons)"
|
|
40
|
+
)
|
|
41
|
+
noise_removed_at_suggested: float | None = Field(
|
|
42
|
+
description="Part des résultats hors document attendu qui tomberaient sous ce seuil"
|
|
43
|
+
)
|
|
44
|
+
results: list[CaseResult]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def load_golden(path: Path) -> list[GoldenCase]:
|
|
48
|
+
if not path.is_file():
|
|
49
|
+
return []
|
|
50
|
+
cases: list[GoldenCase] = []
|
|
51
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
52
|
+
if line.strip():
|
|
53
|
+
cases.append(GoldenCase.model_validate(json.loads(line)))
|
|
54
|
+
return cases
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def append_golden(path: Path, case: GoldenCase) -> None:
|
|
58
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
59
|
+
with path.open("a", encoding="utf-8") as f:
|
|
60
|
+
f.write(case.model_dump_json(exclude_none=True) + "\n")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
async def run_eval(service: MemoryService, cases: list[GoldenCase], k: int = 10) -> EvalReport:
|
|
64
|
+
results: list[CaseResult] = []
|
|
65
|
+
positives: list[float] = []
|
|
66
|
+
negatives: list[float] = []
|
|
67
|
+
for case in cases:
|
|
68
|
+
hits = await service.search(case.query, project=case.project, k=k, min_score=0.0)
|
|
69
|
+
docs = _doc_order(hits)
|
|
70
|
+
rank = docs.index(case.doc_id) + 1 if case.doc_id in docs else None
|
|
71
|
+
expected = next((h.score for h in hits if h.doc_id == case.doc_id), None)
|
|
72
|
+
if expected is not None:
|
|
73
|
+
positives.append(expected)
|
|
74
|
+
negatives.extend(h.score for h in hits if h.doc_id != case.doc_id)
|
|
75
|
+
results.append(
|
|
76
|
+
CaseResult(
|
|
77
|
+
query=case.query,
|
|
78
|
+
doc_id=case.doc_id,
|
|
79
|
+
rank=rank,
|
|
80
|
+
expected_score=expected,
|
|
81
|
+
top_doc_id=hits[0].doc_id if hits else None,
|
|
82
|
+
top_score=hits[0].score if hits else None,
|
|
83
|
+
)
|
|
84
|
+
)
|
|
85
|
+
n = len(results)
|
|
86
|
+
found = [r.rank for r in results if r.rank is not None]
|
|
87
|
+
suggested = round(min(positives), 3) if positives else None
|
|
88
|
+
noise = (
|
|
89
|
+
sum(1 for s in negatives if s < suggested) / len(negatives)
|
|
90
|
+
if suggested is not None and negatives
|
|
91
|
+
else None
|
|
92
|
+
)
|
|
93
|
+
return EvalReport(
|
|
94
|
+
cases=n,
|
|
95
|
+
recall_at_1=(sum(1 for r in found if r == 1) / n) if n else 0.0,
|
|
96
|
+
recall_at_5=(sum(1 for r in found if r <= 5) / n) if n else 0.0,
|
|
97
|
+
mrr=(sum(1 / r for r in found) / n) if n else 0.0,
|
|
98
|
+
suggested_min_score=suggested,
|
|
99
|
+
noise_removed_at_suggested=round(noise, 3) if noise is not None else None,
|
|
100
|
+
results=results,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _doc_order(hits: list[Hit]) -> list[str]:
|
|
105
|
+
seen: dict[str, None] = {}
|
|
106
|
+
for h in hits:
|
|
107
|
+
seen.setdefault(h.doc_id, None)
|
|
108
|
+
return list(seen)
|
loom_notes/ids.py
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
"""Identifiants : doc_id aléatoire, point_id déterministe (réécriture sans orphelins)."""
|
|
2
|
+
|
|
3
|
+
import uuid
|
|
4
|
+
|
|
5
|
+
_NAMESPACE = uuid.UUID("6f1c3a2e-9b7d-4e5a-8c0f-2d1e4b6a7c93")
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def new_doc_id() -> str:
|
|
9
|
+
return str(uuid.uuid4())
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def point_id(doc_id: str, chunk_index: int) -> str:
|
|
13
|
+
return str(uuid.uuid5(_NAMESPACE, f"{doc_id}:{chunk_index}"))
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from loom_notes.ingest.chunk import chunk_document
|
|
2
|
+
from loom_notes.ingest.dedup import content_hash
|
|
3
|
+
from loom_notes.ingest.extract import Extracted, extract_from_html, fetch_url, read_markdown_file
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"Extracted",
|
|
7
|
+
"chunk_document",
|
|
8
|
+
"content_hash",
|
|
9
|
+
"extract_from_html",
|
|
10
|
+
"fetch_url",
|
|
11
|
+
"read_markdown_file",
|
|
12
|
+
]
|