opensolr-haystack 0.2.1__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {opensolr_haystack-0.2.1/opensolr_haystack.egg-info → opensolr_haystack-0.2.3}/PKG-INFO +17 -1
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/README.md +16 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/client.py +137 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/store.py +26 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3/opensolr_haystack.egg-info}/PKG-INFO +17 -1
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/pyproject.toml +1 -1
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/LICENSE +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/components/retrievers/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/components/retrievers/opensolr/retriever.py +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/SOURCES.txt +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/dependency_links.txt +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/requires.txt +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/top_level.txt +0 -0
- {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -121,6 +121,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
|
|
|
121
121
|
WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
|
|
122
122
|
site for you.
|
|
123
123
|
|
|
124
|
+
## Grounded RAG answers
|
|
125
|
+
|
|
126
|
+
One call: hybrid retrieval picks the top hits, whose content becomes the LLM
|
|
127
|
+
context, and Opensolr's server-side LLM answers — no generator component,
|
|
128
|
+
no LLM key:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
answer = store.ai_answer(
|
|
132
|
+
"what does the refund policy say?",
|
|
133
|
+
rag_docs=3, # how many hybrid hits feed the LLM (default 3)
|
|
134
|
+
rag_words=1500, # words of text taken from each hit (default 1500)
|
|
135
|
+
# instruction="Answer in German, cite the exact titles you used", # optional
|
|
136
|
+
)
|
|
137
|
+
```
|
|
138
|
+
|
|
124
139
|
## How it's tested
|
|
125
140
|
|
|
126
141
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -139,6 +154,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
|
|
|
139
154
|
- **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
|
|
140
155
|
extraction (13k+ chars), automatic content-type detection, then retrieved
|
|
141
156
|
with a purely semantic query against its contents.
|
|
157
|
+
- **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
|
|
142
158
|
|
|
143
159
|
The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
|
|
144
160
|
hybrid + lexical retrieval, filters, serde round-trip) before every release.
|
|
@@ -103,6 +103,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
|
|
|
103
103
|
WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
|
|
104
104
|
site for you.
|
|
105
105
|
|
|
106
|
+
## Grounded RAG answers
|
|
107
|
+
|
|
108
|
+
One call: hybrid retrieval picks the top hits, whose content becomes the LLM
|
|
109
|
+
context, and Opensolr's server-side LLM answers — no generator component,
|
|
110
|
+
no LLM key:
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
answer = store.ai_answer(
|
|
114
|
+
"what does the refund policy say?",
|
|
115
|
+
rag_docs=3, # how many hybrid hits feed the LLM (default 3)
|
|
116
|
+
rag_words=1500, # words of text taken from each hit (default 1500)
|
|
117
|
+
# instruction="Answer in German, cite the exact titles you used", # optional
|
|
118
|
+
)
|
|
119
|
+
```
|
|
120
|
+
|
|
106
121
|
## How it's tested
|
|
107
122
|
|
|
108
123
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -121,6 +136,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
|
|
|
121
136
|
- **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
|
|
122
137
|
extraction (13k+ chars), automatic content-type detection, then retrieved
|
|
123
138
|
with a purely semantic query against its contents.
|
|
139
|
+
- **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
|
|
124
140
|
|
|
125
141
|
The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
|
|
126
142
|
hybrid + lexical retrieval, filters, serde round-trip) before every release.
|
|
@@ -264,6 +264,143 @@ class OpensolrClient:
|
|
|
264
264
|
resp.raise_for_status()
|
|
265
265
|
return resp.json()
|
|
266
266
|
|
|
267
|
+
def hybrid_search(
|
|
268
|
+
self,
|
|
269
|
+
index: str,
|
|
270
|
+
query: str,
|
|
271
|
+
rows: int = 5,
|
|
272
|
+
mode: str = "union",
|
|
273
|
+
alpha: float = 0.5,
|
|
274
|
+
fl: str = "*,score",
|
|
275
|
+
fq: Optional[str] = None,
|
|
276
|
+
) -> Dict[str, Any]:
|
|
277
|
+
"""Hybrid (BM25 + kNN) search via the native ``{!hybrid}`` parser.
|
|
278
|
+
|
|
279
|
+
The query is embedded server-side; lexical and vector scores are
|
|
280
|
+
fused per document on the Solr side.
|
|
281
|
+
"""
|
|
282
|
+
clean = query.replace("{", " ").replace("}", " ").replace('"', " ")
|
|
283
|
+
vector = self.embed(index, query, is_query=True)
|
|
284
|
+
compact = json.dumps(vector, separators=(",", ":"))
|
|
285
|
+
params: Dict[str, Any] = {
|
|
286
|
+
"q": (
|
|
287
|
+
f"{{!hybrid lexical=$lexicalRaw vector=$vectorQuery "
|
|
288
|
+
f"mode={mode} alpha={alpha} topN={max(rows, 10)}}}"
|
|
289
|
+
),
|
|
290
|
+
"lexicalRaw": f'{{!edismax qf="title^100 text^1"}}{clean}',
|
|
291
|
+
"vectorQuery": f"{{!knn f=embeddings topK={max(rows, 10)}}}{compact}",
|
|
292
|
+
"rows": rows,
|
|
293
|
+
"fl": fl,
|
|
294
|
+
}
|
|
295
|
+
if fq:
|
|
296
|
+
params["fq"] = fq
|
|
297
|
+
return self.solr_select(index, params)
|
|
298
|
+
|
|
299
|
+
#: RAG context defaults — how many hybrid hits feed the LLM, and how many
|
|
300
|
+
#: words of each hit's text are included. Both overridable per call.
|
|
301
|
+
RAG_DOCS = 3
|
|
302
|
+
RAG_WORDS = 1500
|
|
303
|
+
|
|
304
|
+
def _rag_context(
|
|
305
|
+
self,
|
|
306
|
+
index: str,
|
|
307
|
+
query: str,
|
|
308
|
+
fq: Optional[str] = None,
|
|
309
|
+
docs: Optional[int] = None,
|
|
310
|
+
words: Optional[int] = None,
|
|
311
|
+
) -> str:
|
|
312
|
+
"""Build the LLM context from the top hybrid search hits.
|
|
313
|
+
|
|
314
|
+
Retrieval runs through the server-side ``embed_and_search`` pipeline —
|
|
315
|
+
the platform's own tuned hybrid ranking (field weights, minimum-match,
|
|
316
|
+
quality boosts), the same machinery behind the hosted search UI, so it
|
|
317
|
+
improves automatically with the platform. When a custom ``fq`` is
|
|
318
|
+
given (which that endpoint doesn't accept) — or if it fails —
|
|
319
|
+
retrieval falls back to the client-side ``{!hybrid}`` query.
|
|
320
|
+
"""
|
|
321
|
+
|
|
322
|
+
def _flat(v: Any) -> str:
|
|
323
|
+
if isinstance(v, list):
|
|
324
|
+
v = " ".join(str(x) for x in v)
|
|
325
|
+
return str(v or "")
|
|
326
|
+
|
|
327
|
+
docs = docs or self.RAG_DOCS
|
|
328
|
+
words = words or self.RAG_WORDS
|
|
329
|
+
hits: List[Dict[str, Any]] = []
|
|
330
|
+
if not fq:
|
|
331
|
+
try:
|
|
332
|
+
body = self.embed_and_search(index, query, rows=docs)
|
|
333
|
+
if isinstance(body, dict):
|
|
334
|
+
hits = body.get("results", {}).get("docs", []) or []
|
|
335
|
+
except (OpensolrError, httpx.HTTPError):
|
|
336
|
+
hits = []
|
|
337
|
+
if not hits:
|
|
338
|
+
body = self.hybrid_search(
|
|
339
|
+
index, query, rows=docs, fl="title,description,text", fq=fq
|
|
340
|
+
)
|
|
341
|
+
hits = body.get("response", {}).get("docs", [])
|
|
342
|
+
parts: List[str] = []
|
|
343
|
+
for doc in hits[:docs]:
|
|
344
|
+
text_words = _flat(doc.get("text")).split()[:words]
|
|
345
|
+
parts.append(
|
|
346
|
+
_flat(doc.get("title")) + " - "
|
|
347
|
+
+ _flat(doc.get("description")) + " - "
|
|
348
|
+
+ " ".join(text_words) + " - "
|
|
349
|
+
)
|
|
350
|
+
return "".join(parts)
|
|
351
|
+
|
|
352
|
+
def ai_summary(
|
|
353
|
+
self,
|
|
354
|
+
index: str,
|
|
355
|
+
query: str,
|
|
356
|
+
filter_query: Optional[str] = None,
|
|
357
|
+
rag_docs: Optional[int] = None,
|
|
358
|
+
rag_words: Optional[int] = None,
|
|
359
|
+
instruction: Optional[str] = None,
|
|
360
|
+
**params: Any,
|
|
361
|
+
) -> str:
|
|
362
|
+
"""Grounded RAG answer: hybrid retrieval over the index feeds the LLM.
|
|
363
|
+
|
|
364
|
+
Retrieval runs client-side via ``hybrid_search`` (same pipeline as the
|
|
365
|
+
hosted search UI): the top ``rag_docs`` hits' title/description/text
|
|
366
|
+
(first ``rag_words`` words each) become the LLM context. Pass
|
|
367
|
+
``instruction`` to fully control the prompt (e.g. "Answer in German",
|
|
368
|
+
"Extract a list of people"). If retrieval fails or returns nothing,
|
|
369
|
+
the server falls back to its own retrieval. Returns plain text.
|
|
370
|
+
"""
|
|
371
|
+
data = {
|
|
372
|
+
**self._auth_params(),
|
|
373
|
+
"index_name": index,
|
|
374
|
+
"query": query,
|
|
375
|
+
"stream": "false",
|
|
376
|
+
**params,
|
|
377
|
+
}
|
|
378
|
+
if instruction:
|
|
379
|
+
data["instruction"] = instruction
|
|
380
|
+
if "context" not in data:
|
|
381
|
+
try:
|
|
382
|
+
context = self._rag_context(
|
|
383
|
+
index, query, fq=filter_query, docs=rag_docs, words=rag_words
|
|
384
|
+
)
|
|
385
|
+
except (OpensolrError, httpx.HTTPError):
|
|
386
|
+
context = ""
|
|
387
|
+
if context:
|
|
388
|
+
data["context"] = context
|
|
389
|
+
data.setdefault(
|
|
390
|
+
"instruction",
|
|
391
|
+
"Read and understand the full context below, and formulate "
|
|
392
|
+
f"a clear, concise and factual answer to: '{query}'.\n"
|
|
393
|
+
"Answer ONLY from the context. Format the answer in "
|
|
394
|
+
"Markdown, use bold section headers where they help, and "
|
|
395
|
+
"cite exact titles or names from the context when "
|
|
396
|
+
"referring to them.\n",
|
|
397
|
+
)
|
|
398
|
+
resp = self._http.post(f"{AI_BASE}/ai_summary", data=data)
|
|
399
|
+
if resp.status_code >= 400:
|
|
400
|
+
raise OpensolrError(f"ai_summary: HTTP {resp.status_code}: {resp.text[:200]}")
|
|
401
|
+
# The stream is prefixed with flush-padding whitespace — strip it.
|
|
402
|
+
return resp.text.strip()
|
|
403
|
+
|
|
267
404
|
def solr_update(self, index: str, payload: Any, commit: bool = True) -> Dict[str, Any]:
|
|
268
405
|
base, auth = self.solr_endpoint(index)
|
|
269
406
|
params = {"commit": "true"} if commit else {"commitWithin": "10000"}
|
|
@@ -248,6 +248,32 @@ class OpensolrDocumentStore:
|
|
|
248
248
|
self.client.ingest(self.index, docs[i : i + 50], wait=self.ingest_wait)
|
|
249
249
|
return len(docs)
|
|
250
250
|
|
|
251
|
+
def ai_answer(
|
|
252
|
+
self,
|
|
253
|
+
query: str,
|
|
254
|
+
filters: Optional[Dict[str, Any]] = None,
|
|
255
|
+
rag_docs: int = 3,
|
|
256
|
+
rag_words: int = 1500,
|
|
257
|
+
instruction: Optional[str] = None,
|
|
258
|
+
**kwargs: Any,
|
|
259
|
+
) -> str:
|
|
260
|
+
"""Grounded RAG answer generated only from this index's content.
|
|
261
|
+
|
|
262
|
+
Two-step pattern: hybrid (BM25 + kNN) retrieval picks the top
|
|
263
|
+
``rag_docs`` hits (first ``rag_words`` words of text each), whose
|
|
264
|
+
title/description/text become the LLM context — the same pipeline as
|
|
265
|
+
Opensolr's hosted search UI. Pass ``instruction`` to fully control
|
|
266
|
+
the prompt (e.g. "Answer in German, cite the sources you used").
|
|
267
|
+
Returns plain text.
|
|
268
|
+
"""
|
|
269
|
+
fqs = _filters_to_fq(filters)
|
|
270
|
+
fq = " AND ".join(f"({f})" for f in fqs) if fqs else None
|
|
271
|
+
return self.client.ai_summary(
|
|
272
|
+
self.index, query, filter_query=fq,
|
|
273
|
+
rag_docs=rag_docs, rag_words=rag_words, instruction=instruction,
|
|
274
|
+
**kwargs,
|
|
275
|
+
)
|
|
276
|
+
|
|
251
277
|
def delete_documents(self, document_ids: List[str]) -> None:
|
|
252
278
|
if not document_ids:
|
|
253
279
|
return
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -121,6 +121,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
|
|
|
121
121
|
WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
|
|
122
122
|
site for you.
|
|
123
123
|
|
|
124
|
+
## Grounded RAG answers
|
|
125
|
+
|
|
126
|
+
One call: hybrid retrieval picks the top hits, whose content becomes the LLM
|
|
127
|
+
context, and Opensolr's server-side LLM answers — no generator component,
|
|
128
|
+
no LLM key:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
answer = store.ai_answer(
|
|
132
|
+
"what does the refund policy say?",
|
|
133
|
+
rag_docs=3, # how many hybrid hits feed the LLM (default 3)
|
|
134
|
+
rag_words=1500, # words of text taken from each hit (default 1500)
|
|
135
|
+
# instruction="Answer in German, cite the exact titles you used", # optional
|
|
136
|
+
)
|
|
137
|
+
```
|
|
138
|
+
|
|
124
139
|
## How it's tested
|
|
125
140
|
|
|
126
141
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -139,6 +154,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
|
|
|
139
154
|
- **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
|
|
140
155
|
extraction (13k+ chars), automatic content-type detection, then retrieved
|
|
141
156
|
with a purely semantic query against its contents.
|
|
157
|
+
- **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
|
|
142
158
|
|
|
143
159
|
The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
|
|
144
160
|
hybrid + lexical retrieval, filters, serde round-trip) before every release.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "opensolr-haystack"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.3"
|
|
8
8
|
description = "Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|