opensolr-haystack 0.2.1__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (15) hide show
  1. {opensolr_haystack-0.2.1/opensolr_haystack.egg-info → opensolr_haystack-0.2.3}/PKG-INFO +17 -1
  2. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/README.md +16 -0
  3. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/client.py +137 -0
  4. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/store.py +26 -0
  5. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3/opensolr_haystack.egg-info}/PKG-INFO +17 -1
  6. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/pyproject.toml +1 -1
  7. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/LICENSE +0 -0
  8. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/components/retrievers/opensolr/__init__.py +0 -0
  9. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/components/retrievers/opensolr/retriever.py +0 -0
  10. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/haystack_integrations/document_stores/opensolr/__init__.py +0 -0
  11. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/SOURCES.txt +0 -0
  12. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/dependency_links.txt +0 -0
  13. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/requires.txt +0 -0
  14. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/opensolr_haystack.egg-info/top_level.txt +0 -0
  15. {opensolr_haystack-0.2.1 → opensolr_haystack-0.2.3}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: opensolr-haystack
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
5
5
  Author-email: Opensolr <support@opensolr.com>
6
6
  License: MIT
@@ -121,6 +121,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
121
121
  WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
122
122
  site for you.
123
123
 
124
+ ## Grounded RAG answers
125
+
126
+ One call: hybrid retrieval picks the top hits, whose content becomes the LLM
127
+ context, and Opensolr's server-side LLM answers — no generator component,
128
+ no LLM key:
129
+
130
+ ```python
131
+ answer = store.ai_answer(
132
+ "what does the refund policy say?",
133
+ rag_docs=3, # how many hybrid hits feed the LLM (default 3)
134
+ rag_words=1500, # words of text taken from each hit (default 1500)
135
+ # instruction="Answer in German, cite the exact titles you used", # optional
136
+ )
137
+ ```
138
+
124
139
  ## How it's tested
125
140
 
126
141
  Every release is validated against **live Opensolr infrastructure** — no mocks:
@@ -139,6 +154,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
139
154
  - **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
140
155
  extraction (13k+ chars), automatic content-type detection, then retrieved
141
156
  with a purely semantic query against its contents.
157
+ - **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
142
158
 
143
159
  The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
144
160
  hybrid + lexical retrieval, filters, serde round-trip) before every release.
@@ -103,6 +103,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
103
103
  WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
104
104
  site for you.
105
105
 
106
+ ## Grounded RAG answers
107
+
108
+ One call: hybrid retrieval picks the top hits, whose content becomes the LLM
109
+ context, and Opensolr's server-side LLM answers — no generator component,
110
+ no LLM key:
111
+
112
+ ```python
113
+ answer = store.ai_answer(
114
+ "what does the refund policy say?",
115
+ rag_docs=3, # how many hybrid hits feed the LLM (default 3)
116
+ rag_words=1500, # words of text taken from each hit (default 1500)
117
+ # instruction="Answer in German, cite the exact titles you used", # optional
118
+ )
119
+ ```
120
+
106
121
  ## How it's tested
107
122
 
108
123
  Every release is validated against **live Opensolr infrastructure** — no mocks:
@@ -121,6 +136,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
121
136
  - **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
122
137
  extraction (13k+ chars), automatic content-type detection, then retrieved
123
138
  with a purely semantic query against its contents.
139
+ - **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
124
140
 
125
141
  The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
126
142
  hybrid + lexical retrieval, filters, serde round-trip) before every release.
@@ -264,6 +264,143 @@ class OpensolrClient:
264
264
  resp.raise_for_status()
265
265
  return resp.json()
266
266
 
267
+ def hybrid_search(
268
+ self,
269
+ index: str,
270
+ query: str,
271
+ rows: int = 5,
272
+ mode: str = "union",
273
+ alpha: float = 0.5,
274
+ fl: str = "*,score",
275
+ fq: Optional[str] = None,
276
+ ) -> Dict[str, Any]:
277
+ """Hybrid (BM25 + kNN) search via the native ``{!hybrid}`` parser.
278
+
279
+ The query is embedded server-side; lexical and vector scores are
280
+ fused per document on the Solr side.
281
+ """
282
+ clean = query.replace("{", " ").replace("}", " ").replace('"', " ")
283
+ vector = self.embed(index, query, is_query=True)
284
+ compact = json.dumps(vector, separators=(",", ":"))
285
+ params: Dict[str, Any] = {
286
+ "q": (
287
+ f"{{!hybrid lexical=$lexicalRaw vector=$vectorQuery "
288
+ f"mode={mode} alpha={alpha} topN={max(rows, 10)}}}"
289
+ ),
290
+ "lexicalRaw": f'{{!edismax qf="title^100 text^1"}}{clean}',
291
+ "vectorQuery": f"{{!knn f=embeddings topK={max(rows, 10)}}}{compact}",
292
+ "rows": rows,
293
+ "fl": fl,
294
+ }
295
+ if fq:
296
+ params["fq"] = fq
297
+ return self.solr_select(index, params)
298
+
299
+ #: RAG context defaults — how many hybrid hits feed the LLM, and how many
300
+ #: words of each hit's text are included. Both overridable per call.
301
+ RAG_DOCS = 3
302
+ RAG_WORDS = 1500
303
+
304
+ def _rag_context(
305
+ self,
306
+ index: str,
307
+ query: str,
308
+ fq: Optional[str] = None,
309
+ docs: Optional[int] = None,
310
+ words: Optional[int] = None,
311
+ ) -> str:
312
+ """Build the LLM context from the top hybrid search hits.
313
+
314
+ Retrieval runs through the server-side ``embed_and_search`` pipeline —
315
+ the platform's own tuned hybrid ranking (field weights, minimum-match,
316
+ quality boosts), the same machinery behind the hosted search UI, so it
317
+ improves automatically with the platform. When a custom ``fq`` is
318
+ given (which that endpoint doesn't accept) — or if it fails —
319
+ retrieval falls back to the client-side ``{!hybrid}`` query.
320
+ """
321
+
322
+ def _flat(v: Any) -> str:
323
+ if isinstance(v, list):
324
+ v = " ".join(str(x) for x in v)
325
+ return str(v or "")
326
+
327
+ docs = docs or self.RAG_DOCS
328
+ words = words or self.RAG_WORDS
329
+ hits: List[Dict[str, Any]] = []
330
+ if not fq:
331
+ try:
332
+ body = self.embed_and_search(index, query, rows=docs)
333
+ if isinstance(body, dict):
334
+ hits = body.get("results", {}).get("docs", []) or []
335
+ except (OpensolrError, httpx.HTTPError):
336
+ hits = []
337
+ if not hits:
338
+ body = self.hybrid_search(
339
+ index, query, rows=docs, fl="title,description,text", fq=fq
340
+ )
341
+ hits = body.get("response", {}).get("docs", [])
342
+ parts: List[str] = []
343
+ for doc in hits[:docs]:
344
+ text_words = _flat(doc.get("text")).split()[:words]
345
+ parts.append(
346
+ _flat(doc.get("title")) + " - "
347
+ + _flat(doc.get("description")) + " - "
348
+ + " ".join(text_words) + " - "
349
+ )
350
+ return "".join(parts)
351
+
352
+ def ai_summary(
353
+ self,
354
+ index: str,
355
+ query: str,
356
+ filter_query: Optional[str] = None,
357
+ rag_docs: Optional[int] = None,
358
+ rag_words: Optional[int] = None,
359
+ instruction: Optional[str] = None,
360
+ **params: Any,
361
+ ) -> str:
362
+ """Grounded RAG answer: hybrid retrieval over the index feeds the LLM.
363
+
364
+ Retrieval runs client-side via ``hybrid_search`` (same pipeline as the
365
+ hosted search UI): the top ``rag_docs`` hits' title/description/text
366
+ (first ``rag_words`` words each) become the LLM context. Pass
367
+ ``instruction`` to fully control the prompt (e.g. "Answer in German",
368
+ "Extract a list of people"). If retrieval fails or returns nothing,
369
+ the server falls back to its own retrieval. Returns plain text.
370
+ """
371
+ data = {
372
+ **self._auth_params(),
373
+ "index_name": index,
374
+ "query": query,
375
+ "stream": "false",
376
+ **params,
377
+ }
378
+ if instruction:
379
+ data["instruction"] = instruction
380
+ if "context" not in data:
381
+ try:
382
+ context = self._rag_context(
383
+ index, query, fq=filter_query, docs=rag_docs, words=rag_words
384
+ )
385
+ except (OpensolrError, httpx.HTTPError):
386
+ context = ""
387
+ if context:
388
+ data["context"] = context
389
+ data.setdefault(
390
+ "instruction",
391
+ "Read and understand the full context below, and formulate "
392
+ f"a clear, concise and factual answer to: '{query}'.\n"
393
+ "Answer ONLY from the context. Format the answer in "
394
+ "Markdown, use bold section headers where they help, and "
395
+ "cite exact titles or names from the context when "
396
+ "referring to them.\n",
397
+ )
398
+ resp = self._http.post(f"{AI_BASE}/ai_summary", data=data)
399
+ if resp.status_code >= 400:
400
+ raise OpensolrError(f"ai_summary: HTTP {resp.status_code}: {resp.text[:200]}")
401
+ # The stream is prefixed with flush-padding whitespace — strip it.
402
+ return resp.text.strip()
403
+
267
404
  def solr_update(self, index: str, payload: Any, commit: bool = True) -> Dict[str, Any]:
268
405
  base, auth = self.solr_endpoint(index)
269
406
  params = {"commit": "true"} if commit else {"commitWithin": "10000"}
@@ -248,6 +248,32 @@ class OpensolrDocumentStore:
248
248
  self.client.ingest(self.index, docs[i : i + 50], wait=self.ingest_wait)
249
249
  return len(docs)
250
250
 
251
+ def ai_answer(
252
+ self,
253
+ query: str,
254
+ filters: Optional[Dict[str, Any]] = None,
255
+ rag_docs: int = 3,
256
+ rag_words: int = 1500,
257
+ instruction: Optional[str] = None,
258
+ **kwargs: Any,
259
+ ) -> str:
260
+ """Grounded RAG answer generated only from this index's content.
261
+
262
+ Two-step pattern: hybrid (BM25 + kNN) retrieval picks the top
263
+ ``rag_docs`` hits (first ``rag_words`` words of text each), whose
264
+ title/description/text become the LLM context — the same pipeline as
265
+ Opensolr's hosted search UI. Pass ``instruction`` to fully control
266
+ the prompt (e.g. "Answer in German, cite the sources you used").
267
+ Returns plain text.
268
+ """
269
+ fqs = _filters_to_fq(filters)
270
+ fq = " AND ".join(f"({f})" for f in fqs) if fqs else None
271
+ return self.client.ai_summary(
272
+ self.index, query, filter_query=fq,
273
+ rag_docs=rag_docs, rag_words=rag_words, instruction=instruction,
274
+ **kwargs,
275
+ )
276
+
251
277
  def delete_documents(self, document_ids: List[str]) -> None:
252
278
  if not document_ids:
253
279
  return
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: opensolr-haystack
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
5
5
  Author-email: Opensolr <support@opensolr.com>
6
6
  License: MIT
@@ -121,6 +121,21 @@ entry? Configure the **Web Crawler** in the Control Panel (Index Tools →
121
121
  WebCrawler): add your site URL, validate it, and Opensolr indexes the whole
122
122
  site for you.
123
123
 
124
+ ## Grounded RAG answers
125
+
126
+ One call: hybrid retrieval picks the top hits, whose content becomes the LLM
127
+ context, and Opensolr's server-side LLM answers — no generator component,
128
+ no LLM key:
129
+
130
+ ```python
131
+ answer = store.ai_answer(
132
+ "what does the refund policy say?",
133
+ rag_docs=3, # how many hybrid hits feed the LLM (default 3)
134
+ rag_words=1500, # words of text taken from each hit (default 1500)
135
+ # instruction="Answer in German, cite the exact titles you used", # optional
136
+ )
137
+ ```
138
+
124
139
  ## How it's tested
125
140
 
126
141
  Every release is validated against **live Opensolr infrastructure** — no mocks:
@@ -139,6 +154,7 @@ Every release is validated against **live Opensolr infrastructure** — no mocks
139
154
  - **PDF ingestion**: a real PDF ingested via `rtf:true` — server-side text
140
155
  extraction (13k+ chars), automatic content-type detection, then retrieved
141
156
  with a purely semantic query against its contents.
157
+ - **Grounded RAG answers**: `ai_answer` verified end-to-end — a question answerable only from the ingested PDF returns the correct answer, sourced from the PDF's extracted text via hybrid retrieval.
142
158
 
143
159
  The store is exercised live (write via ingestion, DuplicatePolicy SKIP/FAIL,
144
160
  hybrid + lexical retrieval, filters, serde round-trip) before every release.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "opensolr-haystack"
7
- version = "0.2.1"
7
+ version = "0.2.3"
8
8
  description = "Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval"
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }