opensolr-haystack 0.2.3__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/PKG-INFO +22 -3
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/README.md +21 -2
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/haystack_integrations/document_stores/opensolr/client.py +19 -3
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/haystack_integrations/document_stores/opensolr/store.py +6 -1
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/PKG-INFO +22 -3
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/pyproject.toml +1 -1
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/LICENSE +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/haystack_integrations/components/retrievers/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/haystack_integrations/components/retrievers/opensolr/retriever.py +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/haystack_integrations/document_stores/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/SOURCES.txt +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/dependency_links.txt +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/requires.txt +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/top_level.txt +0 -0
- {opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -98,8 +98,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
98
98
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
99
99
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
100
100
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
101
|
-
documents become searchable within about a minute. Progress is visible in
|
|
102
|
-
|
|
101
|
+
documents become searchable within about a minute. Progress is visible in
|
|
102
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
103
|
+
processing / completed / failed, with processed / success / failed document
|
|
104
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
103
105
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
104
106
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
105
107
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -136,6 +138,23 @@ answer = store.ai_answer(
|
|
|
136
138
|
)
|
|
137
139
|
```
|
|
138
140
|
|
|
141
|
+
|
|
142
|
+
### Search tuning
|
|
143
|
+
|
|
144
|
+
Retrieval (search and RAG grounding) runs through the platform's tuned
|
|
145
|
+
pipeline: global defaults → your index's saved **Search Tuning** (Control
|
|
146
|
+
Panel → Index Settings → Search Tuning: semantic↔lexical balance, field
|
|
147
|
+
weights, minimum match, search mode, vector candidate pool, content quality
|
|
148
|
+
boost) → optional per-call overrides via `tuning`:
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
tuning={"search_mode": "keywords_required", "fw_title": 0.2,
|
|
152
|
+
"mm": "strict", "vector_topk": 500, "quality_boost": 0.3}
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Defaults match the platform's PHP configuration exactly — customize in the
|
|
156
|
+
Control Panel once, or per call from code.
|
|
157
|
+
|
|
139
158
|
## How it's tested
|
|
140
159
|
|
|
141
160
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -80,8 +80,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
80
80
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
81
81
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
82
82
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
83
|
-
documents become searchable within about a minute. Progress is visible in
|
|
84
|
-
|
|
83
|
+
documents become searchable within about a minute. Progress is visible in
|
|
84
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
85
|
+
processing / completed / failed, with processed / success / failed document
|
|
86
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
85
87
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
86
88
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
87
89
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -118,6 +120,23 @@ answer = store.ai_answer(
|
|
|
118
120
|
)
|
|
119
121
|
```
|
|
120
122
|
|
|
123
|
+
|
|
124
|
+
### Search tuning
|
|
125
|
+
|
|
126
|
+
Retrieval (search and RAG grounding) runs through the platform's tuned
|
|
127
|
+
pipeline: global defaults → your index's saved **Search Tuning** (Control
|
|
128
|
+
Panel → Index Settings → Search Tuning: semantic↔lexical balance, field
|
|
129
|
+
weights, minimum match, search mode, vector candidate pool, content quality
|
|
130
|
+
boost) → optional per-call overrides via `tuning`:
|
|
131
|
+
|
|
132
|
+
```
|
|
133
|
+
tuning={"search_mode": "keywords_required", "fw_title": 0.2,
|
|
134
|
+
"mm": "strict", "vector_topk": 500, "quality_boost": 0.3}
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Defaults match the platform's PHP configuration exactly — customize in the
|
|
138
|
+
Control Panel once, or per call from code.
|
|
139
|
+
|
|
121
140
|
## How it's tested
|
|
122
141
|
|
|
123
142
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -233,7 +233,20 @@ class OpensolrClient:
|
|
|
233
233
|
raise OpensolrError(f"ingest_status: non-JSON response: {resp.text[:200]}") from exc
|
|
234
234
|
|
|
235
235
|
def embed_and_search(self, index: str, query: str, rows: int = 10, **params: Any) -> Dict[str, Any]:
|
|
236
|
-
"""Server-side one-shot: embed the query, run
|
|
236
|
+
"""Server-side one-shot: embed the query, run the platform's tuned
|
|
237
|
+
hybrid search, return ranked docs.
|
|
238
|
+
|
|
239
|
+
Retrieval uses the same pipeline as the hosted search UI: global
|
|
240
|
+
defaults, overridden by the index's saved Search Tuning (Control
|
|
241
|
+
Panel → Index Settings → Search Tuning), overridden by any of these
|
|
242
|
+
per-call knobs passed as extra params: ``fw_title``,
|
|
243
|
+
``fw_description``, ``fw_uri``, ``fw_text``, ``fw_text_t``,
|
|
244
|
+
``lexical_weight``, ``vector_weight``, ``vector_topk``,
|
|
245
|
+
``search_mode`` (union / keywords_required / meaning_required /
|
|
246
|
+
intersection), ``quality_boost``, ``min_score``,
|
|
247
|
+
``freshness_boost``, ``lexical_norm_k``, ``mm`` (flexible /
|
|
248
|
+
balanced / strict or raw Solr mm syntax).
|
|
249
|
+
"""
|
|
237
250
|
body = self.ai(
|
|
238
251
|
"embed_and_search",
|
|
239
252
|
index_name=index,
|
|
@@ -308,6 +321,7 @@ class OpensolrClient:
|
|
|
308
321
|
fq: Optional[str] = None,
|
|
309
322
|
docs: Optional[int] = None,
|
|
310
323
|
words: Optional[int] = None,
|
|
324
|
+
tuning: Optional[Dict[str, Any]] = None,
|
|
311
325
|
) -> str:
|
|
312
326
|
"""Build the LLM context from the top hybrid search hits.
|
|
313
327
|
|
|
@@ -329,7 +343,7 @@ class OpensolrClient:
|
|
|
329
343
|
hits: List[Dict[str, Any]] = []
|
|
330
344
|
if not fq:
|
|
331
345
|
try:
|
|
332
|
-
body = self.embed_and_search(index, query, rows=docs)
|
|
346
|
+
body = self.embed_and_search(index, query, rows=docs, **(tuning or {}))
|
|
333
347
|
if isinstance(body, dict):
|
|
334
348
|
hits = body.get("results", {}).get("docs", []) or []
|
|
335
349
|
except (OpensolrError, httpx.HTTPError):
|
|
@@ -357,6 +371,7 @@ class OpensolrClient:
|
|
|
357
371
|
rag_docs: Optional[int] = None,
|
|
358
372
|
rag_words: Optional[int] = None,
|
|
359
373
|
instruction: Optional[str] = None,
|
|
374
|
+
tuning: Optional[Dict[str, Any]] = None,
|
|
360
375
|
**params: Any,
|
|
361
376
|
) -> str:
|
|
362
377
|
"""Grounded RAG answer: hybrid retrieval over the index feeds the LLM.
|
|
@@ -380,7 +395,8 @@ class OpensolrClient:
|
|
|
380
395
|
if "context" not in data:
|
|
381
396
|
try:
|
|
382
397
|
context = self._rag_context(
|
|
383
|
-
index, query, fq=filter_query, docs=rag_docs, words=rag_words
|
|
398
|
+
index, query, fq=filter_query, docs=rag_docs, words=rag_words,
|
|
399
|
+
tuning=tuning,
|
|
384
400
|
)
|
|
385
401
|
except (OpensolrError, httpx.HTTPError):
|
|
386
402
|
context = ""
|
|
@@ -255,6 +255,7 @@ class OpensolrDocumentStore:
|
|
|
255
255
|
rag_docs: int = 3,
|
|
256
256
|
rag_words: int = 1500,
|
|
257
257
|
instruction: Optional[str] = None,
|
|
258
|
+
tuning: Optional[Dict[str, Any]] = None,
|
|
258
259
|
**kwargs: Any,
|
|
259
260
|
) -> str:
|
|
260
261
|
"""Grounded RAG answer generated only from this index's content.
|
|
@@ -264,13 +265,17 @@ class OpensolrDocumentStore:
|
|
|
264
265
|
title/description/text become the LLM context — the same pipeline as
|
|
265
266
|
Opensolr's hosted search UI. Pass ``instruction`` to fully control
|
|
266
267
|
the prompt (e.g. "Answer in German, cite the sources you used").
|
|
267
|
-
|
|
268
|
+
Retrieval uses the platform's tuned pipeline: your index's saved
|
|
269
|
+
Search Tuning (Control Panel) applies automatically; ``tuning``
|
|
270
|
+
overrides any knob per call (fw_title, lexical_weight, search_mode,
|
|
271
|
+
mm, vector_topk, quality_boost, ...). Returns plain text.
|
|
268
272
|
"""
|
|
269
273
|
fqs = _filters_to_fq(filters)
|
|
270
274
|
fq = " AND ".join(f"({f})" for f in fqs) if fqs else None
|
|
271
275
|
return self.client.ai_summary(
|
|
272
276
|
self.index, query, filter_query=fq,
|
|
273
277
|
rag_docs=rag_docs, rag_words=rag_words, instruction=instruction,
|
|
278
|
+
tuning=tuning,
|
|
274
279
|
**kwargs,
|
|
275
280
|
)
|
|
276
281
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -98,8 +98,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
98
98
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
99
99
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
100
100
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
101
|
-
documents become searchable within about a minute. Progress is visible in
|
|
102
|
-
|
|
101
|
+
documents become searchable within about a minute. Progress is visible in
|
|
102
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
103
|
+
processing / completed / failed, with processed / success / failed document
|
|
104
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
103
105
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
104
106
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
105
107
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -136,6 +138,23 @@ answer = store.ai_answer(
|
|
|
136
138
|
)
|
|
137
139
|
```
|
|
138
140
|
|
|
141
|
+
|
|
142
|
+
### Search tuning
|
|
143
|
+
|
|
144
|
+
Retrieval (search and RAG grounding) runs through the platform's tuned
|
|
145
|
+
pipeline: global defaults → your index's saved **Search Tuning** (Control
|
|
146
|
+
Panel → Index Settings → Search Tuning: semantic↔lexical balance, field
|
|
147
|
+
weights, minimum match, search mode, vector candidate pool, content quality
|
|
148
|
+
boost) → optional per-call overrides via `tuning`:
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
tuning={"search_mode": "keywords_required", "fw_title": 0.2,
|
|
152
|
+
"mm": "strict", "vector_topk": 500, "quality_boost": 0.3}
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
Defaults match the platform's PHP configuration exactly — customize in the
|
|
156
|
+
Control Panel once, or per call from code.
|
|
157
|
+
|
|
139
158
|
## How it's tested
|
|
140
159
|
|
|
141
160
|
Every release is validated against **live Opensolr infrastructure** — no mocks:
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "opensolr-haystack"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.5"
|
|
8
8
|
description = "Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.3 → opensolr_haystack-0.2.5}/opensolr_haystack.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|