opensolr-haystack 0.2.4__tar.gz → 0.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/PKG-INFO +7 -3
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/README.md +6 -2
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/haystack_integrations/document_stores/opensolr/client.py +34 -2
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/PKG-INFO +7 -3
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/pyproject.toml +1 -1
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/LICENSE +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/haystack_integrations/components/retrievers/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/haystack_integrations/components/retrievers/opensolr/retriever.py +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/haystack_integrations/document_stores/opensolr/__init__.py +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/haystack_integrations/document_stores/opensolr/store.py +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/SOURCES.txt +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/dependency_links.txt +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/requires.txt +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/top_level.txt +0 -0
- {opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.6
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -22,6 +22,8 @@ Dynamic: license-file
|
|
|
22
22
|
[Opensolr](https://opensolr.com) — managed Apache Solr as a DocumentStore,
|
|
23
23
|
with **server-side embeddings** and native **hybrid (BM25 + kNN) retrieval**.
|
|
24
24
|
|
|
25
|
+
**See it live (real news index, hybrid + AI answer):** https://search.opensolr.com/news__dense?q=how+am+I+supposed+to+save+money%3F
|
|
26
|
+
|
|
25
27
|
No embedder components needed in your pipeline — texts and queries are
|
|
26
28
|
embedded on Opensolr's GPU infrastructure (multilingual E5-large-instruct,
|
|
27
29
|
1024 dimensions, cosine).
|
|
@@ -98,8 +100,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
98
100
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
99
101
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
100
102
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
101
|
-
documents become searchable within about a minute. Progress is visible in
|
|
102
|
-
|
|
103
|
+
documents become searchable within about a minute. Progress is visible in
|
|
104
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
105
|
+
processing / completed / failed, with processed / success / failed document
|
|
106
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
103
107
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
104
108
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
105
109
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -4,6 +4,8 @@
|
|
|
4
4
|
[Opensolr](https://opensolr.com) — managed Apache Solr as a DocumentStore,
|
|
5
5
|
with **server-side embeddings** and native **hybrid (BM25 + kNN) retrieval**.
|
|
6
6
|
|
|
7
|
+
**See it live (real news index, hybrid + AI answer):** https://search.opensolr.com/news__dense?q=how+am+I+supposed+to+save+money%3F
|
|
8
|
+
|
|
7
9
|
No embedder components needed in your pipeline — texts and queries are
|
|
8
10
|
embedded on Opensolr's GPU infrastructure (multilingual E5-large-instruct,
|
|
9
11
|
1024 dimensions, cosine).
|
|
@@ -80,8 +82,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
80
82
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
81
83
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
82
84
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
83
|
-
documents become searchable within about a minute. Progress is visible in
|
|
84
|
-
|
|
85
|
+
documents become searchable within about a minute. Progress is visible in
|
|
86
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
87
|
+
processing / completed / failed, with processed / success / failed document
|
|
88
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
85
89
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
86
90
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
87
91
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -42,6 +42,29 @@ def resolve_location(location: str) -> str:
|
|
|
42
42
|
BATCH_EMBED_MAX = 50
|
|
43
43
|
|
|
44
44
|
|
|
45
|
+
# Default RAG instruction (added 2026-08-25).
|
|
46
|
+
# Without an instruction the server falls back to "Answer the query based on
|
|
47
|
+
# the context", which makes the model hedge: it opens with a disclaimer that
|
|
48
|
+
# the context does not cover the question and then answers it anyway. These
|
|
49
|
+
# rules make it lead with the answer and keep the concrete details. Callers
|
|
50
|
+
# can still override everything by passing ``instruction``.
|
|
51
|
+
DEFAULT_RAG_INSTRUCTION = (
|
|
52
|
+
"Answer the query using only the context below. "
|
|
53
|
+
"Do not repeat or restate the question, and do not print it as a heading. "
|
|
54
|
+
"If the question is a yes or no question and the context supports it, begin with \"Yes\". "
|
|
55
|
+
"For any other question, begin with the fact itself, never with \"Yes\". "
|
|
56
|
+
"Never begin with \"No\" when the context does support the answer. "
|
|
57
|
+
"Start with the answer itself: do not open with a preamble about what the context does "
|
|
58
|
+
"or does not address, and never say the context does not cover the query and then answer "
|
|
59
|
+
"it anyway. "
|
|
60
|
+
"Give a substantive answer with the concrete details from the context: who, what, where "
|
|
61
|
+
"and when. "
|
|
62
|
+
"Do not dismiss the query on a technicality. If the context covers something closely "
|
|
63
|
+
"related rather than the exact wording used, explain what it does say and how it relates. "
|
|
64
|
+
"Only if nothing in the context is relevant at all, say so in one sentence and name what "
|
|
65
|
+
"the context is about instead."
|
|
66
|
+
)
|
|
67
|
+
|
|
45
68
|
class OpensolrError(RuntimeError):
|
|
46
69
|
"""Raised when an Opensolr API call fails."""
|
|
47
70
|
|
|
@@ -353,6 +376,16 @@ class OpensolrClient:
|
|
|
353
376
|
index, query, rows=docs, fl="title,description,text", fq=fq
|
|
354
377
|
)
|
|
355
378
|
hits = body.get("response", {}).get("docs", [])
|
|
379
|
+
# Relevance floor (2026-08-25). Retrieval always returns `docs` hits, so a
|
|
380
|
+
# narrow question arrives with one good match and several unrelated ones.
|
|
381
|
+
# The model then sees that most of its context does not answer the query
|
|
382
|
+
# and hedges. Drop anything scoring below half of the best hit; documents
|
|
383
|
+
# without a score are kept, since only provable weakness is filtered.
|
|
384
|
+
scored = [float(h["score"]) for h in hits if isinstance(h, dict) and h.get("score") is not None]
|
|
385
|
+
if scored:
|
|
386
|
+
floor = max(scored) * 0.5
|
|
387
|
+
hits = [h for h in hits if h.get("score") is None or float(h["score"]) >= floor]
|
|
388
|
+
|
|
356
389
|
parts: List[str] = []
|
|
357
390
|
for doc in hits[:docs]:
|
|
358
391
|
text_words = _flat(doc.get("text")).split()[:words]
|
|
@@ -390,8 +423,7 @@ class OpensolrClient:
|
|
|
390
423
|
"stream": "false",
|
|
391
424
|
**params,
|
|
392
425
|
}
|
|
393
|
-
|
|
394
|
-
data["instruction"] = instruction
|
|
426
|
+
data["instruction"] = instruction or DEFAULT_RAG_INSTRUCTION
|
|
395
427
|
if "context" not in data:
|
|
396
428
|
try:
|
|
397
429
|
context = self._rag_context(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: opensolr-haystack
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.6
|
|
4
4
|
Summary: Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval
|
|
5
5
|
Author-email: Opensolr <support@opensolr.com>
|
|
6
6
|
License: MIT
|
|
@@ -22,6 +22,8 @@ Dynamic: license-file
|
|
|
22
22
|
[Opensolr](https://opensolr.com) — managed Apache Solr as a DocumentStore,
|
|
23
23
|
with **server-side embeddings** and native **hybrid (BM25 + kNN) retrieval**.
|
|
24
24
|
|
|
25
|
+
**See it live (real news index, hybrid + AI answer):** https://search.opensolr.com/news__dense?q=how+am+I+supposed+to+save+money%3F
|
|
26
|
+
|
|
25
27
|
No embedder components needed in your pipeline — texts and queries are
|
|
26
28
|
embedded on Opensolr's GPU infrastructure (multilingual E5-large-instruct,
|
|
27
29
|
1024 dimensions, cosine).
|
|
@@ -98,8 +100,10 @@ Writes go through Opensolr's [Data Ingestion API](https://opensolr.com/learn/api
|
|
|
98
100
|
— the same pipeline the Drupal and WordPress connectors use. It is
|
|
99
101
|
**asynchronous**: documents are queued, then embeddings, sentiment, language
|
|
100
102
|
and all crawler-identical derived fields are computed **server-side**, and
|
|
101
|
-
documents become searchable within about a minute. Progress is visible in
|
|
102
|
-
|
|
103
|
+
documents become searchable within about a minute. Progress is visible in
|
|
104
|
+
**Control Panel → Data Ingestion** — a per-job status board (queued /
|
|
105
|
+
processing / completed / failed, with processed / success / failed document
|
|
106
|
+
counts per job) — and via the `ingest_status` API. Each document's
|
|
103
107
|
identity is its `uri` (the Solr id is `md5(uri)`): pass a real URL in
|
|
104
108
|
metadata (`{"uri": "https://..."}`), or a deterministic one is synthesized
|
|
105
109
|
from your id. Re-submitting the same `uri` updates the document. Pass
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "opensolr-haystack"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.6"
|
|
8
8
|
description = "Haystack integration for Opensolr — managed Apache Solr DocumentStore with server-side embeddings and hybrid BM25+kNN retrieval"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
{opensolr_haystack-0.2.4 → opensolr_haystack-0.2.6}/opensolr_haystack.egg-info/top_level.txt
RENAMED
|
File without changes
|
|
File without changes
|