sie-haystack 0.7.1__tar.gz → 0.7.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/.gitignore +2 -36
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/PKG-INFO +2 -2
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/pyproject.toml +1 -1
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/sie_haystack/rankers.py +49 -8
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/conftest.py +7 -2
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/test_integration.py +1 -1
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/test_rankers.py +71 -5
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/README.md +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/haystack_integrations/components/embedders/sie/__init__.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/haystack_integrations/components/extractors/sie/__init__.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/haystack_integrations/components/rankers/sie/__init__.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/sie_haystack/__init__.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/sie_haystack/embedders.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/src/sie_haystack/extractors.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/__init__.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/test_embedders.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/test_extractors.py +0 -0
- {sie_haystack-0.7.1 → sie_haystack-0.7.3}/tests/test_namespace_aliases.py +0 -0
|
@@ -142,10 +142,8 @@ celerybeat.pid
|
|
|
142
142
|
|
|
143
143
|
# Environments
|
|
144
144
|
.env
|
|
145
|
-
#
|
|
146
|
-
|
|
147
|
-
# repo (audit §1/§14). Belt-and-braces so a local copy can never be committed.
|
|
148
|
-
*.sie-cloud-secrets.env
|
|
145
|
+
# Local service credential bundles must never be committed.
|
|
146
|
+
*-secrets.env
|
|
149
147
|
.env.cloud
|
|
150
148
|
.venv
|
|
151
149
|
env/
|
|
@@ -244,33 +242,6 @@ override.tf.json
|
|
|
244
242
|
# tfvars files may contain secrets
|
|
245
243
|
*.tfvars
|
|
246
244
|
*.tfvars.json
|
|
247
|
-
# Reviewed worker-AMI release contracts contain only immutable public package,
|
|
248
|
-
# AMI, and Terraform backend coordinates; the release workflow accepts only
|
|
249
|
-
# committed files from this exact internal path.
|
|
250
|
-
!deploy/terraform/aws/internal-examples/cloud-benchmark-runner/worker-ami/releases/*.tfvars.json
|
|
251
|
-
# Exception: this edge tfvars carries only a Secrets Manager ARN (a resource
|
|
252
|
-
# identifier, not a secret) so the staging-us edge apply needs no manual -var.
|
|
253
|
-
!deploy/cloud/terraform/aws-edge/staging-us.tfvars
|
|
254
|
-
# This dev adoption contract contains only public AWS resource identifiers; it
|
|
255
|
-
# is checked in so a reviewed plan cannot silently target a different orphan
|
|
256
|
-
# VPC/listener through an operator-local tfvars file.
|
|
257
|
-
!deploy/cloud/terraform/dev-controlplane-https-adoption/dev-us.tfvars
|
|
258
|
-
!deploy/cloud/terraform/dev-controlplane-https-adoption/.terraform.lock.hcl
|
|
259
|
-
# Managed release roots commit their exact provider selections. Release
|
|
260
|
-
# preflights and applies use these read-only instead of selecting a newer
|
|
261
|
-
# compatible provider during a long-running deployment.
|
|
262
|
-
!deploy/cloud/terraform/release-infrastructure/.terraform.lock.hcl
|
|
263
|
-
!deploy/cloud/terraform/aws-controlplane/.terraform.lock.hcl
|
|
264
|
-
!deploy/cloud/terraform/aws-edge/.terraform.lock.hcl
|
|
265
|
-
!deploy/cloud/terraform/s3-state/.terraform.lock.hcl
|
|
266
|
-
!deploy/cloud/terraform/vercel/.terraform.lock.hcl
|
|
267
|
-
!deploy/cloud/terraform/console-preview/.terraform.lock.hcl
|
|
268
|
-
!deploy/cloud/terraform/examples/dev/.terraform.lock.hcl
|
|
269
|
-
!deploy/cloud/terraform/examples/dev-eu/.terraform.lock.hcl
|
|
270
|
-
!deploy/cloud/terraform/examples/staging/.terraform.lock.hcl
|
|
271
|
-
!deploy/cloud/terraform/examples/staging-eu/.terraform.lock.hcl
|
|
272
|
-
!deploy/cloud/terraform/examples/prod-us/.terraform.lock.hcl
|
|
273
|
-
# Keep .terraform.lock.hcl for reproducibility (provider versions)
|
|
274
245
|
|
|
275
246
|
# Node.js
|
|
276
247
|
node_modules/
|
|
@@ -294,12 +265,7 @@ tmp/
|
|
|
294
265
|
.tmp/
|
|
295
266
|
.local/
|
|
296
267
|
|
|
297
|
-
.requirements-modal.txt
|
|
298
|
-
.mpa-i6pn-*-requirements-modal.txt
|
|
299
268
|
b6_results.json
|
|
300
|
-
# Raw dev-cloud load-test reports are local artifacts, not source-controlled baselines.
|
|
301
|
-
packages/sie_cloud/bench/results/dev-us-*/
|
|
302
|
-
|
|
303
269
|
# Personal mise overrides (per-developer toolchain tweaks, e.g. rustup
|
|
304
270
|
# component naming differences across rustup versions).
|
|
305
271
|
mise.local.toml
|
|
@@ -5,11 +5,49 @@ Provides SIERanker for reranking documents by relevance to a query.
|
|
|
5
5
|
|
|
6
6
|
from __future__ import annotations
|
|
7
7
|
|
|
8
|
+
from collections.abc import Mapping
|
|
8
9
|
from typing import Any
|
|
9
10
|
|
|
10
11
|
from haystack import Document, component
|
|
11
12
|
|
|
12
13
|
|
|
14
|
+
def _scores_by_index(results: Mapping[str, Any], count: int) -> list[float]:
|
|
15
|
+
"""Map ScoreResult entries back to input positions by item_id.
|
|
16
|
+
|
|
17
|
+
Parses each entry's ``item_id`` defensively: a missing, non-integer,
|
|
18
|
+
negative, or out-of-range id is skipped (that document keeps its 0.0
|
|
19
|
+
default) so a malformed entry can neither crash the rerank nor mis-assign a
|
|
20
|
+
score to the wrong document.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
results: ScoreResult envelope from ``SIEClient.score()``.
|
|
24
|
+
count: Number of input documents.
|
|
25
|
+
|
|
26
|
+
Returns:
|
|
27
|
+
Scores indexed by input position (0.0 for any unscored/invalid item).
|
|
28
|
+
"""
|
|
29
|
+
scores = [0.0] * count
|
|
30
|
+
for entry in results.get("scores", []):
|
|
31
|
+
item_id = entry.get("item_id", entry.get("index"))
|
|
32
|
+
# Accept only a genuine integer or an integer string. Reject bool (an int
|
|
33
|
+
# subclass, int(True) == 1), float (int(1.5) == 1), and non-integer
|
|
34
|
+
# strings, so a malformed id cannot silently overwrite the wrong position.
|
|
35
|
+
if isinstance(item_id, bool):
|
|
36
|
+
continue
|
|
37
|
+
if isinstance(item_id, int):
|
|
38
|
+
idx = item_id
|
|
39
|
+
elif isinstance(item_id, str):
|
|
40
|
+
try:
|
|
41
|
+
idx = int(item_id)
|
|
42
|
+
except ValueError:
|
|
43
|
+
continue
|
|
44
|
+
else:
|
|
45
|
+
continue
|
|
46
|
+
if 0 <= idx < count:
|
|
47
|
+
scores[idx] = float(entry.get("score", 0.0))
|
|
48
|
+
return scores
|
|
49
|
+
|
|
50
|
+
|
|
13
51
|
@component
|
|
14
52
|
class SIERanker:
|
|
15
53
|
"""Reranks documents by relevance to a query using SIE.
|
|
@@ -108,10 +146,19 @@ class SIERanker:
|
|
|
108
146
|
# Score documents
|
|
109
147
|
results = self.client.score(self._model, query_item, doc_items)
|
|
110
148
|
|
|
149
|
+
# ``results`` is a ScoreResult envelope; ranked entries are under
|
|
150
|
+
# ``results["scores"]`` (each a ScoreEntry keyed by ``item_id`` = input
|
|
151
|
+
# position, plus ``score``). Map scores back to input order by item_id
|
|
152
|
+
# rather than zipping positionally — the entries are sorted by relevance,
|
|
153
|
+
# not by input order. The envelope also carries ``results["request"]``
|
|
154
|
+
# (request id) and ``results["usage"]`` (token usage); those are
|
|
155
|
+
# available but intentionally not surfaced through the Haystack contract.
|
|
156
|
+
score_by_index = _scores_by_index(results, len(documents))
|
|
157
|
+
|
|
111
158
|
# Build scored documents
|
|
112
159
|
scored_docs = []
|
|
113
|
-
for
|
|
114
|
-
score =
|
|
160
|
+
for idx, doc in enumerate(documents):
|
|
161
|
+
score = score_by_index[idx]
|
|
115
162
|
# Store score in document metadata
|
|
116
163
|
doc_with_score = Document(
|
|
117
164
|
id=doc.id,
|
|
@@ -131,9 +178,3 @@ class SIERanker:
|
|
|
131
178
|
ranked_docs = ranked_docs[:effective_top_k]
|
|
132
179
|
|
|
133
180
|
return {"documents": ranked_docs}
|
|
134
|
-
|
|
135
|
-
def _extract_score(self, result: Any) -> float:
|
|
136
|
-
"""Extract score from SDK result."""
|
|
137
|
-
if isinstance(result, dict):
|
|
138
|
-
return float(result.get("score", 0.0))
|
|
139
|
-
return float(getattr(result, "score", 0.0))
|
|
@@ -155,13 +155,18 @@ def mock_sie_client() -> MagicMock:
|
|
|
155
155
|
for item in items
|
|
156
156
|
]
|
|
157
157
|
|
|
158
|
-
def mock_score(_model: str, query: Any, items: list[Any], **kwargs: Any) ->
|
|
158
|
+
def mock_score(_model: str, query: Any, items: list[Any], **kwargs: Any) -> dict[str, Any]:
|
|
159
|
+
# Mirror the real SDK: a ScoreResult envelope with ranked entries under
|
|
160
|
+
# "scores", not a bare list.
|
|
159
161
|
query_text = _get_text(query)
|
|
160
162
|
item_dicts = [
|
|
161
163
|
{"id": i.get("id", str(idx)) if isinstance(i, dict) else str(idx), "text": _get_text(i)}
|
|
162
164
|
for idx, i in enumerate(items)
|
|
163
165
|
]
|
|
164
|
-
return
|
|
166
|
+
return {
|
|
167
|
+
"model": _model,
|
|
168
|
+
"scores": _create_mock_score_result(query_text, item_dicts, kwargs.get("top_k")),
|
|
169
|
+
}
|
|
165
170
|
|
|
166
171
|
def mock_extract(_model: str, items: Any, labels: list[str], **_kwargs: Any) -> list[dict]:
|
|
167
172
|
# Extract always returns a list of entities for Haystack
|
|
@@ -4,7 +4,7 @@ These tests require a running SIE server and serve as runnable examples.
|
|
|
4
4
|
Run with: pytest -m integration integrations/sie_haystack/tests/
|
|
5
5
|
|
|
6
6
|
Prerequisites:
|
|
7
|
-
mise run serve -d cpu -p 8080
|
|
7
|
+
mise run serve -- -d cpu -p 8080
|
|
8
8
|
"""
|
|
9
9
|
|
|
10
10
|
from __future__ import annotations
|
|
@@ -2,9 +2,42 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
from collections import Counter
|
|
6
|
+
from unittest.mock import MagicMock
|
|
7
|
+
|
|
5
8
|
from haystack import Document
|
|
6
9
|
from sie_haystack import SIERanker
|
|
7
10
|
|
|
11
|
+
# Ranked entries reference input positions via item_id, out of input order and
|
|
12
|
+
# sorted by relevance (doc-3 most relevant).
|
|
13
|
+
_RANKED_ENVELOPE = {
|
|
14
|
+
"model": "test-reranker",
|
|
15
|
+
"scores": [
|
|
16
|
+
{"item_id": "3", "score": 0.9, "rank": 0},
|
|
17
|
+
{"item_id": "1", "score": 0.7, "rank": 1},
|
|
18
|
+
{"item_id": "4", "score": 0.5, "rank": 2},
|
|
19
|
+
{"item_id": "0", "score": 0.3, "rank": 3},
|
|
20
|
+
{"item_id": "2", "score": 0.1, "rank": 4},
|
|
21
|
+
],
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Envelope for three inputs where only item_id "1" is usable; the rest are
|
|
25
|
+
# malformed and must be skipped without crashing so their inputs keep the 0.0
|
|
26
|
+
# default. The float 1.5 and bool True come AFTER the valid "1": if int()
|
|
27
|
+
# accepted them (int(1.5) == 1, int(True) == 1) they would overwrite position 1.
|
|
28
|
+
_MALFORMED_ENVELOPE = {
|
|
29
|
+
"model": "test-reranker",
|
|
30
|
+
"scores": [
|
|
31
|
+
{"item_id": "1", "score": 0.8, "rank": 0},
|
|
32
|
+
{"item_id": "not-an-int", "score": 0.95, "rank": 1},
|
|
33
|
+
{"item_id": "-1", "score": 0.9, "rank": 2},
|
|
34
|
+
{"item_id": "99", "score": 0.7, "rank": 3},
|
|
35
|
+
{"score": 0.5, "rank": 4},
|
|
36
|
+
{"item_id": 1.5, "score": 0.99, "rank": 5},
|
|
37
|
+
{"item_id": True, "score": 0.98, "rank": 6},
|
|
38
|
+
],
|
|
39
|
+
}
|
|
40
|
+
|
|
8
41
|
|
|
9
42
|
class TestSIERanker:
|
|
10
43
|
"""Tests for SIERanker component."""
|
|
@@ -22,11 +55,44 @@ class TestSIERanker:
|
|
|
22
55
|
result = ranker.run(query=test_query, documents=haystack_documents)
|
|
23
56
|
|
|
24
57
|
assert "documents" in result
|
|
25
|
-
|
|
26
|
-
#
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
58
|
+
docs = result["documents"]
|
|
59
|
+
# Every input document comes back exactly as many times as it went in
|
|
60
|
+
# (Counter checks multiplicity, not just membership). The envelope bug
|
|
61
|
+
# tried to zip documents against the ScoreResult dict's keys.
|
|
62
|
+
assert len(docs) == len(haystack_documents)
|
|
63
|
+
assert Counter(d.content for d in docs) == Counter(d.content for d in haystack_documents)
|
|
64
|
+
scores = [d.meta["score"] for d in docs]
|
|
65
|
+
assert all(isinstance(s, float) for s in scores)
|
|
66
|
+
# Distinct and descending.
|
|
67
|
+
assert scores == sorted(scores, reverse=True)
|
|
68
|
+
assert len(set(scores)) == len(scores)
|
|
69
|
+
|
|
70
|
+
def test_run_maps_scores_by_item_id(self, mock_sie_client: object) -> None:
|
|
71
|
+
"""Top-ranked document is the relevant one, scores mapped by item_id."""
|
|
72
|
+
documents = [Document(content=f"doc-{i}") for i in range(5)]
|
|
73
|
+
mock_sie_client.score = MagicMock(return_value=_RANKED_ENVELOPE)
|
|
74
|
+
ranker = SIERanker(model="test-reranker")
|
|
75
|
+
ranker._client = mock_sie_client
|
|
76
|
+
|
|
77
|
+
ranked = ranker.run(query="query", documents=documents)["documents"]
|
|
78
|
+
|
|
79
|
+
assert [d.content for d in ranked] == ["doc-3", "doc-1", "doc-4", "doc-0", "doc-2"]
|
|
80
|
+
assert ranked[0].meta["score"] == 0.9
|
|
81
|
+
assert [d.meta["score"] for d in ranked] == [0.9, 0.7, 0.5, 0.3, 0.1]
|
|
82
|
+
|
|
83
|
+
def test_run_skips_malformed_item_id(self, mock_sie_client: object) -> None:
|
|
84
|
+
"""Malformed item_ids are skipped (no crash, no misassignment)."""
|
|
85
|
+
documents = [Document(content=f"doc-{i}") for i in range(3)]
|
|
86
|
+
mock_sie_client.score = MagicMock(return_value=_MALFORMED_ENVELOPE)
|
|
87
|
+
ranker = SIERanker(model="test-reranker")
|
|
88
|
+
ranker._client = mock_sie_client
|
|
89
|
+
|
|
90
|
+
ranked = ranker.run(query="query", documents=documents)["documents"]
|
|
91
|
+
|
|
92
|
+
assert len(ranked) == 3
|
|
93
|
+
by_content = {d.content: d.meta["score"] for d in ranked}
|
|
94
|
+
assert by_content == {"doc-1": 0.8, "doc-0": 0.0, "doc-2": 0.0}
|
|
95
|
+
assert ranked[0].content == "doc-1"
|
|
30
96
|
|
|
31
97
|
def test_run_empty_list(self, mock_sie_client: object) -> None:
|
|
32
98
|
"""Test that run handles empty document list."""
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|