ragsentry 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. ragsentry-0.1.0/LICENSE +21 -0
  2. ragsentry-0.1.0/PKG-INFO +128 -0
  3. ragsentry-0.1.0/README.md +106 -0
  4. ragsentry-0.1.0/pyproject.toml +38 -0
  5. ragsentry-0.1.0/ragsentry/__init__.py +1 -0
  6. ragsentry-0.1.0/ragsentry/__main__.py +4 -0
  7. ragsentry-0.1.0/ragsentry/adapters/__init__.py +1 -0
  8. ragsentry-0.1.0/ragsentry/adapters/base.py +32 -0
  9. ragsentry-0.1.0/ragsentry/adapters/http.py +106 -0
  10. ragsentry-0.1.0/ragsentry/adapters/python_callable.py +74 -0
  11. ragsentry-0.1.0/ragsentry/adapters/shell.py +84 -0
  12. ragsentry-0.1.0/ragsentry/ci.py +150 -0
  13. ragsentry-0.1.0/ragsentry/cli.py +392 -0
  14. ragsentry-0.1.0/ragsentry/diff.py +211 -0
  15. ragsentry-0.1.0/ragsentry/evalset/__init__.py +1 -0
  16. ragsentry-0.1.0/ragsentry/evalset/loader.py +51 -0
  17. ragsentry-0.1.0/ragsentry/evalset/schema.py +13 -0
  18. ragsentry-0.1.0/ragsentry/metrics/__init__.py +1 -0
  19. ragsentry-0.1.0/ragsentry/metrics/judge_config.py +182 -0
  20. ragsentry-0.1.0/ragsentry/metrics/ragas_backend.py +181 -0
  21. ragsentry-0.1.0/ragsentry/report/__init__.py +1 -0
  22. ragsentry-0.1.0/ragsentry/report/console.py +73 -0
  23. ragsentry-0.1.0/ragsentry/report/pr_comment.py +89 -0
  24. ragsentry-0.1.0/ragsentry/runner.py +56 -0
  25. ragsentry-0.1.0/ragsentry/storage/__init__.py +13 -0
  26. ragsentry-0.1.0/ragsentry/storage/base.py +44 -0
  27. ragsentry-0.1.0/ragsentry/storage/local_file.py +56 -0
  28. ragsentry-0.1.0/ragsentry/storage/postgres.py +119 -0
  29. ragsentry-0.1.0/ragsentry.egg-info/PKG-INFO +128 -0
  30. ragsentry-0.1.0/ragsentry.egg-info/SOURCES.txt +33 -0
  31. ragsentry-0.1.0/ragsentry.egg-info/dependency_links.txt +1 -0
  32. ragsentry-0.1.0/ragsentry.egg-info/entry_points.txt +2 -0
  33. ragsentry-0.1.0/ragsentry.egg-info/requires.txt +13 -0
  34. ragsentry-0.1.0/ragsentry.egg-info/top_level.txt +1 -0
  35. ragsentry-0.1.0/setup.cfg +4 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Omar Yasser
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,128 @@
1
+ Metadata-Version: 2.4
2
+ Name: ragsentry
3
+ Version: 0.1.0
4
+ Summary: Regression-testing and evaluation tool for RAG systems ('pytest for RAG')
5
+ Author: RagSentry Team
6
+ License: MIT
7
+ Requires-Python: >=3.10
8
+ Description-Content-Type: text/markdown
9
+ License-File: LICENSE
10
+ Requires-Dist: pydantic>=2.0
11
+ Requires-Dist: click>=8.0
12
+ Requires-Dist: httpx>=0.25.0
13
+ Requires-Dist: ragas>=0.1.0
14
+ Requires-Dist: langchain-openai>=0.1.0
15
+ Requires-Dist: langchain-anthropic>=0.1.0
16
+ Requires-Dist: datasets>=2.14.0
17
+ Provides-Extra: test
18
+ Requires-Dist: pytest>=7.0; extra == "test"
19
+ Provides-Extra: postgres
20
+ Requires-Dist: psycopg[binary]>=3.0; extra == "postgres"
21
+ Dynamic: license-file
22
+
23
+ # RagSentry — pytest for RAG
24
+
25
+ > **Regression testing and evaluation CLI for Retrieval-Augmented Generation (RAG) systems.**
26
+
27
+ RagSentry connects to your RAG application through a thin adapter, evaluates retrieved contexts and generated answers against your dataset, tracks runs over time, diffs runs to catch regressions, and gates CI builds.
28
+
29
+ ---
30
+
31
+ ## 🚀 Quickstart in 5 Minutes
32
+
33
+ ### 1. Install RagSentry
34
+
35
+ ```bash
36
+ pip install ragsentry
37
+ ```
38
+
39
+ *(Optional PostgreSQL storage support: `pip install ragsentry[postgres]`)*
40
+
41
+ ---
42
+
43
+ ### 2. Initialize RagSentry in your project
44
+
45
+ ```bash
46
+ ragsentry init
47
+ ```
48
+
49
+ This scaffolds:
50
+ - `ragsentry_adapter.py` — starter adapter stub
51
+ - `evalset.jsonl` — starter evaluation set
52
+
53
+ ---
54
+
55
+ ### 3. Connect your RAG application
56
+
57
+ Edit `ragsentry_adapter.py` to call your RAG pipeline and return the required contract:
58
+
59
+ ```python
60
+ from my_rag_app import ask_rag
61
+
62
+ def query(question: str) -> dict:
63
+ result = ask_rag(question)
64
+ return {
65
+ "answer": result["answer"],
66
+ "contexts": result["contexts"], # list[str] of retrieved passages
67
+ }
68
+ ```
69
+
70
+ ---
71
+
72
+ ### 4. Run an Evaluation
73
+
74
+ #### Unscored Generation Pass (No LLM required):
75
+ ```bash
76
+ ragsentry run -e evalset.jsonl -a ragsentry_adapter:query --no-scoring
77
+ ```
78
+
79
+ #### Full Evaluation with LLM Judge:
80
+ ```bash
81
+ ragsentry run -e evalset.jsonl -a ragsentry_adapter:query -p openrouter -m openai/gpt-oss-120b
82
+ ```
83
+
84
+ ---
85
+
86
+ ### 5. Diff Runs & Gate CI
87
+
88
+ #### Diff two evaluation runs:
89
+ ```bash
90
+ ragsentry diff run_2026-09-27T10-00-00_a1b2c3d4.json run_2026-09-27T11-00-00_e5f6g7h8.json
91
+ ```
92
+
93
+ #### Quality Gate in CI (Exit code 0 on pass, non-zero on failure):
94
+ ```bash
95
+ ragsentry ci \
96
+ --candidate <candidate_run_id> \
97
+ --baseline <baseline_run_id> \
98
+ -t faithfulness=0.80 \
99
+ -t context_recall=0.85 \
100
+ --max-regressed-questions 1 \
101
+ --pr-comment-out pr_comment.md
102
+ ```
103
+
104
+ ---
105
+
106
+ ## ⚡ Key Features
107
+
108
+ - 🔌 **Plug-and-Play Adapters:** Supports Python callables (`pkg.mod:func`), REST APIs (`http://localhost:8000/query`), or CLI tools (`shell:python rag_cli.py`).
109
+ - 🤖 **Provider-Agnostic Judge Scoring:** Works with OpenAI, Anthropic, Gemini, Groq, xAI, Ollama, DeepSeek, OpenRouter, or custom OpenAI-compatible endpoints.
110
+ - 💾 **Dual Storage Backends:** Local JSON file store or PostgreSQL (`--storage postgresql://...`).
111
+ - 📊 **Rich Terminal Diffs:** Colorized diff tables highlighting improved, regressed, and unchanged questions per metric.
112
+ - 🚦 **CI Quality Gates:** Strict threshold enforcement and regression limits with exportable GitHub PR Markdown summaries.
113
+ - 🧩 **Agent-Usable Scaffolder Skill:** Includes `skills/adapter-scaffolder` so AI coding assistants can inspect unfamiliar RAG repos and scaffold adapters automatically.
114
+
115
+ ---
116
+
117
+ ## 📖 Documentation & Guides
118
+
119
+ - 📘 [Adapter Contract Guide](docs/adapter-contract.md)
120
+ - 📙 [CI / CD Integration Guide](docs/ci-integration.md)
121
+ - 📗 [Configuration Reference](docs/config-reference.md)
122
+ - 📕 [Adapter Scaffolder Skill](skills/adapter-scaffolder/SKILL.md)
123
+
124
+ ---
125
+
126
+ ## 📜 License
127
+
128
+ MIT License.
@@ -0,0 +1,106 @@
1
+ # RagSentry — pytest for RAG
2
+
3
+ > **Regression testing and evaluation CLI for Retrieval-Augmented Generation (RAG) systems.**
4
+
5
+ RagSentry connects to your RAG application through a thin adapter, evaluates retrieved contexts and generated answers against your dataset, tracks runs over time, diffs runs to catch regressions, and gates CI builds.
6
+
7
+ ---
8
+
9
+ ## 🚀 Quickstart in 5 Minutes
10
+
11
+ ### 1. Install RagSentry
12
+
13
+ ```bash
14
+ pip install ragsentry
15
+ ```
16
+
17
+ *(Optional PostgreSQL storage support: `pip install ragsentry[postgres]`)*
18
+
19
+ ---
20
+
21
+ ### 2. Initialize RagSentry in your project
22
+
23
+ ```bash
24
+ ragsentry init
25
+ ```
26
+
27
+ This scaffolds:
28
+ - `ragsentry_adapter.py` — starter adapter stub
29
+ - `evalset.jsonl` — starter evaluation set
30
+
31
+ ---
32
+
33
+ ### 3. Connect your RAG application
34
+
35
+ Edit `ragsentry_adapter.py` to call your RAG pipeline and return the required contract:
36
+
37
+ ```python
38
+ from my_rag_app import ask_rag
39
+
40
+ def query(question: str) -> dict:
41
+ result = ask_rag(question)
42
+ return {
43
+ "answer": result["answer"],
44
+ "contexts": result["contexts"], # list[str] of retrieved passages
45
+ }
46
+ ```
47
+
48
+ ---
49
+
50
+ ### 4. Run an Evaluation
51
+
52
+ #### Unscored Generation Pass (No LLM required):
53
+ ```bash
54
+ ragsentry run -e evalset.jsonl -a ragsentry_adapter:query --no-scoring
55
+ ```
56
+
57
+ #### Full Evaluation with LLM Judge:
58
+ ```bash
59
+ ragsentry run -e evalset.jsonl -a ragsentry_adapter:query -p openrouter -m openai/gpt-oss-120b
60
+ ```
61
+
62
+ ---
63
+
64
+ ### 5. Diff Runs & Gate CI
65
+
66
+ #### Diff two evaluation runs:
67
+ ```bash
68
+ ragsentry diff run_2026-09-27T10-00-00_a1b2c3d4.json run_2026-09-27T11-00-00_e5f6g7h8.json
69
+ ```
70
+
71
+ #### Quality Gate in CI (Exit code 0 on pass, non-zero on failure):
72
+ ```bash
73
+ ragsentry ci \
74
+ --candidate <candidate_run_id> \
75
+ --baseline <baseline_run_id> \
76
+ -t faithfulness=0.80 \
77
+ -t context_recall=0.85 \
78
+ --max-regressed-questions 1 \
79
+ --pr-comment-out pr_comment.md
80
+ ```
81
+
82
+ ---
83
+
84
+ ## ⚡ Key Features
85
+
86
+ - 🔌 **Plug-and-Play Adapters:** Supports Python callables (`pkg.mod:func`), REST APIs (`http://localhost:8000/query`), or CLI tools (`shell:python rag_cli.py`).
87
+ - 🤖 **Provider-Agnostic Judge Scoring:** Works with OpenAI, Anthropic, Gemini, Groq, xAI, Ollama, DeepSeek, OpenRouter, or custom OpenAI-compatible endpoints.
88
+ - 💾 **Dual Storage Backends:** Local JSON file store or PostgreSQL (`--storage postgresql://...`).
89
+ - 📊 **Rich Terminal Diffs:** Colorized diff tables highlighting improved, regressed, and unchanged questions per metric.
90
+ - 🚦 **CI Quality Gates:** Strict threshold enforcement and regression limits with exportable GitHub PR Markdown summaries.
91
+ - 🧩 **Agent-Usable Scaffolder Skill:** Includes `skills/adapter-scaffolder` so AI coding assistants can inspect unfamiliar RAG repos and scaffold adapters automatically.
92
+
93
+ ---
94
+
95
+ ## 📖 Documentation & Guides
96
+
97
+ - 📘 [Adapter Contract Guide](docs/adapter-contract.md)
98
+ - 📙 [CI / CD Integration Guide](docs/ci-integration.md)
99
+ - 📗 [Configuration Reference](docs/config-reference.md)
100
+ - 📕 [Adapter Scaffolder Skill](skills/adapter-scaffolder/SKILL.md)
101
+
102
+ ---
103
+
104
+ ## 📜 License
105
+
106
+ MIT License.
@@ -0,0 +1,38 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "ragsentry"
7
+ version = "0.1.0"
8
+ description = "Regression-testing and evaluation tool for RAG systems ('pytest for RAG')"
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = { text = "MIT" }
12
+ authors = [
13
+ { name = "RagSentry Team" }
14
+ ]
15
+ dependencies = [
16
+ "pydantic>=2.0",
17
+ "click>=8.0",
18
+ "httpx>=0.25.0",
19
+ "ragas>=0.1.0",
20
+ "langchain-openai>=0.1.0",
21
+ "langchain-anthropic>=0.1.0",
22
+ "datasets>=2.14.0",
23
+ ]
24
+
25
+ [project.optional-dependencies]
26
+ test = [
27
+ "pytest>=7.0",
28
+ ]
29
+ postgres = [
30
+ "psycopg[binary]>=3.0",
31
+ ]
32
+
33
+ [project.scripts]
34
+ ragsentry = "ragsentry.cli:main"
35
+
36
+ [tool.setuptools.packages.find]
37
+ where = ["."]
38
+ include = ["ragsentry*"]
@@ -0,0 +1 @@
1
+
@@ -0,0 +1,4 @@
1
+ from ragsentry.cli import main
2
+
3
+ if __name__ == "__main__":
4
+ main()
@@ -0,0 +1,32 @@
1
+ from __future__ import annotations
2
+
3
+
4
+ from dataclasses import dataclass, field
5
+ from typing import Any, Protocol, runtime_checkable
6
+
7
+
8
+ @dataclass
9
+ class AdapterResponse:
10
+ """Standardized response structure returned by any RagSentry adapter."""
11
+
12
+ answer: str
13
+ contexts: list[str]
14
+ metadata: dict[str, Any] = field(default_factory=dict)
15
+
16
+
17
+ def to_dict(self) -> dict[str, Any]:
18
+ return{
19
+ "answer": self.answer,
20
+ "contexts": self.contexts,
21
+ "metadata": self.metadata,
22
+ }
23
+
24
+
25
+
26
+ @runtime_checkable
27
+ class Adapter(Protocol):
28
+ """Protocol defining the adapter contract for external RAG systems."""
29
+
30
+ def query(self, question: str) -> AdapterResponse:
31
+ """Query the target RAG system with a question and return an AdapterResponse."""
32
+ ...
@@ -0,0 +1,106 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any
4
+ import httpx
5
+
6
+
7
+ from ragsentry.adapters.base import Adapter, AdapterResponse
8
+
9
+
10
+ def extract_path(data: Any, path: str) -> Any:
11
+ """Extract a nested value using dotted syntax like 'data.answer' or 'results.0.text'."""
12
+ if not path:
13
+ return data
14
+
15
+ current = data
16
+ for part in path.split("."):
17
+ if isinstance(current, dict):
18
+ current = current.get(part)
19
+
20
+ elif isinstance(current, (list, tuple)) and part.isdigit():
21
+ idx = int(part)
22
+ current = current[idx] if 0<= idx < len(current) else None
23
+
24
+ else:
25
+ return None
26
+
27
+ if current is None:
28
+ break
29
+ return current
30
+
31
+
32
+
33
+
34
+ class HttpAdapter(Adapter):
35
+ """Adapter that queries an external RAG service over HTTP."""
36
+
37
+ def __init__(
38
+ self,
39
+ url: str,
40
+ method: str = "POST",
41
+ headers: dict[str, str] | None = None,
42
+ question_key: str = "question",
43
+ answer_path: str = "answer",
44
+ contexts_path: str = "contexts",
45
+ metadata_path: str | None = "metadata",
46
+ timeout: float =30.0
47
+ ) -> None:
48
+ self.url = url
49
+ self.method = method.upper()
50
+ self.headers = headers or {"Content-Type": "application/json"}
51
+ self.question_key = question_key
52
+ self.answer_path = answer_path
53
+ self.contexts_path = contexts_path
54
+ self.metadata_path = metadata_path
55
+ self.timeout = timeout
56
+
57
+
58
+ def query(self, question: str) -> AdapterResponse:
59
+ payload = {self.question_key: question}
60
+
61
+ with httpx.Client(timeout=self.timeout) as client:
62
+ if self.method == "POST":
63
+ response = client.post(self.url ,json=payload, headers=self.headers)
64
+ elif self.method == "GET":
65
+ response = client.get(self.url ,params=payload, headers=self.headers)
66
+
67
+ else:
68
+ raise ValueError(f"Unsupported HTTP method: {self.method}")
69
+
70
+
71
+ response.raise_for_status()
72
+ data = response.json()
73
+
74
+
75
+ #extract the answer
76
+ raw_answer = extract_path(data, self.answer_path)
77
+ if raw_answer is None:
78
+ raise ValueError(
79
+ f"Failed to extract answer using path '{self.answer_path}' from response: {data}"
80
+ )
81
+
82
+ #extract the contexts
83
+ raw_contexts = extract_path(data, self.contexts_path)
84
+ if raw_contexts is None:
85
+ raw_contexts = []
86
+ elif not isinstance(raw_contexts, list):
87
+ raw_contexts = [raw_contexts]
88
+
89
+ contexts = [
90
+ c if isinstance(c, str) else str(c.get("text", c)) if isinstance(c, dict) else str(c)
91
+ for c in raw_contexts
92
+ ]
93
+
94
+ #extract metadata
95
+ metadata = {}
96
+ if self.metadata_path:
97
+ raw_meta = extract_path(data, self.metadata_path)
98
+ if isinstance(raw_meta, dict):
99
+ metadata = raw_meta
100
+
101
+
102
+ return AdapterResponse(
103
+ answer=str(raw_answer),
104
+ contexts=contexts,
105
+ metadata=metadata
106
+ )
@@ -0,0 +1,74 @@
1
+ from __future__ import annotations
2
+
3
+ import importlib
4
+ from typing import Any, Callable
5
+ import os
6
+ import sys
7
+
8
+ from ragsentry.adapters.base import Adapter, AdapterResponse
9
+
10
+
11
+ class PythonCallableAdapter(Adapter):
12
+ """Adapter that wraps an in-process Python callable."""
13
+
14
+
15
+ def __init__(self, target: Callable[[str], dict[str, Any] | AdapterResponse] | str) -> None:
16
+
17
+ if isinstance(target, str):
18
+ self._callable = self._load_callable(target)
19
+ elif callable(target):
20
+ self._callable = target
21
+
22
+ else:
23
+ raise TypeError(f"Target must be a callable or an import string ('module:func'), got {type(target)}")
24
+
25
+
26
+
27
+ @staticmethod
28
+ def _load_callable(import_str: str) -> Callable[[str], Any]:
29
+ """Resolves an import string of format 'module.submodule:function_name'."""
30
+ if ":" not in import_str:
31
+ raise ValueError(
32
+ f"Invalid import string '{import_str}'. Expected format: 'package.module:function_name'"
33
+ )
34
+ cwd = os.getcwd()
35
+ if cwd not in sys.path:
36
+ sys.path.insert(0, cwd)
37
+
38
+ module_path, func_name = import_str.split(":", 1)
39
+ module = importlib.import_module(module_path)
40
+ target = getattr(module, func_name)
41
+ if not callable(target):
42
+ raise TypeError(f"Resolved object '{import_str}' is not callable.")
43
+
44
+ return target
45
+
46
+
47
+ def query(self, question: str) -> AdapterResponse:
48
+
49
+ raw = self._callable(question)
50
+
51
+ if isinstance(raw, AdapterResponse):
52
+ return raw
53
+
54
+ if isinstance(raw, dict):
55
+ if "answer" not in raw or "contexts" not in raw:
56
+ raise ValueError(
57
+ f"Callable returned a dict missing 'answer' or 'contexts': keys={list(raw.keys())}"
58
+ )
59
+
60
+ contexts = raw["contexts"]
61
+ if not isinstance(contexts, list):
62
+ raise TypeError(f"'contexts' must be a list of strings, got {type(contexts)}")
63
+
64
+
65
+ return AdapterResponse(
66
+ answer=str(raw["answer"]),
67
+ contexts=[str(c) for c in contexts],
68
+ metadata=raw.get("metadata", {}),
69
+ )
70
+
71
+
72
+ raise TypeError(
73
+ f"Expected callable to return dict or AdapterResponse, got {type(raw)}"
74
+ )
@@ -0,0 +1,84 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import shlex
5
+ import subprocess
6
+ from typing import Any
7
+
8
+
9
+ from ragsentry.adapters.base import Adapter, AdapterResponse
10
+
11
+
12
+ class ShellAdapter(Adapter):
13
+ """Adapter that executes an external shell command or script."""
14
+
15
+ def __init__(
16
+ self,
17
+ command: str | list [str],
18
+ pass_as: str = "stdin", #stdin or arg
19
+ timeout: float = 30.0,
20
+ cwd: str | None = None,
21
+ ) -> None:
22
+
23
+ if isinstance(command, str):
24
+ self.command = shlex.split(command, posix=False)
25
+
26
+ else:
27
+ self.command = list(command)
28
+
29
+
30
+ self.pass_as = pass_as.lower()
31
+ self.timeout = timeout
32
+ self.cwd = cwd
33
+
34
+
35
+ if self.pass_as not in ("stdin", "arg"):
36
+ raise ValueError(f"pass_as must be 'stdin' or 'arg', got '{pass_as}'")
37
+
38
+
39
+ def query(self, question: str) -> AdapterResponse:
40
+ cmd = list(self.command)
41
+ input_data = None
42
+
43
+ if self.pass_as == "arg":
44
+ cmd.append(question)
45
+
46
+ else:
47
+ input_data = json.dumps({"question": question})
48
+
49
+ proc = subprocess.run(
50
+ cmd,
51
+ input=input_data,
52
+ capture_output=True,
53
+ text=True,
54
+ timeout=self.timeout,
55
+ cwd=self.cwd,
56
+ )
57
+
58
+ if proc.returncode != 0:
59
+ raise RuntimeError(
60
+ f"Shell command failed with exit code {proc.returncode}.\nStderr: {proc.stderr}"
61
+ )
62
+ stdout_clean = proc.stdout.strip()
63
+ if not stdout_clean:
64
+ raise ValueError(f"Shell command produced no stdout output.\nStderr: {proc.stderr}")
65
+ try:
66
+ data: dict[str, Any] = json.loads(stdout_clean)
67
+ except json.JSONDecodeError as exc:
68
+ raise ValueError(
69
+ f"Shell adapter expected JSON output on stdout, got: {stdout_clean}"
70
+ ) from exc
71
+ if "answer" not in data or "contexts" not in data:
72
+ raise ValueError(
73
+ f"Shell command output missing 'answer' or 'contexts'. Keys found: {list(data.keys())}"
74
+ )
75
+ contexts = data["contexts"]
76
+ if not isinstance(contexts, list):
77
+ raise TypeError(f"'contexts' in shell output must be a list, got {type(contexts)}")
78
+
79
+
80
+ return AdapterResponse(
81
+ answer=str(data["answer"]),
82
+ contexts=[str(c) for c in contexts],
83
+ metadata=data.get("metadata", {}),
84
+ )