lattereview 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Pouria Rouzrokh
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,184 @@
1
+ Metadata-Version: 2.1
2
+ Name: lattereview
3
+ Version: 0.1.0
4
+ Summary: A framework for multi-agent review workflows using large language models
5
+ Home-page: https://github.com/PouriaRouzrokh/LatteReview
6
+ Author: Pouria Rouzrokh
7
+ Author-email: po.rouzrokh@gmail.com
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Science/Research
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Operating System :: OS Independent
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3.9
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: litellm>=1.55.2
19
+ Requires-Dist: nest-asyncio>=1.6.0
20
+ Requires-Dist: ollama>=0.4.4
21
+ Requires-Dist: openai>=1.57.4
22
+ Requires-Dist: pandas>=2.2.3
23
+ Requires-Dist: pydantic>=2.10.3
24
+ Requires-Dist: python-dotenv>=1.0.1
25
+ Requires-Dist: tokencost>=0.1.17
26
+ Requires-Dist: tqdm>=4.67.1
27
+ Requires-Dist: openpyxl>=3.1.5
28
+
29
+ # LatteReview 🤖☕
30
+
31
+ [![PyPI version](https://badge.fury.io/py/lattereview.svg)](https://badge.fury.io/py/lattereview)
32
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
33
+ [![Python 3.9+](https://img.shields.io/badge/python-3.9+-blue.svg)](https://www.python.org/downloads/)
34
+ [![Code style: black](https://img.shields.io/badge/code%20style-black-000000.svg)](https://github.com/psf/black)
35
+ [![Maintained: yes](https://img.shields.io/badge/Maintained%3F-yes-green.svg)](https://github.com/prouzrokh/lattereview)
36
+
37
+ LatteReview is a powerful Python package designed to automate academic literature review processes through AI-powered agents. Just like enjoying a cup of latte ☕, reviewing numerous research articles should be a pleasant, efficient experience that doesn't consume your entire day!
38
+
39
+ ## 🎯 Key Features
40
+
41
+ - Multi-agent review system with customizable roles and expertise
42
+ - Support for multiple review rounds with hierarchical decision-making
43
+ - Flexible model integration (OpenAI, Gemini, Claude, Groq, local models via Ollama)
44
+ - Asynchronous processing for high-performance batch reviews
45
+ - Structured output format with detailed scoring and reasoning
46
+ - Comprehensive cost tracking and memory management
47
+ - Extensible architecture for custom review workflows
48
+
49
+ ## 🛠️ Installation
50
+
51
+ ```bash
52
+ pip install lattereview
53
+ ```
54
+
55
+ ## 🚀 Quick Start
56
+
57
+ Here's a simple example of how to set up a review workflow with two primary reviewers and an expert reviewer for conflict resolution:
58
+
59
+ ```python
60
+ from lattereview.providers import LiteLLMProvider
61
+ from lattereview.agents import ScoringReviewer
62
+ from lattereview.review_workflow import ReviewWorkflow
63
+ import pandas as pd
64
+ import asyncio
65
+
66
+ # First Reviewer: Conservative approach
67
+ reviewer1 = ScoringReviewer(
68
+ provider=LiteLLMProvider(model="gpt-4o-mini"),
69
+ name="Alice",
70
+ backstory="a radiologist with expertise in systematic reviews",
71
+ input_description="article title and abstract",
72
+ scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
73
+ score_set=[1, 2, 3, 4, 5],
74
+ scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
75
+ model_args={"temperature": 0.1} # Low temperature for consistent scoring
76
+ )
77
+
78
+ # Second Reviewer: More exploratory approach
79
+ reviewer2 = ScoringReviewer(
80
+ provider=LiteLLMProvider(model="gemini/gemini-1.5-flash"),
81
+ name="Bob",
82
+ backstory="a computer scientist specializing in medical AI",
83
+ input_description="article title and abstract",
84
+ scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
85
+ score_set=[1, 2, 3, 4, 5],
86
+ scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
87
+ model_args={"temperature": 0.8} # Higher temperature for creative interpretation
88
+ )
89
+
90
+ # Expert Reviewer: Resolves disagreements
91
+ expert = ScoringReviewer(
92
+ provider=LiteLLMProvider(model="gpt-4o"),
93
+ name="Carol",
94
+ backstory="a professor of AI in medical imaging",
95
+ input_description="article title, abstract, and previous reviews",
96
+ scoring_task="Review Alice and Bob's relevance assessments of this article to AI in radiology",
97
+ score_set=[1, 2],
98
+ scoring_rules='Score 1 if you agree with Alice\'s assessment, 2 if you agree with Bob\'s assessment',
99
+ model_args={"temperature": 0.1} # Low temperature for careful judgment
100
+ )
101
+
102
+ # Define the multi-round review workflow
103
+ workflow = ReviewWorkflow([
104
+ {
105
+ "round": 'A', # First round: Initial review by both reviewers
106
+ "reviewers": [reviewer1, reviewer2],
107
+ "inputs": ["title", "abstract"]
108
+ },
109
+ {
110
+ "round": 'B', # Second round: Expert reviews only disagreements
111
+ "reviewers": [expert],
112
+ "inputs": ["title", "abstract", "round-A_Alice_output", "round-A_Bob_output"],
113
+ "filter": lambda row: row["round-A_Alice_score"] != row["round-A_Bob_score"]
114
+ }
115
+ ])
116
+
117
+ # Run the workflow on your data
118
+ data = pd.read_excel("articles.xlsx")
119
+ results = asyncio.run(workflow(data))
120
+ ```
121
+
122
+ ## 🔌 Model Support
123
+
124
+ LatteReview offers flexible model integration through multiple providers:
125
+
126
+ - **LiteLLMProvider** (Recommended): Supports OpenAI, Anthropic (Claude), Gemini, Groq, and more
127
+ - **OpenAIProvider**: Direct integration with OpenAI and Gemini APIs
128
+ - **OllamaProvider**: Optimized for local models via Ollama
129
+
130
+ Note: Models should support async operations and structured JSON outputs for optimal performance.
131
+
132
+ ## 📖 Documentation
133
+
134
+ Full documentation and API reference will be available in our upcoming preprint paper. [Link to be added]
135
+
136
+ ## 🛣️ Roadmap
137
+
138
+ - [ ] Development of `AbstractionReviewer` class for automated paper summarization
139
+ - [ ] Support for image-based inputs and multimodal analysis
140
+ - [ ] Development of a no-code web application
141
+ - [ ] Integration of RAG (Retrieval-Augmented Generation) tools
142
+ - [ ] Addition of graph-based analysis tools
143
+ - [ ] Enhanced visualization capabilities
144
+ - [ ] Support for additional model providers
145
+
146
+ ## 👨‍💻 Author
147
+
148
+ **Pouria Rouzrokh, MD, MPH, MHPE**
149
+ Medical Practitioner and Machine Learning Engineer
150
+ Incoming Radiology Resident @Yale University
151
+ Former Data Scientist @Mayo Clinic AI Lab
152
+
153
+ Find my work:
154
+ [![Twitter Follow](https://img.shields.io/twitter/follow/prouzrokh?style=social)](https://twitter.com/prouzrokh)
155
+ [![LinkedIn](https://img.shields.io/badge/LinkedIn-Connect-blue)](https://linkedin.com/in/pouria-rouzrokh)
156
+ [![Google Scholar](https://img.shields.io/badge/Google%20Scholar-Profile-green)](https://scholar.google.com/citations?user=Ksv9I0sAAAAJ&hl=en)
157
+ [![Email](https://img.shields.io/badge/Email-Contact-red)](mailto:po.rouzrokh@gmail.com)
158
+
159
+ ## ❤️ Support LatteReview
160
+
161
+ If you find LatteReview helpful in your research or work, consider supporting its continued development. Since we're already sharing a virtual coffee break while reviewing papers, maybe you'd like to treat me to a real one? ☕ 😊
162
+
163
+ ### Ways to Support:
164
+
165
+ - [Treat me to a coffee](http://ko-fi.com/pouriarouzrokh) on Ko-fi ☕
166
+ - [Star the repository](https://github.com/PouriaRouzrokh/LatteReview) to help others discover the project
167
+ - Submit bug reports, feature requests, or contribute code
168
+ - Share your experience using LatteReview in your research
169
+
170
+ ## 📜 License
171
+
172
+ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
173
+
174
+ ## 🤝 Contributing
175
+
176
+ We welcome contributions! Please feel free to submit a Pull Request.
177
+
178
+ ## 📚 Citation
179
+
180
+ If you use LatteReview in your research, please cite our paper:
181
+
182
+ ```bibtex
183
+ # Preprint citation to be added
184
+ ```
@@ -0,0 +1,156 @@
1
+ # LatteReview 🤖☕
2
+
3
+ [![PyPI version](https://badge.fury.io/py/lattereview.svg)](https://badge.fury.io/py/lattereview)
4
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
5
+ [![Python 3.9+](https://img.shields.io/badge/python-3.9+-blue.svg)](https://www.python.org/downloads/)
6
+ [![Code style: black](https://img.shields.io/badge/code%20style-black-000000.svg)](https://github.com/psf/black)
7
+ [![Maintained: yes](https://img.shields.io/badge/Maintained%3F-yes-green.svg)](https://github.com/prouzrokh/lattereview)
8
+
9
+ LatteReview is a powerful Python package designed to automate academic literature review processes through AI-powered agents. Just like enjoying a cup of latte ☕, reviewing numerous research articles should be a pleasant, efficient experience that doesn't consume your entire day!
10
+
11
+ ## 🎯 Key Features
12
+
13
+ - Multi-agent review system with customizable roles and expertise
14
+ - Support for multiple review rounds with hierarchical decision-making
15
+ - Flexible model integration (OpenAI, Gemini, Claude, Groq, local models via Ollama)
16
+ - Asynchronous processing for high-performance batch reviews
17
+ - Structured output format with detailed scoring and reasoning
18
+ - Comprehensive cost tracking and memory management
19
+ - Extensible architecture for custom review workflows
20
+
21
+ ## 🛠️ Installation
22
+
23
+ ```bash
24
+ pip install lattereview
25
+ ```
26
+
27
+ ## 🚀 Quick Start
28
+
29
+ Here's a simple example of how to set up a review workflow with two primary reviewers and an expert reviewer for conflict resolution:
30
+
31
+ ```python
32
+ from lattereview.providers import LiteLLMProvider
33
+ from lattereview.agents import ScoringReviewer
34
+ from lattereview.review_workflow import ReviewWorkflow
35
+ import pandas as pd
36
+ import asyncio
37
+
38
+ # First Reviewer: Conservative approach
39
+ reviewer1 = ScoringReviewer(
40
+ provider=LiteLLMProvider(model="gpt-4o-mini"),
41
+ name="Alice",
42
+ backstory="a radiologist with expertise in systematic reviews",
43
+ input_description="article title and abstract",
44
+ scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
45
+ score_set=[1, 2, 3, 4, 5],
46
+ scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
47
+ model_args={"temperature": 0.1} # Low temperature for consistent scoring
48
+ )
49
+
50
+ # Second Reviewer: More exploratory approach
51
+ reviewer2 = ScoringReviewer(
52
+ provider=LiteLLMProvider(model="gemini/gemini-1.5-flash"),
53
+ name="Bob",
54
+ backstory="a computer scientist specializing in medical AI",
55
+ input_description="article title and abstract",
56
+ scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
57
+ score_set=[1, 2, 3, 4, 5],
58
+ scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
59
+ model_args={"temperature": 0.8} # Higher temperature for creative interpretation
60
+ )
61
+
62
+ # Expert Reviewer: Resolves disagreements
63
+ expert = ScoringReviewer(
64
+ provider=LiteLLMProvider(model="gpt-4o"),
65
+ name="Carol",
66
+ backstory="a professor of AI in medical imaging",
67
+ input_description="article title, abstract, and previous reviews",
68
+ scoring_task="Review Alice and Bob's relevance assessments of this article to AI in radiology",
69
+ score_set=[1, 2],
70
+ scoring_rules='Score 1 if you agree with Alice\'s assessment, 2 if you agree with Bob\'s assessment',
71
+ model_args={"temperature": 0.1} # Low temperature for careful judgment
72
+ )
73
+
74
+ # Define the multi-round review workflow
75
+ workflow = ReviewWorkflow([
76
+ {
77
+ "round": 'A', # First round: Initial review by both reviewers
78
+ "reviewers": [reviewer1, reviewer2],
79
+ "inputs": ["title", "abstract"]
80
+ },
81
+ {
82
+ "round": 'B', # Second round: Expert reviews only disagreements
83
+ "reviewers": [expert],
84
+ "inputs": ["title", "abstract", "round-A_Alice_output", "round-A_Bob_output"],
85
+ "filter": lambda row: row["round-A_Alice_score"] != row["round-A_Bob_score"]
86
+ }
87
+ ])
88
+
89
+ # Run the workflow on your data
90
+ data = pd.read_excel("articles.xlsx")
91
+ results = asyncio.run(workflow(data))
92
+ ```
93
+
94
+ ## 🔌 Model Support
95
+
96
+ LatteReview offers flexible model integration through multiple providers:
97
+
98
+ - **LiteLLMProvider** (Recommended): Supports OpenAI, Anthropic (Claude), Gemini, Groq, and more
99
+ - **OpenAIProvider**: Direct integration with OpenAI and Gemini APIs
100
+ - **OllamaProvider**: Optimized for local models via Ollama
101
+
102
+ Note: Models should support async operations and structured JSON outputs for optimal performance.
103
+
104
+ ## 📖 Documentation
105
+
106
+ Full documentation and API reference will be available in our upcoming preprint paper. [Link to be added]
107
+
108
+ ## 🛣️ Roadmap
109
+
110
+ - [ ] Development of `AbstractionReviewer` class for automated paper summarization
111
+ - [ ] Support for image-based inputs and multimodal analysis
112
+ - [ ] Development of a no-code web application
113
+ - [ ] Integration of RAG (Retrieval-Augmented Generation) tools
114
+ - [ ] Addition of graph-based analysis tools
115
+ - [ ] Enhanced visualization capabilities
116
+ - [ ] Support for additional model providers
117
+
118
+ ## 👨‍💻 Author
119
+
120
+ **Pouria Rouzrokh, MD, MPH, MHPE**
121
+ Medical Practitioner and Machine Learning Engineer
122
+ Incoming Radiology Resident @Yale University
123
+ Former Data Scientist @Mayo Clinic AI Lab
124
+
125
+ Find my work:
126
+ [![Twitter Follow](https://img.shields.io/twitter/follow/prouzrokh?style=social)](https://twitter.com/prouzrokh)
127
+ [![LinkedIn](https://img.shields.io/badge/LinkedIn-Connect-blue)](https://linkedin.com/in/pouria-rouzrokh)
128
+ [![Google Scholar](https://img.shields.io/badge/Google%20Scholar-Profile-green)](https://scholar.google.com/citations?user=Ksv9I0sAAAAJ&hl=en)
129
+ [![Email](https://img.shields.io/badge/Email-Contact-red)](mailto:po.rouzrokh@gmail.com)
130
+
131
+ ## ❤️ Support LatteReview
132
+
133
+ If you find LatteReview helpful in your research or work, consider supporting its continued development. Since we're already sharing a virtual coffee break while reviewing papers, maybe you'd like to treat me to a real one? ☕ 😊
134
+
135
+ ### Ways to Support:
136
+
137
+ - [Treat me to a coffee](http://ko-fi.com/pouriarouzrokh) on Ko-fi ☕
138
+ - [Star the repository](https://github.com/PouriaRouzrokh/LatteReview) to help others discover the project
139
+ - Submit bug reports, feature requests, or contribute code
140
+ - Share your experience using LatteReview in your research
141
+
142
+ ## 📜 License
143
+
144
+ This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
145
+
146
+ ## 🤝 Contributing
147
+
148
+ We welcome contributions! Please feel free to submit a Pull Request.
149
+
150
+ ## 📚 Citation
151
+
152
+ If you use LatteReview in your research, please cite our paper:
153
+
154
+ ```bibtex
155
+ # Preprint citation to be added
156
+ ```
File without changes
@@ -0,0 +1 @@
1
+ from .scoring_reviewer import ScoringReviewer
@@ -0,0 +1,160 @@
1
+ """Base agent class with consistent error handling and type safety."""
2
+
3
+ from typing import List, Optional, Dict, Any, Union
4
+ from enum import Enum
5
+ from pydantic import BaseModel, Field
6
+
7
+ DEFAULT_CONCURRENT_REQUESTS = 20
8
+
9
+
10
+ class ReasoningType(Enum):
11
+ """Enumeration for reasoning types."""
12
+
13
+ NONE = "none"
14
+ BRIEF = "brief"
15
+ LONG = "long"
16
+ COT = "cot"
17
+
18
+
19
+ class AgentError(Exception):
20
+ """Base exception for agent-related errors."""
21
+
22
+ pass
23
+
24
+
25
+ class BaseAgent(BaseModel):
26
+ response_format: Dict[str, Any]
27
+ provider: Optional[Any] = None
28
+ model_args: Dict[str, Any] = Field(default_factory=dict)
29
+ max_concurrent_requests: int = DEFAULT_CONCURRENT_REQUESTS
30
+ name: str = "BaseAgent"
31
+ backstory: str = "a generic base agent"
32
+ input_description: str = "article title/abstract"
33
+ examples: Union[str, List[Union[str, Dict[str, Any]]]] = None
34
+ reasoning: ReasoningType = ReasoningType.BRIEF
35
+ system_prompt: Optional[str] = None
36
+ item_prompt: Optional[str] = None
37
+ cost_so_far: float = 0
38
+ memory: List[Dict[str, Any]] = []
39
+ identity: Dict[str, Any] = {}
40
+
41
+ def __init__(self, **data: Any) -> None:
42
+ """Initialize the base agent with error handling."""
43
+ try:
44
+ super().__init__(**data)
45
+ if isinstance(self.reasoning, str):
46
+ self.reasoning = ReasoningType(self.reasoning.lower())
47
+ if self.reasoning == ReasoningType.NONE:
48
+ self.response_format.pop("reasoning", None)
49
+ self.setup()
50
+ except Exception as e:
51
+ raise AgentError(f"Error initializing agent: {str(e)}")
52
+
53
+ def setup(self) -> None:
54
+ """Setup the agent before use."""
55
+ raise NotImplementedError("This method must be implemented by subclasses.")
56
+
57
+ def build_system_prompt(self) -> str:
58
+ """Build the system prompt for the agent."""
59
+ try:
60
+ return self._clean_text(
61
+ f"""
62
+ Your name is <<{self.name}>> and you are <<{self.backstory}>>.
63
+ Your task is to review input itmes with the following description: <<{self.input_description}>>.
64
+ Your final output should have the following keys: \
65
+ {", ".join(f"{k} ({v})" for k, v in self.response_format.items())}.
66
+ """
67
+ )
68
+ except Exception as e:
69
+ raise AgentError(f"Error building system prompt: {str(e)}")
70
+
71
+ def build_item_prompt(self, base_prompt: str, item_dict: Dict[str, Any]) -> str:
72
+ """Build the item prompt with variable substitution."""
73
+ try:
74
+ prompt = base_prompt
75
+ if "examples" in item_dict:
76
+ item_dict["examples"] = self.process_examples(item_dict["examples"])
77
+ if "reasoning" in item_dict:
78
+ item_dict["reasoning"] = self.process_reasoning(item_dict["reasoning"])
79
+
80
+ for key, value in item_dict.items():
81
+ if value is not None:
82
+ prompt = prompt.replace(f"${{{key}}}$", str(value))
83
+ else:
84
+ prompt = prompt.replace(f"${{{key}}}$", "")
85
+
86
+ return self._clean_text(prompt)
87
+ except Exception as e:
88
+ raise AgentError(f"Error building item prompt: {str(e)}")
89
+
90
+ def process_reasoning(self, reasoning: Union[str, ReasoningType]) -> str:
91
+ """Process the reasoning type into a prompt string."""
92
+ try:
93
+ if isinstance(reasoning, str):
94
+ reasoning = ReasoningType(reasoning.lower())
95
+
96
+ reasoning_map = {
97
+ ReasoningType.NONE: "",
98
+ ReasoningType.BRIEF:
99
+ "You must also provide a brief (1 sentence) reasoning for your scoring. First reason then score!",
100
+ ReasoningType.LONG:
101
+ "You must also provide a detailed reasoning for your scoring. First reason then score!",
102
+ ReasoningType.COT:
103
+ "You must also provide a reasoning for your scoring . Think step by step in your reasoning. \
104
+ First reason then score!",
105
+ }
106
+
107
+ return self._clean_text(reasoning_map.get(reasoning, ""))
108
+ except Exception as e:
109
+ raise AgentError(f"Error processing reasoning: {str(e)}")
110
+
111
+ def process_examples(self, examples: Union[str, Dict[str, Any], List[Union[str, Dict[str, Any]]]]) -> str:
112
+ """Process examples into a formatted string."""
113
+ try:
114
+ if not examples:
115
+ return ""
116
+
117
+ if not isinstance(examples, list):
118
+ examples = [examples]
119
+
120
+ examples_str = []
121
+ for example in examples:
122
+ if isinstance(example, dict):
123
+ examples_str.append("***" + "".join(f"{k}: {v}\n" for k, v in example.items()))
124
+ elif isinstance(example, str):
125
+ examples_str.append("***" + example)
126
+ else:
127
+ raise ValueError(f"Invalid example type: {type(example)}")
128
+
129
+ return self._clean_text(
130
+ "<<Here is one or more examples of the performance you are expected to have: \n"
131
+ + "".join(examples_str)
132
+ + ">>"
133
+ )
134
+ except Exception as e:
135
+ raise AgentError(f"Error processing examples: {str(e)}")
136
+
137
+ def reset_memory(self) -> None:
138
+ """Reset the agent's memory and cost tracking."""
139
+ try:
140
+ self.memory = []
141
+ self.cost_so_far = 0
142
+ self.identity = {}
143
+ except Exception as e:
144
+ raise AgentError(f"Error resetting memory: {str(e)}")
145
+
146
+ def _clean_text(self, text: str) -> str:
147
+ """Remove extra spaces and blank lines from text."""
148
+ try:
149
+ lines = [line.strip() for line in text.splitlines() if line.strip()]
150
+ return " ".join(" ".join(line.split()) for line in lines)
151
+ except Exception as e:
152
+ raise AgentError(f"Error cleaning text: {str(e)}")
153
+
154
+ async def review_items(self, items: List[str]) -> List[Dict[str, Any]]:
155
+ """Review a list of items asynchronously."""
156
+ raise NotImplementedError("This method must be implemented by subclasses.")
157
+
158
+ async def review_item(self, item: str) -> Dict[str, Any]:
159
+ """Review a single item asynchronously."""
160
+ raise NotImplementedError("This method must be implemented by subclasses.")
@@ -0,0 +1,129 @@
1
+ """Reviewer agent implementation with consistent error handling and type safety."""
2
+
3
+ import asyncio
4
+ from pathlib import Path
5
+ import datetime
6
+ from typing import List, Dict, Any, Optional
7
+ from pydantic import Field
8
+ from .base_agent import BaseAgent, AgentError, ReasoningType
9
+ from tqdm.asyncio import tqdm
10
+ import warnings
11
+
12
+ DEFAULT_MAX_RETRIES = 3
13
+
14
+
15
+ class ScoringReviewer(BaseAgent):
16
+ response_format: Dict[str, Any] = {
17
+ "reasoning": str,
18
+ "score": int,
19
+ }
20
+ scoring_task: Optional[str] = None
21
+ score_set: List[int] = [1, 2]
22
+ scoring_rules: str = "Your scores should follow the defined schema."
23
+ generic_item_prompt: Optional[str] = Field(default=None)
24
+ reasoning: ReasoningType = ReasoningType.BRIEF
25
+ max_retries: int = DEFAULT_MAX_RETRIES
26
+
27
+ class Config:
28
+ arbitrary_types_allowed = True
29
+
30
+ def model_post_init(self, __context: Any) -> None:
31
+ """Initialize after Pydantic model initialization."""
32
+ try:
33
+ assert self.reasoning != ReasoningType.NONE, "Reasoning type cannot be 'none' for ScoreReviewer"
34
+ assert (
35
+ 0 not in self.score_set
36
+ ), "Score set must not contain 0. This value is reserved for uncertain scorings / errors."
37
+ prompt_path = Path(__file__).parent.parent / "generic_prompts" / "review_prompt.txt"
38
+ if not prompt_path.exists():
39
+ raise FileNotFoundError(f"Review prompt template not found at {prompt_path}")
40
+ self.generic_item_prompt = prompt_path.read_text(encoding="utf-8")
41
+ self.setup()
42
+ except Exception as e:
43
+ raise AgentError(f"Error initializing agent: {str(e)}")
44
+
45
+ def setup(self) -> None:
46
+ """Build the agent's identity and configure the provider."""
47
+ try:
48
+ self.system_prompt = self.build_system_prompt()
49
+ self.score_set = str(self.score_set)
50
+ keys_to_replace = ["scoring_task", "score_set", "scoring_rules", "reasoning", "examples"]
51
+
52
+ self.item_prompt = self.build_item_prompt(
53
+ self.generic_item_prompt, {key: getattr(self, key) for key in keys_to_replace}
54
+ )
55
+
56
+ self.identity = {
57
+ "system_prompt": self.system_prompt,
58
+ "item_prompt": self.item_prompt,
59
+ "model_args": self.model_args,
60
+ }
61
+
62
+ if not self.provider:
63
+ raise AgentError("Provider not initialized")
64
+
65
+ self.provider.set_response_format(self.response_format)
66
+ self.provider.system_prompt = self.system_prompt
67
+ except Exception as e:
68
+ raise AgentError(f"Error in setup: {str(e)}")
69
+
70
+ async def review_items(self, items: List[str], tqdm_keywords: dict = None) -> List[Dict[str, Any]]:
71
+ """Review a list of items asynchronously with concurrency control and progress bar."""
72
+ try:
73
+ self.setup()
74
+ semaphore = asyncio.Semaphore(self.max_concurrent_requests)
75
+
76
+ async def limited_review_item(item: str, index: int) -> tuple[int, Dict[str, Any], Dict[str, float]]:
77
+ async with semaphore:
78
+ response, cost = await self.review_item(item)
79
+ return index, response, cost
80
+
81
+ # Building the tqdm desc
82
+ if tqdm_keywords:
83
+ tqdm_desc = f"""{[f'{k}: {v}' for k, v in tqdm_keywords.items()]} - \
84
+ {datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S')}"""
85
+ else:
86
+ tqdm_desc = f"Reviewing {len(items)} items - {datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S')}"
87
+
88
+ # Create tasks with indices
89
+ tasks = [limited_review_item(item, i) for i, item in enumerate(items)]
90
+
91
+ # Collect results with indices
92
+ responses_costs = []
93
+ async for result in tqdm(asyncio.as_completed(tasks), total=len(items), desc=tqdm_desc):
94
+ responses_costs.append(await result)
95
+
96
+ # Sort by original index and separate response and cost
97
+ responses_costs.sort(key=lambda x: x[0]) # Sort by index
98
+ results = []
99
+
100
+ for i, response, cost in responses_costs:
101
+ if isinstance(cost, dict):
102
+ cost = cost["total_cost"]
103
+ self.cost_so_far += cost
104
+ results.append(response)
105
+ self.memory.append(
106
+ {
107
+ "identity": self.identity,
108
+ "item": items[i],
109
+ "response": response,
110
+ "cost": cost,
111
+ "model_args": self.model_args,
112
+ }
113
+ )
114
+
115
+ return results, cost
116
+ except Exception as e:
117
+ raise AgentError(f"Error reviewing items: {str(e)}")
118
+
119
+ async def review_item(self, item: str) -> tuple[Dict[str, Any], Dict[str, float]]:
120
+ """Review a single item asynchronously with error handling."""
121
+ num_tried = 0
122
+ while num_tried < self.max_retries:
123
+ try:
124
+ item_prompt = self.build_item_prompt(self.item_prompt, {"item": item})
125
+ response, cost = await self.provider.get_json_response(item_prompt, **self.model_args)
126
+ return response, cost
127
+ except Exception as e:
128
+ warnings.warn(f"Error reviewing item: {str(e)}. Retrying {num_tried}/{self.max_retries}")
129
+ raise AgentError("Error reviewing item!")
@@ -0,0 +1,15 @@
1
+ Review the input item below and evaluate it against the following criteria:
2
+
3
+ Scoring task: <<${scoring_task}$>>
4
+
5
+ Input item: <<${item}$>>
6
+
7
+ The possible scores for you to choose from are: ${score_set}$.
8
+
9
+ Your scoring should be based on the following rules: <<${scoring_rules}$>>
10
+
11
+ If you are highly uncertain about what score to return, return a score of "0".
12
+
13
+ ${reasoning}$
14
+
15
+ ${examples}$
@@ -0,0 +1,3 @@
1
+ from .litellm_provider import LiteLLMProvider
2
+ from .ollama_provider import OllamaProvider
3
+ from .openai_provider import OpenAIProvider