lattereview 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lattereview-0.1.0/LICENSE +21 -0
- lattereview-0.1.0/PKG-INFO +184 -0
- lattereview-0.1.0/README.md +156 -0
- lattereview-0.1.0/lattereview/__init__.py +0 -0
- lattereview-0.1.0/lattereview/agents/__init__.py +1 -0
- lattereview-0.1.0/lattereview/agents/base_agent.py +160 -0
- lattereview-0.1.0/lattereview/agents/scoring_reviewer.py +129 -0
- lattereview-0.1.0/lattereview/generic_prompts/__init__.py +0 -0
- lattereview-0.1.0/lattereview/generic_prompts/review_prompt.txt +15 -0
- lattereview-0.1.0/lattereview/providers/__init__.py +3 -0
- lattereview-0.1.0/lattereview/providers/base_provider.py +107 -0
- lattereview-0.1.0/lattereview/providers/litellm_provider.py +129 -0
- lattereview-0.1.0/lattereview/providers/ollama_provider.py +169 -0
- lattereview-0.1.0/lattereview/providers/openai_provider.py +121 -0
- lattereview-0.1.0/lattereview/review_workflow.py +202 -0
- lattereview-0.1.0/lattereview.egg-info/PKG-INFO +184 -0
- lattereview-0.1.0/lattereview.egg-info/SOURCES.txt +23 -0
- lattereview-0.1.0/lattereview.egg-info/dependency_links.txt +1 -0
- lattereview-0.1.0/lattereview.egg-info/requires.txt +10 -0
- lattereview-0.1.0/lattereview.egg-info/top_level.txt +2 -0
- lattereview-0.1.0/notebooks/__init__.py +0 -0
- lattereview-0.1.0/pyproject.toml +7 -0
- lattereview-0.1.0/setup.cfg +9 -0
- lattereview-0.1.0/setup.py +41 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Pouria Rouzrokh
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: lattereview
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: A framework for multi-agent review workflows using large language models
|
|
5
|
+
Home-page: https://github.com/PouriaRouzrokh/LatteReview
|
|
6
|
+
Author: Pouria Rouzrokh
|
|
7
|
+
Author-email: po.rouzrokh@gmail.com
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: litellm>=1.55.2
|
|
19
|
+
Requires-Dist: nest-asyncio>=1.6.0
|
|
20
|
+
Requires-Dist: ollama>=0.4.4
|
|
21
|
+
Requires-Dist: openai>=1.57.4
|
|
22
|
+
Requires-Dist: pandas>=2.2.3
|
|
23
|
+
Requires-Dist: pydantic>=2.10.3
|
|
24
|
+
Requires-Dist: python-dotenv>=1.0.1
|
|
25
|
+
Requires-Dist: tokencost>=0.1.17
|
|
26
|
+
Requires-Dist: tqdm>=4.67.1
|
|
27
|
+
Requires-Dist: openpyxl>=3.1.5
|
|
28
|
+
|
|
29
|
+
# LatteReview 🤖☕
|
|
30
|
+
|
|
31
|
+
[](https://badge.fury.io/py/lattereview)
|
|
32
|
+
[](https://opensource.org/licenses/MIT)
|
|
33
|
+
[](https://www.python.org/downloads/)
|
|
34
|
+
[](https://github.com/psf/black)
|
|
35
|
+
[](https://github.com/prouzrokh/lattereview)
|
|
36
|
+
|
|
37
|
+
LatteReview is a powerful Python package designed to automate academic literature review processes through AI-powered agents. Just like enjoying a cup of latte ☕, reviewing numerous research articles should be a pleasant, efficient experience that doesn't consume your entire day!
|
|
38
|
+
|
|
39
|
+
## 🎯 Key Features
|
|
40
|
+
|
|
41
|
+
- Multi-agent review system with customizable roles and expertise
|
|
42
|
+
- Support for multiple review rounds with hierarchical decision-making
|
|
43
|
+
- Flexible model integration (OpenAI, Gemini, Claude, Groq, local models via Ollama)
|
|
44
|
+
- Asynchronous processing for high-performance batch reviews
|
|
45
|
+
- Structured output format with detailed scoring and reasoning
|
|
46
|
+
- Comprehensive cost tracking and memory management
|
|
47
|
+
- Extensible architecture for custom review workflows
|
|
48
|
+
|
|
49
|
+
## 🛠️ Installation
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install lattereview
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
## 🚀 Quick Start
|
|
56
|
+
|
|
57
|
+
Here's a simple example of how to set up a review workflow with two primary reviewers and an expert reviewer for conflict resolution:
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
from lattereview.providers import LiteLLMProvider
|
|
61
|
+
from lattereview.agents import ScoringReviewer
|
|
62
|
+
from lattereview.review_workflow import ReviewWorkflow
|
|
63
|
+
import pandas as pd
|
|
64
|
+
import asyncio
|
|
65
|
+
|
|
66
|
+
# First Reviewer: Conservative approach
|
|
67
|
+
reviewer1 = ScoringReviewer(
|
|
68
|
+
provider=LiteLLMProvider(model="gpt-4o-mini"),
|
|
69
|
+
name="Alice",
|
|
70
|
+
backstory="a radiologist with expertise in systematic reviews",
|
|
71
|
+
input_description="article title and abstract",
|
|
72
|
+
scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
|
|
73
|
+
score_set=[1, 2, 3, 4, 5],
|
|
74
|
+
scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
|
|
75
|
+
model_args={"temperature": 0.1} # Low temperature for consistent scoring
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
# Second Reviewer: More exploratory approach
|
|
79
|
+
reviewer2 = ScoringReviewer(
|
|
80
|
+
provider=LiteLLMProvider(model="gemini/gemini-1.5-flash"),
|
|
81
|
+
name="Bob",
|
|
82
|
+
backstory="a computer scientist specializing in medical AI",
|
|
83
|
+
input_description="article title and abstract",
|
|
84
|
+
scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
|
|
85
|
+
score_set=[1, 2, 3, 4, 5],
|
|
86
|
+
scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
|
|
87
|
+
model_args={"temperature": 0.8} # Higher temperature for creative interpretation
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# Expert Reviewer: Resolves disagreements
|
|
91
|
+
expert = ScoringReviewer(
|
|
92
|
+
provider=LiteLLMProvider(model="gpt-4o"),
|
|
93
|
+
name="Carol",
|
|
94
|
+
backstory="a professor of AI in medical imaging",
|
|
95
|
+
input_description="article title, abstract, and previous reviews",
|
|
96
|
+
scoring_task="Review Alice and Bob's relevance assessments of this article to AI in radiology",
|
|
97
|
+
score_set=[1, 2],
|
|
98
|
+
scoring_rules='Score 1 if you agree with Alice\'s assessment, 2 if you agree with Bob\'s assessment',
|
|
99
|
+
model_args={"temperature": 0.1} # Low temperature for careful judgment
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
# Define the multi-round review workflow
|
|
103
|
+
workflow = ReviewWorkflow([
|
|
104
|
+
{
|
|
105
|
+
"round": 'A', # First round: Initial review by both reviewers
|
|
106
|
+
"reviewers": [reviewer1, reviewer2],
|
|
107
|
+
"inputs": ["title", "abstract"]
|
|
108
|
+
},
|
|
109
|
+
{
|
|
110
|
+
"round": 'B', # Second round: Expert reviews only disagreements
|
|
111
|
+
"reviewers": [expert],
|
|
112
|
+
"inputs": ["title", "abstract", "round-A_Alice_output", "round-A_Bob_output"],
|
|
113
|
+
"filter": lambda row: row["round-A_Alice_score"] != row["round-A_Bob_score"]
|
|
114
|
+
}
|
|
115
|
+
])
|
|
116
|
+
|
|
117
|
+
# Run the workflow on your data
|
|
118
|
+
data = pd.read_excel("articles.xlsx")
|
|
119
|
+
results = asyncio.run(workflow(data))
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## 🔌 Model Support
|
|
123
|
+
|
|
124
|
+
LatteReview offers flexible model integration through multiple providers:
|
|
125
|
+
|
|
126
|
+
- **LiteLLMProvider** (Recommended): Supports OpenAI, Anthropic (Claude), Gemini, Groq, and more
|
|
127
|
+
- **OpenAIProvider**: Direct integration with OpenAI and Gemini APIs
|
|
128
|
+
- **OllamaProvider**: Optimized for local models via Ollama
|
|
129
|
+
|
|
130
|
+
Note: Models should support async operations and structured JSON outputs for optimal performance.
|
|
131
|
+
|
|
132
|
+
## 📖 Documentation
|
|
133
|
+
|
|
134
|
+
Full documentation and API reference will be available in our upcoming preprint paper. [Link to be added]
|
|
135
|
+
|
|
136
|
+
## 🛣️ Roadmap
|
|
137
|
+
|
|
138
|
+
- [ ] Development of `AbstractionReviewer` class for automated paper summarization
|
|
139
|
+
- [ ] Support for image-based inputs and multimodal analysis
|
|
140
|
+
- [ ] Development of a no-code web application
|
|
141
|
+
- [ ] Integration of RAG (Retrieval-Augmented Generation) tools
|
|
142
|
+
- [ ] Addition of graph-based analysis tools
|
|
143
|
+
- [ ] Enhanced visualization capabilities
|
|
144
|
+
- [ ] Support for additional model providers
|
|
145
|
+
|
|
146
|
+
## 👨💻 Author
|
|
147
|
+
|
|
148
|
+
**Pouria Rouzrokh, MD, MPH, MHPE**
|
|
149
|
+
Medical Practitioner and Machine Learning Engineer
|
|
150
|
+
Incoming Radiology Resident @Yale University
|
|
151
|
+
Former Data Scientist @Mayo Clinic AI Lab
|
|
152
|
+
|
|
153
|
+
Find my work:
|
|
154
|
+
[](https://twitter.com/prouzrokh)
|
|
155
|
+
[](https://linkedin.com/in/pouria-rouzrokh)
|
|
156
|
+
[](https://scholar.google.com/citations?user=Ksv9I0sAAAAJ&hl=en)
|
|
157
|
+
[](mailto:po.rouzrokh@gmail.com)
|
|
158
|
+
|
|
159
|
+
## ❤️ Support LatteReview
|
|
160
|
+
|
|
161
|
+
If you find LatteReview helpful in your research or work, consider supporting its continued development. Since we're already sharing a virtual coffee break while reviewing papers, maybe you'd like to treat me to a real one? ☕ 😊
|
|
162
|
+
|
|
163
|
+
### Ways to Support:
|
|
164
|
+
|
|
165
|
+
- [Treat me to a coffee](http://ko-fi.com/pouriarouzrokh) on Ko-fi ☕
|
|
166
|
+
- [Star the repository](https://github.com/PouriaRouzrokh/LatteReview) to help others discover the project
|
|
167
|
+
- Submit bug reports, feature requests, or contribute code
|
|
168
|
+
- Share your experience using LatteReview in your research
|
|
169
|
+
|
|
170
|
+
## 📜 License
|
|
171
|
+
|
|
172
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
173
|
+
|
|
174
|
+
## 🤝 Contributing
|
|
175
|
+
|
|
176
|
+
We welcome contributions! Please feel free to submit a Pull Request.
|
|
177
|
+
|
|
178
|
+
## 📚 Citation
|
|
179
|
+
|
|
180
|
+
If you use LatteReview in your research, please cite our paper:
|
|
181
|
+
|
|
182
|
+
```bibtex
|
|
183
|
+
# Preprint citation to be added
|
|
184
|
+
```
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
# LatteReview 🤖☕
|
|
2
|
+
|
|
3
|
+
[](https://badge.fury.io/py/lattereview)
|
|
4
|
+
[](https://opensource.org/licenses/MIT)
|
|
5
|
+
[](https://www.python.org/downloads/)
|
|
6
|
+
[](https://github.com/psf/black)
|
|
7
|
+
[](https://github.com/prouzrokh/lattereview)
|
|
8
|
+
|
|
9
|
+
LatteReview is a powerful Python package designed to automate academic literature review processes through AI-powered agents. Just like enjoying a cup of latte ☕, reviewing numerous research articles should be a pleasant, efficient experience that doesn't consume your entire day!
|
|
10
|
+
|
|
11
|
+
## 🎯 Key Features
|
|
12
|
+
|
|
13
|
+
- Multi-agent review system with customizable roles and expertise
|
|
14
|
+
- Support for multiple review rounds with hierarchical decision-making
|
|
15
|
+
- Flexible model integration (OpenAI, Gemini, Claude, Groq, local models via Ollama)
|
|
16
|
+
- Asynchronous processing for high-performance batch reviews
|
|
17
|
+
- Structured output format with detailed scoring and reasoning
|
|
18
|
+
- Comprehensive cost tracking and memory management
|
|
19
|
+
- Extensible architecture for custom review workflows
|
|
20
|
+
|
|
21
|
+
## 🛠️ Installation
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install lattereview
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## 🚀 Quick Start
|
|
28
|
+
|
|
29
|
+
Here's a simple example of how to set up a review workflow with two primary reviewers and an expert reviewer for conflict resolution:
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from lattereview.providers import LiteLLMProvider
|
|
33
|
+
from lattereview.agents import ScoringReviewer
|
|
34
|
+
from lattereview.review_workflow import ReviewWorkflow
|
|
35
|
+
import pandas as pd
|
|
36
|
+
import asyncio
|
|
37
|
+
|
|
38
|
+
# First Reviewer: Conservative approach
|
|
39
|
+
reviewer1 = ScoringReviewer(
|
|
40
|
+
provider=LiteLLMProvider(model="gpt-4o-mini"),
|
|
41
|
+
name="Alice",
|
|
42
|
+
backstory="a radiologist with expertise in systematic reviews",
|
|
43
|
+
input_description="article title and abstract",
|
|
44
|
+
scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
|
|
45
|
+
score_set=[1, 2, 3, 4, 5],
|
|
46
|
+
scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
|
|
47
|
+
model_args={"temperature": 0.1} # Low temperature for consistent scoring
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# Second Reviewer: More exploratory approach
|
|
51
|
+
reviewer2 = ScoringReviewer(
|
|
52
|
+
provider=LiteLLMProvider(model="gemini/gemini-1.5-flash"),
|
|
53
|
+
name="Bob",
|
|
54
|
+
backstory="a computer scientist specializing in medical AI",
|
|
55
|
+
input_description="article title and abstract",
|
|
56
|
+
scoring_task="Evaluate how relevant the article is to artificial intelligence applications in radiology",
|
|
57
|
+
score_set=[1, 2, 3, 4, 5],
|
|
58
|
+
scoring_rules="Rate the relevance on a scale of 1 to 5, where 1 means the article is not at all relevant to AI in radiology, and 5 means it directly focuses on AI applications in radiology.",
|
|
59
|
+
model_args={"temperature": 0.8} # Higher temperature for creative interpretation
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
# Expert Reviewer: Resolves disagreements
|
|
63
|
+
expert = ScoringReviewer(
|
|
64
|
+
provider=LiteLLMProvider(model="gpt-4o"),
|
|
65
|
+
name="Carol",
|
|
66
|
+
backstory="a professor of AI in medical imaging",
|
|
67
|
+
input_description="article title, abstract, and previous reviews",
|
|
68
|
+
scoring_task="Review Alice and Bob's relevance assessments of this article to AI in radiology",
|
|
69
|
+
score_set=[1, 2],
|
|
70
|
+
scoring_rules='Score 1 if you agree with Alice\'s assessment, 2 if you agree with Bob\'s assessment',
|
|
71
|
+
model_args={"temperature": 0.1} # Low temperature for careful judgment
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
# Define the multi-round review workflow
|
|
75
|
+
workflow = ReviewWorkflow([
|
|
76
|
+
{
|
|
77
|
+
"round": 'A', # First round: Initial review by both reviewers
|
|
78
|
+
"reviewers": [reviewer1, reviewer2],
|
|
79
|
+
"inputs": ["title", "abstract"]
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
"round": 'B', # Second round: Expert reviews only disagreements
|
|
83
|
+
"reviewers": [expert],
|
|
84
|
+
"inputs": ["title", "abstract", "round-A_Alice_output", "round-A_Bob_output"],
|
|
85
|
+
"filter": lambda row: row["round-A_Alice_score"] != row["round-A_Bob_score"]
|
|
86
|
+
}
|
|
87
|
+
])
|
|
88
|
+
|
|
89
|
+
# Run the workflow on your data
|
|
90
|
+
data = pd.read_excel("articles.xlsx")
|
|
91
|
+
results = asyncio.run(workflow(data))
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## 🔌 Model Support
|
|
95
|
+
|
|
96
|
+
LatteReview offers flexible model integration through multiple providers:
|
|
97
|
+
|
|
98
|
+
- **LiteLLMProvider** (Recommended): Supports OpenAI, Anthropic (Claude), Gemini, Groq, and more
|
|
99
|
+
- **OpenAIProvider**: Direct integration with OpenAI and Gemini APIs
|
|
100
|
+
- **OllamaProvider**: Optimized for local models via Ollama
|
|
101
|
+
|
|
102
|
+
Note: Models should support async operations and structured JSON outputs for optimal performance.
|
|
103
|
+
|
|
104
|
+
## 📖 Documentation
|
|
105
|
+
|
|
106
|
+
Full documentation and API reference will be available in our upcoming preprint paper. [Link to be added]
|
|
107
|
+
|
|
108
|
+
## 🛣️ Roadmap
|
|
109
|
+
|
|
110
|
+
- [ ] Development of `AbstractionReviewer` class for automated paper summarization
|
|
111
|
+
- [ ] Support for image-based inputs and multimodal analysis
|
|
112
|
+
- [ ] Development of a no-code web application
|
|
113
|
+
- [ ] Integration of RAG (Retrieval-Augmented Generation) tools
|
|
114
|
+
- [ ] Addition of graph-based analysis tools
|
|
115
|
+
- [ ] Enhanced visualization capabilities
|
|
116
|
+
- [ ] Support for additional model providers
|
|
117
|
+
|
|
118
|
+
## 👨💻 Author
|
|
119
|
+
|
|
120
|
+
**Pouria Rouzrokh, MD, MPH, MHPE**
|
|
121
|
+
Medical Practitioner and Machine Learning Engineer
|
|
122
|
+
Incoming Radiology Resident @Yale University
|
|
123
|
+
Former Data Scientist @Mayo Clinic AI Lab
|
|
124
|
+
|
|
125
|
+
Find my work:
|
|
126
|
+
[](https://twitter.com/prouzrokh)
|
|
127
|
+
[](https://linkedin.com/in/pouria-rouzrokh)
|
|
128
|
+
[](https://scholar.google.com/citations?user=Ksv9I0sAAAAJ&hl=en)
|
|
129
|
+
[](mailto:po.rouzrokh@gmail.com)
|
|
130
|
+
|
|
131
|
+
## ❤️ Support LatteReview
|
|
132
|
+
|
|
133
|
+
If you find LatteReview helpful in your research or work, consider supporting its continued development. Since we're already sharing a virtual coffee break while reviewing papers, maybe you'd like to treat me to a real one? ☕ 😊
|
|
134
|
+
|
|
135
|
+
### Ways to Support:
|
|
136
|
+
|
|
137
|
+
- [Treat me to a coffee](http://ko-fi.com/pouriarouzrokh) on Ko-fi ☕
|
|
138
|
+
- [Star the repository](https://github.com/PouriaRouzrokh/LatteReview) to help others discover the project
|
|
139
|
+
- Submit bug reports, feature requests, or contribute code
|
|
140
|
+
- Share your experience using LatteReview in your research
|
|
141
|
+
|
|
142
|
+
## 📜 License
|
|
143
|
+
|
|
144
|
+
This project is licensed under the MIT License - see the [LICENSE](LICENSE) file for details.
|
|
145
|
+
|
|
146
|
+
## 🤝 Contributing
|
|
147
|
+
|
|
148
|
+
We welcome contributions! Please feel free to submit a Pull Request.
|
|
149
|
+
|
|
150
|
+
## 📚 Citation
|
|
151
|
+
|
|
152
|
+
If you use LatteReview in your research, please cite our paper:
|
|
153
|
+
|
|
154
|
+
```bibtex
|
|
155
|
+
# Preprint citation to be added
|
|
156
|
+
```
|
|
File without changes
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .scoring_reviewer import ScoringReviewer
|
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
"""Base agent class with consistent error handling and type safety."""
|
|
2
|
+
|
|
3
|
+
from typing import List, Optional, Dict, Any, Union
|
|
4
|
+
from enum import Enum
|
|
5
|
+
from pydantic import BaseModel, Field
|
|
6
|
+
|
|
7
|
+
DEFAULT_CONCURRENT_REQUESTS = 20
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ReasoningType(Enum):
|
|
11
|
+
"""Enumeration for reasoning types."""
|
|
12
|
+
|
|
13
|
+
NONE = "none"
|
|
14
|
+
BRIEF = "brief"
|
|
15
|
+
LONG = "long"
|
|
16
|
+
COT = "cot"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class AgentError(Exception):
|
|
20
|
+
"""Base exception for agent-related errors."""
|
|
21
|
+
|
|
22
|
+
pass
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class BaseAgent(BaseModel):
|
|
26
|
+
response_format: Dict[str, Any]
|
|
27
|
+
provider: Optional[Any] = None
|
|
28
|
+
model_args: Dict[str, Any] = Field(default_factory=dict)
|
|
29
|
+
max_concurrent_requests: int = DEFAULT_CONCURRENT_REQUESTS
|
|
30
|
+
name: str = "BaseAgent"
|
|
31
|
+
backstory: str = "a generic base agent"
|
|
32
|
+
input_description: str = "article title/abstract"
|
|
33
|
+
examples: Union[str, List[Union[str, Dict[str, Any]]]] = None
|
|
34
|
+
reasoning: ReasoningType = ReasoningType.BRIEF
|
|
35
|
+
system_prompt: Optional[str] = None
|
|
36
|
+
item_prompt: Optional[str] = None
|
|
37
|
+
cost_so_far: float = 0
|
|
38
|
+
memory: List[Dict[str, Any]] = []
|
|
39
|
+
identity: Dict[str, Any] = {}
|
|
40
|
+
|
|
41
|
+
def __init__(self, **data: Any) -> None:
|
|
42
|
+
"""Initialize the base agent with error handling."""
|
|
43
|
+
try:
|
|
44
|
+
super().__init__(**data)
|
|
45
|
+
if isinstance(self.reasoning, str):
|
|
46
|
+
self.reasoning = ReasoningType(self.reasoning.lower())
|
|
47
|
+
if self.reasoning == ReasoningType.NONE:
|
|
48
|
+
self.response_format.pop("reasoning", None)
|
|
49
|
+
self.setup()
|
|
50
|
+
except Exception as e:
|
|
51
|
+
raise AgentError(f"Error initializing agent: {str(e)}")
|
|
52
|
+
|
|
53
|
+
def setup(self) -> None:
|
|
54
|
+
"""Setup the agent before use."""
|
|
55
|
+
raise NotImplementedError("This method must be implemented by subclasses.")
|
|
56
|
+
|
|
57
|
+
def build_system_prompt(self) -> str:
|
|
58
|
+
"""Build the system prompt for the agent."""
|
|
59
|
+
try:
|
|
60
|
+
return self._clean_text(
|
|
61
|
+
f"""
|
|
62
|
+
Your name is <<{self.name}>> and you are <<{self.backstory}>>.
|
|
63
|
+
Your task is to review input itmes with the following description: <<{self.input_description}>>.
|
|
64
|
+
Your final output should have the following keys: \
|
|
65
|
+
{", ".join(f"{k} ({v})" for k, v in self.response_format.items())}.
|
|
66
|
+
"""
|
|
67
|
+
)
|
|
68
|
+
except Exception as e:
|
|
69
|
+
raise AgentError(f"Error building system prompt: {str(e)}")
|
|
70
|
+
|
|
71
|
+
def build_item_prompt(self, base_prompt: str, item_dict: Dict[str, Any]) -> str:
|
|
72
|
+
"""Build the item prompt with variable substitution."""
|
|
73
|
+
try:
|
|
74
|
+
prompt = base_prompt
|
|
75
|
+
if "examples" in item_dict:
|
|
76
|
+
item_dict["examples"] = self.process_examples(item_dict["examples"])
|
|
77
|
+
if "reasoning" in item_dict:
|
|
78
|
+
item_dict["reasoning"] = self.process_reasoning(item_dict["reasoning"])
|
|
79
|
+
|
|
80
|
+
for key, value in item_dict.items():
|
|
81
|
+
if value is not None:
|
|
82
|
+
prompt = prompt.replace(f"${{{key}}}$", str(value))
|
|
83
|
+
else:
|
|
84
|
+
prompt = prompt.replace(f"${{{key}}}$", "")
|
|
85
|
+
|
|
86
|
+
return self._clean_text(prompt)
|
|
87
|
+
except Exception as e:
|
|
88
|
+
raise AgentError(f"Error building item prompt: {str(e)}")
|
|
89
|
+
|
|
90
|
+
def process_reasoning(self, reasoning: Union[str, ReasoningType]) -> str:
|
|
91
|
+
"""Process the reasoning type into a prompt string."""
|
|
92
|
+
try:
|
|
93
|
+
if isinstance(reasoning, str):
|
|
94
|
+
reasoning = ReasoningType(reasoning.lower())
|
|
95
|
+
|
|
96
|
+
reasoning_map = {
|
|
97
|
+
ReasoningType.NONE: "",
|
|
98
|
+
ReasoningType.BRIEF:
|
|
99
|
+
"You must also provide a brief (1 sentence) reasoning for your scoring. First reason then score!",
|
|
100
|
+
ReasoningType.LONG:
|
|
101
|
+
"You must also provide a detailed reasoning for your scoring. First reason then score!",
|
|
102
|
+
ReasoningType.COT:
|
|
103
|
+
"You must also provide a reasoning for your scoring . Think step by step in your reasoning. \
|
|
104
|
+
First reason then score!",
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
return self._clean_text(reasoning_map.get(reasoning, ""))
|
|
108
|
+
except Exception as e:
|
|
109
|
+
raise AgentError(f"Error processing reasoning: {str(e)}")
|
|
110
|
+
|
|
111
|
+
def process_examples(self, examples: Union[str, Dict[str, Any], List[Union[str, Dict[str, Any]]]]) -> str:
|
|
112
|
+
"""Process examples into a formatted string."""
|
|
113
|
+
try:
|
|
114
|
+
if not examples:
|
|
115
|
+
return ""
|
|
116
|
+
|
|
117
|
+
if not isinstance(examples, list):
|
|
118
|
+
examples = [examples]
|
|
119
|
+
|
|
120
|
+
examples_str = []
|
|
121
|
+
for example in examples:
|
|
122
|
+
if isinstance(example, dict):
|
|
123
|
+
examples_str.append("***" + "".join(f"{k}: {v}\n" for k, v in example.items()))
|
|
124
|
+
elif isinstance(example, str):
|
|
125
|
+
examples_str.append("***" + example)
|
|
126
|
+
else:
|
|
127
|
+
raise ValueError(f"Invalid example type: {type(example)}")
|
|
128
|
+
|
|
129
|
+
return self._clean_text(
|
|
130
|
+
"<<Here is one or more examples of the performance you are expected to have: \n"
|
|
131
|
+
+ "".join(examples_str)
|
|
132
|
+
+ ">>"
|
|
133
|
+
)
|
|
134
|
+
except Exception as e:
|
|
135
|
+
raise AgentError(f"Error processing examples: {str(e)}")
|
|
136
|
+
|
|
137
|
+
def reset_memory(self) -> None:
|
|
138
|
+
"""Reset the agent's memory and cost tracking."""
|
|
139
|
+
try:
|
|
140
|
+
self.memory = []
|
|
141
|
+
self.cost_so_far = 0
|
|
142
|
+
self.identity = {}
|
|
143
|
+
except Exception as e:
|
|
144
|
+
raise AgentError(f"Error resetting memory: {str(e)}")
|
|
145
|
+
|
|
146
|
+
def _clean_text(self, text: str) -> str:
|
|
147
|
+
"""Remove extra spaces and blank lines from text."""
|
|
148
|
+
try:
|
|
149
|
+
lines = [line.strip() for line in text.splitlines() if line.strip()]
|
|
150
|
+
return " ".join(" ".join(line.split()) for line in lines)
|
|
151
|
+
except Exception as e:
|
|
152
|
+
raise AgentError(f"Error cleaning text: {str(e)}")
|
|
153
|
+
|
|
154
|
+
async def review_items(self, items: List[str]) -> List[Dict[str, Any]]:
|
|
155
|
+
"""Review a list of items asynchronously."""
|
|
156
|
+
raise NotImplementedError("This method must be implemented by subclasses.")
|
|
157
|
+
|
|
158
|
+
async def review_item(self, item: str) -> Dict[str, Any]:
|
|
159
|
+
"""Review a single item asynchronously."""
|
|
160
|
+
raise NotImplementedError("This method must be implemented by subclasses.")
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Reviewer agent implementation with consistent error handling and type safety."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import datetime
|
|
6
|
+
from typing import List, Dict, Any, Optional
|
|
7
|
+
from pydantic import Field
|
|
8
|
+
from .base_agent import BaseAgent, AgentError, ReasoningType
|
|
9
|
+
from tqdm.asyncio import tqdm
|
|
10
|
+
import warnings
|
|
11
|
+
|
|
12
|
+
DEFAULT_MAX_RETRIES = 3
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class ScoringReviewer(BaseAgent):
|
|
16
|
+
response_format: Dict[str, Any] = {
|
|
17
|
+
"reasoning": str,
|
|
18
|
+
"score": int,
|
|
19
|
+
}
|
|
20
|
+
scoring_task: Optional[str] = None
|
|
21
|
+
score_set: List[int] = [1, 2]
|
|
22
|
+
scoring_rules: str = "Your scores should follow the defined schema."
|
|
23
|
+
generic_item_prompt: Optional[str] = Field(default=None)
|
|
24
|
+
reasoning: ReasoningType = ReasoningType.BRIEF
|
|
25
|
+
max_retries: int = DEFAULT_MAX_RETRIES
|
|
26
|
+
|
|
27
|
+
class Config:
|
|
28
|
+
arbitrary_types_allowed = True
|
|
29
|
+
|
|
30
|
+
def model_post_init(self, __context: Any) -> None:
|
|
31
|
+
"""Initialize after Pydantic model initialization."""
|
|
32
|
+
try:
|
|
33
|
+
assert self.reasoning != ReasoningType.NONE, "Reasoning type cannot be 'none' for ScoreReviewer"
|
|
34
|
+
assert (
|
|
35
|
+
0 not in self.score_set
|
|
36
|
+
), "Score set must not contain 0. This value is reserved for uncertain scorings / errors."
|
|
37
|
+
prompt_path = Path(__file__).parent.parent / "generic_prompts" / "review_prompt.txt"
|
|
38
|
+
if not prompt_path.exists():
|
|
39
|
+
raise FileNotFoundError(f"Review prompt template not found at {prompt_path}")
|
|
40
|
+
self.generic_item_prompt = prompt_path.read_text(encoding="utf-8")
|
|
41
|
+
self.setup()
|
|
42
|
+
except Exception as e:
|
|
43
|
+
raise AgentError(f"Error initializing agent: {str(e)}")
|
|
44
|
+
|
|
45
|
+
def setup(self) -> None:
|
|
46
|
+
"""Build the agent's identity and configure the provider."""
|
|
47
|
+
try:
|
|
48
|
+
self.system_prompt = self.build_system_prompt()
|
|
49
|
+
self.score_set = str(self.score_set)
|
|
50
|
+
keys_to_replace = ["scoring_task", "score_set", "scoring_rules", "reasoning", "examples"]
|
|
51
|
+
|
|
52
|
+
self.item_prompt = self.build_item_prompt(
|
|
53
|
+
self.generic_item_prompt, {key: getattr(self, key) for key in keys_to_replace}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
self.identity = {
|
|
57
|
+
"system_prompt": self.system_prompt,
|
|
58
|
+
"item_prompt": self.item_prompt,
|
|
59
|
+
"model_args": self.model_args,
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
if not self.provider:
|
|
63
|
+
raise AgentError("Provider not initialized")
|
|
64
|
+
|
|
65
|
+
self.provider.set_response_format(self.response_format)
|
|
66
|
+
self.provider.system_prompt = self.system_prompt
|
|
67
|
+
except Exception as e:
|
|
68
|
+
raise AgentError(f"Error in setup: {str(e)}")
|
|
69
|
+
|
|
70
|
+
async def review_items(self, items: List[str], tqdm_keywords: dict = None) -> List[Dict[str, Any]]:
|
|
71
|
+
"""Review a list of items asynchronously with concurrency control and progress bar."""
|
|
72
|
+
try:
|
|
73
|
+
self.setup()
|
|
74
|
+
semaphore = asyncio.Semaphore(self.max_concurrent_requests)
|
|
75
|
+
|
|
76
|
+
async def limited_review_item(item: str, index: int) -> tuple[int, Dict[str, Any], Dict[str, float]]:
|
|
77
|
+
async with semaphore:
|
|
78
|
+
response, cost = await self.review_item(item)
|
|
79
|
+
return index, response, cost
|
|
80
|
+
|
|
81
|
+
# Building the tqdm desc
|
|
82
|
+
if tqdm_keywords:
|
|
83
|
+
tqdm_desc = f"""{[f'{k}: {v}' for k, v in tqdm_keywords.items()]} - \
|
|
84
|
+
{datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S')}"""
|
|
85
|
+
else:
|
|
86
|
+
tqdm_desc = f"Reviewing {len(items)} items - {datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S')}"
|
|
87
|
+
|
|
88
|
+
# Create tasks with indices
|
|
89
|
+
tasks = [limited_review_item(item, i) for i, item in enumerate(items)]
|
|
90
|
+
|
|
91
|
+
# Collect results with indices
|
|
92
|
+
responses_costs = []
|
|
93
|
+
async for result in tqdm(asyncio.as_completed(tasks), total=len(items), desc=tqdm_desc):
|
|
94
|
+
responses_costs.append(await result)
|
|
95
|
+
|
|
96
|
+
# Sort by original index and separate response and cost
|
|
97
|
+
responses_costs.sort(key=lambda x: x[0]) # Sort by index
|
|
98
|
+
results = []
|
|
99
|
+
|
|
100
|
+
for i, response, cost in responses_costs:
|
|
101
|
+
if isinstance(cost, dict):
|
|
102
|
+
cost = cost["total_cost"]
|
|
103
|
+
self.cost_so_far += cost
|
|
104
|
+
results.append(response)
|
|
105
|
+
self.memory.append(
|
|
106
|
+
{
|
|
107
|
+
"identity": self.identity,
|
|
108
|
+
"item": items[i],
|
|
109
|
+
"response": response,
|
|
110
|
+
"cost": cost,
|
|
111
|
+
"model_args": self.model_args,
|
|
112
|
+
}
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
return results, cost
|
|
116
|
+
except Exception as e:
|
|
117
|
+
raise AgentError(f"Error reviewing items: {str(e)}")
|
|
118
|
+
|
|
119
|
+
async def review_item(self, item: str) -> tuple[Dict[str, Any], Dict[str, float]]:
|
|
120
|
+
"""Review a single item asynchronously with error handling."""
|
|
121
|
+
num_tried = 0
|
|
122
|
+
while num_tried < self.max_retries:
|
|
123
|
+
try:
|
|
124
|
+
item_prompt = self.build_item_prompt(self.item_prompt, {"item": item})
|
|
125
|
+
response, cost = await self.provider.get_json_response(item_prompt, **self.model_args)
|
|
126
|
+
return response, cost
|
|
127
|
+
except Exception as e:
|
|
128
|
+
warnings.warn(f"Error reviewing item: {str(e)}. Retrying {num_tried}/{self.max_retries}")
|
|
129
|
+
raise AgentError("Error reviewing item!")
|
|
File without changes
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
Review the input item below and evaluate it against the following criteria:
|
|
2
|
+
|
|
3
|
+
Scoring task: <<${scoring_task}$>>
|
|
4
|
+
|
|
5
|
+
Input item: <<${item}$>>
|
|
6
|
+
|
|
7
|
+
The possible scores for you to choose from are: ${score_set}$.
|
|
8
|
+
|
|
9
|
+
Your scoring should be based on the following rules: <<${scoring_rules}$>>
|
|
10
|
+
|
|
11
|
+
If you are highly uncertain about what score to return, return a score of "0".
|
|
12
|
+
|
|
13
|
+
${reasoning}$
|
|
14
|
+
|
|
15
|
+
${examples}$
|