code-review-ai-cli 1.1.0__tar.gz → 1.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/PKG-INFO +113 -7
  2. code_review_ai_cli-1.3.0/README.md +289 -0
  3. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/PKG-INFO +113 -7
  4. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/SOURCES.txt +3 -1
  5. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/pyproject.toml +1 -1
  6. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/ai_review.py +779 -93
  7. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/config.py +20 -4
  8. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/formatter.py +3 -3
  9. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/git_utils.py +112 -2
  10. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/llm_client.py +448 -72
  11. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/prompts/config.yaml.template +17 -5
  12. code_review_ai_cli-1.3.0/src/prompts/review_context.example.md +59 -0
  13. code_review_ai_cli-1.3.0/src/rag_engine.py +162 -0
  14. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/tfs_client.py +290 -202
  15. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/usage_tracker.py +10 -0
  16. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_ai_review.py +302 -41
  17. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_config.py +6 -2
  18. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_git_utils.py +153 -1
  19. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_llm_client.py +147 -13
  20. code_review_ai_cli-1.3.0/tests/test_rag_engine.py +250 -0
  21. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_tfs_client.py +253 -75
  22. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_usage_tracker.py +9 -0
  23. code_review_ai_cli-1.1.0/README.md +0 -183
  24. code_review_ai_cli-1.1.0/src/prompts/review_prompt.md.template +0 -42
  25. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/dependency_links.txt +0 -0
  26. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/entry_points.txt +0 -0
  27. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/requires.txt +0 -0
  28. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/code_review_ai_cli.egg-info/top_level.txt +0 -0
  29. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/setup.cfg +0 -0
  30. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/src/__init__.py +0 -0
  31. {code_review_ai_cli-1.1.0 → code_review_ai_cli-1.3.0}/tests/test_formatter.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: code-review-ai-cli
3
- Version: 1.1.0
3
+ Version: 1.3.0
4
4
  Summary: Automated AI-powered code review CLI for Azure DevOps / TFS Pull Requests
5
5
  License: MIT
6
6
  Keywords: code-review,ai,azure-devops,tfs,pull-request,llm
@@ -48,10 +48,48 @@ Automated code review tool with Pull Request integration for Azure DevOps/TFS an
48
48
  - **Multiple LLM Providers** — OpenAI, Azure OpenAI, Gemini, Claude, Ollama, GitHub Copilot, AWS Bedrock
49
49
  - **Smart Filtering** — Filter by file extensions, limit diff size
50
50
  - **Project-aware PR Context** — Sends repository and linked work item context while restricting findings to modified PR lines
51
- - **Customizable Prompts** — Markdown-based review guidelines
51
+ - **RAG Context** — Enriches the review with related code snippets found via `git grep` in the local repository
52
+ - **Single Reviewer Context** — One Markdown context file with local override support
52
53
  - **Usage Tracking** — Store per-PR token usage and optional cost estimates
53
54
  - **Interactive CLI** — Menu-driven selection and confirmation
54
55
 
56
+ ## Local Repository Requirement
57
+
58
+ > **The CLI must be run from inside the local clone of the repository being reviewed.**
59
+
60
+ The PR diff is fetched from Azure DevOps/TFS, but several features depend on the local git state:
61
+
62
+ | Feature | Local dependency |
63
+ |---|---|
64
+ | PR diff | `git fetch` + three-dot diff against `origin/<branch>` |
65
+ | RAG context | `git grep` on the working tree |
66
+ | Branch validation | `git branch --show-current` |
67
+
68
+ ### Branch requirement when RAG is enabled
69
+
70
+ When `review.rag.enabled: true`, the local repository **must be checked out on the PR target branch** (e.g. `development`). If the local branch differs from the target, the review is blocked:
71
+
72
+ ```
73
+ Local branch mismatch: you are on 'main' but this PR targets 'development'.
74
+ RAG context would be built from the wrong branch, which may corrupt the review.
75
+ Please run: git checkout development
76
+ ```
77
+
78
+ **Example:**
79
+ ```bash
80
+ # PR: merge 'feature/my-feature' → 'development'
81
+ cd /path/to/my-repo # must be inside the repo
82
+ git checkout development # must match PR target branch
83
+ ai-review pr-review 42
84
+ ```
85
+
86
+ To skip the branch check entirely, disable RAG:
87
+ ```yaml
88
+ review:
89
+ rag:
90
+ enabled: false
91
+ ```
92
+
55
93
  ## Installation & Quick Start
56
94
 
57
95
  ### 1. Install Package
@@ -89,7 +127,9 @@ ai-review init
89
127
 
90
128
  Creates:
91
129
  - `config.yaml` — LLM and review settings
92
- - `review_prompt.md` — Customizable review guidelines
130
+ - `review_context.example.md` — Canonical example reviewer context
131
+ - `review_context.local.md` — Local reviewer context override, ignored by git
132
+ - `.gitignore` entry for `review_context.local.md`
93
133
 
94
134
  ### 3. Run Your First Review
95
135
 
@@ -127,9 +167,14 @@ tfs:
127
167
  review:
128
168
  language: pt
129
169
  verbosity: detailed # or: quick, security
130
- scope: diff_with_context # diff_with_context, diff_only, full_code
170
+ scope: diff_with_context # diff_with_context, diff_only
131
171
  file_extensions_filter: [".cs", ".ts", ".py"]
132
172
  max_diff_files: 50
173
+ max_comments_to_post: 20
174
+ custom_prompt_file: review_context.local.md
175
+ rag:
176
+ enabled: true # requires local branch == PR target
177
+ max_chars: 40000
133
178
  project_context:
134
179
  enabled: true
135
180
  mode: on_demand # on_demand, full
@@ -143,14 +188,27 @@ review:
143
188
  max_items: 20
144
189
  ```
145
190
 
146
- By default, repository context is loaded on demand: the prompt includes the PR
147
- diff, full changed-file contents, linked work item documentation, and a
148
- repository manifest, then the model requests any extra files it needs. Set
191
+ By default, repository context is loaded on demand: the prompt includes explicit
192
+ source-branch change packets, full changed-file contents as read-only context,
193
+ linked work item documentation, and a repository manifest, then the model
194
+ requests any extra files it needs. Inline PR comments must still be grounded in
195
+ actual changed lines from the change packets. Set
149
196
  `review.project_context.mode: full` to send the full eligible repository
150
197
  snapshot instead. Bedrock uses a default estimated prompt budget of 180000
151
198
  tokens; override it with `llm.max_prompt_tokens` if your model supports more or
152
199
  less.
153
200
 
201
+ Reviewer context is loaded from exactly one Markdown file. By default the CLI
202
+ uses `review_context.local.md` when it exists, otherwise it falls back to the
203
+ packaged `src/prompts/review_context.example.md`. Keep local team tweaks in
204
+ `review_context.local.md`; it is gitignored so the canonical context cannot
205
+ drift across machines.
206
+
207
+ For Copilot-backed reviews, Claude Sonnet models are often strong choices for
208
+ large PR validation, for example `llm.provider: copilot` with a Claude Sonnet
209
+ model available to your organization. The tool still enforces the same
210
+ source-branch grounding, duplicate checks, and comment cap regardless of model.
211
+
154
212
  ## Development & Testing
155
213
 
156
214
  This is a standard Python project with the following structure:
@@ -163,11 +221,59 @@ src/
163
221
  tfs_client.py # Azure DevOps integration
164
222
  git_utils.py # Git diff processing
165
223
  formatter.py # Output formatting (terminal, markdown, JSON)
224
+ rag_engine.py # RAG context via git grep
166
225
 
167
226
  tests/
168
227
  test_*.py # Unit and integration tests
169
228
  ```
170
229
 
230
+ ## RAG Context
231
+
232
+ ### Current Implementation
233
+
234
+ The RAG engine enriches the LLM prompt with related code snippets found in the local repository:
235
+
236
+ 1. **Extract identifiers** — function and class names are parsed from the PR diff
237
+ 2. **Search** — `git grep` finds files containing those identifiers
238
+ 3. **Extract snippets** — ±10 lines around each match are included as read-only context
239
+
240
+ All operations run locally with no extra dependencies. The quality of RAG context depends entirely on the local repository state, which is why the **local branch must match the PR target branch**.
241
+
242
+ ### Recommended Stack for Enhanced RAG (Local & Open Source)
243
+
244
+ For teams wanting semantic similarity instead of keyword search, the recommended local stack is:
245
+
246
+ | Component | Library | Reason |
247
+ |---|---|---|
248
+ | **Vector database** | [ChromaDB](https://www.trychroma.com/) | Runs in-memory or persists to a local SQLite file — no server needed, `pip install chromadb` |
249
+ | **Embeddings** | [sentence-transformers](https://www.sbert.net/) | Generates vectors locally on CPU — no API calls, no cost |
250
+
251
+ Example integration pattern:
252
+
253
+ ```python
254
+ from sentence_transformers import SentenceTransformer
255
+ import chromadb
256
+
257
+ model = SentenceTransformer("all-MiniLM-L6-v2") # ~80 MB, CPU-friendly
258
+ client = chromadb.Client() # in-memory
259
+ collection = client.create_collection("repo-index")
260
+
261
+ # Index
262
+ collection.add(
263
+ documents=[snippet_text],
264
+ embeddings=model.encode([snippet_text]).tolist(),
265
+ ids=["file:line"],
266
+ )
267
+
268
+ # Query
269
+ results = collection.query(
270
+ query_embeddings=model.encode([query]).tolist(),
271
+ n_results=5,
272
+ )
273
+ ```
274
+
275
+ This stack keeps the CLI lightweight (`pip install`) and requires no external services or paid APIs.
276
+
171
277
  ### Run Tests
172
278
 
173
279
  ```bash
@@ -0,0 +1,289 @@
1
+ # AI Code Review CLI
2
+
3
+ Automated code review tool with Pull Request integration for Azure DevOps/TFS and support for multiple LLM providers.
4
+ **Documentation where** [docs/index.md](docs/index.md) for complete guides on CLI usage, LLM configuration, and architecture.
5
+
6
+ ## Features
7
+
8
+ - **AI Pull Request Review** — Automated code analysis with configurable LLM providers
9
+ - **Structured Comments** — Inline suggestions + general summary comments
10
+ - **Dry-run Mode** — Validate reviews before posting
11
+ - **Multiple LLM Providers** — OpenAI, Azure OpenAI, Gemini, Claude, Ollama, GitHub Copilot, AWS Bedrock
12
+ - **Smart Filtering** — Filter by file extensions, limit diff size
13
+ - **Project-aware PR Context** — Sends repository and linked work item context while restricting findings to modified PR lines
14
+ - **RAG Context** — Enriches the review with related code snippets found via `git grep` in the local repository
15
+ - **Single Reviewer Context** — One Markdown context file with local override support
16
+ - **Usage Tracking** — Store per-PR token usage and optional cost estimates
17
+ - **Interactive CLI** — Menu-driven selection and confirmation
18
+
19
+ ## Local Repository Requirement
20
+
21
+ > **The CLI must be run from inside the local clone of the repository being reviewed.**
22
+
23
+ The PR diff is fetched from Azure DevOps/TFS, but several features depend on the local git state:
24
+
25
+ | Feature | Local dependency |
26
+ |---|---|
27
+ | PR diff | `git fetch` + three-dot diff against `origin/<branch>` |
28
+ | RAG context | `git grep` on the working tree |
29
+ | Branch validation | `git branch --show-current` |
30
+
31
+ ### Branch requirement when RAG is enabled
32
+
33
+ When `review.rag.enabled: true`, the local repository **must be checked out on the PR target branch** (e.g. `development`). If the local branch differs from the target, the review is blocked:
34
+
35
+ ```
36
+ Local branch mismatch: you are on 'main' but this PR targets 'development'.
37
+ RAG context would be built from the wrong branch, which may corrupt the review.
38
+ Please run: git checkout development
39
+ ```
40
+
41
+ **Example:**
42
+ ```bash
43
+ # PR: merge 'feature/my-feature' → 'development'
44
+ cd /path/to/my-repo # must be inside the repo
45
+ git checkout development # must match PR target branch
46
+ ai-review pr-review 42
47
+ ```
48
+
49
+ To skip the branch check entirely, disable RAG:
50
+ ```yaml
51
+ review:
52
+ rag:
53
+ enabled: false
54
+ ```
55
+
56
+ ## Installation & Quick Start
57
+
58
+ ### 1. Install Package
59
+
60
+ From PyPI:
61
+
62
+ ```bash
63
+ pip install code-review-ai-cli
64
+ ```
65
+
66
+ With optional LLM SDK extras:
67
+
68
+ ```bash
69
+ pip install "code-review-ai-cli[bedrock]" # AWS Bedrock
70
+ pip install "code-review-ai-cli[openai]" # OpenAI SDK
71
+ pip install "code-review-ai-cli[gemini]" # Google Gemini SDK
72
+ pip install "code-review-ai-cli[claude]" # Anthropic Claude SDK
73
+ pip install "code-review-ai-cli[all]" # All optional SDKs
74
+ ```
75
+
76
+ For development:
77
+
78
+ ```bash
79
+ pip install -r requirements.txt
80
+ pip install "code-review-ai-cli[dev]" # Test suite + linting
81
+ ```
82
+
83
+ ### 2. Initialize Configuration
84
+
85
+ Generate config templates in your project directory:
86
+
87
+ ```bash
88
+ ai-review init
89
+ ```
90
+
91
+ Creates:
92
+ - `config.yaml` — LLM and review settings
93
+ - `review_context.example.md` — Canonical example reviewer context
94
+ - `review_context.local.md` — Local reviewer context override, ignored by git
95
+ - `.gitignore` entry for `review_context.local.md`
96
+
97
+ ### 3. Run Your First Review
98
+
99
+ ```bash
100
+ # List active PRs
101
+ ai-review list-prs
102
+
103
+ # Review a specific PR
104
+ ai-review pr-review 42
105
+
106
+ # Dry-run (preview without posting)
107
+ ai-review pr-review 42 --dry-run
108
+
109
+ # Check stored token/cost usage
110
+ ai-review usage
111
+ ```
112
+
113
+ ## Configuration File
114
+
115
+ After `ai-review init`, edit `config.yaml`:
116
+
117
+ ```yaml
118
+ llm:
119
+ provider: bedrock # or: openai, gemini, claude, etc.
120
+ model: anthropic.claude-3-5-sonnet-20240620-v1:0
121
+
122
+ bedrock:
123
+ region: us-east-1
124
+
125
+ tfs:
126
+ base_url: https://dev.azure.com/your-org
127
+ project: YourProject
128
+ pat: xxxxxxxxx
129
+
130
+ review:
131
+ language: pt
132
+ verbosity: detailed # or: quick, security
133
+ scope: diff_with_context # diff_with_context, diff_only
134
+ file_extensions_filter: [".cs", ".ts", ".py"]
135
+ max_diff_files: 50
136
+ max_comments_to_post: 20
137
+ custom_prompt_file: review_context.local.md
138
+ rag:
139
+ enabled: true # requires local branch == PR target
140
+ max_chars: 40000
141
+ project_context:
142
+ enabled: true
143
+ mode: on_demand # on_demand, full
144
+ manifest_max_chars: 60000
145
+ retrieval_max_rounds: 2
146
+ retrieval_max_files: 20
147
+ retrieval_max_chars: 120000
148
+ retrieval_file_max_chars: 30000
149
+ work_item_context:
150
+ enabled: true
151
+ max_items: 20
152
+ ```
153
+
154
+ By default, repository context is loaded on demand: the prompt includes explicit
155
+ source-branch change packets, full changed-file contents as read-only context,
156
+ linked work item documentation, and a repository manifest, then the model
157
+ requests any extra files it needs. Inline PR comments must still be grounded in
158
+ actual changed lines from the change packets. Set
159
+ `review.project_context.mode: full` to send the full eligible repository
160
+ snapshot instead. Bedrock uses a default estimated prompt budget of 180000
161
+ tokens; override it with `llm.max_prompt_tokens` if your model supports more or
162
+ less.
163
+
164
+ Reviewer context is loaded from exactly one Markdown file. By default the CLI
165
+ uses `review_context.local.md` when it exists, otherwise it falls back to the
166
+ packaged `src/prompts/review_context.example.md`. Keep local team tweaks in
167
+ `review_context.local.md`; it is gitignored so the canonical context cannot
168
+ drift across machines.
169
+
170
+ For Copilot-backed reviews, Claude Sonnet models are often strong choices for
171
+ large PR validation, for example `llm.provider: copilot` with a Claude Sonnet
172
+ model available to your organization. The tool still enforces the same
173
+ source-branch grounding, duplicate checks, and comment cap regardless of model.
174
+
175
+ ## Development & Testing
176
+
177
+ This is a standard Python project with the following structure:
178
+
179
+ ```
180
+ src/
181
+ ai_review.py # CLI entry point
182
+ config.py # Configuration management
183
+ llm_client.py # LLM provider abstraction
184
+ tfs_client.py # Azure DevOps integration
185
+ git_utils.py # Git diff processing
186
+ formatter.py # Output formatting (terminal, markdown, JSON)
187
+ rag_engine.py # RAG context via git grep
188
+
189
+ tests/
190
+ test_*.py # Unit and integration tests
191
+ ```
192
+
193
+ ## RAG Context
194
+
195
+ ### Current Implementation
196
+
197
+ The RAG engine enriches the LLM prompt with related code snippets found in the local repository:
198
+
199
+ 1. **Extract identifiers** — function and class names are parsed from the PR diff
200
+ 2. **Search** — `git grep` finds files containing those identifiers
201
+ 3. **Extract snippets** — ±10 lines around each match are included as read-only context
202
+
203
+ All operations run locally with no extra dependencies. The quality of RAG context depends entirely on the local repository state, which is why the **local branch must match the PR target branch**.
204
+
205
+ ### Recommended Stack for Enhanced RAG (Local & Open Source)
206
+
207
+ For teams wanting semantic similarity instead of keyword search, the recommended local stack is:
208
+
209
+ | Component | Library | Reason |
210
+ |---|---|---|
211
+ | **Vector database** | [ChromaDB](https://www.trychroma.com/) | Runs in-memory or persists to a local SQLite file — no server needed, `pip install chromadb` |
212
+ | **Embeddings** | [sentence-transformers](https://www.sbert.net/) | Generates vectors locally on CPU — no API calls, no cost |
213
+
214
+ Example integration pattern:
215
+
216
+ ```python
217
+ from sentence_transformers import SentenceTransformer
218
+ import chromadb
219
+
220
+ model = SentenceTransformer("all-MiniLM-L6-v2") # ~80 MB, CPU-friendly
221
+ client = chromadb.Client() # in-memory
222
+ collection = client.create_collection("repo-index")
223
+
224
+ # Index
225
+ collection.add(
226
+ documents=[snippet_text],
227
+ embeddings=model.encode([snippet_text]).tolist(),
228
+ ids=["file:line"],
229
+ )
230
+
231
+ # Query
232
+ results = collection.query(
233
+ query_embeddings=model.encode([query]).tolist(),
234
+ n_results=5,
235
+ )
236
+ ```
237
+
238
+ This stack keeps the CLI lightweight (`pip install`) and requires no external services or paid APIs.
239
+
240
+ ### Run Tests
241
+
242
+ ```bash
243
+ python -m pytest --cov=src --cov-report=term
244
+ ```
245
+
246
+ ### Code Quality
247
+
248
+ Project follows PEP8 with Black formatter and Ruff linter:
249
+
250
+ ```bash
251
+ # Format code
252
+ black src/ tests/
253
+
254
+ # Lint
255
+ ruff check src/ tests/
256
+
257
+ # Type checking (if using mypy)
258
+ mypy src/
259
+ ```
260
+
261
+
262
+ ## VS Code Integration
263
+
264
+ Predefined tasks for quick execution in VS Code:
265
+
266
+ ### Available Tasks
267
+
268
+ 1. **AI Review: Pull Request (Interactive)**
269
+ - Interactive mode with main menu
270
+ - Run with: `Ctrl+Shift+B` → Select task
271
+
272
+ 2. **AI Review: PR (Dry-Run)**
273
+ - Dry-run for a specific PR
274
+ - Prompts for PR ID
275
+
276
+ 3. **AI Review: List Active PRs**
277
+ - Lists active PRs
278
+ - Quick diagnostics
279
+
280
+ 4. **AI Review: Interactive Mode**
281
+ - Full tool menu
282
+ - Runs in background
283
+
284
+ ### How to run tasks
285
+
286
+ In VS Code:
287
+ 1. `Ctrl+Shift+P` → "Tasks: Run Task"
288
+ 2. Select the desired task
289
+ 3. Fill in parameters if needed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: code-review-ai-cli
3
- Version: 1.1.0
3
+ Version: 1.3.0
4
4
  Summary: Automated AI-powered code review CLI for Azure DevOps / TFS Pull Requests
5
5
  License: MIT
6
6
  Keywords: code-review,ai,azure-devops,tfs,pull-request,llm
@@ -48,10 +48,48 @@ Automated code review tool with Pull Request integration for Azure DevOps/TFS an
48
48
  - **Multiple LLM Providers** — OpenAI, Azure OpenAI, Gemini, Claude, Ollama, GitHub Copilot, AWS Bedrock
49
49
  - **Smart Filtering** — Filter by file extensions, limit diff size
50
50
  - **Project-aware PR Context** — Sends repository and linked work item context while restricting findings to modified PR lines
51
- - **Customizable Prompts** — Markdown-based review guidelines
51
+ - **RAG Context** — Enriches the review with related code snippets found via `git grep` in the local repository
52
+ - **Single Reviewer Context** — One Markdown context file with local override support
52
53
  - **Usage Tracking** — Store per-PR token usage and optional cost estimates
53
54
  - **Interactive CLI** — Menu-driven selection and confirmation
54
55
 
56
+ ## Local Repository Requirement
57
+
58
+ > **The CLI must be run from inside the local clone of the repository being reviewed.**
59
+
60
+ The PR diff is fetched from Azure DevOps/TFS, but several features depend on the local git state:
61
+
62
+ | Feature | Local dependency |
63
+ |---|---|
64
+ | PR diff | `git fetch` + three-dot diff against `origin/<branch>` |
65
+ | RAG context | `git grep` on the working tree |
66
+ | Branch validation | `git branch --show-current` |
67
+
68
+ ### Branch requirement when RAG is enabled
69
+
70
+ When `review.rag.enabled: true`, the local repository **must be checked out on the PR target branch** (e.g. `development`). If the local branch differs from the target, the review is blocked:
71
+
72
+ ```
73
+ Local branch mismatch: you are on 'main' but this PR targets 'development'.
74
+ RAG context would be built from the wrong branch, which may corrupt the review.
75
+ Please run: git checkout development
76
+ ```
77
+
78
+ **Example:**
79
+ ```bash
80
+ # PR: merge 'feature/my-feature' → 'development'
81
+ cd /path/to/my-repo # must be inside the repo
82
+ git checkout development # must match PR target branch
83
+ ai-review pr-review 42
84
+ ```
85
+
86
+ To skip the branch check entirely, disable RAG:
87
+ ```yaml
88
+ review:
89
+ rag:
90
+ enabled: false
91
+ ```
92
+
55
93
  ## Installation & Quick Start
56
94
 
57
95
  ### 1. Install Package
@@ -89,7 +127,9 @@ ai-review init
89
127
 
90
128
  Creates:
91
129
  - `config.yaml` — LLM and review settings
92
- - `review_prompt.md` — Customizable review guidelines
130
+ - `review_context.example.md` — Canonical example reviewer context
131
+ - `review_context.local.md` — Local reviewer context override, ignored by git
132
+ - `.gitignore` entry for `review_context.local.md`
93
133
 
94
134
  ### 3. Run Your First Review
95
135
 
@@ -127,9 +167,14 @@ tfs:
127
167
  review:
128
168
  language: pt
129
169
  verbosity: detailed # or: quick, security
130
- scope: diff_with_context # diff_with_context, diff_only, full_code
170
+ scope: diff_with_context # diff_with_context, diff_only
131
171
  file_extensions_filter: [".cs", ".ts", ".py"]
132
172
  max_diff_files: 50
173
+ max_comments_to_post: 20
174
+ custom_prompt_file: review_context.local.md
175
+ rag:
176
+ enabled: true # requires local branch == PR target
177
+ max_chars: 40000
133
178
  project_context:
134
179
  enabled: true
135
180
  mode: on_demand # on_demand, full
@@ -143,14 +188,27 @@ review:
143
188
  max_items: 20
144
189
  ```
145
190
 
146
- By default, repository context is loaded on demand: the prompt includes the PR
147
- diff, full changed-file contents, linked work item documentation, and a
148
- repository manifest, then the model requests any extra files it needs. Set
191
+ By default, repository context is loaded on demand: the prompt includes explicit
192
+ source-branch change packets, full changed-file contents as read-only context,
193
+ linked work item documentation, and a repository manifest, then the model
194
+ requests any extra files it needs. Inline PR comments must still be grounded in
195
+ actual changed lines from the change packets. Set
149
196
  `review.project_context.mode: full` to send the full eligible repository
150
197
  snapshot instead. Bedrock uses a default estimated prompt budget of 180000
151
198
  tokens; override it with `llm.max_prompt_tokens` if your model supports more or
152
199
  less.
153
200
 
201
+ Reviewer context is loaded from exactly one Markdown file. By default the CLI
202
+ uses `review_context.local.md` when it exists, otherwise it falls back to the
203
+ packaged `src/prompts/review_context.example.md`. Keep local team tweaks in
204
+ `review_context.local.md`; it is gitignored so the canonical context cannot
205
+ drift across machines.
206
+
207
+ For Copilot-backed reviews, Claude Sonnet models are often strong choices for
208
+ large PR validation, for example `llm.provider: copilot` with a Claude Sonnet
209
+ model available to your organization. The tool still enforces the same
210
+ source-branch grounding, duplicate checks, and comment cap regardless of model.
211
+
154
212
  ## Development & Testing
155
213
 
156
214
  This is a standard Python project with the following structure:
@@ -163,11 +221,59 @@ src/
163
221
  tfs_client.py # Azure DevOps integration
164
222
  git_utils.py # Git diff processing
165
223
  formatter.py # Output formatting (terminal, markdown, JSON)
224
+ rag_engine.py # RAG context via git grep
166
225
 
167
226
  tests/
168
227
  test_*.py # Unit and integration tests
169
228
  ```
170
229
 
230
+ ## RAG Context
231
+
232
+ ### Current Implementation
233
+
234
+ The RAG engine enriches the LLM prompt with related code snippets found in the local repository:
235
+
236
+ 1. **Extract identifiers** — function and class names are parsed from the PR diff
237
+ 2. **Search** — `git grep` finds files containing those identifiers
238
+ 3. **Extract snippets** — ±10 lines around each match are included as read-only context
239
+
240
+ All operations run locally with no extra dependencies. The quality of RAG context depends entirely on the local repository state, which is why the **local branch must match the PR target branch**.
241
+
242
+ ### Recommended Stack for Enhanced RAG (Local & Open Source)
243
+
244
+ For teams wanting semantic similarity instead of keyword search, the recommended local stack is:
245
+
246
+ | Component | Library | Reason |
247
+ |---|---|---|
248
+ | **Vector database** | [ChromaDB](https://www.trychroma.com/) | Runs in-memory or persists to a local SQLite file — no server needed, `pip install chromadb` |
249
+ | **Embeddings** | [sentence-transformers](https://www.sbert.net/) | Generates vectors locally on CPU — no API calls, no cost |
250
+
251
+ Example integration pattern:
252
+
253
+ ```python
254
+ from sentence_transformers import SentenceTransformer
255
+ import chromadb
256
+
257
+ model = SentenceTransformer("all-MiniLM-L6-v2") # ~80 MB, CPU-friendly
258
+ client = chromadb.Client() # in-memory
259
+ collection = client.create_collection("repo-index")
260
+
261
+ # Index
262
+ collection.add(
263
+ documents=[snippet_text],
264
+ embeddings=model.encode([snippet_text]).tolist(),
265
+ ids=["file:line"],
266
+ )
267
+
268
+ # Query
269
+ results = collection.query(
270
+ query_embeddings=model.encode([query]).tolist(),
271
+ n_results=5,
272
+ )
273
+ ```
274
+
275
+ This stack keeps the CLI lightweight (`pip install`) and requires no external services or paid APIs.
276
+
171
277
  ### Run Tests
172
278
 
173
279
  ```bash
@@ -12,14 +12,16 @@ src/config.py
12
12
  src/formatter.py
13
13
  src/git_utils.py
14
14
  src/llm_client.py
15
+ src/rag_engine.py
15
16
  src/tfs_client.py
16
17
  src/usage_tracker.py
17
18
  src/prompts/config.yaml.template
18
- src/prompts/review_prompt.md.template
19
+ src/prompts/review_context.example.md
19
20
  tests/test_ai_review.py
20
21
  tests/test_config.py
21
22
  tests/test_formatter.py
22
23
  tests/test_git_utils.py
23
24
  tests/test_llm_client.py
25
+ tests/test_rag_engine.py
24
26
  tests/test_tfs_client.py
25
27
  tests/test_usage_tracker.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "code-review-ai-cli"
7
- version = "1.1.0"
7
+ version = "1.3.0"
8
8
  description = "Automated AI-powered code review CLI for Azure DevOps / TFS Pull Requests"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"