scrutai 0.4.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scrutai-0.4.0/.gitignore +12 -0
- scrutai-0.4.0/.scrutai.yml +21 -0
- scrutai-0.4.0/DEVELOPMENT.md +500 -0
- scrutai-0.4.0/LICENSE +191 -0
- scrutai-0.4.0/PKG-INFO +734 -0
- scrutai-0.4.0/README.md +672 -0
- scrutai-0.4.0/RELEASES.md +297 -0
- scrutai-0.4.0/action.yml +94 -0
- scrutai-0.4.0/docs/ARCHITECTURE.md +136 -0
- scrutai-0.4.0/docs/MCP.md +267 -0
- scrutai-0.4.0/docs/MCP_PLAN.md +156 -0
- scrutai-0.4.0/examples/mcp/README.md +24 -0
- scrutai-0.4.0/examples/mcp/claude-code.sh +12 -0
- scrutai-0.4.0/examples/mcp/claude_desktop_config.json +16 -0
- scrutai-0.4.0/examples/mcp/cursor-mcp.json +12 -0
- scrutai-0.4.0/examples/mcp/try_it.py +89 -0
- scrutai-0.4.0/examples/scrutai-workflow.yml +30 -0
- scrutai-0.4.0/pyproject.toml +112 -0
- scrutai-0.4.0/src/scrutai/__init__.py +9 -0
- scrutai-0.4.0/src/scrutai/agents/__init__.py +45 -0
- scrutai-0.4.0/src/scrutai/agents/base.py +221 -0
- scrutai-0.4.0/src/scrutai/agents/correctness.py +21 -0
- scrutai-0.4.0/src/scrutai/agents/crewai_backend.py +235 -0
- scrutai-0.4.0/src/scrutai/agents/performance.py +20 -0
- scrutai-0.4.0/src/scrutai/agents/security.py +46 -0
- scrutai-0.4.0/src/scrutai/agents/style.py +20 -0
- scrutai-0.4.0/src/scrutai/agents/tests.py +21 -0
- scrutai-0.4.0/src/scrutai/cli.py +362 -0
- scrutai-0.4.0/src/scrutai/concurrency.py +17 -0
- scrutai-0.4.0/src/scrutai/config.py +127 -0
- scrutai-0.4.0/src/scrutai/critic.py +150 -0
- scrutai-0.4.0/src/scrutai/demo.py +47 -0
- scrutai-0.4.0/src/scrutai/diff.py +102 -0
- scrutai-0.4.0/src/scrutai/eval/__init__.py +3 -0
- scrutai-0.4.0/src/scrutai/eval/cases.jsonl +50 -0
- scrutai-0.4.0/src/scrutai/eval/harness.py +321 -0
- scrutai-0.4.0/src/scrutai/github.py +190 -0
- scrutai-0.4.0/src/scrutai/inputs.py +99 -0
- scrutai-0.4.0/src/scrutai/llm.py +213 -0
- scrutai-0.4.0/src/scrutai/mcp/__init__.py +1 -0
- scrutai-0.4.0/src/scrutai/mcp/compat.py +160 -0
- scrutai-0.4.0/src/scrutai/mcp/schemas.py +191 -0
- scrutai-0.4.0/src/scrutai/mcp/security.py +136 -0
- scrutai-0.4.0/src/scrutai/mcp/server.py +847 -0
- scrutai-0.4.0/src/scrutai/mock.py +666 -0
- scrutai-0.4.0/src/scrutai/models.py +206 -0
- scrutai-0.4.0/src/scrutai/orchestrator.py +237 -0
- scrutai-0.4.0/src/scrutai/patch.py +174 -0
- scrutai-0.4.0/src/scrutai/progress.py +62 -0
- scrutai-0.4.0/src/scrutai/py.typed +0 -0
- scrutai-0.4.0/src/scrutai/report.py +137 -0
- scrutai-0.4.0/src/scrutai/router.py +92 -0
- scrutai-0.4.0/src/scrutai/rules/semgrep.yml +69 -0
- scrutai-0.4.0/src/scrutai/runs.py +109 -0
- scrutai-0.4.0/src/scrutai/tools/__init__.py +15 -0
- scrutai-0.4.0/src/scrutai/tools/repo.py +123 -0
- scrutai-0.4.0/src/scrutai/tools/semgrep.py +131 -0
- scrutai-0.4.0/src/scrutai/tools/toolbox.py +133 -0
- scrutai-0.4.0/src/scrutai/trace.py +221 -0
- scrutai-0.4.0/src/scrutai/web/__init__.py +1 -0
- scrutai-0.4.0/src/scrutai/web/server.py +179 -0
- scrutai-0.4.0/src/scrutai/web/static/assets/index-BOOR7lCF.css +1 -0
- scrutai-0.4.0/src/scrutai/web/static/assets/index-Bz6jMS2_.js +9 -0
- scrutai-0.4.0/src/scrutai/web/static/index.html +15 -0
- scrutai-0.4.0/tests/conftest.py +44 -0
- scrutai-0.4.0/tests/fake_github.py +128 -0
- scrutai-0.4.0/tests/mcp_util.py +89 -0
- scrutai-0.4.0/tests/test_agents.py +159 -0
- scrutai-0.4.0/tests/test_budget.py +60 -0
- scrutai-0.4.0/tests/test_cancel_progress.py +107 -0
- scrutai-0.4.0/tests/test_chunking.py +102 -0
- scrutai-0.4.0/tests/test_cli.py +109 -0
- scrutai-0.4.0/tests/test_crewai_backend.py +169 -0
- scrutai-0.4.0/tests/test_critic.py +192 -0
- scrutai-0.4.0/tests/test_diff.py +105 -0
- scrutai-0.4.0/tests/test_eval.py +90 -0
- scrutai-0.4.0/tests/test_github.py +127 -0
- scrutai-0.4.0/tests/test_inputs.py +69 -0
- scrutai-0.4.0/tests/test_live_client.py +97 -0
- scrutai-0.4.0/tests/test_mcp_benchmark.py +108 -0
- scrutai-0.4.0/tests/test_mcp_github.py +153 -0
- scrutai-0.4.0/tests/test_mcp_prompts.py +69 -0
- scrutai-0.4.0/tests/test_mcp_server.py +357 -0
- scrutai-0.4.0/tests/test_mcp_transports.py +145 -0
- scrutai-0.4.0/tests/test_parallel.py +75 -0
- scrutai-0.4.0/tests/test_perf_style.py +68 -0
- scrutai-0.4.0/tests/test_router.py +88 -0
- scrutai-0.4.0/tests/test_semgrep.py +121 -0
- scrutai-0.4.0/tests/test_server.py +119 -0
- scrutai-0.4.0/tests/test_smoke.py +35 -0
- scrutai-0.4.0/tests/test_trace.py +145 -0
- scrutai-0.4.0/tests/test_ui_e2e.py +116 -0
scrutai-0.4.0/.gitignore
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Scrutai configuration. All values shown are defaults — delete any you don't override.
|
|
2
|
+
enabled_agents: [security, correctness, tests, performance, style]
|
|
3
|
+
min_severity: low
|
|
4
|
+
min_confidence: 0.6
|
|
5
|
+
max_critic_rounds: 2
|
|
6
|
+
token_budget: 200000 # hard cap per review; 0 disables
|
|
7
|
+
max_cost_usd: 0.0 # optional dollar cap (live mode); 0 disables
|
|
8
|
+
concurrency: 4 # parallel critic/defense calls
|
|
9
|
+
fail_on: high
|
|
10
|
+
llm_mode: mock # switch to "live" and export a provider key to run for real
|
|
11
|
+
routing: heuristic # or "llm": the router model may narrow the selection further
|
|
12
|
+
max_agent_steps: 4 # ReAct budget per specialist (tool calls + final answer)
|
|
13
|
+
semgrep: auto # auto | off | required
|
|
14
|
+
semgrep_config: bundled # or any semgrep --config value, e.g. p/default
|
|
15
|
+
tracing: none # none | otel | langfuse (and: scrutai review --trace run.jsonl)
|
|
16
|
+
include: ["**/*"]
|
|
17
|
+
exclude: ["**/vendor/**", "**/*.lock", "**/dist/**"]
|
|
18
|
+
models: # any LiteLLM model string; cheap router, strong critic
|
|
19
|
+
router: anthropic/claude-haiku-4-5
|
|
20
|
+
specialist: anthropic/claude-sonnet-5-5
|
|
21
|
+
critic: anthropic/claude-opus-5-5
|
|
@@ -0,0 +1,500 @@
|
|
|
1
|
+
# Developing Scrutai
|
|
2
|
+
|
|
3
|
+
Thanks for wanting to make Scrutai better. This guide gets you from a fresh clone to a merged pull
|
|
4
|
+
request: setup, how the code fits together, step-by-step recipes for the most common changes, how
|
|
5
|
+
the tests work, and what reviewers look for.
|
|
6
|
+
|
|
7
|
+
New here? Read [README.md](README.md) first for *what* Scrutai does, and
|
|
8
|
+
[docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) for *why* it is built the way it is.
|
|
9
|
+
|
|
10
|
+
## Contents
|
|
11
|
+
|
|
12
|
+
- [Ways to contribute](#ways-to-contribute)
|
|
13
|
+
- [Setup](#setup)
|
|
14
|
+
- [Running the checks](#running-the-checks)
|
|
15
|
+
- [The architecture in five minutes](#the-architecture-in-five-minutes)
|
|
16
|
+
- [Code map](#code-map)
|
|
17
|
+
- [Recipes](#recipes)
|
|
18
|
+
- [Add a specialist agent](#add-a-specialist-agent)
|
|
19
|
+
- [Add a repository tool](#add-a-repository-tool)
|
|
20
|
+
- [Add a Semgrep rule](#add-a-semgrep-rule)
|
|
21
|
+
- [Add benchmark cases](#add-benchmark-cases)
|
|
22
|
+
- [Add a framework backend](#add-a-framework-backend)
|
|
23
|
+
- [Add or change a trace event](#add-or-change-a-trace-event)
|
|
24
|
+
- [Work on the web UI](#work-on-the-web-ui)
|
|
25
|
+
- [Add an MCP tool](#add-an-mcp-tool)
|
|
26
|
+
- [Update the README's screenshots and recording](#update-the-readmes-screenshots-and-recording)
|
|
27
|
+
- [Testing guide](#testing-guide)
|
|
28
|
+
- [Conventions](#conventions)
|
|
29
|
+
- [Pull requests](#pull-requests)
|
|
30
|
+
- [Releasing](#releasing)
|
|
31
|
+
- [Debugging tips](#debugging-tips)
|
|
32
|
+
- [Getting help](#getting-help)
|
|
33
|
+
|
|
34
|
+
## Ways to contribute
|
|
35
|
+
|
|
36
|
+
- **Benchmark cases:** the most valuable contribution. Especially *traps*: code that looks like a
|
|
37
|
+
bug but isn't, so the critic is forced to prove itself. See
|
|
38
|
+
[Add benchmark cases](#add-benchmark-cases).
|
|
39
|
+
- **Semgrep rules** for the bundled offline ruleset, ideally for languages beyond Python.
|
|
40
|
+
- **New specialists** (e.g. accessibility, i18n, infrastructure-as-code) or new categories for
|
|
41
|
+
existing ones.
|
|
42
|
+
- **Framework backends** (e.g. Google ADK) behind the `Specialist._loop` seam.
|
|
43
|
+
- **UI improvements** to the agent theater.
|
|
44
|
+
- **Bug reports** with a failing diff attached: `scrutai review --diff bug.patch --trace bug.jsonl`
|
|
45
|
+
gives us everything we need to reproduce it.
|
|
46
|
+
|
|
47
|
+
## Setup
|
|
48
|
+
|
|
49
|
+
**You need:** Python 3.12+, `git`, and Node.js 22+ (only for the web UI). Optional:
|
|
50
|
+
[`uv`](https://docs.astral.sh/uv/) (faster installs), ripgrep, Semgrep.
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
git clone https://github.com/imkarthiknr/Scrutai.git
|
|
54
|
+
cd Scrutai
|
|
55
|
+
|
|
56
|
+
# Python: with uv (recommended)...
|
|
57
|
+
uv venv --python 3.12
|
|
58
|
+
uv pip install -e ".[dev]"
|
|
59
|
+
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
60
|
+
|
|
61
|
+
# ...or with plain pip
|
|
62
|
+
python -m venv .venv && source .venv/bin/activate
|
|
63
|
+
pip install -e ".[dev]"
|
|
64
|
+
|
|
65
|
+
# Browser for the end-to-end UI tests (skipped automatically if missing)
|
|
66
|
+
playwright install chromium
|
|
67
|
+
|
|
68
|
+
# Optional: run lint and format checks on every commit
|
|
69
|
+
pre-commit install
|
|
70
|
+
|
|
71
|
+
# Web UI (only if you work on it)
|
|
72
|
+
cd web && npm ci && cd ..
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
The `dev` extra installs everything the test suite exercises: pytest, ruff, mypy, the web server,
|
|
76
|
+
Playwright, OpenTelemetry, CrewAI and the MCP SDK. CrewAI pins the MCP SDK to 1.28, so the
|
|
77
|
+
main environment tests MCP on 1.x; CI tests it on 2.x as well (see below).
|
|
78
|
+
|
|
79
|
+
Check that it works:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
scrutai review --demo # should report "5 issue(s) upheld" and exit with code 1
|
|
83
|
+
pytest -q # should report 200+ passed
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
You never need an API key to develop. Everything, including the whole test suite and CI, runs
|
|
87
|
+
against the offline mock model, so no contributor or CI run spends money or needs a secret.
|
|
88
|
+
|
|
89
|
+
**Optional: trying a change against a real model.** Use your own provider key (see the README's
|
|
90
|
+
[Running with a real model](README.md#running-with-a-real-model-bring-your-own-key) for providers
|
|
91
|
+
and variable names):
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
export ANTHROPIC_API_KEY=... # your key, in your shell only
|
|
95
|
+
scrutai review --demo --config my-live.yml # a copy of .scrutai.yml with llm_mode: live
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
- Keep live configs and keys out of commits: put keys only in the environment, and set a
|
|
99
|
+
`max_cost_usd` cap in live configs.
|
|
100
|
+
- Never add a test that needs a real key. Use `MockLLMClient`, a scripted client, or LiteLLM's
|
|
101
|
+
`mock_response` (see the [testing guide](#testing-guide)).
|
|
102
|
+
- When a PR changes prompts or the critic, say in its description whether you checked it live,
|
|
103
|
+
with which model.
|
|
104
|
+
|
|
105
|
+
## Running the checks
|
|
106
|
+
|
|
107
|
+
CI runs exactly these. Run them before you push:
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
# Python job
|
|
111
|
+
ruff check .
|
|
112
|
+
ruff format --check .
|
|
113
|
+
mypy src # strict mode
|
|
114
|
+
pytest -q # incl. browser E2E if Chromium is installed
|
|
115
|
+
scrutai eval --min-precision 0.95 --min-recall 0.8 # the benchmark gate
|
|
116
|
+
|
|
117
|
+
# MCP job: the MCP tests again, on MCP SDK 2.x (a separate venv: CrewAI pins 1.x)
|
|
118
|
+
uv venv .venv-mcp2 --python 3.12
|
|
119
|
+
uv pip install --python .venv-mcp2 -e . "mcp>=2" pytest httpx pyyaml
|
|
120
|
+
.venv-mcp2/bin/pytest -q tests/test_mcp_*.py
|
|
121
|
+
|
|
122
|
+
# Web job (from web/)
|
|
123
|
+
npm run typecheck
|
|
124
|
+
npm test # vitest
|
|
125
|
+
npm run build && git diff --exit-code -- ../src/scrutai/web/static # bundle must be committed
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Useful variations:
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
pytest -q tests/test_critic.py # one file
|
|
132
|
+
pytest -q -k "debate or withdraw" # by name
|
|
133
|
+
pytest -q --ignore=tests/test_ui_e2e.py # skip browser tests for a fast loop
|
|
134
|
+
ruff check . --fix && ruff format . # auto-fix lint and formatting
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## The architecture in five minutes
|
|
138
|
+
|
|
139
|
+
Everything goes through one function, `review_diff(diff, config, llm)` in
|
|
140
|
+
`src/scrutai/orchestrator.py`. The CLI, the GitHub Action, the web server and the benchmark are
|
|
141
|
+
thin wrappers around it.
|
|
142
|
+
|
|
143
|
+
```text
|
|
144
|
+
DiffContext ──► route ──► specialist × (agent, chunk) ──► collect ──► critic ⇄ defend ──► verdict ──► ReviewResult
|
|
145
|
+
│ │ │ │
|
|
146
|
+
router.py agents/*.py dedupe() critic.py
|
|
147
|
+
(ReAct over tools/)
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
A review goes like this:
|
|
151
|
+
|
|
152
|
+
1. **Diff in.** `diff.py` turns a git range, a patch or a PR into a `DiffContext` (a list of
|
|
153
|
+
`ChangedFile`s with raw patches). `patch.py` does the pure parsing: added lines with real
|
|
154
|
+
line numbers, slicing, globs.
|
|
155
|
+
2. **Route.** `router.py` decides which specialists a change needs, using file kinds and the "risk
|
|
156
|
+
surface" of added lines. The orchestrator splits the diff into chunks (`chunk_diff`) and routes
|
|
157
|
+
each chunk on its own.
|
|
158
|
+
3. **Fan out.** LangGraph `Send` starts one branch per (agent, chunk). Each branch builds a
|
|
159
|
+
specialist and calls `review()`.
|
|
160
|
+
4. **ReAct.** `Specialist.review()` (in `agents/base.py`) builds a prompt, then calls `_loop()`: the
|
|
161
|
+
model answers with either a tool action (`{"action": {...}}`) or findings
|
|
162
|
+
(`{"findings": [...]}`). Tools run through a sandboxed `Toolbox` (`tools/`).
|
|
163
|
+
5. **Collect.** Findings from all branches are sorted and deduplicated by `(file, line, category)`.
|
|
164
|
+
6. **Critic.** `critic.py` judges each finding against the cited code: uphold, downgrade, kill or
|
|
165
|
+
challenge. Challenged findings go to `defend` (the specialist's `defend()`), then back to the
|
|
166
|
+
critic. `max_critic_rounds` bounds the loop.
|
|
167
|
+
7. **Verdict.** Survivors become `ReviewResult.findings`, everything else goes to `dropped` with
|
|
168
|
+
the reason.
|
|
169
|
+
|
|
170
|
+
Three cross-cutting pieces wrap every LLM call:
|
|
171
|
+
|
|
172
|
+
- **`BudgetedClient`** (`llm.py`) enforces `token_budget` and `max_cost_usd`.
|
|
173
|
+
- **`TracingClient`** (`trace.py`) records spans when a tracer is active.
|
|
174
|
+
- **`MockLLMClient`** (`mock.py`) speaks the exact same protocol offline, which is what makes the
|
|
175
|
+
whole system testable without a model.
|
|
176
|
+
|
|
177
|
+
**The protocol is JSON, line-anchored.** Reviewed code always appears as `L<n>: <code>`, and protocol
|
|
178
|
+
markers (`--- STEP`, `--- FINAL`, `OBSERVATION:`) are only recognised at the start of a line. That
|
|
179
|
+
way, a diff that happens to *contain* those strings (Scrutai reviewing itself, for example) can't
|
|
180
|
+
spoof the protocol.
|
|
181
|
+
|
|
182
|
+
## Code map
|
|
183
|
+
|
|
184
|
+
| Path | Responsibility |
|
|
185
|
+
|---|---|
|
|
186
|
+
| `src/scrutai/cli.py` | `review`, `serve`, `mcp` and `eval` commands; output formats; exit codes. |
|
|
187
|
+
| `src/scrutai/config.py` | `ScrutaiConfig`: every option, defaults and validation. |
|
|
188
|
+
| `src/scrutai/models.py` | Pydantic models: `ChangedFile`, `DiffContext`, `Finding`, `ReviewResult`. |
|
|
189
|
+
| `src/scrutai/diff.py` | Git and patch input, ref validation, include/exclude filters, chunking. |
|
|
190
|
+
| `src/scrutai/patch.py` | Pure unified-diff parsing; no I/O. |
|
|
191
|
+
| `src/scrutai/router.py` | Heuristic and optional LLM routing. |
|
|
192
|
+
| `src/scrutai/orchestrator.py` | The LangGraph graph and `review_diff()`. |
|
|
193
|
+
| `src/scrutai/agents/base.py` | `Specialist`: prompts, the `_loop` seam, finding parsing, `defend()`. |
|
|
194
|
+
| `src/scrutai/agents/<name>.py` | One specialist each: role, categories, tools, file kinds. |
|
|
195
|
+
| `src/scrutai/agents/crewai_backend.py` | The CrewAI backend: tool adapters and the protocol bridge. |
|
|
196
|
+
| `src/scrutai/critic.py` | The critic prompt, decisions, and per-round judging. |
|
|
197
|
+
| `src/scrutai/tools/` | `repo.py` (the functions), `toolbox.py` (registry + sandbox), `semgrep.py`. |
|
|
198
|
+
| `src/scrutai/llm.py` | `LiteLLMClient`, `BudgetedClient`, `extract_json`, `make_client`. |
|
|
199
|
+
| `src/scrutai/mock.py` | The offline model: specialist rules, the mock critic, mock defenses. |
|
|
200
|
+
| `src/scrutai/report.py` | Markdown and SARIF rendering; finding fingerprints are in `models.py`. |
|
|
201
|
+
| `src/scrutai/github.py` | GitHub REST client; idempotent `publish()`. |
|
|
202
|
+
| `src/scrutai/trace.py` | Tracer, spans, events, OpenTelemetry and Langfuse hooks. |
|
|
203
|
+
| `src/scrutai/eval/harness.py` | Benchmark loading, hermetic case runs, metrics, `--compare`. |
|
|
204
|
+
| `src/scrutai/inputs.py` | `Source` + `prepare()`: the one input path (demo, git range, patch, file, PR) every front end uses. |
|
|
205
|
+
| `src/scrutai/runs.py` | `Run` / `RunStore`: reviews in flight and finished, shared by the web and MCP servers. |
|
|
206
|
+
| `src/scrutai/progress.py` | `ProgressListener`: trace events → "step N of M" progress. |
|
|
207
|
+
| `src/scrutai/mcp/server.py` | The MCP server: tools, resources, prompts; `Reviewer` runs and keeps reviews. |
|
|
208
|
+
| `src/scrutai/mcp/compat.py` | The only module that imports the MCP SDK; hides 1.x vs 2.x differences. |
|
|
209
|
+
| `src/scrutai/mcp/schemas.py` | Structured tool output (`ReviewSummary`, `FindingExplanation`, …). |
|
|
210
|
+
| `src/scrutai/mcp/security.py` | Root allowlist, input caps, HTTP bearer-token and `Host` guard. |
|
|
211
|
+
| `src/scrutai/web/server.py` | FastAPI app: runs, SSE event stream, replay. |
|
|
212
|
+
| `src/scrutai/web/static/` | **Built** UI bundle (generated; never edit by hand). |
|
|
213
|
+
| `web/src/` | The React UI source: `reduce.ts` (all UI state), `components/`, `api.ts`. |
|
|
214
|
+
| `action.yml` | The composite GitHub Action. |
|
|
215
|
+
| `src/scrutai/eval/cases.jsonl` | The labelled benchmark (shipped in the package). |
|
|
216
|
+
|
|
217
|
+
## Recipes
|
|
218
|
+
|
|
219
|
+
### Add a specialist agent
|
|
220
|
+
|
|
221
|
+
Example: an `accessibility` agent for front-end code.
|
|
222
|
+
|
|
223
|
+
1. **Create `src/scrutai/agents/accessibility.py`:**
|
|
224
|
+
|
|
225
|
+
```python
|
|
226
|
+
from __future__ import annotations
|
|
227
|
+
|
|
228
|
+
from typing import ClassVar
|
|
229
|
+
|
|
230
|
+
from .base import Specialist
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
class AccessibilityAgent(Specialist):
|
|
234
|
+
name = "accessibility"
|
|
235
|
+
role = "You find accessibility problems in UI code: missing labels, alt text, focus traps."
|
|
236
|
+
categories: ClassVar[dict[str, str]] = {
|
|
237
|
+
"missing_alt_text": "an <img> without meaningful alt text",
|
|
238
|
+
"unlabelled_control": "an input or button with no accessible name",
|
|
239
|
+
}
|
|
240
|
+
tools: ClassVar[list[str]] = ["read_file", "grep"]
|
|
241
|
+
kinds = ("code",) # which ChangedFile.kind values this agent sees
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
2. **Register it** in `src/scrutai/agents/__init__.py`: add it to `REGISTRY` and `__all__`.
|
|
245
|
+
3. **Route it** in `src/scrutai/router.py`: add an entry to the `wants` dict in
|
|
246
|
+
`heuristic_route()`. Without one, the agent runs on every reviewable change.
|
|
247
|
+
4. **Enable it by default** (optional): add it to `enabled_agents` in `config.py` **and**
|
|
248
|
+
`.scrutai.yml`. `test_default_config_matches_shipped_yaml` keeps the two in sync.
|
|
249
|
+
5. **Teach the mock** in `src/scrutai/mock.py`: add `Rule(...)`s with your agent name, so the agent
|
|
250
|
+
does something offline. If a rule needs the critic to kill a known false-positive pattern, add
|
|
251
|
+
that check in `MockLLMClient._critic`.
|
|
252
|
+
6. **Add benchmark cases** for each category, including at least one trap (see below).
|
|
253
|
+
7. **Show it in the UI:** add the name to `ALL_AGENTS` in `web/src/reduce.ts`, then rebuild the UI.
|
|
254
|
+
8. **Test it:** a unit test in the style of `tests/test_perf_style.py`, and make sure
|
|
255
|
+
`scrutai eval` still passes the gate.
|
|
256
|
+
|
|
257
|
+
### Add a repository tool
|
|
258
|
+
|
|
259
|
+
1. Write the function in `src/scrutai/tools/repo.py`. Treat **every argument as hostile**: confine
|
|
260
|
+
paths with `_confine()`, pass patterns as data (`-e PATTERN`, never as a flag), bound the output.
|
|
261
|
+
2. Register it in `TOOLS` in `src/scrutai/tools/toolbox.py` with a `Tool(name, signature,
|
|
262
|
+
description, run)`. The signature and description go into every prompt that offers the tool.
|
|
263
|
+
3. Allow it on the agents that should use it (their `tools` class variable).
|
|
264
|
+
4. Add an argument schema to `_SCHEMAS` in `src/scrutai/agents/crewai_backend.py`, so CrewAI can
|
|
265
|
+
validate calls.
|
|
266
|
+
5. Test it directly, *and* test that a model can't escape the sandbox through it (see
|
|
267
|
+
`test_read_file_is_confined_to_repo`).
|
|
268
|
+
|
|
269
|
+
### Add a Semgrep rule
|
|
270
|
+
|
|
271
|
+
1. Add the rule to `src/scrutai/rules/semgrep.yml`. Use the id `scrutai.<category>.<short-name>`:
|
|
272
|
+
the category becomes the finding category automatically.
|
|
273
|
+
2. Use `pattern-not` to exclude the safe forms (constant arguments, safe loaders, explicit opt-outs).
|
|
274
|
+
3. Verify it against the real binary on positives *and* traps:
|
|
275
|
+
|
|
276
|
+
```bash
|
|
277
|
+
pip install semgrep
|
|
278
|
+
semgrep scan --json --metrics=off --config src/scrutai/rules/semgrep.yml path/to/samples.py
|
|
279
|
+
```
|
|
280
|
+
|
|
281
|
+
4. Tests use a fake `semgrep` binary (see the `fake_semgrep` fixture in `tests/test_semgrep.py`), so
|
|
282
|
+
CI doesn't need Semgrep installed.
|
|
283
|
+
|
|
284
|
+
### Add benchmark cases
|
|
285
|
+
|
|
286
|
+
Append JSON lines to `src/scrutai/eval/cases.jsonl`:
|
|
287
|
+
|
|
288
|
+
```json
|
|
289
|
+
{"id": "trap-secret-env", "file": "settings.py", "patch": "+API_KEY = os.environ[\"API_KEY\"]\n", "labels": [], "note": "read from the environment"}
|
|
290
|
+
```
|
|
291
|
+
|
|
292
|
+
- `labels` is a **multiset of categories** a correct reviewer reports; `[]` means the change is clean.
|
|
293
|
+
- `repo` (optional) adds other files to the case's temporary repository, e.g. an existing test that
|
|
294
|
+
should stop a `missing_tests` finding.
|
|
295
|
+
- Prefer **traps**: realistic code that a naive reviewer would flag.
|
|
296
|
+
- If the mock model can't catch something a good reviewer should, keep the case and label it
|
|
297
|
+
honestly; it becomes a documented expected miss (ids starting with `miss-`). **Never tune the
|
|
298
|
+
mock to the benchmark, and never delete a case to raise a number.**
|
|
299
|
+
|
|
300
|
+
Run `scrutai eval --report eval.md` and read the "Cases with errors" table.
|
|
301
|
+
|
|
302
|
+
### Add a framework backend
|
|
303
|
+
|
|
304
|
+
Specialists share everything except one method: `Specialist._loop(system, transcript, toolbox,
|
|
305
|
+
answer_keys, final_note)`, which runs the ReAct loop and returns the answer payload, or `None`.
|
|
306
|
+
|
|
307
|
+
1. Write a mixin that overrides `_loop` (see `CrewAIMixin` in `agents/crewai_backend.py`). Drive
|
|
308
|
+
your framework's agent with the given `system` prompt and `transcript`, expose
|
|
309
|
+
`toolbox.allowed` as the framework's tools (each calling `toolbox.run(name, args)`), and return
|
|
310
|
+
a dict containing one of `answer_keys`.
|
|
311
|
+
2. Reach the model through `self.llm` (Scrutai's client) rather than letting the framework call a
|
|
312
|
+
provider directly. That keeps budgets, cost tracking, tracing and mock mode working, and keeps
|
|
313
|
+
`--compare` fair.
|
|
314
|
+
3. Add the backend name to `BACKENDS` and to `agent_class()` in `agents/__init__.py`, and add an
|
|
315
|
+
optional extra in `pyproject.toml`.
|
|
316
|
+
4. Prove parity: `scrutai eval --compare <backend>` should report 100% per-case agreement in mock
|
|
317
|
+
mode. Copy the structure of `tests/test_crewai_backend.py`.
|
|
318
|
+
|
|
319
|
+
### Add or change a trace event
|
|
320
|
+
|
|
321
|
+
Trace events feed `--trace` files, OpenTelemetry and the agent theater.
|
|
322
|
+
|
|
323
|
+
1. Emit it with `emit("kind", **fields)` (a point event) or `span("kind", name, ...)` (start/end
|
|
324
|
+
pair) from `scrutai.trace`.
|
|
325
|
+
2. Add its shape to the `TraceEvent` union in `web/src/types.ts` and handle it in `apply()` in
|
|
326
|
+
`web/src/reduce.ts`.
|
|
327
|
+
3. **Regenerate the recorded fixture** that the UI tests run on. The file is opened in append mode,
|
|
328
|
+
so delete it first:
|
|
329
|
+
|
|
330
|
+
```bash
|
|
331
|
+
rm web/src/__fixtures__/demo.jsonl
|
|
332
|
+
scrutai review --demo --trace web/src/__fixtures__/demo.jsonl
|
|
333
|
+
```
|
|
334
|
+
|
|
335
|
+
4. Update the event-kind assertions in `tests/test_trace.py`.
|
|
336
|
+
|
|
337
|
+
### Work on the web UI
|
|
338
|
+
|
|
339
|
+
```bash
|
|
340
|
+
scrutai serve # terminal 1: the API on http://127.0.0.1:8765
|
|
341
|
+
cd web && npm run dev # terminal 2: Vite dev server with hot reload; /api is proxied
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
- **All UI state comes from `reduce(events)`** in `web/src/reduce.ts`, a pure function. Put logic
|
|
345
|
+
there, test it in `reduce.test.ts`, and keep components presentational.
|
|
346
|
+
- The production bundle is built into `src/scrutai/web/static/` and **committed**, so
|
|
347
|
+
`pip install` users don't need Node. After any UI change, run `npm run build` and commit the
|
|
348
|
+
result. CI fails if the committed bundle doesn't match the source; the build is deterministic.
|
|
349
|
+
- Support light and dark themes (CSS variables in `styles.css`), keyboard focus, and phone widths:
|
|
350
|
+
the page must never scroll horizontally. `tests/test_ui_e2e.py` checks the last point.
|
|
351
|
+
|
|
352
|
+
### Add an MCP tool
|
|
353
|
+
|
|
354
|
+
The MCP server is a thin adapter, so a new tool is usually a few lines in `build_server()` in
|
|
355
|
+
`src/scrutai/mcp/server.py`.
|
|
356
|
+
|
|
357
|
+
1. **Register it with the typed helpers** from `compat.py`:
|
|
358
|
+
`@tool(server, name=..., title=..., description=..., annotations=..., structured_output=True)`.
|
|
359
|
+
Never import `mcp.server` anywhere else: `compat.py` keeps 1.x and 2.x working.
|
|
360
|
+
2. **Return a Pydantic model** from `schemas.py`, so clients get an output schema.
|
|
361
|
+
3. **Declare honest annotations** with `tool_annotations(...)`. Anything that writes outside the
|
|
362
|
+
server is `destructive=True` and must ask the user with `ask_user()` (see `post_review`).
|
|
363
|
+
4. **Trust no argument.**
|
|
364
|
+
- Resolve paths with `reviewer.roots.resolve()`.
|
|
365
|
+
- Cap sizes.
|
|
366
|
+
- Raise `ToolError` with a clear message for bad input.
|
|
367
|
+
- Inside a resource or prompt, raise `ResourceError` or `invalid_params()` instead: mcp 2 hides
|
|
368
|
+
the text of other exceptions.
|
|
369
|
+
5. **Keep blocking work off the event loop:** `await anyio.to_thread.run_sync(...,
|
|
370
|
+
abandon_on_cancel=True)`, and honour cancellation (see `Reviewer.review`).
|
|
371
|
+
6. **Test it through a real client.** Use `in_memory()` from `tests/mcp_util.py`, which works on
|
|
372
|
+
both SDK majors and can answer elicitation. Add a case to `tests/test_mcp_*.py`, and run the
|
|
373
|
+
MCP job above so it also passes on mcp 2.
|
|
374
|
+
7. **Document it** in the tools table in `docs/MCP.md`.
|
|
375
|
+
|
|
376
|
+
### Update the README's screenshots and recording
|
|
377
|
+
|
|
378
|
+
The README's images and screen recording are captured from real runs in mock mode. After a change
|
|
379
|
+
to the CLI output or the UI, regenerate them:
|
|
380
|
+
|
|
381
|
+
```bash
|
|
382
|
+
python scripts/capture_media.py # needs the dev extra, Chromium for Playwright, and ffmpeg
|
|
383
|
+
```
|
|
384
|
+
|
|
385
|
+
The script writes `docs/images/*.png`, `theater.gif` and `theater.mp4`. Check the result by eye
|
|
386
|
+
before committing; keep the GIF under about 5 MB.
|
|
387
|
+
|
|
388
|
+
## Testing guide
|
|
389
|
+
|
|
390
|
+
The test suite runs **fully offline** and never calls a real model. The main tools:
|
|
391
|
+
|
|
392
|
+
| Need | Use |
|
|
393
|
+
|---|---|
|
|
394
|
+
| A model that behaves realistically | `MockLLMClient` (`scrutai.llm`): deterministic, speaks the real protocol, deliberately noisy. |
|
|
395
|
+
| A model that says exactly what a test needs | A small scripted client with `complete()`, `tokens_used` and `cost_usd`; see `Scripted` in `tests/test_agents.py` and `Critic` in `tests/test_critic.py`. |
|
|
396
|
+
| A real git repository | The `git_repo` fixture in `tests/conftest.py`: pass `{path: content}`, get a repo with a `feature` branch over `main`. |
|
|
397
|
+
| Semgrep without installing it | The `fake_semgrep` fixture in `tests/test_semgrep.py`. |
|
|
398
|
+
| GitHub without the network | The `github` fixture (`FakeGitHub` in `tests/fake_github.py`): an in-process HTTP server; also runs the Action's real shell step. |
|
|
399
|
+
| An MCP client | `tests/mcp_util.py`: `in_memory(server, elicit=...)` on either SDK major, plus `over_http()` for a real `scrutai mcp` process; see `tests/test_mcp_transports.py`. |
|
|
400
|
+
| The live LiteLLM path | `tests/test_live_client.py`: LiteLLM's `mock_response` builds real response objects. |
|
|
401
|
+
| The UI in a browser | `tests/test_ui_e2e.py`: a real server and Chromium; skipped if no browser. Set `SCRUTAI_CHROMIUM` to use a specific binary. |
|
|
402
|
+
| UI state logic | `web/src/reduce.test.ts` (vitest), against a trace recorded from the real backend. |
|
|
403
|
+
|
|
404
|
+
Ground rules:
|
|
405
|
+
|
|
406
|
+
- **Every bug fix gets a regression test** that fails before the fix.
|
|
407
|
+
- **Never skip, disable or loosen a test to get green.** If a test is wrong, fix the test and say why
|
|
408
|
+
in the commit message.
|
|
409
|
+
- **Keep tests deterministic.** Concurrency is real (agents run in parallel), so assert on sorted or
|
|
410
|
+
set results, never on completion order.
|
|
411
|
+
- **Hermetic paths.** Use `tmp_path` or `git_repo`, never the working directory: the agents' tools
|
|
412
|
+
search whatever repository they are pointed at.
|
|
413
|
+
|
|
414
|
+
## Conventions
|
|
415
|
+
|
|
416
|
+
- **Python 3.12+**, fully typed: `mypy --strict` must pass. Use modern syntax: `X | None`, PEP 695
|
|
417
|
+
generics, `StrEnum`.
|
|
418
|
+
- **ruff** for lint and formatting (line length 100). Don't fight the formatter.
|
|
419
|
+
- **Comments explain *why*,** not what. Match the density of the code around you.
|
|
420
|
+
- **Model output is untrusted input.** Anything a model returns (tool arguments, file paths,
|
|
421
|
+
JSON) is validated before use. Tool paths are confined to the repository; patterns are data.
|
|
422
|
+
- **Nothing leaves the machine** except calls to the configured model provider. No telemetry; turn
|
|
423
|
+
off third-party telemetry when integrating a library.
|
|
424
|
+
- **Honest numbers.** README and benchmark claims must be reproducible with a command, and mock-mode
|
|
425
|
+
results must be labelled as such.
|
|
426
|
+
- **Dependencies:** core dependencies stay small. Anything heavy goes behind an optional extra
|
|
427
|
+
and is imported lazily.
|
|
428
|
+
|
|
429
|
+
## Pull requests
|
|
430
|
+
|
|
431
|
+
1. **Branch from `main`**, and keep each commit to one logical change. We use
|
|
432
|
+
[Conventional Commits](https://www.conventionalcommits.org/): `feat(critic): ...`,
|
|
433
|
+
`fix(tools): ...`, `docs: ...`, `test: ...`, `chore: ...`. The body says *why*.
|
|
434
|
+
2. **Run [the checks](#running-the-checks)** locally.
|
|
435
|
+
3. **Open a PR** describing what changed, why, and how you tested it. Include before/after output for
|
|
436
|
+
behaviour changes, and a screenshot for UI changes.
|
|
437
|
+
|
|
438
|
+
Checklist:
|
|
439
|
+
|
|
440
|
+
- [ ] `ruff check`, `ruff format --check`, `mypy src` and `pytest` pass.
|
|
441
|
+
- [ ] `scrutai eval --min-precision 0.95 --min-recall 0.8` passes.
|
|
442
|
+
- [ ] New behaviour has tests; bug fixes have a regression test.
|
|
443
|
+
- [ ] UI changes: `npm run typecheck`, `npm test`, and the rebuilt bundle is committed.
|
|
444
|
+
- [ ] Trace event changes: types, reducer and fixture updated.
|
|
445
|
+
- [ ] Docs updated (README, config reference, `.scrutai.yml`) when user-facing behaviour changes.
|
|
446
|
+
- [ ] An entry under **Unreleased** in [RELEASES.md](RELEASES.md).
|
|
447
|
+
|
|
448
|
+
CI must be green before merge. Reviewers look hardest at the critic, the tool sandbox and anything
|
|
449
|
+
that changes benchmark numbers.
|
|
450
|
+
|
|
451
|
+
## Releasing
|
|
452
|
+
|
|
453
|
+
Versions follow [Semantic Versioning](https://semver.org/). While Scrutai is `0.x`, a minor version
|
|
454
|
+
may change behaviour; RELEASES.md calls those changes out.
|
|
455
|
+
|
|
456
|
+
1. Move the **Unreleased** notes in [RELEASES.md](RELEASES.md) under a new version heading
|
|
457
|
+
(`## 0.X.0`; the release workflow copies that section into the GitHub release).
|
|
458
|
+
2. Bump the version in all three places: `pyproject.toml`, `src/scrutai/__init__.py` and
|
|
459
|
+
`web/package.json` (`cd web && npm version 0.X.0 --no-git-tag-version`). The README's PyPI
|
|
460
|
+
badge updates itself.
|
|
461
|
+
3. Rebuild the UI (`cd web && npm run build`) and commit the bundle.
|
|
462
|
+
4. Merge to `main` with CI green (the `package` job builds the wheel and runs it outside the
|
|
463
|
+
repository), then tag the release: `git tag v0.X.0 && git push origin v0.X.0`.
|
|
464
|
+
|
|
465
|
+
The tag starts `.github/workflows/release.yml`:
|
|
466
|
+
|
|
467
|
+
1. It checks that the tag matches `pyproject.toml`'s version.
|
|
468
|
+
2. It builds the wheel and sdist, runs `twine check`, and smoke-tests the wheel in a clean
|
|
469
|
+
environment.
|
|
470
|
+
3. It publishes to **TestPyPI**, then to **PyPI**, through trusted publishing (no API tokens
|
|
471
|
+
exist).
|
|
472
|
+
4. It creates the GitHub release with the built files attached.
|
|
473
|
+
|
|
474
|
+
To rehearse without publishing to PyPI, run the workflow manually (Actions → release → Run
|
|
475
|
+
workflow): that run stops after TestPyPI. A version number can be uploaded to PyPI only once,
|
|
476
|
+
even if it is deleted later.
|
|
477
|
+
|
|
478
|
+
One-time setup, done by the PyPI project owner: on pypi.org and test.pypi.org add a trusted
|
|
479
|
+
publisher with owner `imkarthiknr`, repository `Scrutai`, workflow `release.yml`, and
|
|
480
|
+
environment `pypi` or `testpypi` respectively.
|
|
481
|
+
|
|
482
|
+
## Debugging tips
|
|
483
|
+
|
|
484
|
+
- **See everything a review did:** `scrutai review ... --trace run.jsonl`, then open it with
|
|
485
|
+
`scrutai serve --replay run.jsonl`, or read the JSONL directly (one event per line, with `seq`).
|
|
486
|
+
- **See what the critic killed:** `--show-dropped`, or the `dropped` array in `--json` output; each
|
|
487
|
+
entry has `critic_note` and a per-round `history`.
|
|
488
|
+
- **A finding is missing:** check routing first. The `plan` event in the trace lists which agents ran
|
|
489
|
+
on which chunk.
|
|
490
|
+
- **The mock model is behaving strangely on a real diff:** remember it is a rule-based stand-in. If
|
|
491
|
+
reviewed code contains protocol-looking text, check that markers are matched line-anchored.
|
|
492
|
+
- **Reproduce CI without ripgrep:** CI runners don't have it, so `grep` falls back to `git grep` or
|
|
493
|
+
pure Python. Run tests with a `PATH` that hides `rg` to check that fallback.
|
|
494
|
+
|
|
495
|
+
## Getting help
|
|
496
|
+
|
|
497
|
+
Open a [GitHub issue](https://github.com/imkarthiknr/Scrutai/issues) for bugs and proposals, or a
|
|
498
|
+
draft pull request early if you want feedback on an approach. For security issues, use
|
|
499
|
+
[private advisories](https://github.com/imkarthiknr/Scrutai/security/advisories/new) instead of a
|
|
500
|
+
public issue.
|