scrutai 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. scrutai-0.4.0/.gitignore +12 -0
  2. scrutai-0.4.0/.scrutai.yml +21 -0
  3. scrutai-0.4.0/DEVELOPMENT.md +500 -0
  4. scrutai-0.4.0/LICENSE +191 -0
  5. scrutai-0.4.0/PKG-INFO +734 -0
  6. scrutai-0.4.0/README.md +672 -0
  7. scrutai-0.4.0/RELEASES.md +297 -0
  8. scrutai-0.4.0/action.yml +94 -0
  9. scrutai-0.4.0/docs/ARCHITECTURE.md +136 -0
  10. scrutai-0.4.0/docs/MCP.md +267 -0
  11. scrutai-0.4.0/docs/MCP_PLAN.md +156 -0
  12. scrutai-0.4.0/examples/mcp/README.md +24 -0
  13. scrutai-0.4.0/examples/mcp/claude-code.sh +12 -0
  14. scrutai-0.4.0/examples/mcp/claude_desktop_config.json +16 -0
  15. scrutai-0.4.0/examples/mcp/cursor-mcp.json +12 -0
  16. scrutai-0.4.0/examples/mcp/try_it.py +89 -0
  17. scrutai-0.4.0/examples/scrutai-workflow.yml +30 -0
  18. scrutai-0.4.0/pyproject.toml +112 -0
  19. scrutai-0.4.0/src/scrutai/__init__.py +9 -0
  20. scrutai-0.4.0/src/scrutai/agents/__init__.py +45 -0
  21. scrutai-0.4.0/src/scrutai/agents/base.py +221 -0
  22. scrutai-0.4.0/src/scrutai/agents/correctness.py +21 -0
  23. scrutai-0.4.0/src/scrutai/agents/crewai_backend.py +235 -0
  24. scrutai-0.4.0/src/scrutai/agents/performance.py +20 -0
  25. scrutai-0.4.0/src/scrutai/agents/security.py +46 -0
  26. scrutai-0.4.0/src/scrutai/agents/style.py +20 -0
  27. scrutai-0.4.0/src/scrutai/agents/tests.py +21 -0
  28. scrutai-0.4.0/src/scrutai/cli.py +362 -0
  29. scrutai-0.4.0/src/scrutai/concurrency.py +17 -0
  30. scrutai-0.4.0/src/scrutai/config.py +127 -0
  31. scrutai-0.4.0/src/scrutai/critic.py +150 -0
  32. scrutai-0.4.0/src/scrutai/demo.py +47 -0
  33. scrutai-0.4.0/src/scrutai/diff.py +102 -0
  34. scrutai-0.4.0/src/scrutai/eval/__init__.py +3 -0
  35. scrutai-0.4.0/src/scrutai/eval/cases.jsonl +50 -0
  36. scrutai-0.4.0/src/scrutai/eval/harness.py +321 -0
  37. scrutai-0.4.0/src/scrutai/github.py +190 -0
  38. scrutai-0.4.0/src/scrutai/inputs.py +99 -0
  39. scrutai-0.4.0/src/scrutai/llm.py +213 -0
  40. scrutai-0.4.0/src/scrutai/mcp/__init__.py +1 -0
  41. scrutai-0.4.0/src/scrutai/mcp/compat.py +160 -0
  42. scrutai-0.4.0/src/scrutai/mcp/schemas.py +191 -0
  43. scrutai-0.4.0/src/scrutai/mcp/security.py +136 -0
  44. scrutai-0.4.0/src/scrutai/mcp/server.py +847 -0
  45. scrutai-0.4.0/src/scrutai/mock.py +666 -0
  46. scrutai-0.4.0/src/scrutai/models.py +206 -0
  47. scrutai-0.4.0/src/scrutai/orchestrator.py +237 -0
  48. scrutai-0.4.0/src/scrutai/patch.py +174 -0
  49. scrutai-0.4.0/src/scrutai/progress.py +62 -0
  50. scrutai-0.4.0/src/scrutai/py.typed +0 -0
  51. scrutai-0.4.0/src/scrutai/report.py +137 -0
  52. scrutai-0.4.0/src/scrutai/router.py +92 -0
  53. scrutai-0.4.0/src/scrutai/rules/semgrep.yml +69 -0
  54. scrutai-0.4.0/src/scrutai/runs.py +109 -0
  55. scrutai-0.4.0/src/scrutai/tools/__init__.py +15 -0
  56. scrutai-0.4.0/src/scrutai/tools/repo.py +123 -0
  57. scrutai-0.4.0/src/scrutai/tools/semgrep.py +131 -0
  58. scrutai-0.4.0/src/scrutai/tools/toolbox.py +133 -0
  59. scrutai-0.4.0/src/scrutai/trace.py +221 -0
  60. scrutai-0.4.0/src/scrutai/web/__init__.py +1 -0
  61. scrutai-0.4.0/src/scrutai/web/server.py +179 -0
  62. scrutai-0.4.0/src/scrutai/web/static/assets/index-BOOR7lCF.css +1 -0
  63. scrutai-0.4.0/src/scrutai/web/static/assets/index-Bz6jMS2_.js +9 -0
  64. scrutai-0.4.0/src/scrutai/web/static/index.html +15 -0
  65. scrutai-0.4.0/tests/conftest.py +44 -0
  66. scrutai-0.4.0/tests/fake_github.py +128 -0
  67. scrutai-0.4.0/tests/mcp_util.py +89 -0
  68. scrutai-0.4.0/tests/test_agents.py +159 -0
  69. scrutai-0.4.0/tests/test_budget.py +60 -0
  70. scrutai-0.4.0/tests/test_cancel_progress.py +107 -0
  71. scrutai-0.4.0/tests/test_chunking.py +102 -0
  72. scrutai-0.4.0/tests/test_cli.py +109 -0
  73. scrutai-0.4.0/tests/test_crewai_backend.py +169 -0
  74. scrutai-0.4.0/tests/test_critic.py +192 -0
  75. scrutai-0.4.0/tests/test_diff.py +105 -0
  76. scrutai-0.4.0/tests/test_eval.py +90 -0
  77. scrutai-0.4.0/tests/test_github.py +127 -0
  78. scrutai-0.4.0/tests/test_inputs.py +69 -0
  79. scrutai-0.4.0/tests/test_live_client.py +97 -0
  80. scrutai-0.4.0/tests/test_mcp_benchmark.py +108 -0
  81. scrutai-0.4.0/tests/test_mcp_github.py +153 -0
  82. scrutai-0.4.0/tests/test_mcp_prompts.py +69 -0
  83. scrutai-0.4.0/tests/test_mcp_server.py +357 -0
  84. scrutai-0.4.0/tests/test_mcp_transports.py +145 -0
  85. scrutai-0.4.0/tests/test_parallel.py +75 -0
  86. scrutai-0.4.0/tests/test_perf_style.py +68 -0
  87. scrutai-0.4.0/tests/test_router.py +88 -0
  88. scrutai-0.4.0/tests/test_semgrep.py +121 -0
  89. scrutai-0.4.0/tests/test_server.py +119 -0
  90. scrutai-0.4.0/tests/test_smoke.py +35 -0
  91. scrutai-0.4.0/tests/test_trace.py +145 -0
  92. scrutai-0.4.0/tests/test_ui_e2e.py +116 -0
@@ -0,0 +1,12 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ venv/
5
+ *.egg-info/
6
+ dist/
7
+ build/
8
+ .pytest_cache/
9
+ .mypy_cache/
10
+ .ruff_cache/
11
+ .env
12
+ .langfuse/
@@ -0,0 +1,21 @@
1
+ # Scrutai configuration. All values shown are defaults — delete any you don't override.
2
+ enabled_agents: [security, correctness, tests, performance, style]
3
+ min_severity: low
4
+ min_confidence: 0.6
5
+ max_critic_rounds: 2
6
+ token_budget: 200000 # hard cap per review; 0 disables
7
+ max_cost_usd: 0.0 # optional dollar cap (live mode); 0 disables
8
+ concurrency: 4 # parallel critic/defense calls
9
+ fail_on: high
10
+ llm_mode: mock # switch to "live" and export a provider key to run for real
11
+ routing: heuristic # or "llm": the router model may narrow the selection further
12
+ max_agent_steps: 4 # ReAct budget per specialist (tool calls + final answer)
13
+ semgrep: auto # auto | off | required
14
+ semgrep_config: bundled # or any semgrep --config value, e.g. p/default
15
+ tracing: none # none | otel | langfuse (and: scrutai review --trace run.jsonl)
16
+ include: ["**/*"]
17
+ exclude: ["**/vendor/**", "**/*.lock", "**/dist/**"]
18
+ models: # any LiteLLM model string; cheap router, strong critic
19
+ router: anthropic/claude-haiku-4-5
20
+ specialist: anthropic/claude-sonnet-5-5
21
+ critic: anthropic/claude-opus-5-5
@@ -0,0 +1,500 @@
1
+ # Developing Scrutai
2
+
3
+ Thanks for wanting to make Scrutai better. This guide gets you from a fresh clone to a merged pull
4
+ request: setup, how the code fits together, step-by-step recipes for the most common changes, how
5
+ the tests work, and what reviewers look for.
6
+
7
+ New here? Read [README.md](README.md) first for *what* Scrutai does, and
8
+ [docs/ARCHITECTURE.md](docs/ARCHITECTURE.md) for *why* it is built the way it is.
9
+
10
+ ## Contents
11
+
12
+ - [Ways to contribute](#ways-to-contribute)
13
+ - [Setup](#setup)
14
+ - [Running the checks](#running-the-checks)
15
+ - [The architecture in five minutes](#the-architecture-in-five-minutes)
16
+ - [Code map](#code-map)
17
+ - [Recipes](#recipes)
18
+ - [Add a specialist agent](#add-a-specialist-agent)
19
+ - [Add a repository tool](#add-a-repository-tool)
20
+ - [Add a Semgrep rule](#add-a-semgrep-rule)
21
+ - [Add benchmark cases](#add-benchmark-cases)
22
+ - [Add a framework backend](#add-a-framework-backend)
23
+ - [Add or change a trace event](#add-or-change-a-trace-event)
24
+ - [Work on the web UI](#work-on-the-web-ui)
25
+ - [Add an MCP tool](#add-an-mcp-tool)
26
+ - [Update the README's screenshots and recording](#update-the-readmes-screenshots-and-recording)
27
+ - [Testing guide](#testing-guide)
28
+ - [Conventions](#conventions)
29
+ - [Pull requests](#pull-requests)
30
+ - [Releasing](#releasing)
31
+ - [Debugging tips](#debugging-tips)
32
+ - [Getting help](#getting-help)
33
+
34
+ ## Ways to contribute
35
+
36
+ - **Benchmark cases:** the most valuable contribution. Especially *traps*: code that looks like a
37
+ bug but isn't, so the critic is forced to prove itself. See
38
+ [Add benchmark cases](#add-benchmark-cases).
39
+ - **Semgrep rules** for the bundled offline ruleset, ideally for languages beyond Python.
40
+ - **New specialists** (e.g. accessibility, i18n, infrastructure-as-code) or new categories for
41
+ existing ones.
42
+ - **Framework backends** (e.g. Google ADK) behind the `Specialist._loop` seam.
43
+ - **UI improvements** to the agent theater.
44
+ - **Bug reports** with a failing diff attached: `scrutai review --diff bug.patch --trace bug.jsonl`
45
+ gives us everything we need to reproduce it.
46
+
47
+ ## Setup
48
+
49
+ **You need:** Python 3.12+, `git`, and Node.js 22+ (only for the web UI). Optional:
50
+ [`uv`](https://docs.astral.sh/uv/) (faster installs), ripgrep, Semgrep.
51
+
52
+ ```bash
53
+ git clone https://github.com/imkarthiknr/Scrutai.git
54
+ cd Scrutai
55
+
56
+ # Python: with uv (recommended)...
57
+ uv venv --python 3.12
58
+ uv pip install -e ".[dev]"
59
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
60
+
61
+ # ...or with plain pip
62
+ python -m venv .venv && source .venv/bin/activate
63
+ pip install -e ".[dev]"
64
+
65
+ # Browser for the end-to-end UI tests (skipped automatically if missing)
66
+ playwright install chromium
67
+
68
+ # Optional: run lint and format checks on every commit
69
+ pre-commit install
70
+
71
+ # Web UI (only if you work on it)
72
+ cd web && npm ci && cd ..
73
+ ```
74
+
75
+ The `dev` extra installs everything the test suite exercises: pytest, ruff, mypy, the web server,
76
+ Playwright, OpenTelemetry, CrewAI and the MCP SDK. CrewAI pins the MCP SDK to 1.28, so the
77
+ main environment tests MCP on 1.x; CI tests it on 2.x as well (see below).
78
+
79
+ Check that it works:
80
+
81
+ ```bash
82
+ scrutai review --demo # should report "5 issue(s) upheld" and exit with code 1
83
+ pytest -q # should report 200+ passed
84
+ ```
85
+
86
+ You never need an API key to develop. Everything, including the whole test suite and CI, runs
87
+ against the offline mock model, so no contributor or CI run spends money or needs a secret.
88
+
89
+ **Optional: trying a change against a real model.** Use your own provider key (see the README's
90
+ [Running with a real model](README.md#running-with-a-real-model-bring-your-own-key) for providers
91
+ and variable names):
92
+
93
+ ```bash
94
+ export ANTHROPIC_API_KEY=... # your key, in your shell only
95
+ scrutai review --demo --config my-live.yml # a copy of .scrutai.yml with llm_mode: live
96
+ ```
97
+
98
+ - Keep live configs and keys out of commits: put keys only in the environment, and set a
99
+ `max_cost_usd` cap in live configs.
100
+ - Never add a test that needs a real key. Use `MockLLMClient`, a scripted client, or LiteLLM's
101
+ `mock_response` (see the [testing guide](#testing-guide)).
102
+ - When a PR changes prompts or the critic, say in its description whether you checked it live,
103
+ with which model.
104
+
105
+ ## Running the checks
106
+
107
+ CI runs exactly these. Run them before you push:
108
+
109
+ ```bash
110
+ # Python job
111
+ ruff check .
112
+ ruff format --check .
113
+ mypy src # strict mode
114
+ pytest -q # incl. browser E2E if Chromium is installed
115
+ scrutai eval --min-precision 0.95 --min-recall 0.8 # the benchmark gate
116
+
117
+ # MCP job: the MCP tests again, on MCP SDK 2.x (a separate venv: CrewAI pins 1.x)
118
+ uv venv .venv-mcp2 --python 3.12
119
+ uv pip install --python .venv-mcp2 -e . "mcp>=2" pytest httpx pyyaml
120
+ .venv-mcp2/bin/pytest -q tests/test_mcp_*.py
121
+
122
+ # Web job (from web/)
123
+ npm run typecheck
124
+ npm test # vitest
125
+ npm run build && git diff --exit-code -- ../src/scrutai/web/static # bundle must be committed
126
+ ```
127
+
128
+ Useful variations:
129
+
130
+ ```bash
131
+ pytest -q tests/test_critic.py # one file
132
+ pytest -q -k "debate or withdraw" # by name
133
+ pytest -q --ignore=tests/test_ui_e2e.py # skip browser tests for a fast loop
134
+ ruff check . --fix && ruff format . # auto-fix lint and formatting
135
+ ```
136
+
137
+ ## The architecture in five minutes
138
+
139
+ Everything goes through one function, `review_diff(diff, config, llm)` in
140
+ `src/scrutai/orchestrator.py`. The CLI, the GitHub Action, the web server and the benchmark are
141
+ thin wrappers around it.
142
+
143
+ ```text
144
+ DiffContext ──► route ──► specialist × (agent, chunk) ──► collect ──► critic ⇄ defend ──► verdict ──► ReviewResult
145
+ │ │ │ │
146
+ router.py agents/*.py dedupe() critic.py
147
+ (ReAct over tools/)
148
+ ```
149
+
150
+ A review goes like this:
151
+
152
+ 1. **Diff in.** `diff.py` turns a git range, a patch or a PR into a `DiffContext` (a list of
153
+ `ChangedFile`s with raw patches). `patch.py` does the pure parsing: added lines with real
154
+ line numbers, slicing, globs.
155
+ 2. **Route.** `router.py` decides which specialists a change needs, using file kinds and the "risk
156
+ surface" of added lines. The orchestrator splits the diff into chunks (`chunk_diff`) and routes
157
+ each chunk on its own.
158
+ 3. **Fan out.** LangGraph `Send` starts one branch per (agent, chunk). Each branch builds a
159
+ specialist and calls `review()`.
160
+ 4. **ReAct.** `Specialist.review()` (in `agents/base.py`) builds a prompt, then calls `_loop()`: the
161
+ model answers with either a tool action (`{"action": {...}}`) or findings
162
+ (`{"findings": [...]}`). Tools run through a sandboxed `Toolbox` (`tools/`).
163
+ 5. **Collect.** Findings from all branches are sorted and deduplicated by `(file, line, category)`.
164
+ 6. **Critic.** `critic.py` judges each finding against the cited code: uphold, downgrade, kill or
165
+ challenge. Challenged findings go to `defend` (the specialist's `defend()`), then back to the
166
+ critic. `max_critic_rounds` bounds the loop.
167
+ 7. **Verdict.** Survivors become `ReviewResult.findings`, everything else goes to `dropped` with
168
+ the reason.
169
+
170
+ Three cross-cutting pieces wrap every LLM call:
171
+
172
+ - **`BudgetedClient`** (`llm.py`) enforces `token_budget` and `max_cost_usd`.
173
+ - **`TracingClient`** (`trace.py`) records spans when a tracer is active.
174
+ - **`MockLLMClient`** (`mock.py`) speaks the exact same protocol offline, which is what makes the
175
+ whole system testable without a model.
176
+
177
+ **The protocol is JSON, line-anchored.** Reviewed code always appears as `L<n>: <code>`, and protocol
178
+ markers (`--- STEP`, `--- FINAL`, `OBSERVATION:`) are only recognised at the start of a line. That
179
+ way, a diff that happens to *contain* those strings (Scrutai reviewing itself, for example) can't
180
+ spoof the protocol.
181
+
182
+ ## Code map
183
+
184
+ | Path | Responsibility |
185
+ |---|---|
186
+ | `src/scrutai/cli.py` | `review`, `serve`, `mcp` and `eval` commands; output formats; exit codes. |
187
+ | `src/scrutai/config.py` | `ScrutaiConfig`: every option, defaults and validation. |
188
+ | `src/scrutai/models.py` | Pydantic models: `ChangedFile`, `DiffContext`, `Finding`, `ReviewResult`. |
189
+ | `src/scrutai/diff.py` | Git and patch input, ref validation, include/exclude filters, chunking. |
190
+ | `src/scrutai/patch.py` | Pure unified-diff parsing; no I/O. |
191
+ | `src/scrutai/router.py` | Heuristic and optional LLM routing. |
192
+ | `src/scrutai/orchestrator.py` | The LangGraph graph and `review_diff()`. |
193
+ | `src/scrutai/agents/base.py` | `Specialist`: prompts, the `_loop` seam, finding parsing, `defend()`. |
194
+ | `src/scrutai/agents/<name>.py` | One specialist each: role, categories, tools, file kinds. |
195
+ | `src/scrutai/agents/crewai_backend.py` | The CrewAI backend: tool adapters and the protocol bridge. |
196
+ | `src/scrutai/critic.py` | The critic prompt, decisions, and per-round judging. |
197
+ | `src/scrutai/tools/` | `repo.py` (the functions), `toolbox.py` (registry + sandbox), `semgrep.py`. |
198
+ | `src/scrutai/llm.py` | `LiteLLMClient`, `BudgetedClient`, `extract_json`, `make_client`. |
199
+ | `src/scrutai/mock.py` | The offline model: specialist rules, the mock critic, mock defenses. |
200
+ | `src/scrutai/report.py` | Markdown and SARIF rendering; finding fingerprints are in `models.py`. |
201
+ | `src/scrutai/github.py` | GitHub REST client; idempotent `publish()`. |
202
+ | `src/scrutai/trace.py` | Tracer, spans, events, OpenTelemetry and Langfuse hooks. |
203
+ | `src/scrutai/eval/harness.py` | Benchmark loading, hermetic case runs, metrics, `--compare`. |
204
+ | `src/scrutai/inputs.py` | `Source` + `prepare()`: the one input path (demo, git range, patch, file, PR) every front end uses. |
205
+ | `src/scrutai/runs.py` | `Run` / `RunStore`: reviews in flight and finished, shared by the web and MCP servers. |
206
+ | `src/scrutai/progress.py` | `ProgressListener`: trace events → "step N of M" progress. |
207
+ | `src/scrutai/mcp/server.py` | The MCP server: tools, resources, prompts; `Reviewer` runs and keeps reviews. |
208
+ | `src/scrutai/mcp/compat.py` | The only module that imports the MCP SDK; hides 1.x vs 2.x differences. |
209
+ | `src/scrutai/mcp/schemas.py` | Structured tool output (`ReviewSummary`, `FindingExplanation`, …). |
210
+ | `src/scrutai/mcp/security.py` | Root allowlist, input caps, HTTP bearer-token and `Host` guard. |
211
+ | `src/scrutai/web/server.py` | FastAPI app: runs, SSE event stream, replay. |
212
+ | `src/scrutai/web/static/` | **Built** UI bundle (generated; never edit by hand). |
213
+ | `web/src/` | The React UI source: `reduce.ts` (all UI state), `components/`, `api.ts`. |
214
+ | `action.yml` | The composite GitHub Action. |
215
+ | `src/scrutai/eval/cases.jsonl` | The labelled benchmark (shipped in the package). |
216
+
217
+ ## Recipes
218
+
219
+ ### Add a specialist agent
220
+
221
+ Example: an `accessibility` agent for front-end code.
222
+
223
+ 1. **Create `src/scrutai/agents/accessibility.py`:**
224
+
225
+ ```python
226
+ from __future__ import annotations
227
+
228
+ from typing import ClassVar
229
+
230
+ from .base import Specialist
231
+
232
+
233
+ class AccessibilityAgent(Specialist):
234
+ name = "accessibility"
235
+ role = "You find accessibility problems in UI code: missing labels, alt text, focus traps."
236
+ categories: ClassVar[dict[str, str]] = {
237
+ "missing_alt_text": "an <img> without meaningful alt text",
238
+ "unlabelled_control": "an input or button with no accessible name",
239
+ }
240
+ tools: ClassVar[list[str]] = ["read_file", "grep"]
241
+ kinds = ("code",) # which ChangedFile.kind values this agent sees
242
+ ```
243
+
244
+ 2. **Register it** in `src/scrutai/agents/__init__.py`: add it to `REGISTRY` and `__all__`.
245
+ 3. **Route it** in `src/scrutai/router.py`: add an entry to the `wants` dict in
246
+ `heuristic_route()`. Without one, the agent runs on every reviewable change.
247
+ 4. **Enable it by default** (optional): add it to `enabled_agents` in `config.py` **and**
248
+ `.scrutai.yml`. `test_default_config_matches_shipped_yaml` keeps the two in sync.
249
+ 5. **Teach the mock** in `src/scrutai/mock.py`: add `Rule(...)`s with your agent name, so the agent
250
+ does something offline. If a rule needs the critic to kill a known false-positive pattern, add
251
+ that check in `MockLLMClient._critic`.
252
+ 6. **Add benchmark cases** for each category, including at least one trap (see below).
253
+ 7. **Show it in the UI:** add the name to `ALL_AGENTS` in `web/src/reduce.ts`, then rebuild the UI.
254
+ 8. **Test it:** a unit test in the style of `tests/test_perf_style.py`, and make sure
255
+ `scrutai eval` still passes the gate.
256
+
257
+ ### Add a repository tool
258
+
259
+ 1. Write the function in `src/scrutai/tools/repo.py`. Treat **every argument as hostile**: confine
260
+ paths with `_confine()`, pass patterns as data (`-e PATTERN`, never as a flag), bound the output.
261
+ 2. Register it in `TOOLS` in `src/scrutai/tools/toolbox.py` with a `Tool(name, signature,
262
+ description, run)`. The signature and description go into every prompt that offers the tool.
263
+ 3. Allow it on the agents that should use it (their `tools` class variable).
264
+ 4. Add an argument schema to `_SCHEMAS` in `src/scrutai/agents/crewai_backend.py`, so CrewAI can
265
+ validate calls.
266
+ 5. Test it directly, *and* test that a model can't escape the sandbox through it (see
267
+ `test_read_file_is_confined_to_repo`).
268
+
269
+ ### Add a Semgrep rule
270
+
271
+ 1. Add the rule to `src/scrutai/rules/semgrep.yml`. Use the id `scrutai.<category>.<short-name>`:
272
+ the category becomes the finding category automatically.
273
+ 2. Use `pattern-not` to exclude the safe forms (constant arguments, safe loaders, explicit opt-outs).
274
+ 3. Verify it against the real binary on positives *and* traps:
275
+
276
+ ```bash
277
+ pip install semgrep
278
+ semgrep scan --json --metrics=off --config src/scrutai/rules/semgrep.yml path/to/samples.py
279
+ ```
280
+
281
+ 4. Tests use a fake `semgrep` binary (see the `fake_semgrep` fixture in `tests/test_semgrep.py`), so
282
+ CI doesn't need Semgrep installed.
283
+
284
+ ### Add benchmark cases
285
+
286
+ Append JSON lines to `src/scrutai/eval/cases.jsonl`:
287
+
288
+ ```json
289
+ {"id": "trap-secret-env", "file": "settings.py", "patch": "+API_KEY = os.environ[\"API_KEY\"]\n", "labels": [], "note": "read from the environment"}
290
+ ```
291
+
292
+ - `labels` is a **multiset of categories** a correct reviewer reports; `[]` means the change is clean.
293
+ - `repo` (optional) adds other files to the case's temporary repository, e.g. an existing test that
294
+ should stop a `missing_tests` finding.
295
+ - Prefer **traps**: realistic code that a naive reviewer would flag.
296
+ - If the mock model can't catch something a good reviewer should, keep the case and label it
297
+ honestly; it becomes a documented expected miss (ids starting with `miss-`). **Never tune the
298
+ mock to the benchmark, and never delete a case to raise a number.**
299
+
300
+ Run `scrutai eval --report eval.md` and read the "Cases with errors" table.
301
+
302
+ ### Add a framework backend
303
+
304
+ Specialists share everything except one method: `Specialist._loop(system, transcript, toolbox,
305
+ answer_keys, final_note)`, which runs the ReAct loop and returns the answer payload, or `None`.
306
+
307
+ 1. Write a mixin that overrides `_loop` (see `CrewAIMixin` in `agents/crewai_backend.py`). Drive
308
+ your framework's agent with the given `system` prompt and `transcript`, expose
309
+ `toolbox.allowed` as the framework's tools (each calling `toolbox.run(name, args)`), and return
310
+ a dict containing one of `answer_keys`.
311
+ 2. Reach the model through `self.llm` (Scrutai's client) rather than letting the framework call a
312
+ provider directly. That keeps budgets, cost tracking, tracing and mock mode working, and keeps
313
+ `--compare` fair.
314
+ 3. Add the backend name to `BACKENDS` and to `agent_class()` in `agents/__init__.py`, and add an
315
+ optional extra in `pyproject.toml`.
316
+ 4. Prove parity: `scrutai eval --compare <backend>` should report 100% per-case agreement in mock
317
+ mode. Copy the structure of `tests/test_crewai_backend.py`.
318
+
319
+ ### Add or change a trace event
320
+
321
+ Trace events feed `--trace` files, OpenTelemetry and the agent theater.
322
+
323
+ 1. Emit it with `emit("kind", **fields)` (a point event) or `span("kind", name, ...)` (start/end
324
+ pair) from `scrutai.trace`.
325
+ 2. Add its shape to the `TraceEvent` union in `web/src/types.ts` and handle it in `apply()` in
326
+ `web/src/reduce.ts`.
327
+ 3. **Regenerate the recorded fixture** that the UI tests run on. The file is opened in append mode,
328
+ so delete it first:
329
+
330
+ ```bash
331
+ rm web/src/__fixtures__/demo.jsonl
332
+ scrutai review --demo --trace web/src/__fixtures__/demo.jsonl
333
+ ```
334
+
335
+ 4. Update the event-kind assertions in `tests/test_trace.py`.
336
+
337
+ ### Work on the web UI
338
+
339
+ ```bash
340
+ scrutai serve # terminal 1: the API on http://127.0.0.1:8765
341
+ cd web && npm run dev # terminal 2: Vite dev server with hot reload; /api is proxied
342
+ ```
343
+
344
+ - **All UI state comes from `reduce(events)`** in `web/src/reduce.ts`, a pure function. Put logic
345
+ there, test it in `reduce.test.ts`, and keep components presentational.
346
+ - The production bundle is built into `src/scrutai/web/static/` and **committed**, so
347
+ `pip install` users don't need Node. After any UI change, run `npm run build` and commit the
348
+ result. CI fails if the committed bundle doesn't match the source; the build is deterministic.
349
+ - Support light and dark themes (CSS variables in `styles.css`), keyboard focus, and phone widths:
350
+ the page must never scroll horizontally. `tests/test_ui_e2e.py` checks the last point.
351
+
352
+ ### Add an MCP tool
353
+
354
+ The MCP server is a thin adapter, so a new tool is usually a few lines in `build_server()` in
355
+ `src/scrutai/mcp/server.py`.
356
+
357
+ 1. **Register it with the typed helpers** from `compat.py`:
358
+ `@tool(server, name=..., title=..., description=..., annotations=..., structured_output=True)`.
359
+ Never import `mcp.server` anywhere else: `compat.py` keeps 1.x and 2.x working.
360
+ 2. **Return a Pydantic model** from `schemas.py`, so clients get an output schema.
361
+ 3. **Declare honest annotations** with `tool_annotations(...)`. Anything that writes outside the
362
+ server is `destructive=True` and must ask the user with `ask_user()` (see `post_review`).
363
+ 4. **Trust no argument.**
364
+ - Resolve paths with `reviewer.roots.resolve()`.
365
+ - Cap sizes.
366
+ - Raise `ToolError` with a clear message for bad input.
367
+ - Inside a resource or prompt, raise `ResourceError` or `invalid_params()` instead: mcp 2 hides
368
+ the text of other exceptions.
369
+ 5. **Keep blocking work off the event loop:** `await anyio.to_thread.run_sync(...,
370
+ abandon_on_cancel=True)`, and honour cancellation (see `Reviewer.review`).
371
+ 6. **Test it through a real client.** Use `in_memory()` from `tests/mcp_util.py`, which works on
372
+ both SDK majors and can answer elicitation. Add a case to `tests/test_mcp_*.py`, and run the
373
+ MCP job above so it also passes on mcp 2.
374
+ 7. **Document it** in the tools table in `docs/MCP.md`.
375
+
376
+ ### Update the README's screenshots and recording
377
+
378
+ The README's images and screen recording are captured from real runs in mock mode. After a change
379
+ to the CLI output or the UI, regenerate them:
380
+
381
+ ```bash
382
+ python scripts/capture_media.py # needs the dev extra, Chromium for Playwright, and ffmpeg
383
+ ```
384
+
385
+ The script writes `docs/images/*.png`, `theater.gif` and `theater.mp4`. Check the result by eye
386
+ before committing; keep the GIF under about 5 MB.
387
+
388
+ ## Testing guide
389
+
390
+ The test suite runs **fully offline** and never calls a real model. The main tools:
391
+
392
+ | Need | Use |
393
+ |---|---|
394
+ | A model that behaves realistically | `MockLLMClient` (`scrutai.llm`): deterministic, speaks the real protocol, deliberately noisy. |
395
+ | A model that says exactly what a test needs | A small scripted client with `complete()`, `tokens_used` and `cost_usd`; see `Scripted` in `tests/test_agents.py` and `Critic` in `tests/test_critic.py`. |
396
+ | A real git repository | The `git_repo` fixture in `tests/conftest.py`: pass `{path: content}`, get a repo with a `feature` branch over `main`. |
397
+ | Semgrep without installing it | The `fake_semgrep` fixture in `tests/test_semgrep.py`. |
398
+ | GitHub without the network | The `github` fixture (`FakeGitHub` in `tests/fake_github.py`): an in-process HTTP server; also runs the Action's real shell step. |
399
+ | An MCP client | `tests/mcp_util.py`: `in_memory(server, elicit=...)` on either SDK major, plus `over_http()` for a real `scrutai mcp` process; see `tests/test_mcp_transports.py`. |
400
+ | The live LiteLLM path | `tests/test_live_client.py`: LiteLLM's `mock_response` builds real response objects. |
401
+ | The UI in a browser | `tests/test_ui_e2e.py`: a real server and Chromium; skipped if no browser. Set `SCRUTAI_CHROMIUM` to use a specific binary. |
402
+ | UI state logic | `web/src/reduce.test.ts` (vitest), against a trace recorded from the real backend. |
403
+
404
+ Ground rules:
405
+
406
+ - **Every bug fix gets a regression test** that fails before the fix.
407
+ - **Never skip, disable or loosen a test to get green.** If a test is wrong, fix the test and say why
408
+ in the commit message.
409
+ - **Keep tests deterministic.** Concurrency is real (agents run in parallel), so assert on sorted or
410
+ set results, never on completion order.
411
+ - **Hermetic paths.** Use `tmp_path` or `git_repo`, never the working directory: the agents' tools
412
+ search whatever repository they are pointed at.
413
+
414
+ ## Conventions
415
+
416
+ - **Python 3.12+**, fully typed: `mypy --strict` must pass. Use modern syntax: `X | None`, PEP 695
417
+ generics, `StrEnum`.
418
+ - **ruff** for lint and formatting (line length 100). Don't fight the formatter.
419
+ - **Comments explain *why*,** not what. Match the density of the code around you.
420
+ - **Model output is untrusted input.** Anything a model returns (tool arguments, file paths,
421
+ JSON) is validated before use. Tool paths are confined to the repository; patterns are data.
422
+ - **Nothing leaves the machine** except calls to the configured model provider. No telemetry; turn
423
+ off third-party telemetry when integrating a library.
424
+ - **Honest numbers.** README and benchmark claims must be reproducible with a command, and mock-mode
425
+ results must be labelled as such.
426
+ - **Dependencies:** core dependencies stay small. Anything heavy goes behind an optional extra
427
+ and is imported lazily.
428
+
429
+ ## Pull requests
430
+
431
+ 1. **Branch from `main`**, and keep each commit to one logical change. We use
432
+ [Conventional Commits](https://www.conventionalcommits.org/): `feat(critic): ...`,
433
+ `fix(tools): ...`, `docs: ...`, `test: ...`, `chore: ...`. The body says *why*.
434
+ 2. **Run [the checks](#running-the-checks)** locally.
435
+ 3. **Open a PR** describing what changed, why, and how you tested it. Include before/after output for
436
+ behaviour changes, and a screenshot for UI changes.
437
+
438
+ Checklist:
439
+
440
+ - [ ] `ruff check`, `ruff format --check`, `mypy src` and `pytest` pass.
441
+ - [ ] `scrutai eval --min-precision 0.95 --min-recall 0.8` passes.
442
+ - [ ] New behaviour has tests; bug fixes have a regression test.
443
+ - [ ] UI changes: `npm run typecheck`, `npm test`, and the rebuilt bundle is committed.
444
+ - [ ] Trace event changes: types, reducer and fixture updated.
445
+ - [ ] Docs updated (README, config reference, `.scrutai.yml`) when user-facing behaviour changes.
446
+ - [ ] An entry under **Unreleased** in [RELEASES.md](RELEASES.md).
447
+
448
+ CI must be green before merge. Reviewers look hardest at the critic, the tool sandbox and anything
449
+ that changes benchmark numbers.
450
+
451
+ ## Releasing
452
+
453
+ Versions follow [Semantic Versioning](https://semver.org/). While Scrutai is `0.x`, a minor version
454
+ may change behaviour; RELEASES.md calls those changes out.
455
+
456
+ 1. Move the **Unreleased** notes in [RELEASES.md](RELEASES.md) under a new version heading
457
+ (`## 0.X.0`; the release workflow copies that section into the GitHub release).
458
+ 2. Bump the version in all three places: `pyproject.toml`, `src/scrutai/__init__.py` and
459
+ `web/package.json` (`cd web && npm version 0.X.0 --no-git-tag-version`). The README's PyPI
460
+ badge updates itself.
461
+ 3. Rebuild the UI (`cd web && npm run build`) and commit the bundle.
462
+ 4. Merge to `main` with CI green (the `package` job builds the wheel and runs it outside the
463
+ repository), then tag the release: `git tag v0.X.0 && git push origin v0.X.0`.
464
+
465
+ The tag starts `.github/workflows/release.yml`:
466
+
467
+ 1. It checks that the tag matches `pyproject.toml`'s version.
468
+ 2. It builds the wheel and sdist, runs `twine check`, and smoke-tests the wheel in a clean
469
+ environment.
470
+ 3. It publishes to **TestPyPI**, then to **PyPI**, through trusted publishing (no API tokens
471
+ exist).
472
+ 4. It creates the GitHub release with the built files attached.
473
+
474
+ To rehearse without publishing to PyPI, run the workflow manually (Actions → release → Run
475
+ workflow): that run stops after TestPyPI. A version number can be uploaded to PyPI only once,
476
+ even if it is deleted later.
477
+
478
+ One-time setup, done by the PyPI project owner: on pypi.org and test.pypi.org add a trusted
479
+ publisher with owner `imkarthiknr`, repository `Scrutai`, workflow `release.yml`, and
480
+ environment `pypi` or `testpypi` respectively.
481
+
482
+ ## Debugging tips
483
+
484
+ - **See everything a review did:** `scrutai review ... --trace run.jsonl`, then open it with
485
+ `scrutai serve --replay run.jsonl`, or read the JSONL directly (one event per line, with `seq`).
486
+ - **See what the critic killed:** `--show-dropped`, or the `dropped` array in `--json` output; each
487
+ entry has `critic_note` and a per-round `history`.
488
+ - **A finding is missing:** check routing first. The `plan` event in the trace lists which agents ran
489
+ on which chunk.
490
+ - **The mock model is behaving strangely on a real diff:** remember it is a rule-based stand-in. If
491
+ reviewed code contains protocol-looking text, check that markers are matched line-anchored.
492
+ - **Reproduce CI without ripgrep:** CI runners don't have it, so `grep` falls back to `git grep` or
493
+ pure Python. Run tests with a `PATH` that hides `rg` to check that fallback.
494
+
495
+ ## Getting help
496
+
497
+ Open a [GitHub issue](https://github.com/imkarthiknr/Scrutai/issues) for bugs and proposals, or a
498
+ draft pull request early if you want feedback on an approach. For security issues, use
499
+ [private advisories](https://github.com/imkarthiknr/Scrutai/security/advisories/new) instead of a
500
+ public issue.