argus-code-review 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (150) hide show
  1. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/CHANGELOG.md +43 -1
  2. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/PKG-INFO +7 -7
  3. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/README.md +5 -4
  4. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/bench.py +203 -32
  5. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/bench_default.toml +7 -5
  6. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/cli.py +41 -11
  7. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/config.py +9 -15
  8. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/github_client.py +86 -0
  9. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/graph.py +297 -6
  10. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/helpers.py +110 -0
  11. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/models.py +2 -1
  12. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/runners.py +9 -9
  13. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/pyproject.toml +2 -9
  14. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/conftest.py +1 -0
  15. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/golden/review_response.schema.json +1 -1
  16. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_argus_review_local.py +23 -0
  17. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_bench.py +757 -63
  18. argus_code_review-0.2.3/tests/test_bench_config_guard.py +710 -0
  19. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_cli_preflight.py +63 -10
  20. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_gemini_runner.py +5 -6
  21. argus_code_review-0.2.3/tests/test_graph_fetch_diff.py +441 -0
  22. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_preflight_image_bump.py +31 -1
  23. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_multi_round.py +4 -2
  24. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_packaging.py +5 -11
  25. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_runners_helpers.py +191 -2
  26. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_runners_new.py +20 -0
  27. argus_code_review-0.2.2/tests/test_graph_fetch_diff.py +0 -228
  28. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/.gitignore +0 -0
  29. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/LICENSE +0 -0
  30. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/__init__.py +0 -0
  31. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/coverage.py +0 -0
  32. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/dotenv_utils.py +0 -0
  33. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/gemini_cache.py +0 -0
  34. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/gemini_runner.py +0 -0
  35. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/llm/models.py +0 -0
  36. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/llm/output_models.py +0 -0
  37. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/llm/pricing.py +0 -0
  38. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/openai_client.py +0 -0
  39. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/openai_runner.py +0 -0
  40. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/pipeline_models.py +0 -0
  41. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/__init__.py +0 -0
  42. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/actions_scanner.py +0 -0
  43. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/engine.py +0 -0
  44. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/eslint_bundle/.gitignore +0 -0
  45. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/eslint_bundle/README.md +0 -0
  46. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
  47. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/eslint_bundle/package-lock.json +0 -0
  48. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/eslint_bundle/package.json +0 -0
  49. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/js_scanner.py +0 -0
  50. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/migration_scanner.py +0 -0
  51. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/rules/README.md +0 -0
  52. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/sarif.py +0 -0
  53. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/scanner_utils.py +0 -0
  54. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/secrets_scanner.py +0 -0
  55. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/shadow.py +0 -0
  56. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/terraform_scanner.py +0 -0
  57. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/precheck/workflow_lint_scanner.py +0 -0
  58. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/__init__.py +0 -0
  59. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-blocking-validator.md +0 -0
  60. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-coverage-check.md +0 -0
  61. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-cross-cutting.md +0 -0
  62. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-feedback-verifier.md +0 -0
  63. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-lite.md +0 -0
  64. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-planner.md +0 -0
  65. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-preflight-router.md +0 -0
  66. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-prior-art.md +0 -0
  67. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-deployment.md +0 -0
  68. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-frontend.md +0 -0
  69. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-infra.md +0 -0
  70. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
  71. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-observability.md +0 -0
  72. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
  73. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-security.md +0 -0
  74. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
  75. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-specialist-sql.md +0 -0
  76. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-subagent.md +0 -0
  77. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-tests-and-docs.md +0 -0
  78. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts/pr-review-writer.md +0 -0
  79. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/prompts_runtime.py +0 -0
  80. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/repo_provision.py +0 -0
  81. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/review_tools.py +0 -0
  82. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/__init__.py +0 -0
  83. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/http.py +0 -0
  84. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/models.py +0 -0
  85. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/precheck.py +0 -0
  86. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/resolver.py +0 -0
  87. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/session.py +0 -0
  88. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/sql.py +0 -0
  89. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/argus/storage/sqlite.py +0 -0
  90. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/008_add_code_reviews.sql +0 -0
  91. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/009_add_reviewer_version.sql +0 -0
  92. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/010_add_review_patterns.sql +0 -0
  93. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/011_add_review_progress_columns.sql +0 -0
  94. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/015_create_agent_runs.sql +0 -0
  95. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/016_add_agent_runs_failure_reason.sql +0 -0
  96. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/schema/017_add_precheck_rules.sql +0 -0
  97. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/__init__.py +0 -0
  98. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/__init__.py +0 -0
  99. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_backend_contract.py +0 -0
  100. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_http.py +0 -0
  101. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_models.py +0 -0
  102. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_resolver.py +0 -0
  103. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_session.py +0 -0
  104. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_sql.py +0 -0
  105. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/storage/test_sqlite_backend.py +0 -0
  106. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_actions_scanner.py +0 -0
  107. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_catchup_gate.py +0 -0
  108. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_cli_args.py +0 -0
  109. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_cli_output_contract.py +0 -0
  110. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_cli_post_review.py +0 -0
  111. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_cli_prompts.py +0 -0
  112. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_config.py +0 -0
  113. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_conftest.py +0 -0
  114. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_gemini_cache.py +0 -0
  115. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_github_client_checks_signal.py +0 -0
  116. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_github_client_write.py +0 -0
  117. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_get_llm_temperature.py +0 -0
  118. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_http_guards.py +0 -0
  119. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_precheck.py +0 -0
  120. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_progress.py +0 -0
  121. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_storage_resolution.py +0 -0
  122. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_graph_timeout_surfacing.py +0 -0
  123. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_js_scanner.py +0 -0
  124. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_lite_review_cost.py +0 -0
  125. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_llm_models.py +0 -0
  126. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_llm_models_override.py +0 -0
  127. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_llm_pricing.py +0 -0
  128. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_migration_scanner.py +0 -0
  129. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_models.py +0 -0
  130. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_openai_client.py +0 -0
  131. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_openai_runner.py +0 -0
  132. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_plan_review.py +0 -0
  133. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_precheck_engine.py +0 -0
  134. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_precheck_engine_integration.py +0 -0
  135. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_precheck_sarif.py +0 -0
  136. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_precheck_shadow.py +0 -0
  137. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_precheck_shadow_integration.py +0 -0
  138. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_prompts.py +0 -0
  139. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_repo_provision.py +0 -0
  140. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_review_patterns_integration.py +0 -0
  141. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_review_tools.py +0 -0
  142. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_runners_context7.py +0 -0
  143. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_runners_context_usage.py +0 -0
  144. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_scanner_utils.py +0 -0
  145. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_secrets_scanner.py +0 -0
  146. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_specialist_validation.py +0 -0
  147. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_storage_precheck.py +0 -0
  148. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_terraform_scanner.py +0 -0
  149. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_terraform_scanner_integration.py +0 -0
  150. {argus_code_review-0.2.2 → argus_code_review-0.2.3}/tests/test_workflow_lint_scanner.py +0 -0
@@ -7,6 +7,47 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.2.3] - 2026-09-14
11
+
12
+ ### Added
13
+
14
+ - Reviewer bench configuration change guard (TECH-6282): PRs touching bench
15
+ configuration files (`.argus/bench.toml`, `argus/bench_default.toml`) or adding
16
+ lines modifying bench routing environment variables (`ARGUS_BENCH_FILE`,
17
+ `ARGUS_NO_BENCH_OVERRIDES`) are deterministically force-BLOCKED with a finding
18
+ in category `argus-self-config` requiring explicit human sign-off. The guard
19
+ evaluates against the full-PR diff scope across multi-round reviews to prevent
20
+ bypasses on subsequent commits.
21
+
22
+ ### Changed
23
+
24
+ - Defaulted `[bulk_reviewer]` in `argus/bench_default.toml` to `platform = "gemini"`,
25
+ `model = "gemini-mini"`, and `caching = "auto"` (TECH-6281). System-generalist,
26
+ specialist, and tests-and-docs reviewers now route to Gemini (`gemini-3.8-flash`)
27
+ out of the box for cost optimization across PR review fan-outs, while individual
28
+ roles (`cross-cutting`, `blocking-validator`, `feedback-verifier`) remain on
29
+ `claude-sdk`.
30
+ - Moved `google-genai` from the optional `[gemini]` extra into core `dependencies`
31
+ in `pyproject.toml`, ensuring the default Gemini bulk reviewer works out of
32
+ the box without requiring an extra install.
33
+ - In `argus/bench.py`, sparse model-only bench overrides (e.g. `model = "claude-mini"`)
34
+ now automatically infer their compatible platform (`claude-sdk`, `gemini`, or
35
+ `openai-responses`) so model overrides do not inherit an incompatible platform
36
+ from lower bench layers. Explicit platform/model mismatches fail validation early
37
+ with clear migration guidance.
38
+ - Made `--specialist-model` / `ARGUS_SPECIALIST_MODEL` apply as a highest-priority
39
+ override forcing `[bulk_reviewer]` to `claude-sdk` with `claude-default` so
40
+ the CLI flag continues to control system and specialist reviewers as documented.
41
+ - Made CLI preflight credential validation conditional on effective bench requirements:
42
+ `GOOGLE_API_KEY` is now required when the resolved bench config includes `gemini`.
43
+
44
+ ### Fixed
45
+
46
+ - Fixed `bench.load_bench()`/`_check_settings()` requiring full credential validation (`GITHUB_TOKEN_RO`, `OPENAI_API_KEY`) just to resolve bench/platform routing config, which broke settings dependency-injection in tests and CI (regression from TECH-6281).
47
+ - Regenerated `tests/golden/review_response.schema.json` to match the
48
+ `argus-self-config` category description added in TECH-6282 — the
49
+ golden-snapshot test was left failing after that PR merged.
50
+
10
51
  ## [0.2.2] - 2026-09-11
11
52
 
12
53
  ### Added
@@ -203,7 +244,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
203
244
  packaged set.
204
245
  - `argus --version`, `argus prompts list`, and `argus prompts export`.
205
246
 
206
- [Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.2...HEAD
247
+ [Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.3...HEAD
248
+ [0.2.3]: https://github.com/redesignhealth/argus-review/compare/v0.2.2...v0.2.3
207
249
  [0.2.2]: https://github.com/redesignhealth/argus-review/compare/v0.2.1...v0.2.2
208
250
  [0.2.1]: https://github.com/redesignhealth/argus-review/compare/v0.2.0...v0.2.1
209
251
  [0.2.0]: https://github.com/redesignhealth/argus-review/compare/v0.1.5...v0.2.0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: argus-code-review
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
5
5
  Project-URL: Repository, https://github.com/redesignhealth/argus-review
6
6
  Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
@@ -17,6 +17,7 @@ Classifier: Topic :: Software Development :: Quality Assurance
17
17
  Requires-Python: <3.14,>=3.12
18
18
  Requires-Dist: asyncpg>=0.29.0
19
19
  Requires-Dist: claude-agent-sdk==0.1.81
20
+ Requires-Dist: google-genai<2,>=1.55
20
21
  Requires-Dist: httpx>=0.27.0
21
22
  Requires-Dist: langchain-anthropic>=1.2.0
22
23
  Requires-Dist: langchain<2,>=1.2.10
@@ -41,8 +42,6 @@ Requires-Dist: pytest>=8.0.0; extra == 'dev'
41
42
  Requires-Dist: ruff>=0.8.0; extra == 'dev'
42
43
  Requires-Dist: types-pyyaml>=6.0; extra == 'dev'
43
44
  Requires-Dist: types-regex>=2024.4.16; extra == 'dev'
44
- Provides-Extra: gemini
45
- Requires-Dist: google-genai<2,>=1.55; extra == 'gemini'
46
45
  Provides-Extra: prechecks
47
46
  Requires-Dist: checkov<4.0.0,>=3.3.0; extra == 'prechecks'
48
47
  Requires-Dist: pyyaml>=6.0; extra == 'prechecks'
@@ -207,10 +206,11 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
207
206
  # --frontier-model controls both the planner/coverage tier AND the cross-cutting
208
207
  # reviewer -- claude-fable-5 here is already the planner/coverage default, but
209
208
  # it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
210
- # Cost note: --specialist-model here also moves the highest-volume path (system
211
- # reviewer, specialists, writer, lite-review) onto the pricier Opus tier -- both
212
- # flags in this example trade cost for reasoning headroom, don't use them together
213
- # as a low-cost default. Any non-empty --specialist-model value also withholds
209
+ # Cost note: --specialist-model here overrides the bulk reviewer path (system
210
+ # reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
211
+ # claude-sdk with the specified model, and also updates the writer and lite-review
212
+ # paths -- both flags in this example trade cost for reasoning headroom, don't use
213
+ # them together as a low-cost default. Any non-empty --specialist-model value also withholds
214
214
  # the 1M-context beta from that same highest-volume path, since the beta is
215
215
  # only verified against the unoverridden default (autocompact may thrash on
216
216
  # long reviews under this override -- see argus/runners.py for the tradeoff).
@@ -155,10 +155,11 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
155
155
  # --frontier-model controls both the planner/coverage tier AND the cross-cutting
156
156
  # reviewer -- claude-fable-5 here is already the planner/coverage default, but
157
157
  # it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
158
- # Cost note: --specialist-model here also moves the highest-volume path (system
159
- # reviewer, specialists, writer, lite-review) onto the pricier Opus tier -- both
160
- # flags in this example trade cost for reasoning headroom, don't use them together
161
- # as a low-cost default. Any non-empty --specialist-model value also withholds
158
+ # Cost note: --specialist-model here overrides the bulk reviewer path (system
159
+ # reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
160
+ # claude-sdk with the specified model, and also updates the writer and lite-review
161
+ # paths -- both flags in this example trade cost for reasoning headroom, don't use
162
+ # them together as a low-cost default. Any non-empty --specialist-model value also withholds
162
163
  # the 1M-context beta from that same highest-volume path, since the beta is
163
164
  # only verified against the unoverridden default (autocompact may thrash on
164
165
  # long reviews under this override -- see argus/runners.py for the tradeoff).
@@ -1,11 +1,10 @@
1
1
  """Bench configuration: which platform/model a leaf reviewer runs on.
2
2
 
3
3
  A "bench" is a lightweight, human-edited TOML config that decides which
4
- LLM *platform* (Claude Agent SDK is the packaged default; Gemini has a
5
- real runner too, see ``argus.gemini_runner``; OpenAI Responses has a
6
- real runner too, see ``argus.openai_runner``; both are opt-in only) and
7
- *model* each leaf reviewer in the review pipeline runs on. This is
8
- deliberately NOT a dynamic/adaptive routing system -- it is a static,
4
+ LLM *platform* (Claude Agent SDK and Gemini have packaged defaults;
5
+ OpenAI Responses has a real runner too, see ``argus.openai_runner``;
6
+ opt-in) and *model* each leaf reviewer in the review pipeline runs on.
7
+ This is deliberately NOT a dynamic/adaptive routing system -- it is a static,
9
8
  PR-reviewed config with a human-editable override chain, in the same
10
9
  spirit as ``argus.prompts_runtime``'s prompt override chain.
11
10
 
@@ -25,9 +24,10 @@ Two kinds of config unit, not a per-role table:
25
24
  Override chain (mirrors ``argus.prompts_runtime`` exactly, including its
26
25
  opt-out convention), lowest to highest priority:
27
26
 
28
- 1. Packaged ``argus/bench_default.toml`` -- the base. Every entry resolves
29
- to ``platform="claude-sdk"`` with today's actual models, so shipping
30
- this file is a behavior-preserving no-op for a default install.
27
+ 1. Packaged ``argus/bench_default.toml`` -- the base. ``[bulk_reviewer]``
28
+ defaults to ``platform="gemini"`` (``model="gemini-mini"``,
29
+ ``caching="auto"``), while individual roles (``cross-cutting``,
30
+ ``blocking-validator``, ``feedback-verifier``) default to ``claude-sdk``.
31
31
  2. ``~/.config/argus/bench.toml`` (respecting ``XDG_CONFIG_HOME``) -- a
32
32
  user-global sparse overlay: only the keys it specifies are overridden;
33
33
  everything else falls through to the layer below.
@@ -82,7 +82,8 @@ from importlib import resources
82
82
  from pathlib import Path
83
83
  from typing import Any, Final, Literal, get_args
84
84
 
85
- from argus.config import get_settings
85
+ from pydantic import TypeAdapter, ValidationError
86
+
86
87
  from argus.llm import models as model_aliases
87
88
  from argus.pipeline_models import SpecialistName
88
89
  from argus.prompts_runtime import known_packaged_prompts
@@ -104,6 +105,23 @@ _TOP_LEVEL_KEYS: Final[frozenset[str]] = frozenset({"bulk_reviewer", "roles"})
104
105
  _BULK_KEYS: Final[frozenset[str]] = frozenset({"platform", "model", "caching"})
105
106
  _ROLE_KEYS: Final[frozenset[str]] = frozenset({"platform", "model", "prompt_name", "caching"})
106
107
 
108
+ _MODEL_FAMILY_TO_PLATFORM: Final[dict[str, Platform]] = {
109
+ "claude": "claude-sdk",
110
+ "gemini": "gemini",
111
+ "gpt": "openai-responses",
112
+ }
113
+
114
+
115
+ def infer_platform_for_model(model: str) -> Platform | None:
116
+ """Infer the execution platform from a model alias or model name.
117
+
118
+ Returns the Platform matching the model family prefix ('claude-*' -> 'claude-sdk',
119
+ 'gemini-*' -> 'gemini', 'gpt-*' -> 'openai-responses'), or None if unknown.
120
+ """
121
+ family = model.lower().split("-", 1)[0]
122
+ return _MODEL_FAMILY_TO_PLATFORM.get(family)
123
+
124
+
107
125
  # Roles a runner function actually resolves via bench.resolve(). Every bulk
108
126
  # role and every individual role is wired to its corresponding runner function.
109
127
  _WIRED_ROLES: Final[frozenset[str]] = frozenset(
@@ -182,7 +200,7 @@ def _load_toml_file(path: Path) -> dict[str, Any]:
182
200
  return _load_toml_bytes(path.read_bytes())
183
201
 
184
202
 
185
- def _overlay_layers(settings: Any) -> list[Path]:
203
+ def _overlay_layers(settings_or_bench_file: Any = None) -> list[Path]:
186
204
  """Ordered overlay file paths, lowest to highest priority.
187
205
 
188
206
  Only existing files are included for the two standard locations
@@ -202,51 +220,192 @@ def _overlay_layers(settings: Any) -> list[Path]:
202
220
  if repo_local.is_file():
203
221
  layers.append(repo_local)
204
222
 
205
- if settings.ARGUS_BENCH_FILE:
206
- bench_file = Path(settings.ARGUS_BENCH_FILE)
207
- if not bench_file.is_file():
208
- raise ValueError(
209
- f"ARGUS_BENCH_FILE={settings.ARGUS_BENCH_FILE!r} does not exist or is not a file"
210
- )
211
- layers.append(bench_file)
223
+ bench_file: str | None = None
224
+ if isinstance(settings_or_bench_file, (str, Path)):
225
+ bench_file = str(settings_or_bench_file)
226
+ elif settings_or_bench_file is not None and hasattr(settings_or_bench_file, "ARGUS_BENCH_FILE"):
227
+ val = getattr(settings_or_bench_file, "ARGUS_BENCH_FILE", None)
228
+ if isinstance(val, (str, Path)) and str(val):
229
+ bench_file = str(val)
230
+
231
+ if bench_file:
232
+ bench_path = Path(bench_file)
233
+ if not bench_path.is_file():
234
+ raise ValueError(f"ARGUS_BENCH_FILE={bench_file!r} does not exist or is not a file")
235
+ layers.append(bench_path)
212
236
 
213
237
  return layers
214
238
 
215
239
 
240
+ def _infer_platforms_for_overlay(overlay: dict[str, Any]) -> dict[str, Any]:
241
+ """Infer the compatible platform for any table that specifies `model` but omits `platform`.
242
+
243
+ Prevents sparse model-only overrides (such as `model = "claude-mini"`) from inheriting
244
+ an incompatible platform (like the default `platform = "gemini"`) from a lower layer.
245
+ """
246
+ result = dict(overlay)
247
+ if "bulk_reviewer" in result and isinstance(result["bulk_reviewer"], dict):
248
+ bulk = dict(result["bulk_reviewer"])
249
+ if "model" in bulk and isinstance(bulk["model"], str) and "platform" not in bulk:
250
+ inferred = infer_platform_for_model(bulk["model"])
251
+ if inferred is not None:
252
+ bulk["platform"] = inferred
253
+ result["bulk_reviewer"] = bulk
254
+ if "roles" in result and isinstance(result["roles"], dict):
255
+ roles = dict(result["roles"])
256
+ for role_name, role_table in roles.items():
257
+ if isinstance(role_table, dict):
258
+ r = dict(role_table)
259
+ if "model" in r and isinstance(r["model"], str) and "platform" not in r:
260
+ inferred = infer_platform_for_model(r["model"])
261
+ if inferred is not None:
262
+ r["platform"] = inferred
263
+ roles[role_name] = r
264
+ result["roles"] = roles
265
+ return result
266
+
267
+
216
268
  _WARNED_ROLES: set[str] = set()
217
269
  _WARNED_ROLES_LOCK: threading.Lock = threading.Lock()
218
270
 
271
+ _BOOL_ADAPTER: Final[TypeAdapter[bool]] = TypeAdapter(bool)
219
272
 
220
- @lru_cache(maxsize=1)
221
- def load_bench() -> dict[str, Any]:
222
- """Load, merge, and validate the effective bench config.
223
273
 
224
- Cached for the life of the process; call :func:`clear_cache` to force
225
- a reload (e.g. in tests, or after mutating ``os.environ``).
274
+ def _parse_bool(val: Any, var_name: str = "ARGUS_NO_BENCH_OVERRIDES") -> bool:
275
+ """Parse a boolean value matching Pydantic's coercion rules.
276
+
277
+ When val is None (i.e. env var unset), returns False default.
278
+ Otherwise delegates directly to Pydantic's TypeAdapter(bool)
279
+ and converts ValidationError to ValueError with a clear message.
226
280
  """
227
- settings = get_settings()
281
+ if val is None:
282
+ return False
283
+ from unittest.mock import NonCallableMock
284
+
285
+ if isinstance(val, NonCallableMock):
286
+ return False
287
+ try:
288
+ return _BOOL_ADAPTER.validate_python(val)
289
+ except ValidationError as exc:
290
+ raise ValueError(
291
+ f"Invalid boolean value for {var_name}: {val!r}. "
292
+ f"Expected a valid boolean (e.g. true/false, yes/no, 1/0, on/off)."
293
+ ) from exc
294
+
295
+
296
+ def _extract_bench_settings(settings: Any) -> tuple[bool, str | None, str | None]:
297
+ """Extract (no_bench_overrides, bench_file, specialist_model) from a settings-like object.
298
+
299
+ When settings is provided, its fields are authoritative and never fall back to
300
+ ambient os.environ (preserving dependency injection). For unit test mocks
301
+ (MagicMock), unconfigured attributes that were never set on the mock fall back to
302
+ os.environ so ambient test fixtures can supply values.
303
+ """
304
+ from unittest.mock import NonCallableMock
305
+
306
+ if isinstance(settings, NonCallableMock):
307
+ if "ARGUS_NO_BENCH_OVERRIDES" in settings.__dict__:
308
+ no_overrides = _parse_bool(
309
+ settings.ARGUS_NO_BENCH_OVERRIDES, "ARGUS_NO_BENCH_OVERRIDES"
310
+ )
311
+ else:
312
+ no_overrides = _parse_bool(
313
+ os.environ.get("ARGUS_NO_BENCH_OVERRIDES"), "ARGUS_NO_BENCH_OVERRIDES"
314
+ )
315
+
316
+ if "ARGUS_BENCH_FILE" in settings.__dict__:
317
+ bf = settings.ARGUS_BENCH_FILE
318
+ bench_file = str(bf) if isinstance(bf, (str, Path)) and str(bf) else None
319
+ else:
320
+ bench_file = os.environ.get("ARGUS_BENCH_FILE") or None
321
+
322
+ if "ARGUS_SPECIALIST_MODEL" in settings.__dict__:
323
+ sm = settings.ARGUS_SPECIALIST_MODEL
324
+ specialist_model = str(sm) if isinstance(sm, str) and sm else None
325
+ else:
326
+ specialist_model = os.environ.get("ARGUS_SPECIALIST_MODEL") or None
327
+
328
+ return (no_overrides, bench_file, specialist_model)
329
+
330
+ # Real Settings or dataclass: strictly authoritative, NEVER fall back to os.environ
331
+ no_overrides = _parse_bool(
332
+ getattr(settings, "ARGUS_NO_BENCH_OVERRIDES", False), "ARGUS_NO_BENCH_OVERRIDES"
333
+ )
334
+ bf = getattr(settings, "ARGUS_BENCH_FILE", None)
335
+ bench_file = str(bf) if isinstance(bf, (str, Path)) and str(bf) else None
336
+ sm = getattr(settings, "ARGUS_SPECIALIST_MODEL", None)
337
+ specialist_model = str(sm) if isinstance(sm, str) and sm else None
338
+ return (no_overrides, bench_file, specialist_model)
339
+
340
+
341
+ def _resolve_bench_settings(settings: Any = None) -> tuple[bool, str | None, str | None]:
342
+ """Resolve bench-routing settings from an injected object or ambient environment."""
343
+ if settings is not None:
344
+ return _extract_bench_settings(settings)
345
+
346
+ # Read os.environ directly from the ambient environment without
347
+ # invoking get_settings() or any dotenv loaders, ensuring bench loading
348
+ # never requires credentials, never mutates os.environ, and never loads
349
+ # untrusted repo-local .env files into the process.
350
+ no_overrides = _parse_bool(
351
+ os.environ.get("ARGUS_NO_BENCH_OVERRIDES"), "ARGUS_NO_BENCH_OVERRIDES"
352
+ )
353
+ bench_file = os.environ.get("ARGUS_BENCH_FILE") or None
354
+ specialist_model = os.environ.get("ARGUS_SPECIALIST_MODEL") or None
355
+ return (no_overrides, bench_file, specialist_model)
356
+
357
+
358
+ @lru_cache(maxsize=16)
359
+ def _load_bench_cached(
360
+ no_bench_overrides: bool,
361
+ bench_file: str | None,
362
+ specialist_model: str | None,
363
+ ) -> dict[str, Any]:
228
364
  merged = _load_packaged_default()
229
365
 
230
- if not settings.ARGUS_NO_BENCH_OVERRIDES:
231
- for layer_path in _overlay_layers(settings):
366
+ if not no_bench_overrides:
367
+ for layer_path in _overlay_layers(bench_file):
232
368
  overlay = _load_toml_file(layer_path)
233
369
  roles = overlay.get("roles", {})
234
370
  if isinstance(roles, dict):
235
371
  for role in roles:
236
372
  _warn_if_not_wired(role)
237
- merged = _deep_merge(merged, overlay)
373
+ prepared = _infer_platforms_for_overlay(overlay)
374
+ merged = _deep_merge(merged, prepared)
375
+
376
+ # --specialist-model / ARGUS_SPECIALIST_MODEL is a CLI/env knob that
377
+ # overrides the system reviewer and specialist reviewers. When set,
378
+ # force bulk_reviewer to claude-sdk with claude-default so the override
379
+ # controls the bulk reviewer roles as documented.
380
+ if specialist_model:
381
+ merged["bulk_reviewer"]["platform"] = "claude-sdk"
382
+ merged["bulk_reviewer"]["model"] = "claude-default"
238
383
 
239
384
  _validate_raw_bench(merged)
240
385
  return merged
241
386
 
242
387
 
388
+ def load_bench(settings: Any = None) -> dict[str, Any]:
389
+ """Load, merge, and validate the effective bench config.
390
+
391
+ Cached for the life of the process; call :func:`clear_cache` to force
392
+ a reload (e.g. in tests, or after mutating ``os.environ``).
393
+ """
394
+ no_overrides, bench_file, specialist_model = _resolve_bench_settings(settings)
395
+ return _load_bench_cached(no_overrides, bench_file, specialist_model)
396
+
397
+
243
398
  def clear_cache() -> None:
244
399
  """Clear the cached bench config, forcing the next call to reload."""
245
400
  with _WARNED_ROLES_LOCK:
246
- load_bench.cache_clear()
401
+ _load_bench_cached.cache_clear()
247
402
  _WARNED_ROLES.clear()
248
403
 
249
404
 
405
+ load_bench.cache_clear = _load_bench_cached.cache_clear # type: ignore[attr-defined]
406
+ load_bench.cache_info = _load_bench_cached.cache_info # type: ignore[attr-defined]
407
+
408
+
250
409
  # ---------------------------------------------------------------------------
251
410
  # Validation
252
411
  # ---------------------------------------------------------------------------
@@ -288,6 +447,16 @@ def _validate_prompt_name(name: Any, where: str) -> None:
288
447
  )
289
448
 
290
449
 
450
+ def _validate_model_platform_compatibility(platform: str, model: str, where: str) -> None:
451
+ expected_platform = infer_platform_for_model(model)
452
+ if expected_platform is not None and platform != expected_platform:
453
+ raise ValueError(
454
+ f"{where}: model {model!r} is not compatible with platform {platform!r} "
455
+ f"(expected platform={expected_platform!r}). "
456
+ f"Set platform = {expected_platform!r} or select a model compatible with {platform!r}."
457
+ )
458
+
459
+
291
460
  def _validate_raw_bench(raw: dict[str, Any]) -> None:
292
461
  """Validate the fully-merged bench config, raising loudly on any problem.
293
462
 
@@ -319,6 +488,7 @@ def _validate_raw_bench(raw: dict[str, Any]) -> None:
319
488
  _validate_platform(bulk["platform"], "[bulk_reviewer]")
320
489
  _validate_model(bulk["model"], "[bulk_reviewer]")
321
490
  _validate_caching(bulk.get("caching", "auto"), "[bulk_reviewer]")
491
+ _validate_model_platform_compatibility(bulk["platform"], bulk["model"], "[bulk_reviewer]")
322
492
 
323
493
  roles = raw.get("roles", {})
324
494
  if not isinstance(roles, dict):
@@ -340,6 +510,7 @@ def _validate_raw_bench(raw: dict[str, Any]) -> None:
340
510
  _validate_model(role_table["model"], where)
341
511
  _validate_caching(role_table.get("caching", "auto"), where)
342
512
  _validate_prompt_name(role_table["prompt_name"], where)
513
+ _validate_model_platform_compatibility(role_table["platform"], role_table["model"], where)
343
514
 
344
515
  # The bulk bucket's own known prompt names are hardcoded (not
345
516
  # TOML-supplied), but validate them too: a packaged/override prompt
@@ -381,7 +552,7 @@ def _warn_if_not_wired(role: str) -> None:
381
552
  )
382
553
 
383
554
 
384
- def resolve(role: str) -> BenchEntry:
555
+ def resolve(role: str, settings: Any = None) -> BenchEntry:
385
556
  """Resolve ``role`` to its effective :class:`BenchEntry`.
386
557
 
387
558
  Bulk-bucket roles (see ``BULK_ROLE_PROMPTS``) route through
@@ -395,7 +566,7 @@ def resolve(role: str) -> BenchEntry:
395
566
  docstring section for why most roles currently have no runner
396
567
  consulting them.
397
568
  """
398
- raw = load_bench()
569
+ raw = load_bench(settings=settings)
399
570
 
400
571
  if role in BULK_ROLE_PROMPTS:
401
572
  bulk = raw["bulk_reviewer"]
@@ -529,9 +700,9 @@ async def _gemini_runner(
529
700
 
530
701
  Imports ``argus.config`` and ``argus.gemini_runner`` lazily (function
531
702
  body, not module level) for the same reason ``_claude_sdk_runner``
532
- imports lazily: avoids a needless import of the (optional,
533
- ``google-genai``-dependent) Gemini runner module for every caller of
534
- this module, even ones that never touch the ``gemini`` platform.
703
+ imports lazily: avoids a needless import of the Gemini runner module
704
+ for every caller of this module, even ones that never touch the
705
+ ``gemini`` platform.
535
706
  """
536
707
  from argus.config import get_settings
537
708
  from argus.gemini_runner import run_session_gemini
@@ -2,9 +2,11 @@
2
2
  #
3
3
  # See argus/bench.py's module docstring for the full override chain and
4
4
  # the bulk_reviewer vs. roles.<name> distinction. This file is the base
5
- # of that chain -- every entry below resolves to platform="claude-sdk"
6
- # with the exact models the pipeline used before the bench existed, so
7
- # shipping this file is a behavior-preserving no-op for a default install.
5
+ # of that chain: [bulk_reviewer] defaults to platform="gemini"
6
+ # (model="gemini-mini", caching="auto") for cost reasons across
7
+ # generalist, specialist, and tests-and-docs reviewers, while individual
8
+ # roles ([roles.cross-cutting], [roles.blocking-validator], and
9
+ # [roles.feedback-verifier]) remain on platform="claude-sdk".
8
10
  #
9
11
  # To customize: create ./.argus/bench.toml or ~/.config/argus/bench.toml
10
12
  # with ONLY the keys you want to change (sparse overlay -- you never need
@@ -22,8 +24,8 @@
22
24
  # Applies as one unit to run_system_reviewer (per system group, including
23
25
  # gap-fill reviewers), every named specialist, and the tests-and-docs
24
26
  # reviewer. Each of those still resolves its own existing prompt file.
25
- platform = "claude-sdk"
26
- model = "claude-default"
27
+ platform = "gemini"
28
+ model = "gemini-mini"
27
29
  caching = "auto"
28
30
 
29
31
  # The three [roles.*] tables below configure independent roles:
@@ -47,8 +47,9 @@ Environment variables (HTTP-mode opt-in):
47
47
 
48
48
  No storage env vars are required: with neither ``ARGUS_DB_URL`` nor the
49
49
  HTTP-shim URLs set, round history and checkpoints default to local SQLite.
50
- Required secrets are only one of ANTHROPIC_API_KEY / ANTHROPIC_AUTH_TOKEN,
51
- GITHUB_TOKEN_RO, and OPENAI_API_KEY. ANTHROPIC_AUTH_TOKEN is the standard
50
+ Required secrets are one of ANTHROPIC_API_KEY / ANTHROPIC_AUTH_TOKEN,
51
+ GITHUB_TOKEN_RO, OPENAI_API_KEY, and (by default, since bulk reviewers default
52
+ to Gemini) GOOGLE_API_KEY. ANTHROPIC_AUTH_TOKEN is the standard
52
53
  mechanism for routing through a corporate LLM gateway/proxy instead of a
53
54
  real Anthropic API key (sent as ``Authorization: Bearer`` rather than
54
55
  ``x-api-key``) — the same convention the Anthropic SDK and Claude Code
@@ -177,11 +178,13 @@ def _load_settings() -> "Settings":
177
178
  def _check_settings(settings: "Settings") -> None:
178
179
  """Validate that critical secrets were loaded, fail loudly otherwise.
179
180
 
180
- Only the three API credentials are required: one of ``ANTHROPIC_API_KEY``
181
- / ``ANTHROPIC_AUTH_TOKEN`` (Agent SDK + LangChain), ``GITHUB_TOKEN_RO``
182
- (diff fetch + clone), and ``OPENAI_API_KEY`` (plan-extraction path). No
183
- storage configuration is required — with neither ``ARGUS_DB_URL`` nor
184
- the HTTP-shim URLs set, history and checkpoints default to local SQLite.
181
+ Always required:
182
+ - ANTHROPIC_API_KEY (or ANTHROPIC_AUTH_TOKEN) for Agent SDK + LangChain
183
+ - GITHUB_TOKEN_RO for diff fetch + clone
184
+ - OPENAI_API_KEY for plan-extraction path
185
+
186
+ Additionally, if any leaf reviewer in the effective resolved bench config
187
+ requires the Gemini platform, GOOGLE_API_KEY is required.
185
188
 
186
189
  Args:
187
190
  settings: Resolved settings object.
@@ -193,6 +196,27 @@ def _check_settings(settings: "Settings") -> None:
193
196
  missing.append("GITHUB_TOKEN_RO")
194
197
  if not settings.OPENAI_API_KEY:
195
198
  missing.append("OPENAI_API_KEY")
199
+
200
+ from argus import bench
201
+
202
+ bench.clear_cache()
203
+ try:
204
+ raw_bench = bench.load_bench(settings=settings)
205
+ except Exception as exc:
206
+ logger.error("Invalid bench configuration: %s", exc)
207
+ sys.exit(1)
208
+
209
+ platforms_needed: set[str] = set()
210
+ bulk = raw_bench.get("bulk_reviewer")
211
+ if isinstance(bulk, dict) and "platform" in bulk:
212
+ platforms_needed.add(bulk["platform"])
213
+ for role_cfg in raw_bench.get("roles", {}).values():
214
+ if isinstance(role_cfg, dict) and "platform" in role_cfg:
215
+ platforms_needed.add(role_cfg["platform"])
216
+
217
+ if "gemini" in platforms_needed and not settings.GOOGLE_API_KEY:
218
+ missing.append("GOOGLE_API_KEY")
219
+
196
220
  if missing:
197
221
  logger.error(
198
222
  "Missing required secrets: %s. Add them to your shell or .env file.",
@@ -330,9 +354,10 @@ def _add_review_args(parser: argparse.ArgumentParser) -> None:
330
354
  default=_MODEL_OVERRIDE_UNSET,
331
355
  help=(
332
356
  "Override the model used by the system reviewer, specialist "
333
- "reviewers, the writer, and the lite-review path (default: "
334
- "claude-sonnet-4-6, or ARGUS_SPECIALIST_MODEL if already set in "
335
- "the environment). Same effect as setting ARGUS_SPECIALIST_MODEL. "
357
+ "reviewers, the writer, and the lite-review path (forcing "
358
+ "bulk reviewers to claude-sdk with the specified model; default "
359
+ "without this flag routes bulk reviewers to gemini-mini). "
360
+ "Same effect as setting ARGUS_SPECIALIST_MODEL. "
336
361
  "Pass an empty string to clear an already-set "
337
362
  "ARGUS_SPECIALIST_MODEL for this run."
338
363
  ),
@@ -572,7 +597,6 @@ def _run_review(parser: argparse.ArgumentParser, args: argparse.Namespace) -> No
572
597
  except Exception as exc: # pydantic ValidationError for missing required vars
573
598
  logger.error("Failed to load settings: %s", exc)
574
599
  sys.exit(1)
575
- _check_settings(settings)
576
600
 
577
601
  # Applied AFTER _load_settings(), not before: _load_settings() calls
578
602
  # load_dotenv_early(..., override=False), which repopulates any env var
@@ -587,6 +611,12 @@ def _run_review(parser: argparse.ArgumentParser, args: argparse.Namespace) -> No
587
611
  _apply_model_override_flag("ARGUS_SPECIALIST_MODEL", args.specialist_model)
588
612
  _apply_model_override_flag("ARGUS_FRONTIER_MODEL", args.frontier_model)
589
613
 
614
+ from argus.config import clear_cache as clear_config_cache, get_settings
615
+
616
+ clear_config_cache()
617
+ settings = get_settings()
618
+ _check_settings(settings)
619
+
590
620
  from argus.models import ReviewRequest
591
621
 
592
622
  request = ReviewRequest(
@@ -92,20 +92,15 @@ class Settings(BaseSettings):
92
92
  path (``argus.llm.models.CLAUDE_DEFAULT``, default
93
93
  ``claude-sonnet-4-6``). Set via ``--specialist-model``; read
94
94
  directly from ``os.environ`` by ``argus.llm.models`` at import
95
- time (module-level constant resolution, not a per-request
96
- Settings lookup), so this field is never actually read off a
97
- ``Settings`` instance anywhere in the codebase -- unlike
98
- ``ARGUS_NO_PROMPT_OVERRIDES`` above, which genuinely is
99
- (``prompts_runtime.override_dirs`` reads
100
- ``settings.ARGUS_NO_PROMPT_OVERRIDES``). It's declared here
101
- purely for documentation/discoverability, same as any other
102
- env var this class's docstring covers.
95
+ time and by ``argus.bench`` (where setting it forces bulk reviewers
96
+ to claude-sdk with claude-default). Also read off a ``Settings``
97
+ instance during bench resolution when dependency-injected.
103
98
  ARGUS_FRONTIER_MODEL: Override the model used by the planner,
104
99
  coverage check, and cross-cutting reviewer
105
100
  (``argus.llm.models.CLAUDE_FRONTIER`` / ``CLAUDE_OPUS``). Set via
106
- ``--frontier-model``; same read pattern (and same
107
- documentation-only Settings field) as ``ARGUS_SPECIALIST_MODEL``
108
- above.
101
+ ``--frontier-model``; read directly from ``os.environ`` by
102
+ ``argus.llm.models`` at import time (module-level constant
103
+ resolution, not a per-request Settings lookup).
109
104
  LANGSMITH_API_KEY / LANGSMITH_PROJECT: Optional tracing.
110
105
  OPENAI_BASE_URL: Optional override for OpenAI API endpoint (e.g. for proxying).
111
106
  Note that OPENAI_API_KEY will be forwarded as a Bearer token to whatever
@@ -131,10 +126,9 @@ class Settings(BaseSettings):
131
126
  additional headroom across all three reviewer platforms.
132
127
  GOOGLE_API_KEY: Gemini platform credential, consumed via the
133
128
  ``google_credential`` property by ``argus.gemini_runner``
134
- (Track 3) whenever a role's bench entry resolves to
135
- ``platform = "gemini"``. Only required if you actually
136
- opt a role into that platform (see ``argus.bench``'s
137
- override chain) -- the packaged default bench never does.
129
+ whenever a role's bench entry resolves to
130
+ ``platform = "gemini"``. Required for default review runs
131
+ since ``[bulk_reviewer]`` defaults to Gemini.
138
132
  GOOGLE_BASE_URL: Optional override for the Gemini API endpoint (e.g. for
139
133
  proxying). Mirrors ``OPENAI_BASE_URL`` above. Note that GOOGLE_API_KEY
140
134
  will be forwarded to whatever endpoint is configured here; point only