argus-code-review 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/CHANGELOG.md +59 -1
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/PKG-INFO +7 -7
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/README.md +5 -4
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/bench.py +203 -32
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/bench_default.toml +7 -5
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/cli.py +41 -11
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/config.py +9 -15
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/gemini_runner.py +8 -6
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/github_client.py +86 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/graph.py +297 -6
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/helpers.py +110 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/models.py +2 -1
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/runners.py +9 -9
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/pyproject.toml +2 -9
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/conftest.py +1 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/golden/review_response.schema.json +1 -1
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_argus_review_local.py +23 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_bench.py +757 -63
- argus_code_review-0.2.4/tests/test_bench_config_guard.py +710 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_cli_preflight.py +63 -10
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_gemini_runner.py +11 -11
- argus_code_review-0.2.4/tests/test_graph_fetch_diff.py +441 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_preflight_image_bump.py +31 -1
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_multi_round.py +4 -2
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_packaging.py +5 -11
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_runners_helpers.py +191 -2
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_runners_new.py +20 -0
- argus_code_review-0.2.2/tests/test_graph_fetch_diff.py +0 -228
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/.gitignore +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/LICENSE +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/coverage.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/llm/models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/llm/output_models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/openai_runner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/pipeline_models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/engine.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/sql.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/argus/storage/sqlite.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_backend_contract.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_http.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_sql.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/storage/test_sqlite_backend.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_config.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_get_llm_temperature.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_precheck.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_graph_timeout_surfacing.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_lite_review_cost.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_llm_models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_llm_models_override.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_models.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_openai_client.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_openai_runner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_precheck_engine.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_runners_context_usage.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.2 → argus_code_review-0.2.4}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,62 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.4] - 2026-09-17
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Decoupled the Gemini reviewer's turn budget from the shared
|
|
15
|
+
`argus.runners._MAX_TURNS` constant (TECH-6453). `argus/gemini_runner.py`
|
|
16
|
+
now has its own `_MAX_TURNS_GEMINI = 45` (30 \* 1.5), independent of the
|
|
17
|
+
Claude Agent SDK and OpenAI runner paths, which remain at `_MAX_TURNS = 30`
|
|
18
|
+
unchanged. Gemini-backed bulk reviewer sessions were exhausting the
|
|
19
|
+
previously-shared 30-turn budget before calling `finish_review`,
|
|
20
|
+
truncating exploration; raising only Gemini's budget avoids widening the
|
|
21
|
+
cost/blast-radius ceiling on the more expensive Claude-Opus-tier
|
|
22
|
+
cross-cutting/blocking-validator/feedback-verifier roles, which showed no
|
|
23
|
+
evidence of needing more turns.
|
|
24
|
+
|
|
25
|
+
## [0.2.3] - 2026-09-14
|
|
26
|
+
|
|
27
|
+
### Added
|
|
28
|
+
|
|
29
|
+
- Reviewer bench configuration change guard (TECH-6282): PRs touching bench
|
|
30
|
+
configuration files (`.argus/bench.toml`, `argus/bench_default.toml`) or adding
|
|
31
|
+
lines modifying bench routing environment variables (`ARGUS_BENCH_FILE`,
|
|
32
|
+
`ARGUS_NO_BENCH_OVERRIDES`) are deterministically force-BLOCKED with a finding
|
|
33
|
+
in category `argus-self-config` requiring explicit human sign-off. The guard
|
|
34
|
+
evaluates against the full-PR diff scope across multi-round reviews to prevent
|
|
35
|
+
bypasses on subsequent commits.
|
|
36
|
+
|
|
37
|
+
### Changed
|
|
38
|
+
|
|
39
|
+
- Defaulted `[bulk_reviewer]` in `argus/bench_default.toml` to `platform = "gemini"`,
|
|
40
|
+
`model = "gemini-mini"`, and `caching = "auto"` (TECH-6281). System-generalist,
|
|
41
|
+
specialist, and tests-and-docs reviewers now route to Gemini (`gemini-3.8-flash`)
|
|
42
|
+
out of the box for cost optimization across PR review fan-outs, while individual
|
|
43
|
+
roles (`cross-cutting`, `blocking-validator`, `feedback-verifier`) remain on
|
|
44
|
+
`claude-sdk`.
|
|
45
|
+
- Moved `google-genai` from the optional `[gemini]` extra into core `dependencies`
|
|
46
|
+
in `pyproject.toml`, ensuring the default Gemini bulk reviewer works out of
|
|
47
|
+
the box without requiring an extra install.
|
|
48
|
+
- In `argus/bench.py`, sparse model-only bench overrides (e.g. `model = "claude-mini"`)
|
|
49
|
+
now automatically infer their compatible platform (`claude-sdk`, `gemini`, or
|
|
50
|
+
`openai-responses`) so model overrides do not inherit an incompatible platform
|
|
51
|
+
from lower bench layers. Explicit platform/model mismatches fail validation early
|
|
52
|
+
with clear migration guidance.
|
|
53
|
+
- Made `--specialist-model` / `ARGUS_SPECIALIST_MODEL` apply as a highest-priority
|
|
54
|
+
override forcing `[bulk_reviewer]` to `claude-sdk` with `claude-default` so
|
|
55
|
+
the CLI flag continues to control system and specialist reviewers as documented.
|
|
56
|
+
- Made CLI preflight credential validation conditional on effective bench requirements:
|
|
57
|
+
`GOOGLE_API_KEY` is now required when the resolved bench config includes `gemini`.
|
|
58
|
+
|
|
59
|
+
### Fixed
|
|
60
|
+
|
|
61
|
+
- Fixed `bench.load_bench()`/`_check_settings()` requiring full credential validation (`GITHUB_TOKEN_RO`, `OPENAI_API_KEY`) just to resolve bench/platform routing config, which broke settings dependency-injection in tests and CI (regression from TECH-6281).
|
|
62
|
+
- Regenerated `tests/golden/review_response.schema.json` to match the
|
|
63
|
+
`argus-self-config` category description added in TECH-6282 — the
|
|
64
|
+
golden-snapshot test was left failing after that PR merged.
|
|
65
|
+
|
|
10
66
|
## [0.2.2] - 2026-09-11
|
|
11
67
|
|
|
12
68
|
### Added
|
|
@@ -203,7 +259,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
203
259
|
packaged set.
|
|
204
260
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
205
261
|
|
|
206
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
262
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...HEAD
|
|
263
|
+
[0.2.4]: https://github.com/redesignhealth/argus-review/compare/v0.2.3...v0.2.4
|
|
264
|
+
[0.2.3]: https://github.com/redesignhealth/argus-review/compare/v0.2.2...v0.2.3
|
|
207
265
|
[0.2.2]: https://github.com/redesignhealth/argus-review/compare/v0.2.1...v0.2.2
|
|
208
266
|
[0.2.1]: https://github.com/redesignhealth/argus-review/compare/v0.2.0...v0.2.1
|
|
209
267
|
[0.2.0]: https://github.com/redesignhealth/argus-review/compare/v0.1.5...v0.2.0
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -17,6 +17,7 @@ Classifier: Topic :: Software Development :: Quality Assurance
|
|
|
17
17
|
Requires-Python: <3.14,>=3.12
|
|
18
18
|
Requires-Dist: asyncpg>=0.29.0
|
|
19
19
|
Requires-Dist: claude-agent-sdk==0.1.81
|
|
20
|
+
Requires-Dist: google-genai<2,>=1.55
|
|
20
21
|
Requires-Dist: httpx>=0.27.0
|
|
21
22
|
Requires-Dist: langchain-anthropic>=1.2.0
|
|
22
23
|
Requires-Dist: langchain<2,>=1.2.10
|
|
@@ -41,8 +42,6 @@ Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
|
41
42
|
Requires-Dist: ruff>=0.8.0; extra == 'dev'
|
|
42
43
|
Requires-Dist: types-pyyaml>=6.0; extra == 'dev'
|
|
43
44
|
Requires-Dist: types-regex>=2024.4.16; extra == 'dev'
|
|
44
|
-
Provides-Extra: gemini
|
|
45
|
-
Requires-Dist: google-genai<2,>=1.55; extra == 'gemini'
|
|
46
45
|
Provides-Extra: prechecks
|
|
47
46
|
Requires-Dist: checkov<4.0.0,>=3.3.0; extra == 'prechecks'
|
|
48
47
|
Requires-Dist: pyyaml>=6.0; extra == 'prechecks'
|
|
@@ -207,10 +206,11 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
207
206
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
208
207
|
# reviewer -- claude-fable-5 here is already the planner/coverage default, but
|
|
209
208
|
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
210
|
-
# Cost note: --specialist-model here
|
|
211
|
-
# reviewer, specialists,
|
|
212
|
-
#
|
|
213
|
-
#
|
|
209
|
+
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
210
|
+
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
211
|
+
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
212
|
+
# paths -- both flags in this example trade cost for reasoning headroom, don't use
|
|
213
|
+
# them together as a low-cost default. Any non-empty --specialist-model value also withholds
|
|
214
214
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
215
215
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
216
216
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
@@ -155,10 +155,11 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
155
155
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
156
156
|
# reviewer -- claude-fable-5 here is already the planner/coverage default, but
|
|
157
157
|
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
158
|
-
# Cost note: --specialist-model here
|
|
159
|
-
# reviewer, specialists,
|
|
160
|
-
#
|
|
161
|
-
#
|
|
158
|
+
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
159
|
+
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
160
|
+
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
161
|
+
# paths -- both flags in this example trade cost for reasoning headroom, don't use
|
|
162
|
+
# them together as a low-cost default. Any non-empty --specialist-model value also withholds
|
|
162
163
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
163
164
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
164
165
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
@@ -1,11 +1,10 @@
|
|
|
1
1
|
"""Bench configuration: which platform/model a leaf reviewer runs on.
|
|
2
2
|
|
|
3
3
|
A "bench" is a lightweight, human-edited TOML config that decides which
|
|
4
|
-
LLM *platform* (Claude Agent SDK
|
|
5
|
-
real runner too, see ``argus.
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
deliberately NOT a dynamic/adaptive routing system -- it is a static,
|
|
4
|
+
LLM *platform* (Claude Agent SDK and Gemini have packaged defaults;
|
|
5
|
+
OpenAI Responses has a real runner too, see ``argus.openai_runner``;
|
|
6
|
+
opt-in) and *model* each leaf reviewer in the review pipeline runs on.
|
|
7
|
+
This is deliberately NOT a dynamic/adaptive routing system -- it is a static,
|
|
9
8
|
PR-reviewed config with a human-editable override chain, in the same
|
|
10
9
|
spirit as ``argus.prompts_runtime``'s prompt override chain.
|
|
11
10
|
|
|
@@ -25,9 +24,10 @@ Two kinds of config unit, not a per-role table:
|
|
|
25
24
|
Override chain (mirrors ``argus.prompts_runtime`` exactly, including its
|
|
26
25
|
opt-out convention), lowest to highest priority:
|
|
27
26
|
|
|
28
|
-
1. Packaged ``argus/bench_default.toml`` -- the base.
|
|
29
|
-
to ``platform="
|
|
30
|
-
|
|
27
|
+
1. Packaged ``argus/bench_default.toml`` -- the base. ``[bulk_reviewer]``
|
|
28
|
+
defaults to ``platform="gemini"`` (``model="gemini-mini"``,
|
|
29
|
+
``caching="auto"``), while individual roles (``cross-cutting``,
|
|
30
|
+
``blocking-validator``, ``feedback-verifier``) default to ``claude-sdk``.
|
|
31
31
|
2. ``~/.config/argus/bench.toml`` (respecting ``XDG_CONFIG_HOME``) -- a
|
|
32
32
|
user-global sparse overlay: only the keys it specifies are overridden;
|
|
33
33
|
everything else falls through to the layer below.
|
|
@@ -82,7 +82,8 @@ from importlib import resources
|
|
|
82
82
|
from pathlib import Path
|
|
83
83
|
from typing import Any, Final, Literal, get_args
|
|
84
84
|
|
|
85
|
-
from
|
|
85
|
+
from pydantic import TypeAdapter, ValidationError
|
|
86
|
+
|
|
86
87
|
from argus.llm import models as model_aliases
|
|
87
88
|
from argus.pipeline_models import SpecialistName
|
|
88
89
|
from argus.prompts_runtime import known_packaged_prompts
|
|
@@ -104,6 +105,23 @@ _TOP_LEVEL_KEYS: Final[frozenset[str]] = frozenset({"bulk_reviewer", "roles"})
|
|
|
104
105
|
_BULK_KEYS: Final[frozenset[str]] = frozenset({"platform", "model", "caching"})
|
|
105
106
|
_ROLE_KEYS: Final[frozenset[str]] = frozenset({"platform", "model", "prompt_name", "caching"})
|
|
106
107
|
|
|
108
|
+
_MODEL_FAMILY_TO_PLATFORM: Final[dict[str, Platform]] = {
|
|
109
|
+
"claude": "claude-sdk",
|
|
110
|
+
"gemini": "gemini",
|
|
111
|
+
"gpt": "openai-responses",
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def infer_platform_for_model(model: str) -> Platform | None:
|
|
116
|
+
"""Infer the execution platform from a model alias or model name.
|
|
117
|
+
|
|
118
|
+
Returns the Platform matching the model family prefix ('claude-*' -> 'claude-sdk',
|
|
119
|
+
'gemini-*' -> 'gemini', 'gpt-*' -> 'openai-responses'), or None if unknown.
|
|
120
|
+
"""
|
|
121
|
+
family = model.lower().split("-", 1)[0]
|
|
122
|
+
return _MODEL_FAMILY_TO_PLATFORM.get(family)
|
|
123
|
+
|
|
124
|
+
|
|
107
125
|
# Roles a runner function actually resolves via bench.resolve(). Every bulk
|
|
108
126
|
# role and every individual role is wired to its corresponding runner function.
|
|
109
127
|
_WIRED_ROLES: Final[frozenset[str]] = frozenset(
|
|
@@ -182,7 +200,7 @@ def _load_toml_file(path: Path) -> dict[str, Any]:
|
|
|
182
200
|
return _load_toml_bytes(path.read_bytes())
|
|
183
201
|
|
|
184
202
|
|
|
185
|
-
def _overlay_layers(
|
|
203
|
+
def _overlay_layers(settings_or_bench_file: Any = None) -> list[Path]:
|
|
186
204
|
"""Ordered overlay file paths, lowest to highest priority.
|
|
187
205
|
|
|
188
206
|
Only existing files are included for the two standard locations
|
|
@@ -202,51 +220,192 @@ def _overlay_layers(settings: Any) -> list[Path]:
|
|
|
202
220
|
if repo_local.is_file():
|
|
203
221
|
layers.append(repo_local)
|
|
204
222
|
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
223
|
+
bench_file: str | None = None
|
|
224
|
+
if isinstance(settings_or_bench_file, (str, Path)):
|
|
225
|
+
bench_file = str(settings_or_bench_file)
|
|
226
|
+
elif settings_or_bench_file is not None and hasattr(settings_or_bench_file, "ARGUS_BENCH_FILE"):
|
|
227
|
+
val = getattr(settings_or_bench_file, "ARGUS_BENCH_FILE", None)
|
|
228
|
+
if isinstance(val, (str, Path)) and str(val):
|
|
229
|
+
bench_file = str(val)
|
|
230
|
+
|
|
231
|
+
if bench_file:
|
|
232
|
+
bench_path = Path(bench_file)
|
|
233
|
+
if not bench_path.is_file():
|
|
234
|
+
raise ValueError(f"ARGUS_BENCH_FILE={bench_file!r} does not exist or is not a file")
|
|
235
|
+
layers.append(bench_path)
|
|
212
236
|
|
|
213
237
|
return layers
|
|
214
238
|
|
|
215
239
|
|
|
240
|
+
def _infer_platforms_for_overlay(overlay: dict[str, Any]) -> dict[str, Any]:
|
|
241
|
+
"""Infer the compatible platform for any table that specifies `model` but omits `platform`.
|
|
242
|
+
|
|
243
|
+
Prevents sparse model-only overrides (such as `model = "claude-mini"`) from inheriting
|
|
244
|
+
an incompatible platform (like the default `platform = "gemini"`) from a lower layer.
|
|
245
|
+
"""
|
|
246
|
+
result = dict(overlay)
|
|
247
|
+
if "bulk_reviewer" in result and isinstance(result["bulk_reviewer"], dict):
|
|
248
|
+
bulk = dict(result["bulk_reviewer"])
|
|
249
|
+
if "model" in bulk and isinstance(bulk["model"], str) and "platform" not in bulk:
|
|
250
|
+
inferred = infer_platform_for_model(bulk["model"])
|
|
251
|
+
if inferred is not None:
|
|
252
|
+
bulk["platform"] = inferred
|
|
253
|
+
result["bulk_reviewer"] = bulk
|
|
254
|
+
if "roles" in result and isinstance(result["roles"], dict):
|
|
255
|
+
roles = dict(result["roles"])
|
|
256
|
+
for role_name, role_table in roles.items():
|
|
257
|
+
if isinstance(role_table, dict):
|
|
258
|
+
r = dict(role_table)
|
|
259
|
+
if "model" in r and isinstance(r["model"], str) and "platform" not in r:
|
|
260
|
+
inferred = infer_platform_for_model(r["model"])
|
|
261
|
+
if inferred is not None:
|
|
262
|
+
r["platform"] = inferred
|
|
263
|
+
roles[role_name] = r
|
|
264
|
+
result["roles"] = roles
|
|
265
|
+
return result
|
|
266
|
+
|
|
267
|
+
|
|
216
268
|
_WARNED_ROLES: set[str] = set()
|
|
217
269
|
_WARNED_ROLES_LOCK: threading.Lock = threading.Lock()
|
|
218
270
|
|
|
271
|
+
_BOOL_ADAPTER: Final[TypeAdapter[bool]] = TypeAdapter(bool)
|
|
219
272
|
|
|
220
|
-
@lru_cache(maxsize=1)
|
|
221
|
-
def load_bench() -> dict[str, Any]:
|
|
222
|
-
"""Load, merge, and validate the effective bench config.
|
|
223
273
|
|
|
224
|
-
|
|
225
|
-
a
|
|
274
|
+
def _parse_bool(val: Any, var_name: str = "ARGUS_NO_BENCH_OVERRIDES") -> bool:
|
|
275
|
+
"""Parse a boolean value matching Pydantic's coercion rules.
|
|
276
|
+
|
|
277
|
+
When val is None (i.e. env var unset), returns False default.
|
|
278
|
+
Otherwise delegates directly to Pydantic's TypeAdapter(bool)
|
|
279
|
+
and converts ValidationError to ValueError with a clear message.
|
|
226
280
|
"""
|
|
227
|
-
|
|
281
|
+
if val is None:
|
|
282
|
+
return False
|
|
283
|
+
from unittest.mock import NonCallableMock
|
|
284
|
+
|
|
285
|
+
if isinstance(val, NonCallableMock):
|
|
286
|
+
return False
|
|
287
|
+
try:
|
|
288
|
+
return _BOOL_ADAPTER.validate_python(val)
|
|
289
|
+
except ValidationError as exc:
|
|
290
|
+
raise ValueError(
|
|
291
|
+
f"Invalid boolean value for {var_name}: {val!r}. "
|
|
292
|
+
f"Expected a valid boolean (e.g. true/false, yes/no, 1/0, on/off)."
|
|
293
|
+
) from exc
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _extract_bench_settings(settings: Any) -> tuple[bool, str | None, str | None]:
|
|
297
|
+
"""Extract (no_bench_overrides, bench_file, specialist_model) from a settings-like object.
|
|
298
|
+
|
|
299
|
+
When settings is provided, its fields are authoritative and never fall back to
|
|
300
|
+
ambient os.environ (preserving dependency injection). For unit test mocks
|
|
301
|
+
(MagicMock), unconfigured attributes that were never set on the mock fall back to
|
|
302
|
+
os.environ so ambient test fixtures can supply values.
|
|
303
|
+
"""
|
|
304
|
+
from unittest.mock import NonCallableMock
|
|
305
|
+
|
|
306
|
+
if isinstance(settings, NonCallableMock):
|
|
307
|
+
if "ARGUS_NO_BENCH_OVERRIDES" in settings.__dict__:
|
|
308
|
+
no_overrides = _parse_bool(
|
|
309
|
+
settings.ARGUS_NO_BENCH_OVERRIDES, "ARGUS_NO_BENCH_OVERRIDES"
|
|
310
|
+
)
|
|
311
|
+
else:
|
|
312
|
+
no_overrides = _parse_bool(
|
|
313
|
+
os.environ.get("ARGUS_NO_BENCH_OVERRIDES"), "ARGUS_NO_BENCH_OVERRIDES"
|
|
314
|
+
)
|
|
315
|
+
|
|
316
|
+
if "ARGUS_BENCH_FILE" in settings.__dict__:
|
|
317
|
+
bf = settings.ARGUS_BENCH_FILE
|
|
318
|
+
bench_file = str(bf) if isinstance(bf, (str, Path)) and str(bf) else None
|
|
319
|
+
else:
|
|
320
|
+
bench_file = os.environ.get("ARGUS_BENCH_FILE") or None
|
|
321
|
+
|
|
322
|
+
if "ARGUS_SPECIALIST_MODEL" in settings.__dict__:
|
|
323
|
+
sm = settings.ARGUS_SPECIALIST_MODEL
|
|
324
|
+
specialist_model = str(sm) if isinstance(sm, str) and sm else None
|
|
325
|
+
else:
|
|
326
|
+
specialist_model = os.environ.get("ARGUS_SPECIALIST_MODEL") or None
|
|
327
|
+
|
|
328
|
+
return (no_overrides, bench_file, specialist_model)
|
|
329
|
+
|
|
330
|
+
# Real Settings or dataclass: strictly authoritative, NEVER fall back to os.environ
|
|
331
|
+
no_overrides = _parse_bool(
|
|
332
|
+
getattr(settings, "ARGUS_NO_BENCH_OVERRIDES", False), "ARGUS_NO_BENCH_OVERRIDES"
|
|
333
|
+
)
|
|
334
|
+
bf = getattr(settings, "ARGUS_BENCH_FILE", None)
|
|
335
|
+
bench_file = str(bf) if isinstance(bf, (str, Path)) and str(bf) else None
|
|
336
|
+
sm = getattr(settings, "ARGUS_SPECIALIST_MODEL", None)
|
|
337
|
+
specialist_model = str(sm) if isinstance(sm, str) and sm else None
|
|
338
|
+
return (no_overrides, bench_file, specialist_model)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def _resolve_bench_settings(settings: Any = None) -> tuple[bool, str | None, str | None]:
|
|
342
|
+
"""Resolve bench-routing settings from an injected object or ambient environment."""
|
|
343
|
+
if settings is not None:
|
|
344
|
+
return _extract_bench_settings(settings)
|
|
345
|
+
|
|
346
|
+
# Read os.environ directly from the ambient environment without
|
|
347
|
+
# invoking get_settings() or any dotenv loaders, ensuring bench loading
|
|
348
|
+
# never requires credentials, never mutates os.environ, and never loads
|
|
349
|
+
# untrusted repo-local .env files into the process.
|
|
350
|
+
no_overrides = _parse_bool(
|
|
351
|
+
os.environ.get("ARGUS_NO_BENCH_OVERRIDES"), "ARGUS_NO_BENCH_OVERRIDES"
|
|
352
|
+
)
|
|
353
|
+
bench_file = os.environ.get("ARGUS_BENCH_FILE") or None
|
|
354
|
+
specialist_model = os.environ.get("ARGUS_SPECIALIST_MODEL") or None
|
|
355
|
+
return (no_overrides, bench_file, specialist_model)
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
@lru_cache(maxsize=16)
|
|
359
|
+
def _load_bench_cached(
|
|
360
|
+
no_bench_overrides: bool,
|
|
361
|
+
bench_file: str | None,
|
|
362
|
+
specialist_model: str | None,
|
|
363
|
+
) -> dict[str, Any]:
|
|
228
364
|
merged = _load_packaged_default()
|
|
229
365
|
|
|
230
|
-
if not
|
|
231
|
-
for layer_path in _overlay_layers(
|
|
366
|
+
if not no_bench_overrides:
|
|
367
|
+
for layer_path in _overlay_layers(bench_file):
|
|
232
368
|
overlay = _load_toml_file(layer_path)
|
|
233
369
|
roles = overlay.get("roles", {})
|
|
234
370
|
if isinstance(roles, dict):
|
|
235
371
|
for role in roles:
|
|
236
372
|
_warn_if_not_wired(role)
|
|
237
|
-
|
|
373
|
+
prepared = _infer_platforms_for_overlay(overlay)
|
|
374
|
+
merged = _deep_merge(merged, prepared)
|
|
375
|
+
|
|
376
|
+
# --specialist-model / ARGUS_SPECIALIST_MODEL is a CLI/env knob that
|
|
377
|
+
# overrides the system reviewer and specialist reviewers. When set,
|
|
378
|
+
# force bulk_reviewer to claude-sdk with claude-default so the override
|
|
379
|
+
# controls the bulk reviewer roles as documented.
|
|
380
|
+
if specialist_model:
|
|
381
|
+
merged["bulk_reviewer"]["platform"] = "claude-sdk"
|
|
382
|
+
merged["bulk_reviewer"]["model"] = "claude-default"
|
|
238
383
|
|
|
239
384
|
_validate_raw_bench(merged)
|
|
240
385
|
return merged
|
|
241
386
|
|
|
242
387
|
|
|
388
|
+
def load_bench(settings: Any = None) -> dict[str, Any]:
|
|
389
|
+
"""Load, merge, and validate the effective bench config.
|
|
390
|
+
|
|
391
|
+
Cached for the life of the process; call :func:`clear_cache` to force
|
|
392
|
+
a reload (e.g. in tests, or after mutating ``os.environ``).
|
|
393
|
+
"""
|
|
394
|
+
no_overrides, bench_file, specialist_model = _resolve_bench_settings(settings)
|
|
395
|
+
return _load_bench_cached(no_overrides, bench_file, specialist_model)
|
|
396
|
+
|
|
397
|
+
|
|
243
398
|
def clear_cache() -> None:
|
|
244
399
|
"""Clear the cached bench config, forcing the next call to reload."""
|
|
245
400
|
with _WARNED_ROLES_LOCK:
|
|
246
|
-
|
|
401
|
+
_load_bench_cached.cache_clear()
|
|
247
402
|
_WARNED_ROLES.clear()
|
|
248
403
|
|
|
249
404
|
|
|
405
|
+
load_bench.cache_clear = _load_bench_cached.cache_clear # type: ignore[attr-defined]
|
|
406
|
+
load_bench.cache_info = _load_bench_cached.cache_info # type: ignore[attr-defined]
|
|
407
|
+
|
|
408
|
+
|
|
250
409
|
# ---------------------------------------------------------------------------
|
|
251
410
|
# Validation
|
|
252
411
|
# ---------------------------------------------------------------------------
|
|
@@ -288,6 +447,16 @@ def _validate_prompt_name(name: Any, where: str) -> None:
|
|
|
288
447
|
)
|
|
289
448
|
|
|
290
449
|
|
|
450
|
+
def _validate_model_platform_compatibility(platform: str, model: str, where: str) -> None:
|
|
451
|
+
expected_platform = infer_platform_for_model(model)
|
|
452
|
+
if expected_platform is not None and platform != expected_platform:
|
|
453
|
+
raise ValueError(
|
|
454
|
+
f"{where}: model {model!r} is not compatible with platform {platform!r} "
|
|
455
|
+
f"(expected platform={expected_platform!r}). "
|
|
456
|
+
f"Set platform = {expected_platform!r} or select a model compatible with {platform!r}."
|
|
457
|
+
)
|
|
458
|
+
|
|
459
|
+
|
|
291
460
|
def _validate_raw_bench(raw: dict[str, Any]) -> None:
|
|
292
461
|
"""Validate the fully-merged bench config, raising loudly on any problem.
|
|
293
462
|
|
|
@@ -319,6 +488,7 @@ def _validate_raw_bench(raw: dict[str, Any]) -> None:
|
|
|
319
488
|
_validate_platform(bulk["platform"], "[bulk_reviewer]")
|
|
320
489
|
_validate_model(bulk["model"], "[bulk_reviewer]")
|
|
321
490
|
_validate_caching(bulk.get("caching", "auto"), "[bulk_reviewer]")
|
|
491
|
+
_validate_model_platform_compatibility(bulk["platform"], bulk["model"], "[bulk_reviewer]")
|
|
322
492
|
|
|
323
493
|
roles = raw.get("roles", {})
|
|
324
494
|
if not isinstance(roles, dict):
|
|
@@ -340,6 +510,7 @@ def _validate_raw_bench(raw: dict[str, Any]) -> None:
|
|
|
340
510
|
_validate_model(role_table["model"], where)
|
|
341
511
|
_validate_caching(role_table.get("caching", "auto"), where)
|
|
342
512
|
_validate_prompt_name(role_table["prompt_name"], where)
|
|
513
|
+
_validate_model_platform_compatibility(role_table["platform"], role_table["model"], where)
|
|
343
514
|
|
|
344
515
|
# The bulk bucket's own known prompt names are hardcoded (not
|
|
345
516
|
# TOML-supplied), but validate them too: a packaged/override prompt
|
|
@@ -381,7 +552,7 @@ def _warn_if_not_wired(role: str) -> None:
|
|
|
381
552
|
)
|
|
382
553
|
|
|
383
554
|
|
|
384
|
-
def resolve(role: str) -> BenchEntry:
|
|
555
|
+
def resolve(role: str, settings: Any = None) -> BenchEntry:
|
|
385
556
|
"""Resolve ``role`` to its effective :class:`BenchEntry`.
|
|
386
557
|
|
|
387
558
|
Bulk-bucket roles (see ``BULK_ROLE_PROMPTS``) route through
|
|
@@ -395,7 +566,7 @@ def resolve(role: str) -> BenchEntry:
|
|
|
395
566
|
docstring section for why most roles currently have no runner
|
|
396
567
|
consulting them.
|
|
397
568
|
"""
|
|
398
|
-
raw = load_bench()
|
|
569
|
+
raw = load_bench(settings=settings)
|
|
399
570
|
|
|
400
571
|
if role in BULK_ROLE_PROMPTS:
|
|
401
572
|
bulk = raw["bulk_reviewer"]
|
|
@@ -529,9 +700,9 @@ async def _gemini_runner(
|
|
|
529
700
|
|
|
530
701
|
Imports ``argus.config`` and ``argus.gemini_runner`` lazily (function
|
|
531
702
|
body, not module level) for the same reason ``_claude_sdk_runner``
|
|
532
|
-
imports lazily: avoids a needless import of the
|
|
533
|
-
|
|
534
|
-
|
|
703
|
+
imports lazily: avoids a needless import of the Gemini runner module
|
|
704
|
+
for every caller of this module, even ones that never touch the
|
|
705
|
+
``gemini`` platform.
|
|
535
706
|
"""
|
|
536
707
|
from argus.config import get_settings
|
|
537
708
|
from argus.gemini_runner import run_session_gemini
|
|
@@ -2,9 +2,11 @@
|
|
|
2
2
|
#
|
|
3
3
|
# See argus/bench.py's module docstring for the full override chain and
|
|
4
4
|
# the bulk_reviewer vs. roles.<name> distinction. This file is the base
|
|
5
|
-
# of that chain
|
|
6
|
-
#
|
|
7
|
-
#
|
|
5
|
+
# of that chain: [bulk_reviewer] defaults to platform="gemini"
|
|
6
|
+
# (model="gemini-mini", caching="auto") for cost reasons across
|
|
7
|
+
# generalist, specialist, and tests-and-docs reviewers, while individual
|
|
8
|
+
# roles ([roles.cross-cutting], [roles.blocking-validator], and
|
|
9
|
+
# [roles.feedback-verifier]) remain on platform="claude-sdk".
|
|
8
10
|
#
|
|
9
11
|
# To customize: create ./.argus/bench.toml or ~/.config/argus/bench.toml
|
|
10
12
|
# with ONLY the keys you want to change (sparse overlay -- you never need
|
|
@@ -22,8 +24,8 @@
|
|
|
22
24
|
# Applies as one unit to run_system_reviewer (per system group, including
|
|
23
25
|
# gap-fill reviewers), every named specialist, and the tests-and-docs
|
|
24
26
|
# reviewer. Each of those still resolves its own existing prompt file.
|
|
25
|
-
platform = "
|
|
26
|
-
model = "
|
|
27
|
+
platform = "gemini"
|
|
28
|
+
model = "gemini-mini"
|
|
27
29
|
caching = "auto"
|
|
28
30
|
|
|
29
31
|
# The three [roles.*] tables below configure independent roles:
|
|
@@ -47,8 +47,9 @@ Environment variables (HTTP-mode opt-in):
|
|
|
47
47
|
|
|
48
48
|
No storage env vars are required: with neither ``ARGUS_DB_URL`` nor the
|
|
49
49
|
HTTP-shim URLs set, round history and checkpoints default to local SQLite.
|
|
50
|
-
Required secrets are
|
|
51
|
-
GITHUB_TOKEN_RO, and
|
|
50
|
+
Required secrets are one of ANTHROPIC_API_KEY / ANTHROPIC_AUTH_TOKEN,
|
|
51
|
+
GITHUB_TOKEN_RO, OPENAI_API_KEY, and (by default, since bulk reviewers default
|
|
52
|
+
to Gemini) GOOGLE_API_KEY. ANTHROPIC_AUTH_TOKEN is the standard
|
|
52
53
|
mechanism for routing through a corporate LLM gateway/proxy instead of a
|
|
53
54
|
real Anthropic API key (sent as ``Authorization: Bearer`` rather than
|
|
54
55
|
``x-api-key``) — the same convention the Anthropic SDK and Claude Code
|
|
@@ -177,11 +178,13 @@ def _load_settings() -> "Settings":
|
|
|
177
178
|
def _check_settings(settings: "Settings") -> None:
|
|
178
179
|
"""Validate that critical secrets were loaded, fail loudly otherwise.
|
|
179
180
|
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
181
|
+
Always required:
|
|
182
|
+
- ANTHROPIC_API_KEY (or ANTHROPIC_AUTH_TOKEN) for Agent SDK + LangChain
|
|
183
|
+
- GITHUB_TOKEN_RO for diff fetch + clone
|
|
184
|
+
- OPENAI_API_KEY for plan-extraction path
|
|
185
|
+
|
|
186
|
+
Additionally, if any leaf reviewer in the effective resolved bench config
|
|
187
|
+
requires the Gemini platform, GOOGLE_API_KEY is required.
|
|
185
188
|
|
|
186
189
|
Args:
|
|
187
190
|
settings: Resolved settings object.
|
|
@@ -193,6 +196,27 @@ def _check_settings(settings: "Settings") -> None:
|
|
|
193
196
|
missing.append("GITHUB_TOKEN_RO")
|
|
194
197
|
if not settings.OPENAI_API_KEY:
|
|
195
198
|
missing.append("OPENAI_API_KEY")
|
|
199
|
+
|
|
200
|
+
from argus import bench
|
|
201
|
+
|
|
202
|
+
bench.clear_cache()
|
|
203
|
+
try:
|
|
204
|
+
raw_bench = bench.load_bench(settings=settings)
|
|
205
|
+
except Exception as exc:
|
|
206
|
+
logger.error("Invalid bench configuration: %s", exc)
|
|
207
|
+
sys.exit(1)
|
|
208
|
+
|
|
209
|
+
platforms_needed: set[str] = set()
|
|
210
|
+
bulk = raw_bench.get("bulk_reviewer")
|
|
211
|
+
if isinstance(bulk, dict) and "platform" in bulk:
|
|
212
|
+
platforms_needed.add(bulk["platform"])
|
|
213
|
+
for role_cfg in raw_bench.get("roles", {}).values():
|
|
214
|
+
if isinstance(role_cfg, dict) and "platform" in role_cfg:
|
|
215
|
+
platforms_needed.add(role_cfg["platform"])
|
|
216
|
+
|
|
217
|
+
if "gemini" in platforms_needed and not settings.GOOGLE_API_KEY:
|
|
218
|
+
missing.append("GOOGLE_API_KEY")
|
|
219
|
+
|
|
196
220
|
if missing:
|
|
197
221
|
logger.error(
|
|
198
222
|
"Missing required secrets: %s. Add them to your shell or .env file.",
|
|
@@ -330,9 +354,10 @@ def _add_review_args(parser: argparse.ArgumentParser) -> None:
|
|
|
330
354
|
default=_MODEL_OVERRIDE_UNSET,
|
|
331
355
|
help=(
|
|
332
356
|
"Override the model used by the system reviewer, specialist "
|
|
333
|
-
"reviewers, the writer, and the lite-review path (
|
|
334
|
-
"claude-
|
|
335
|
-
"
|
|
357
|
+
"reviewers, the writer, and the lite-review path (forcing "
|
|
358
|
+
"bulk reviewers to claude-sdk with the specified model; default "
|
|
359
|
+
"without this flag routes bulk reviewers to gemini-mini). "
|
|
360
|
+
"Same effect as setting ARGUS_SPECIALIST_MODEL. "
|
|
336
361
|
"Pass an empty string to clear an already-set "
|
|
337
362
|
"ARGUS_SPECIALIST_MODEL for this run."
|
|
338
363
|
),
|
|
@@ -572,7 +597,6 @@ def _run_review(parser: argparse.ArgumentParser, args: argparse.Namespace) -> No
|
|
|
572
597
|
except Exception as exc: # pydantic ValidationError for missing required vars
|
|
573
598
|
logger.error("Failed to load settings: %s", exc)
|
|
574
599
|
sys.exit(1)
|
|
575
|
-
_check_settings(settings)
|
|
576
600
|
|
|
577
601
|
# Applied AFTER _load_settings(), not before: _load_settings() calls
|
|
578
602
|
# load_dotenv_early(..., override=False), which repopulates any env var
|
|
@@ -587,6 +611,12 @@ def _run_review(parser: argparse.ArgumentParser, args: argparse.Namespace) -> No
|
|
|
587
611
|
_apply_model_override_flag("ARGUS_SPECIALIST_MODEL", args.specialist_model)
|
|
588
612
|
_apply_model_override_flag("ARGUS_FRONTIER_MODEL", args.frontier_model)
|
|
589
613
|
|
|
614
|
+
from argus.config import clear_cache as clear_config_cache, get_settings
|
|
615
|
+
|
|
616
|
+
clear_config_cache()
|
|
617
|
+
settings = get_settings()
|
|
618
|
+
_check_settings(settings)
|
|
619
|
+
|
|
590
620
|
from argus.models import ReviewRequest
|
|
591
621
|
|
|
592
622
|
request = ReviewRequest(
|
|
@@ -92,20 +92,15 @@ class Settings(BaseSettings):
|
|
|
92
92
|
path (``argus.llm.models.CLAUDE_DEFAULT``, default
|
|
93
93
|
``claude-sonnet-4-6``). Set via ``--specialist-model``; read
|
|
94
94
|
directly from ``os.environ`` by ``argus.llm.models`` at import
|
|
95
|
-
time (
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
``ARGUS_NO_PROMPT_OVERRIDES`` above, which genuinely is
|
|
99
|
-
(``prompts_runtime.override_dirs`` reads
|
|
100
|
-
``settings.ARGUS_NO_PROMPT_OVERRIDES``). It's declared here
|
|
101
|
-
purely for documentation/discoverability, same as any other
|
|
102
|
-
env var this class's docstring covers.
|
|
95
|
+
time and by ``argus.bench`` (where setting it forces bulk reviewers
|
|
96
|
+
to claude-sdk with claude-default). Also read off a ``Settings``
|
|
97
|
+
instance during bench resolution when dependency-injected.
|
|
103
98
|
ARGUS_FRONTIER_MODEL: Override the model used by the planner,
|
|
104
99
|
coverage check, and cross-cutting reviewer
|
|
105
100
|
(``argus.llm.models.CLAUDE_FRONTIER`` / ``CLAUDE_OPUS``). Set via
|
|
106
|
-
``--frontier-model``;
|
|
107
|
-
|
|
108
|
-
|
|
101
|
+
``--frontier-model``; read directly from ``os.environ`` by
|
|
102
|
+
``argus.llm.models`` at import time (module-level constant
|
|
103
|
+
resolution, not a per-request Settings lookup).
|
|
109
104
|
LANGSMITH_API_KEY / LANGSMITH_PROJECT: Optional tracing.
|
|
110
105
|
OPENAI_BASE_URL: Optional override for OpenAI API endpoint (e.g. for proxying).
|
|
111
106
|
Note that OPENAI_API_KEY will be forwarded as a Bearer token to whatever
|
|
@@ -131,10 +126,9 @@ class Settings(BaseSettings):
|
|
|
131
126
|
additional headroom across all three reviewer platforms.
|
|
132
127
|
GOOGLE_API_KEY: Gemini platform credential, consumed via the
|
|
133
128
|
``google_credential`` property by ``argus.gemini_runner``
|
|
134
|
-
|
|
135
|
-
``platform = "gemini"``.
|
|
136
|
-
|
|
137
|
-
override chain) -- the packaged default bench never does.
|
|
129
|
+
whenever a role's bench entry resolves to
|
|
130
|
+
``platform = "gemini"``. Required for default review runs
|
|
131
|
+
since ``[bulk_reviewer]`` defaults to Gemini.
|
|
138
132
|
GOOGLE_BASE_URL: Optional override for the Gemini API endpoint (e.g. for
|
|
139
133
|
proxying). Mirrors ``OPENAI_BASE_URL`` above. Note that GOOGLE_API_KEY
|
|
140
134
|
will be forwarded to whatever endpoint is configured here; point only
|