argus-code-review 0.2.7__tar.gz → 0.2.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/CHANGELOG.md +27 -1
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/PKG-INFO +3 -4
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/README.md +2 -3
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/bench_default.toml +1 -3
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/cli.py +1 -4
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/config.py +17 -10
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/gemini_runner.py +3 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/llm/models.py +24 -22
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/openai_runner.py +8 -7
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-llm-patterns.md +2 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/runners.py +208 -28
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_bench.py +1 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_config.py +2 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_llm_models.py +50 -16
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_llm_models_override.py +14 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_openai_runner.py +38 -28
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_runners_context_usage.py +2 -0
- argus_code_review-0.2.9/tests/test_runners_turn_budget.py +750 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/.gitignore +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/LICENSE +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/bench.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/coverage.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/github_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/graph.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/helpers.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/llm/output_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/llm/usage.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/pipeline_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/engine.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/sql.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/argus/storage/sqlite.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/pyproject.toml +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/schema/018_widen_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/conftest.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/golden/review_response.schema.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_backend_contract.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_http.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_sql.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/storage/test_sqlite_backend.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_argus_review_local.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_bench_config_guard.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_cli_preflight.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_gemini_runner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_fetch_diff.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_get_llm_temperature.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_preflight_image_bump.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_graph_timeout_surfacing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_multi_round.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_openai_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_output_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_packaging.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_precheck_engine.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_runners_helpers.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_runners_new.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_stage_costs.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.9}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,30 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.9] - 2026-10-02
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Upgraded frontier and default model aliases in `argus.llm.models` (TECH-7124, #31):
|
|
15
|
+
- `claude-frontier` and `claude-opus` upgraded to `claude-opus-5-5` (previously `claude-fable-5` and `claude-opus-5`).
|
|
16
|
+
- `claude-default` upgraded to `claude-sonnet-5-5` (previously `claude-sonnet-4-6`).
|
|
17
|
+
- `gpt-frontier` upgraded to `gpt-6.1-sol` (previously `gpt-5.6-sol`).
|
|
18
|
+
- Updated model pricing rates in `argus.llm.models` for `claude-sonnet-5-5`, `claude-opus-5-5`, and `gpt-6.1-sol` (TECH-7124, #31).
|
|
19
|
+
- Re-verified the Anthropic 1M context beta (`context-1m-2025-08-07`) against `claude-sonnet-5-5` via the Redesign Health Anthropic proxy (TECH-7124, #31).
|
|
20
|
+
- Updated approved model policies in `pr-review-specialist-llm-patterns.md`, runner warning logs, bench defaults, and CLI help text for the new model aliases (TECH-7124, #31).
|
|
21
|
+
|
|
22
|
+
## [0.2.8] - 2026-10-02
|
|
23
|
+
|
|
24
|
+
### Changed
|
|
25
|
+
|
|
26
|
+
- Raised the Claude-path turn budget from 30 to 50 and the default reviewer session
|
|
27
|
+
timeout from 900 to 1500 seconds globally across all runner platforms through
|
|
28
|
+
`ARGUS_SESSION_TIMEOUT` (TECH-7093), preserving headroom for longer sessions.
|
|
29
|
+
- Added Claude's budget disclosure and supported `PostToolUse`/`PostToolUseFailure`
|
|
30
|
+
hook-based convergence warnings (derived from budget constants; 37 and 47 tool calls
|
|
31
|
+
by default), using truthful top-level tool-call counts, including failed calls.
|
|
32
|
+
- Decoupled the Claude and OpenAI turn-budget constants; OpenAI remains at 30 turns.
|
|
33
|
+
|
|
10
34
|
## [0.2.7] - 2026-09-26
|
|
11
35
|
|
|
12
36
|
### Fixed
|
|
@@ -356,7 +380,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
356
380
|
packaged set.
|
|
357
381
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
358
382
|
|
|
359
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
383
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.9...HEAD
|
|
384
|
+
[0.2.9]: https://github.com/redesignhealth/argus-review/compare/v0.2.8...v0.2.9
|
|
385
|
+
[0.2.8]: https://github.com/redesignhealth/argus-review/compare/v0.2.7...v0.2.8
|
|
360
386
|
[0.2.7]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...v0.2.7
|
|
361
387
|
[0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
|
|
362
388
|
[0.2.5]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...v0.2.5
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.9
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -204,8 +204,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
204
204
|
|
|
205
205
|
# Override the reviewer models (or set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL instead).
|
|
206
206
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
207
|
-
# reviewer
|
|
208
|
-
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
207
|
+
# reviewer (defaults to claude-opus-5-5).
|
|
209
208
|
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
210
209
|
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
211
210
|
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
@@ -214,7 +213,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
214
213
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
215
214
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
216
215
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
217
|
-
argus review owner/repo --pr 123 --specialist-model claude-opus-5 --frontier-model claude-
|
|
216
|
+
argus review owner/repo --pr 123 --specialist-model claude-opus-5-5 --frontier-model claude-opus-5-5
|
|
218
217
|
|
|
219
218
|
# Clear an already-set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL for just this
|
|
220
219
|
# run by passing an empty string -- useful when the env var is set globally
|
|
@@ -153,8 +153,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
153
153
|
|
|
154
154
|
# Override the reviewer models (or set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL instead).
|
|
155
155
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
156
|
-
# reviewer
|
|
157
|
-
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
156
|
+
# reviewer (defaults to claude-opus-5-5).
|
|
158
157
|
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
159
158
|
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
160
159
|
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
@@ -163,7 +162,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
163
162
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
164
163
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
165
164
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
166
|
-
argus review owner/repo --pr 123 --specialist-model claude-opus-5 --frontier-model claude-
|
|
165
|
+
argus review owner/repo --pr 123 --specialist-model claude-opus-5-5 --frontier-model claude-opus-5-5
|
|
167
166
|
|
|
168
167
|
# Clear an already-set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL for just this
|
|
169
168
|
# run by passing an empty string -- useful when the env var is set globally
|
|
@@ -33,9 +33,7 @@ caching = "auto"
|
|
|
33
33
|
|
|
34
34
|
[roles.cross-cutting]
|
|
35
35
|
platform = "claude-sdk"
|
|
36
|
-
#
|
|
37
|
-
# evals showed no measurable quality gain from frontier on this stage, at
|
|
38
|
-
# ~2x the per-token cost.
|
|
36
|
+
# Matches argus.runners._CROSS_CUTTING_MODEL (resolves to claude-opus-5-5).
|
|
39
37
|
model = "claude-opus"
|
|
40
38
|
prompt_name = "pr-review-cross-cutting"
|
|
41
39
|
caching = "auto"
|
|
@@ -368,10 +368,7 @@ def _add_review_args(parser: argparse.ArgumentParser) -> None:
|
|
|
368
368
|
default=_MODEL_OVERRIDE_UNSET,
|
|
369
369
|
help=(
|
|
370
370
|
"Override the model used by the planner, coverage check, and "
|
|
371
|
-
"cross-cutting reviewer (
|
|
372
|
-
"two, claude-opus-5 for cross-cutting -- note this one flag "
|
|
373
|
-
"collapses both onto the SAME model when set, moving "
|
|
374
|
-
"cross-cutting off its cheaper default; or ARGUS_FRONTIER_MODEL "
|
|
371
|
+
"cross-cutting reviewer (default: claude-opus-5-5, or ARGUS_FRONTIER_MODEL "
|
|
375
372
|
"if already set in the environment). Same effect as setting "
|
|
376
373
|
"ARGUS_FRONTIER_MODEL. Pass an empty string to clear an "
|
|
377
374
|
"already-set ARGUS_FRONTIER_MODEL for this run."
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
A slim ``pydantic-settings`` implementation. All configuration comes from
|
|
4
4
|
environment variables (or a local ``.env`` file loaded via
|
|
5
|
-
``argus.dotenv_utils``)
|
|
5
|
+
``argus.dotenv_utils``) -- no AWS, no SSM.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
@@ -15,7 +15,7 @@ from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
|
15
15
|
# ``argus.runners._SUBPROCESS_TIMEOUT_S`` (the fallback used when no Settings
|
|
16
16
|
# instance is available, mostly tests) imports this same constant rather than
|
|
17
17
|
# hardcoding its own copy, so the two can never drift out of sync.
|
|
18
|
-
DEFAULT_ARGUS_SESSION_TIMEOUT_S =
|
|
18
|
+
DEFAULT_ARGUS_SESSION_TIMEOUT_S = 1500
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
class Settings(BaseSettings):
|
|
@@ -62,7 +62,7 @@ class Settings(BaseSettings):
|
|
|
62
62
|
``argus.precheck``.
|
|
63
63
|
ARGUS_STOCK_SEMGREP_PACKS: Comma-separated semgrep registry pack IDs
|
|
64
64
|
(e.g. ``"p/secrets"``) to run alongside (or instead of) a custom
|
|
65
|
-
``ARGUS_RULES_DIR``
|
|
65
|
+
``ARGUS_RULES_DIR`` -- unlike a local rules directory, each pack
|
|
66
66
|
is fetched over the network by semgrep itself on first use
|
|
67
67
|
(cached locally after). Unset by default: this is an opt-in
|
|
68
68
|
addition of vetted, community-maintained rules, not a silent
|
|
@@ -74,8 +74,8 @@ class Settings(BaseSettings):
|
|
|
74
74
|
(``PrecheckResult.failed_scanners``/``missing_scanners``
|
|
75
75
|
non-empty) instead of the default fail-open
|
|
76
76
|
behavior (surface it in the review comment's degraded-coverage
|
|
77
|
-
section
|
|
78
|
-
|
|
77
|
+
section -- see ``argus.helpers.build_degraded_coverage_labels``
|
|
78
|
+
-- but let the review's own verdict stand on its own merits).
|
|
79
79
|
False by default: every other part of this module's design is
|
|
80
80
|
deliberately fail-open (a broken scanner should never be the
|
|
81
81
|
reason a review can't complete), and this flag exists for
|
|
@@ -91,7 +91,7 @@ class Settings(BaseSettings):
|
|
|
91
91
|
ARGUS_SPECIALIST_MODEL: Override the model used by the system
|
|
92
92
|
reviewer, specialist reviewers, the writer, and the lite-review
|
|
93
93
|
path (``argus.llm.models.CLAUDE_DEFAULT``, default
|
|
94
|
-
``claude-sonnet-
|
|
94
|
+
``claude-sonnet-5-5``). Set via ``--specialist-model``; read
|
|
95
95
|
directly from ``os.environ`` by ``argus.llm.models`` at import
|
|
96
96
|
time and by ``argus.bench`` (where setting it forces bulk reviewers
|
|
97
97
|
to claude-sdk with claude-default). Also read off a ``Settings``
|
|
@@ -117,14 +117,21 @@ class Settings(BaseSettings):
|
|
|
117
117
|
Context7 call.
|
|
118
118
|
ARGUS_SESSION_TIMEOUT: Wall-clock seconds a reviewer subprocess is
|
|
119
119
|
allowed to run before it is killed and reported as a failure.
|
|
120
|
-
Defaults to
|
|
120
|
+
Defaults to 1500 (25 minutes) -- first raised from 300 to 420 after
|
|
121
121
|
production logs showed legitimate (non-runaway) specialist
|
|
122
122
|
reviewers finishing as late as 294s, right at the old timeout's
|
|
123
123
|
edge; raised again to 600 to match rh-data-platform's
|
|
124
124
|
production-proven value ahead of this package taking over as the
|
|
125
125
|
actual production reviewer (rh-data-platform's review_service is
|
|
126
|
-
being retired in its favor); raised
|
|
127
|
-
|
|
126
|
+
being retired in its favor); raised to 900 for additional headroom
|
|
127
|
+
across all three reviewer platforms; raised to 1500 in TECH-7093.
|
|
128
|
+
This timeout is deliberately global to all runner platforms through
|
|
129
|
+
ARGUS_SESSION_TIMEOUT, rather than scoped only to Claude, to preserve
|
|
130
|
+
a single operator control across runners. In production, an observed
|
|
131
|
+
cross-cutting session ran 533.8s under the 30-turn cap; linear
|
|
132
|
+
scaling 30 to 50 projects ~890s, meaning 900s leaves virtually zero
|
|
133
|
+
headroom. 1500s preserves roughly 1.69x margin, critical because a
|
|
134
|
+
timeout kill discards all reviewer output.
|
|
128
135
|
GOOGLE_API_KEY: Gemini platform credential, consumed via the
|
|
129
136
|
``google_credential`` property by ``argus.gemini_runner``
|
|
130
137
|
whenever a role's bench entry resolves to
|
|
@@ -206,7 +213,7 @@ class Settings(BaseSettings):
|
|
|
206
213
|
|
|
207
214
|
Callers that need to hand this credential to something that itself
|
|
208
215
|
reads an env var (the spawned ``claude`` CLI subprocess) must set
|
|
209
|
-
the SAME variable name the caller configured
|
|
216
|
+
the SAME variable name the caller configured -- forcing everything
|
|
210
217
|
to ``ANTHROPIC_API_KEY`` would send a proxy/gateway bearer token as
|
|
211
218
|
an ``x-api-key``, which not every gateway accepts. ``ANTHROPIC_API_KEY``
|
|
212
219
|
wins when both are set, matching the Anthropic SDK's own precedence.
|
|
@@ -41,8 +41,9 @@ function call the model requests each turn (there can be more than
|
|
|
41
41
|
one), feeding all of their results back as a single follow-up turn --
|
|
42
42
|
until the model stops requesting function calls, calls
|
|
43
43
|
``finish_review``, or the turn budget (``_MAX_TURNS_GEMINI``, this
|
|
44
|
-
module's own budget, independent of the Claude
|
|
45
|
-
``argus.runners.
|
|
44
|
+
module's own budget, independent of the Claude path's
|
|
45
|
+
``argus.runners._MAX_TURNS_CLAUDE`` and the OpenAI path's
|
|
46
|
+
``argus.openai_runner._MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
46
47
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
47
48
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
48
49
|
task's blast radius contained to this file plus ``argus.bench``'s
|
|
@@ -13,9 +13,9 @@ gpt-5.5 was NOT on the approved model list (see
|
|
|
13
13
|
gpt-5.4, with a regression-guard test
|
|
14
14
|
(``test_gpt_frontier_pinned_to_approved_model`` in
|
|
15
15
|
``tests/test_llm_models.py``) added specifically to catch a repeat. The
|
|
16
|
-
alias has since been bumped again, this time to ``gpt-
|
|
17
|
-
the gpt-5.5 attempt, the ``gpt-
|
|
18
|
-
model list (the
|
|
16
|
+
alias has since been bumped again, this time to ``gpt-6.1-sol`` (TECH-7124) -- unlike
|
|
17
|
+
the gpt-5.5 attempt, the ``gpt-6.1-sol`` model genuinely IS on the approved
|
|
18
|
+
model list (the model is approved in
|
|
19
19
|
``pr-review-specialist-llm-patterns.md`` alongside this change), so this is
|
|
20
20
|
not a repeat of that mistake. Any future bump of this alias must likewise
|
|
21
21
|
confirm the target model is on the approved list -- and update the policy
|
|
@@ -48,11 +48,15 @@ Per-token model pricing is sourced centrally from ``argus.llm.pricing``
|
|
|
48
48
|
|
|
49
49
|
Tier semantics:
|
|
50
50
|
*frontier* -- best reasoning available in the family; slow / expensive.
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
51
|
+
Under Claude 5.5, ``claude-frontier`` and ``claude-opus``
|
|
52
|
+
intentionally share ``claude-opus-5-5``.
|
|
53
|
+
*opus* -- historically the next tier down from frontier -- strong reasoning
|
|
54
|
+
at roughly half frontier's per-token cost (e.g. Opus 5 vs Fable 5).
|
|
55
|
+
Under Claude 5.5, ``claude-frontier`` and ``claude-opus``
|
|
56
|
+
intentionally share ``claude-opus-5-5``, so the prior cost
|
|
57
|
+
differentiation rationale no longer applies while preserving
|
|
58
|
+
the distinct alias keys for call sites (e.g. the cross-cutting
|
|
59
|
+
reviewer) and override compatibility.
|
|
56
60
|
*default* -- workhorse balance of cost and capability.
|
|
57
61
|
*mini* -- fast and cheap; suitable for high-volume, low-stakes calls.
|
|
58
62
|
|
|
@@ -63,9 +67,10 @@ Runtime overrides:
|
|
|
63
67
|
|
|
64
68
|
``ARGUS_FRONTIER_MODEL`` (``--frontier-model``) overrides both
|
|
65
69
|
``CLAUDE_FRONTIER`` (planner, coverage) and ``CLAUDE_OPUS``
|
|
66
|
-
(cross-cutting) -- there is a single frontier-tier knob at the CLI
|
|
67
|
-
|
|
68
|
-
|
|
70
|
+
(cross-cutting) -- there is a single frontier-tier knob at the CLI.
|
|
71
|
+
Under Claude 5.5, both aliases intentionally share the same default
|
|
72
|
+
(``claude-opus-5-5``), and setting ``ARGUS_FRONTIER_MODEL`` overrides both
|
|
73
|
+
onto the specified model.
|
|
69
74
|
|
|
70
75
|
Both env vars must be set before this module is first imported --
|
|
71
76
|
``cli.py`` sets them from the CLI flags ahead of any deferred import of
|
|
@@ -82,21 +87,18 @@ from argus.llm.pricing import estimate_cost_usd
|
|
|
82
87
|
logger = logging.getLogger(__name__)
|
|
83
88
|
|
|
84
89
|
ALIAS_MAP: Final[dict[str, str]] = {
|
|
85
|
-
# OpenAI -- gpt-5.4 / gpt-5.6 families
|
|
86
|
-
# gpt-frontier is bumped to gpt-
|
|
87
|
-
# gpt-5.6-luna, as
|
|
88
|
-
# (see pr-review-specialist-llm-patterns.md)
|
|
89
|
-
# updated to gpt-5.6-luna in the same change. gpt-5.5/gpt-5.5-mini
|
|
90
|
-
# remain NOT on the approved list -- do not bump either alias to them
|
|
91
|
-
# until the policy table is updated (see the module docstring above).
|
|
90
|
+
# OpenAI -- gpt-5.4 / gpt-5.6 / gpt-6 families
|
|
91
|
+
# gpt-frontier is bumped to gpt-6.1-sol (TECH-7124), and gpt-mini is bumped to
|
|
92
|
+
# gpt-5.6-luna, as both are on the approved model list
|
|
93
|
+
# (see pr-review-specialist-llm-patterns.md).
|
|
92
94
|
# test_llm_models.py pins both of these exact values as a regression
|
|
93
95
|
# guard.
|
|
94
|
-
"gpt-frontier": "gpt-
|
|
96
|
+
"gpt-frontier": "gpt-6.1-sol",
|
|
95
97
|
"gpt-mini": "gpt-5.6-luna",
|
|
96
98
|
# Anthropic
|
|
97
|
-
"claude-frontier": "claude-
|
|
98
|
-
"claude-opus": "claude-opus-5",
|
|
99
|
-
"claude-default": "claude-sonnet-
|
|
99
|
+
"claude-frontier": "claude-opus-5-5",
|
|
100
|
+
"claude-opus": "claude-opus-5-5",
|
|
101
|
+
"claude-default": "claude-sonnet-5-5",
|
|
100
102
|
"claude-mini": "claude-haiku-4-5",
|
|
101
103
|
# Google -- gemini-3 family. Real call site: argus.gemini_runner
|
|
102
104
|
# (Track 3), dispatched via argus.bench's "gemini" platform.
|
|
@@ -40,7 +40,7 @@ the model requests each turn (there can be more than one), chaining the
|
|
|
40
40
|
conversation state forward via ``previous_response_id`` and feeding back
|
|
41
41
|
all tool outputs as ``function_call_output`` items -- until the model stops
|
|
42
42
|
requesting function calls, calls ``finish_review``, or the turn budget
|
|
43
|
-
(``
|
|
43
|
+
(``_MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
44
44
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
45
45
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
46
46
|
task's blast radius contained, the final ``SessionResult.result_text`` is
|
|
@@ -110,7 +110,6 @@ from argus.bench import BenchEntry
|
|
|
110
110
|
from argus.llm.models import estimate_cost_usd
|
|
111
111
|
from argus.llm.models import resolve as resolve_model_alias
|
|
112
112
|
from argus.runners import (
|
|
113
|
-
_MAX_TURNS,
|
|
114
113
|
_MID_BUDGET_NUDGE,
|
|
115
114
|
_NUDGE_TURNS_BEFORE_BUDGET,
|
|
116
115
|
_SUBPROCESS_TIMEOUT_S,
|
|
@@ -123,6 +122,8 @@ from argus.runners import (
|
|
|
123
122
|
|
|
124
123
|
logger = logging.getLogger(__name__)
|
|
125
124
|
|
|
125
|
+
_MAX_TURNS_OPENAI = 30
|
|
126
|
+
|
|
126
127
|
_DEFAULT_READ_LIMIT = 2000 # mirrors argus.review_tools._DEFAULT_READ_LIMIT
|
|
127
128
|
|
|
128
129
|
# Matches the fenced-json extraction argus.helpers.parse_review_result
|
|
@@ -417,7 +418,7 @@ async def _run_turns(
|
|
|
417
418
|
whatever those completed turns actually produced and billed.
|
|
418
419
|
"""
|
|
419
420
|
system_prompt = (
|
|
420
|
-
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=
|
|
421
|
+
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_OPENAI)
|
|
421
422
|
)
|
|
422
423
|
api_key = getattr(settings, "OPENAI_API_KEY", None)
|
|
423
424
|
base_url = getattr(settings, "OPENAI_BASE_URL", None)
|
|
@@ -504,13 +505,13 @@ async def _run_turns(
|
|
|
504
505
|
)
|
|
505
506
|
|
|
506
507
|
exhausted = False
|
|
507
|
-
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(
|
|
508
|
+
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_OPENAI)
|
|
508
509
|
try:
|
|
509
510
|
async with asyncio.timeout(timeout_s):
|
|
510
511
|
with review_tools.review_session(repo_root) as findings_sink:
|
|
511
512
|
previous_response_id: str | None = None
|
|
512
513
|
tool_outputs: list[dict[str, Any]] = []
|
|
513
|
-
for _turn in range(
|
|
514
|
+
for _turn in range(_MAX_TURNS_OPENAI):
|
|
514
515
|
create_kwargs: dict[str, Any] = {
|
|
515
516
|
"model": model,
|
|
516
517
|
"instructions": system_prompt,
|
|
@@ -625,7 +626,7 @@ async def _run_turns(
|
|
|
625
626
|
if finished:
|
|
626
627
|
break
|
|
627
628
|
|
|
628
|
-
if _turn + 1 ==
|
|
629
|
+
if _turn + 1 == _MAX_TURNS_OPENAI - _NUDGE_TURNS_BEFORE_BUDGET:
|
|
629
630
|
tool_outputs.append({"role": "user", "content": _TURN_BUDGET_NUDGE})
|
|
630
631
|
|
|
631
632
|
# Not an `elif` -- these are two independent checkpoints
|
|
@@ -643,7 +644,7 @@ async def _run_turns(
|
|
|
643
644
|
logger.warning(
|
|
644
645
|
"OpenAI session [%s] exhausted its %d-turn budget without a finish_review call",
|
|
645
646
|
label or "unlabeled",
|
|
646
|
-
|
|
647
|
+
_MAX_TURNS_OPENAI,
|
|
647
648
|
)
|
|
648
649
|
exhausted = True
|
|
649
650
|
|
|
@@ -7,8 +7,8 @@ Only approved model families. Check model strings in code against this table:
|
|
|
7
7
|
|
|
8
8
|
| Provider | Approved | Default |
|
|
9
9
|
|----------|----------|---------|
|
|
10
|
-
| Anthropic | claude-opus-4, claude-sonnet-4, claude-haiku-4 (use `CLAUDE_MINI`), claude-
|
|
11
|
-
| OpenAI | gpt-5.4 family (gpt-5.4, gpt-5.4-mini)
|
|
10
|
+
| Anthropic | claude-opus-4, claude-sonnet-4, claude-haiku-4 (use `CLAUDE_MINI`), claude-opus-5-5 (use `CLAUDE_FRONTIER` / `CLAUDE_OPUS`), claude-sonnet-5-5 (use `CLAUDE_DEFAULT`) families; legacy Claude models (claude-fable-5, claude-opus-5, claude-sonnet-4-6, claude-sonnet-5) remain valid only as explicit CLI/env runtime overrides (`ARGUS_SPECIALIST_MODEL`/`ARGUS_FRONTIER_MODEL` or `--specialist-model`/`--frontier-model`) for backward compatibility, not as new hardcoded defaults | claude-sonnet-5-5 |
|
|
11
|
+
| OpenAI | gpt-5.4 family (gpt-5.4, gpt-5.4-mini), gpt-5.6 family, and gpt-6.1-sol (use `GPT_FRONTIER`) | gpt-5.6-luna |
|
|
12
12
|
| Google | gemini-3 family | gemini-3.8-flash |
|
|
13
13
|
|
|
14
14
|
Flag any use of: o3, o1, gpt-4 family, gpt-5-mini, claude-3/3.5 family, gemini-1.5/2.5 family.
|