argus-code-review 0.2.8__tar.gz → 0.2.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/CHANGELOG.md +14 -1
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/PKG-INFO +3 -4
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/README.md +2 -3
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/bench_default.toml +1 -3
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/cli.py +1 -4
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/config.py +1 -1
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/llm/models.py +24 -22
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-llm-patterns.md +2 -2
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/runners.py +26 -20
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_bench.py +1 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_llm_models.py +50 -16
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_llm_models_override.py +14 -2
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_runners_context_usage.py +2 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/.gitignore +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/LICENSE +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/bench.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/coverage.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/gemini_runner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/github_client.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/graph.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/helpers.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/llm/output_models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/llm/usage.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/openai_runner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/pipeline_models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/engine.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/sql.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/storage/sqlite.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/pyproject.toml +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/018_widen_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/conftest.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/golden/review_response.schema.json +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_backend_contract.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_http.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_sql.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/storage/test_sqlite_backend.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_argus_review_local.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_bench_config_guard.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_cli_preflight.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_config.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_gemini_runner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_fetch_diff.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_get_llm_temperature.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_precheck.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_preflight_image_bump.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_timeout_surfacing.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_multi_round.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_openai_client.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_openai_runner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_output_models.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_packaging.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_engine.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_runners_helpers.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_runners_new.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_runners_turn_budget.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_stage_costs.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,18 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.9] - 2026-10-02
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Upgraded frontier and default model aliases in `argus.llm.models` (TECH-7124, #31):
|
|
15
|
+
- `claude-frontier` and `claude-opus` upgraded to `claude-opus-5-5` (previously `claude-fable-5` and `claude-opus-5`).
|
|
16
|
+
- `claude-default` upgraded to `claude-sonnet-5-5` (previously `claude-sonnet-4-6`).
|
|
17
|
+
- `gpt-frontier` upgraded to `gpt-6.1-sol` (previously `gpt-5.6-sol`).
|
|
18
|
+
- Updated model pricing rates in `argus.llm.models` for `claude-sonnet-5-5`, `claude-opus-5-5`, and `gpt-6.1-sol` (TECH-7124, #31).
|
|
19
|
+
- Re-verified the Anthropic 1M context beta (`context-1m-2025-08-07`) against `claude-sonnet-5-5` via the Redesign Health Anthropic proxy (TECH-7124, #31).
|
|
20
|
+
- Updated approved model policies in `pr-review-specialist-llm-patterns.md`, runner warning logs, bench defaults, and CLI help text for the new model aliases (TECH-7124, #31).
|
|
21
|
+
|
|
10
22
|
## [0.2.8] - 2026-10-02
|
|
11
23
|
|
|
12
24
|
### Changed
|
|
@@ -368,7 +380,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
368
380
|
packaged set.
|
|
369
381
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
370
382
|
|
|
371
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
383
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.9...HEAD
|
|
384
|
+
[0.2.9]: https://github.com/redesignhealth/argus-review/compare/v0.2.8...v0.2.9
|
|
372
385
|
[0.2.8]: https://github.com/redesignhealth/argus-review/compare/v0.2.7...v0.2.8
|
|
373
386
|
[0.2.7]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...v0.2.7
|
|
374
387
|
[0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.9
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -204,8 +204,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
204
204
|
|
|
205
205
|
# Override the reviewer models (or set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL instead).
|
|
206
206
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
207
|
-
# reviewer
|
|
208
|
-
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
207
|
+
# reviewer (defaults to claude-opus-5-5).
|
|
209
208
|
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
210
209
|
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
211
210
|
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
@@ -214,7 +213,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
214
213
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
215
214
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
216
215
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
217
|
-
argus review owner/repo --pr 123 --specialist-model claude-opus-5 --frontier-model claude-
|
|
216
|
+
argus review owner/repo --pr 123 --specialist-model claude-opus-5-5 --frontier-model claude-opus-5-5
|
|
218
217
|
|
|
219
218
|
# Clear an already-set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL for just this
|
|
220
219
|
# run by passing an empty string -- useful when the env var is set globally
|
|
@@ -153,8 +153,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
153
153
|
|
|
154
154
|
# Override the reviewer models (or set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL instead).
|
|
155
155
|
# --frontier-model controls both the planner/coverage tier AND the cross-cutting
|
|
156
|
-
# reviewer
|
|
157
|
-
# it also moves cross-cutting OFF its cheaper claude-opus-5 default onto fable-5.
|
|
156
|
+
# reviewer (defaults to claude-opus-5-5).
|
|
158
157
|
# Cost note: --specialist-model here overrides the bulk reviewer path (system
|
|
159
158
|
# reviewer, specialists, tests-and-docs) from its low-cost Gemini default onto
|
|
160
159
|
# claude-sdk with the specified model, and also updates the writer and lite-review
|
|
@@ -163,7 +162,7 @@ argus review owner/repo --pr 123 --dismiss "B2 -- pre-existing, not from this PR
|
|
|
163
162
|
# the 1M-context beta from that same highest-volume path, since the beta is
|
|
164
163
|
# only verified against the unoverridden default (autocompact may thrash on
|
|
165
164
|
# long reviews under this override -- see argus/runners.py for the tradeoff).
|
|
166
|
-
argus review owner/repo --pr 123 --specialist-model claude-opus-5 --frontier-model claude-
|
|
165
|
+
argus review owner/repo --pr 123 --specialist-model claude-opus-5-5 --frontier-model claude-opus-5-5
|
|
167
166
|
|
|
168
167
|
# Clear an already-set ARGUS_SPECIALIST_MODEL/ARGUS_FRONTIER_MODEL for just this
|
|
169
168
|
# run by passing an empty string -- useful when the env var is set globally
|
|
@@ -33,9 +33,7 @@ caching = "auto"
|
|
|
33
33
|
|
|
34
34
|
[roles.cross-cutting]
|
|
35
35
|
platform = "claude-sdk"
|
|
36
|
-
#
|
|
37
|
-
# evals showed no measurable quality gain from frontier on this stage, at
|
|
38
|
-
# ~2x the per-token cost.
|
|
36
|
+
# Matches argus.runners._CROSS_CUTTING_MODEL (resolves to claude-opus-5-5).
|
|
39
37
|
model = "claude-opus"
|
|
40
38
|
prompt_name = "pr-review-cross-cutting"
|
|
41
39
|
caching = "auto"
|
|
@@ -368,10 +368,7 @@ def _add_review_args(parser: argparse.ArgumentParser) -> None:
|
|
|
368
368
|
default=_MODEL_OVERRIDE_UNSET,
|
|
369
369
|
help=(
|
|
370
370
|
"Override the model used by the planner, coverage check, and "
|
|
371
|
-
"cross-cutting reviewer (
|
|
372
|
-
"two, claude-opus-5 for cross-cutting -- note this one flag "
|
|
373
|
-
"collapses both onto the SAME model when set, moving "
|
|
374
|
-
"cross-cutting off its cheaper default; or ARGUS_FRONTIER_MODEL "
|
|
371
|
+
"cross-cutting reviewer (default: claude-opus-5-5, or ARGUS_FRONTIER_MODEL "
|
|
375
372
|
"if already set in the environment). Same effect as setting "
|
|
376
373
|
"ARGUS_FRONTIER_MODEL. Pass an empty string to clear an "
|
|
377
374
|
"already-set ARGUS_FRONTIER_MODEL for this run."
|
|
@@ -91,7 +91,7 @@ class Settings(BaseSettings):
|
|
|
91
91
|
ARGUS_SPECIALIST_MODEL: Override the model used by the system
|
|
92
92
|
reviewer, specialist reviewers, the writer, and the lite-review
|
|
93
93
|
path (``argus.llm.models.CLAUDE_DEFAULT``, default
|
|
94
|
-
``claude-sonnet-
|
|
94
|
+
``claude-sonnet-5-5``). Set via ``--specialist-model``; read
|
|
95
95
|
directly from ``os.environ`` by ``argus.llm.models`` at import
|
|
96
96
|
time and by ``argus.bench`` (where setting it forces bulk reviewers
|
|
97
97
|
to claude-sdk with claude-default). Also read off a ``Settings``
|
|
@@ -13,9 +13,9 @@ gpt-5.5 was NOT on the approved model list (see
|
|
|
13
13
|
gpt-5.4, with a regression-guard test
|
|
14
14
|
(``test_gpt_frontier_pinned_to_approved_model`` in
|
|
15
15
|
``tests/test_llm_models.py``) added specifically to catch a repeat. The
|
|
16
|
-
alias has since been bumped again, this time to ``gpt-
|
|
17
|
-
the gpt-5.5 attempt, the ``gpt-
|
|
18
|
-
model list (the
|
|
16
|
+
alias has since been bumped again, this time to ``gpt-6.1-sol`` (TECH-7124) -- unlike
|
|
17
|
+
the gpt-5.5 attempt, the ``gpt-6.1-sol`` model genuinely IS on the approved
|
|
18
|
+
model list (the model is approved in
|
|
19
19
|
``pr-review-specialist-llm-patterns.md`` alongside this change), so this is
|
|
20
20
|
not a repeat of that mistake. Any future bump of this alias must likewise
|
|
21
21
|
confirm the target model is on the approved list -- and update the policy
|
|
@@ -48,11 +48,15 @@ Per-token model pricing is sourced centrally from ``argus.llm.pricing``
|
|
|
48
48
|
|
|
49
49
|
Tier semantics:
|
|
50
50
|
*frontier* -- best reasoning available in the family; slow / expensive.
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
51
|
+
Under Claude 5.5, ``claude-frontier`` and ``claude-opus``
|
|
52
|
+
intentionally share ``claude-opus-5-5``.
|
|
53
|
+
*opus* -- historically the next tier down from frontier -- strong reasoning
|
|
54
|
+
at roughly half frontier's per-token cost (e.g. Opus 5 vs Fable 5).
|
|
55
|
+
Under Claude 5.5, ``claude-frontier`` and ``claude-opus``
|
|
56
|
+
intentionally share ``claude-opus-5-5``, so the prior cost
|
|
57
|
+
differentiation rationale no longer applies while preserving
|
|
58
|
+
the distinct alias keys for call sites (e.g. the cross-cutting
|
|
59
|
+
reviewer) and override compatibility.
|
|
56
60
|
*default* -- workhorse balance of cost and capability.
|
|
57
61
|
*mini* -- fast and cheap; suitable for high-volume, low-stakes calls.
|
|
58
62
|
|
|
@@ -63,9 +67,10 @@ Runtime overrides:
|
|
|
63
67
|
|
|
64
68
|
``ARGUS_FRONTIER_MODEL`` (``--frontier-model``) overrides both
|
|
65
69
|
``CLAUDE_FRONTIER`` (planner, coverage) and ``CLAUDE_OPUS``
|
|
66
|
-
(cross-cutting) -- there is a single frontier-tier knob at the CLI
|
|
67
|
-
|
|
68
|
-
|
|
70
|
+
(cross-cutting) -- there is a single frontier-tier knob at the CLI.
|
|
71
|
+
Under Claude 5.5, both aliases intentionally share the same default
|
|
72
|
+
(``claude-opus-5-5``), and setting ``ARGUS_FRONTIER_MODEL`` overrides both
|
|
73
|
+
onto the specified model.
|
|
69
74
|
|
|
70
75
|
Both env vars must be set before this module is first imported --
|
|
71
76
|
``cli.py`` sets them from the CLI flags ahead of any deferred import of
|
|
@@ -82,21 +87,18 @@ from argus.llm.pricing import estimate_cost_usd
|
|
|
82
87
|
logger = logging.getLogger(__name__)
|
|
83
88
|
|
|
84
89
|
ALIAS_MAP: Final[dict[str, str]] = {
|
|
85
|
-
# OpenAI -- gpt-5.4 / gpt-5.6 families
|
|
86
|
-
# gpt-frontier is bumped to gpt-
|
|
87
|
-
# gpt-5.6-luna, as
|
|
88
|
-
# (see pr-review-specialist-llm-patterns.md)
|
|
89
|
-
# updated to gpt-5.6-luna in the same change. gpt-5.5/gpt-5.5-mini
|
|
90
|
-
# remain NOT on the approved list -- do not bump either alias to them
|
|
91
|
-
# until the policy table is updated (see the module docstring above).
|
|
90
|
+
# OpenAI -- gpt-5.4 / gpt-5.6 / gpt-6 families
|
|
91
|
+
# gpt-frontier is bumped to gpt-6.1-sol (TECH-7124), and gpt-mini is bumped to
|
|
92
|
+
# gpt-5.6-luna, as both are on the approved model list
|
|
93
|
+
# (see pr-review-specialist-llm-patterns.md).
|
|
92
94
|
# test_llm_models.py pins both of these exact values as a regression
|
|
93
95
|
# guard.
|
|
94
|
-
"gpt-frontier": "gpt-
|
|
96
|
+
"gpt-frontier": "gpt-6.1-sol",
|
|
95
97
|
"gpt-mini": "gpt-5.6-luna",
|
|
96
98
|
# Anthropic
|
|
97
|
-
"claude-frontier": "claude-
|
|
98
|
-
"claude-opus": "claude-opus-5",
|
|
99
|
-
"claude-default": "claude-sonnet-
|
|
99
|
+
"claude-frontier": "claude-opus-5-5",
|
|
100
|
+
"claude-opus": "claude-opus-5-5",
|
|
101
|
+
"claude-default": "claude-sonnet-5-5",
|
|
100
102
|
"claude-mini": "claude-haiku-4-5",
|
|
101
103
|
# Google -- gemini-3 family. Real call site: argus.gemini_runner
|
|
102
104
|
# (Track 3), dispatched via argus.bench's "gemini" platform.
|
|
@@ -7,8 +7,8 @@ Only approved model families. Check model strings in code against this table:
|
|
|
7
7
|
|
|
8
8
|
| Provider | Approved | Default |
|
|
9
9
|
|----------|----------|---------|
|
|
10
|
-
| Anthropic | claude-opus-4, claude-sonnet-4, claude-haiku-4 (use `CLAUDE_MINI`), claude-
|
|
11
|
-
| OpenAI | gpt-5.4 family (gpt-5.4, gpt-5.4-mini)
|
|
10
|
+
| Anthropic | claude-opus-4, claude-sonnet-4, claude-haiku-4 (use `CLAUDE_MINI`), claude-opus-5-5 (use `CLAUDE_FRONTIER` / `CLAUDE_OPUS`), claude-sonnet-5-5 (use `CLAUDE_DEFAULT`) families; legacy Claude models (claude-fable-5, claude-opus-5, claude-sonnet-4-6, claude-sonnet-5) remain valid only as explicit CLI/env runtime overrides (`ARGUS_SPECIALIST_MODEL`/`ARGUS_FRONTIER_MODEL` or `--specialist-model`/`--frontier-model`) for backward compatibility, not as new hardcoded defaults | claude-sonnet-5-5 |
|
|
11
|
+
| OpenAI | gpt-5.4 family (gpt-5.4, gpt-5.4-mini), gpt-5.6 family, and gpt-6.1-sol (use `GPT_FRONTIER`) | gpt-5.6-luna |
|
|
12
12
|
| Google | gemini-3 family | gemini-3.8-flash |
|
|
13
13
|
|
|
14
14
|
Flag any use of: o3, o1, gpt-4 family, gpt-5-mini, claude-3/3.5 family, gemini-1.5/2.5 family.
|
|
@@ -85,8 +85,7 @@ logger = logging.getLogger(__name__)
|
|
|
85
85
|
# ---------------------------------------------------------------------------
|
|
86
86
|
|
|
87
87
|
_SYSTEM_REVIEWER_MODEL = CLAUDE_DEFAULT
|
|
88
|
-
#
|
|
89
|
-
# frontier on this stage, at ~2x the per-token cost.
|
|
88
|
+
# Matches argus.bench_default.toml (resolves to claude-opus-5-5).
|
|
90
89
|
_CROSS_CUTTING_MODEL = CLAUDE_OPUS
|
|
91
90
|
# Whether the system reviewer's current model (following any
|
|
92
91
|
# --specialist-model/ARGUS_SPECIALIST_MODEL override) still matches
|
|
@@ -97,19 +96,21 @@ _CROSS_CUTTING_MODEL = CLAUDE_OPUS
|
|
|
97
96
|
# for why a string-equality gate can't be made sound once overrides exist.
|
|
98
97
|
#
|
|
99
98
|
# Caveat shared with argus/graph.py's _TEMPERATURE_UNSUPPORTED_MODELS
|
|
100
|
-
# comment: the empirical verification below was run against
|
|
101
|
-
# ALIAS_MAP["claude-default"]
|
|
102
|
-
#
|
|
103
|
-
#
|
|
104
|
-
#
|
|
105
|
-
#
|
|
106
|
-
# the
|
|
107
|
-
#
|
|
108
|
-
#
|
|
109
|
-
#
|
|
110
|
-
#
|
|
111
|
-
#
|
|
112
|
-
#
|
|
99
|
+
# comment: the empirical verification below was initially run against
|
|
100
|
+
# ALIAS_MAP["claude-default"] on 2026-07-31 (claude-sonnet-5 at the time)
|
|
101
|
+
# and was live re-verified for claude-sonnet-5-5 on 2026-10-02 via the RH
|
|
102
|
+
# Anthropic proxy: POST /v1/messages with model claude-sonnet-5-5, header
|
|
103
|
+
# `anthropic-beta: context-1m-2025-08-07`, max_tokens=16 returned HTTP 200,
|
|
104
|
+
# message model claude-sonnet-5-5, stop_reason=end_turn, no error. Gating
|
|
105
|
+
# on the alias means the beta keeps applying across future pin bumps
|
|
106
|
+
# automatically, without a fresh re-verification each time -- the
|
|
107
|
+
# alternative (pinning this gate to a literal model string instead)
|
|
108
|
+
# would have silently withheld the beta from the system reviewer's new
|
|
109
|
+
# default entirely, reinstating the TECH-4734 autocompact-thrashing problem
|
|
110
|
+
# for the common no-override case. Tracking the alias was the deliberate
|
|
111
|
+
# tradeoff; re-verify the beta empirically against the current pin whenever
|
|
112
|
+
# ALIAS_MAP["claude-default"] moves, and correct this comment if a future
|
|
113
|
+
# pin ever fails the probe described below.
|
|
113
114
|
_SYSTEM_REVIEWER_UNOVERRIDDEN = _SYSTEM_REVIEWER_MODEL == ALIAS_MAP["claude-default"]
|
|
114
115
|
|
|
115
116
|
# Logged once at import time, not per-session (an earlier per-call version
|
|
@@ -132,19 +133,19 @@ else:
|
|
|
132
133
|
_SYSTEM_REVIEWER_MODEL,
|
|
133
134
|
)
|
|
134
135
|
|
|
135
|
-
# Same one-time-at-import treatment for the
|
|
136
|
-
# override mechanism can cause: --frontier-model/ARGUS_FRONTIER_MODEL
|
|
136
|
+
# Same one-time-at-import treatment for the model/default identity change
|
|
137
|
+
# this override mechanism can cause: --frontier-model/ARGUS_FRONTIER_MODEL
|
|
137
138
|
# repoints both CLAUDE_FRONTIER and CLAUDE_OPUS (see argus/llm/models.py),
|
|
138
139
|
# so a frontier override picked for planning/coverage purposes also moves
|
|
139
|
-
# the cross-cutting reviewer off its
|
|
140
|
-
# runtime signal that happened.
|
|
140
|
+
# the cross-cutting reviewer off its separately named claude-opus
|
|
141
|
+
# default with no other runtime signal that happened.
|
|
141
142
|
if _CROSS_CUTTING_MODEL != ALIAS_MAP["claude-opus"]:
|
|
142
143
|
logger.warning(
|
|
143
144
|
"Cross-cutting reviewer moved off its default model %r onto %r due to "
|
|
144
145
|
"ARGUS_FRONTIER_MODEL -- this env var/--frontier-model repoints both "
|
|
145
146
|
"the frontier tier and the cross-cutting model together, so a "
|
|
146
147
|
"frontier override for planning purposes also moves cross-cutting "
|
|
147
|
-
"off its
|
|
148
|
+
"off its separately named claude-opus alias.",
|
|
148
149
|
ALIAS_MAP["claude-opus"],
|
|
149
150
|
_CROSS_CUTTING_MODEL,
|
|
150
151
|
)
|
|
@@ -1781,6 +1782,11 @@ async def _run_claude_session(
|
|
|
1781
1782
|
# proxy's own header-allowlist (which lives in a different repo, rh-mcp's
|
|
1782
1783
|
# main.py, and could drift independently of this comment).
|
|
1783
1784
|
#
|
|
1785
|
+
# Live re-verification on 2026-10-02 via RH Anthropic proxy: POST /v1/messages
|
|
1786
|
+
# with model claude-sonnet-5-5, header `anthropic-beta: context-1m-2025-08-07`,
|
|
1787
|
+
# max_tokens=16 returned HTTP 200, message model claude-sonnet-5-5,
|
|
1788
|
+
# stop_reason=end_turn, no error.
|
|
1789
|
+
#
|
|
1784
1790
|
# Only applied for _SYSTEM_REVIEWER_MODEL (sonnet), not
|
|
1785
1791
|
# _CROSS_CUTTING_MODEL (opus): sonnet reviewer sessions are the ones
|
|
1786
1792
|
# observed thrashing at the ~155-180k fixed-prefix floor (TECH-4734
|
|
@@ -107,6 +107,7 @@ class TestPackagedDefaultResolutions:
|
|
|
107
107
|
def test_cross_cutting_matches_packaged_default(self) -> None:
|
|
108
108
|
entry = bench.resolve("cross-cutting")
|
|
109
109
|
assert entry.platform == "claude-sdk"
|
|
110
|
+
assert entry.model == "claude-opus"
|
|
110
111
|
assert resolve_alias(entry.model) == runners_module._CROSS_CUTTING_MODEL
|
|
111
112
|
assert entry.prompt_name == "pr-review-cross-cutting"
|
|
112
113
|
|
|
@@ -11,7 +11,6 @@ import pytest
|
|
|
11
11
|
|
|
12
12
|
from argus.llm.models import (
|
|
13
13
|
ALIAS_MAP,
|
|
14
|
-
CLAUDE_DEFAULT,
|
|
15
14
|
EXPERIMENTAL_MODELS,
|
|
16
15
|
GEMINI_FRONTIER,
|
|
17
16
|
GEMINI_MINI,
|
|
@@ -49,13 +48,31 @@ class TestPricingLookup:
|
|
|
49
48
|
)
|
|
50
49
|
|
|
51
50
|
def test_claude_default_pricing_rates(self) -> None:
|
|
52
|
-
"""Regression guard: claude-sonnet-
|
|
53
|
-
expected rates ($
|
|
54
|
-
cost = get_token_cost(
|
|
51
|
+
"""Regression guard: claude-sonnet-5-5 rates in litellm must match
|
|
52
|
+
expected rates ($2/$10/$0.20 per Mtok)."""
|
|
53
|
+
cost = get_token_cost(ALIAS_MAP["claude-default"])
|
|
55
54
|
assert cost is not None
|
|
56
|
-
assert cost.input_cost_per_token == pytest.approx(
|
|
57
|
-
assert cost.output_cost_per_token == pytest.approx(
|
|
58
|
-
assert cost.cache_read_cost_per_token == pytest.approx(0.
|
|
55
|
+
assert cost.input_cost_per_token == pytest.approx(2e-6)
|
|
56
|
+
assert cost.output_cost_per_token == pytest.approx(10e-6)
|
|
57
|
+
assert cost.cache_read_cost_per_token == pytest.approx(0.2e-6)
|
|
58
|
+
|
|
59
|
+
def test_claude_opus_5_5_pricing_rates(self) -> None:
|
|
60
|
+
"""Regression guard: claude-opus-5-5 rates in litellm must match
|
|
61
|
+
expected rates ($4/$20/$0.20 per Mtok)."""
|
|
62
|
+
cost = get_token_cost(ALIAS_MAP["claude-opus"])
|
|
63
|
+
assert cost is not None
|
|
64
|
+
assert cost.input_cost_per_token == pytest.approx(4e-6)
|
|
65
|
+
assert cost.output_cost_per_token == pytest.approx(20e-6)
|
|
66
|
+
assert cost.cache_read_cost_per_token == pytest.approx(0.2e-6)
|
|
67
|
+
|
|
68
|
+
def test_gpt_frontier_pricing_rates(self) -> None:
|
|
69
|
+
"""Regression guard: gpt-6.1-sol rates in litellm must match
|
|
70
|
+
expected rates ($2/$10/$0.10 per Mtok)."""
|
|
71
|
+
cost = get_token_cost(ALIAS_MAP["gpt-frontier"])
|
|
72
|
+
assert cost is not None
|
|
73
|
+
assert cost.input_cost_per_token == pytest.approx(2e-6)
|
|
74
|
+
assert cost.output_cost_per_token == pytest.approx(10e-6)
|
|
75
|
+
assert cost.cache_read_cost_per_token == pytest.approx(0.1e-6)
|
|
59
76
|
|
|
60
77
|
def test_gemini_pricing_is_real_not_a_placeholder(self) -> None:
|
|
61
78
|
"""Gemini has a real, reachable runner (argus.gemini_runner) -- its
|
|
@@ -70,19 +87,21 @@ class TestPricingLookup:
|
|
|
70
87
|
|
|
71
88
|
class TestEstimateCostUsd:
|
|
72
89
|
def test_computes_expected_cost_for_claude_default(self) -> None:
|
|
73
|
-
cost = estimate_cost_usd(
|
|
74
|
-
|
|
90
|
+
cost = estimate_cost_usd(
|
|
91
|
+
ALIAS_MAP["claude-default"], input_tokens=1_000_000, output_tokens=1_000_000
|
|
92
|
+
)
|
|
93
|
+
assert cost == pytest.approx(2.00 + 10.00)
|
|
75
94
|
|
|
76
95
|
def test_cached_input_tokens_billed_separately_at_cache_read_rate(self) -> None:
|
|
77
96
|
"""cached_input_tokens is additive (billed at the cache-read rate),
|
|
78
97
|
not a subset subtracted from input_tokens."""
|
|
79
98
|
cost = estimate_cost_usd(
|
|
80
|
-
|
|
99
|
+
ALIAS_MAP["claude-default"],
|
|
81
100
|
input_tokens=1_000_000,
|
|
82
101
|
output_tokens=0,
|
|
83
102
|
cached_input_tokens=1_000_000,
|
|
84
103
|
)
|
|
85
|
-
assert cost == pytest.approx(
|
|
104
|
+
assert cost == pytest.approx(2.00 + 0.20)
|
|
86
105
|
|
|
87
106
|
def test_cache_creation_tokens_billed(self) -> None:
|
|
88
107
|
"""cache_creation_tokens is billed at the cache-creation rate."""
|
|
@@ -92,7 +111,7 @@ class TestEstimateCostUsd:
|
|
|
92
111
|
output_tokens=0,
|
|
93
112
|
cache_creation_tokens=1_000_000,
|
|
94
113
|
)
|
|
95
|
-
assert cost == pytest.approx(
|
|
114
|
+
assert cost == pytest.approx(2.50)
|
|
96
115
|
|
|
97
116
|
def test_zero_tokens_is_zero_cost(self) -> None:
|
|
98
117
|
assert (
|
|
@@ -136,13 +155,28 @@ class TestEstimateCostUsd:
|
|
|
136
155
|
"""Regression guard for a round-1 Argus BLOCKING finding on this
|
|
137
156
|
PR: gpt-frontier previously resolved to gpt-5.5, which was not on
|
|
138
157
|
the approved model list and may not have been shipped by OpenAI
|
|
139
|
-
yet. The alias has since been legitimately re-bumped to gpt-
|
|
140
|
-
(the
|
|
158
|
+
yet. The alias has since been legitimately re-bumped to gpt-6.1-sol
|
|
159
|
+
(the gpt-6.1-sol model is on the approved model list -- see
|
|
141
160
|
pr-review-specialist-llm-patterns.md). Pin the alias to that
|
|
142
161
|
approved value so a future accidental re-bump to an unapproved
|
|
143
162
|
model string is caught here instead of at review time."""
|
|
144
|
-
assert ALIAS_MAP["gpt-frontier"] == "gpt-
|
|
145
|
-
assert GPT_FRONTIER == "gpt-
|
|
163
|
+
assert ALIAS_MAP["gpt-frontier"] == "gpt-6.1-sol"
|
|
164
|
+
assert GPT_FRONTIER == "gpt-6.1-sol"
|
|
165
|
+
|
|
166
|
+
def test_claude_frontier_pinned_to_approved_model(self) -> None:
|
|
167
|
+
"""Regression guard: claude-frontier must pin to claude-opus-5-5
|
|
168
|
+
under Claude 5.5 per TECH-7124."""
|
|
169
|
+
assert ALIAS_MAP["claude-frontier"] == "claude-opus-5-5"
|
|
170
|
+
|
|
171
|
+
def test_claude_opus_pinned_to_approved_model(self) -> None:
|
|
172
|
+
"""Regression guard: claude-opus must pin to claude-opus-5-5
|
|
173
|
+
under Claude 5.5 per TECH-7124."""
|
|
174
|
+
assert ALIAS_MAP["claude-opus"] == "claude-opus-5-5"
|
|
175
|
+
|
|
176
|
+
def test_claude_default_pinned_to_approved_model(self) -> None:
|
|
177
|
+
"""Regression guard: claude-default must pin to claude-sonnet-5-5
|
|
178
|
+
per TECH-7124."""
|
|
179
|
+
assert ALIAS_MAP["claude-default"] == "claude-sonnet-5-5"
|
|
146
180
|
|
|
147
181
|
def test_gpt_mini_pinned_to_approved_model(self) -> None:
|
|
148
182
|
"""Parallel regression guard for gpt-mini, alongside gpt-frontier's
|
|
@@ -44,10 +44,22 @@ def _reload_after_test() -> Iterator[None]:
|
|
|
44
44
|
importlib.reload(models)
|
|
45
45
|
|
|
46
46
|
|
|
47
|
-
def
|
|
47
|
+
def test_claude_default_is_sonnet_5_5_when_unset(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
48
48
|
monkeypatch.delenv("ARGUS_SPECIALIST_MODEL", raising=False)
|
|
49
49
|
importlib.reload(models)
|
|
50
|
-
assert models.CLAUDE_DEFAULT == "claude-sonnet-
|
|
50
|
+
assert models.CLAUDE_DEFAULT == "claude-sonnet-5-5"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_claude_frontier_is_opus_5_5_when_unset(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
54
|
+
monkeypatch.delenv("ARGUS_FRONTIER_MODEL", raising=False)
|
|
55
|
+
importlib.reload(models)
|
|
56
|
+
assert models.CLAUDE_FRONTIER == "claude-opus-5-5"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_claude_opus_is_opus_5_5_when_unset(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
60
|
+
monkeypatch.delenv("ARGUS_FRONTIER_MODEL", raising=False)
|
|
61
|
+
importlib.reload(models)
|
|
62
|
+
assert models.CLAUDE_OPUS == "claude-opus-5-5"
|
|
51
63
|
|
|
52
64
|
|
|
53
65
|
def test_specialist_override_only_affects_claude_default(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
@@ -724,8 +724,10 @@ class TestOneTimeImportLogs:
|
|
|
724
724
|
importlib.reload(runners_module)
|
|
725
725
|
assert any(
|
|
726
726
|
"Cross-cutting reviewer moved off its default model" in record.message
|
|
727
|
+
and "off its separately named claude-opus alias" in record.message
|
|
727
728
|
for record in caplog.records
|
|
728
729
|
)
|
|
730
|
+
assert not any("cheaper Opus default" in record.message for record in caplog.records)
|
|
729
731
|
finally:
|
|
730
732
|
monkeypatch.delenv("ARGUS_FRONTIER_MODEL", raising=False)
|
|
731
733
|
importlib.reload(models_module)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/eslint.config.js
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package-lock.json
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/precheck/eslint_bundle/package.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-blocking-validator.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-coverage-check.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-cross-cutting.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-feedback-verifier.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-preflight-router.md
RENAMED
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-deployment.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-frontend.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-infra.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-security.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-slackbot.md
RENAMED
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-specialist-sql.md
RENAMED
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/argus/prompts/pr-review-tests-and-docs.md
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/011_add_review_progress_columns.sql
RENAMED
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/016_add_agent_runs_failure_reason.sql
RENAMED
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/schema/018_widen_agent_runs_failure_reason.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/golden/review_response.schema.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_github_client_checks_signal.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_graph_preflight_image_bump.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_engine_integration.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_precheck_shadow_integration.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_review_patterns_integration.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{argus_code_review-0.2.8 → argus_code_review-0.2.9}/tests/test_terraform_scanner_integration.py
RENAMED
|
File without changes
|
|
File without changes
|