argus-code-review 0.2.7__tar.gz → 0.2.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/CHANGELOG.md +14 -1
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/PKG-INFO +1 -1
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/config.py +16 -9
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/gemini_runner.py +3 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/openai_runner.py +8 -7
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/runners.py +182 -8
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_config.py +2 -2
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_openai_runner.py +38 -28
- argus_code_review-0.2.8/tests/test_runners_turn_budget.py +750 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/.gitignore +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/LICENSE +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/README.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/bench.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/bench_default.toml +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/cli.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/coverage.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/github_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/graph.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/helpers.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/llm/models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/llm/output_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/llm/usage.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/pipeline_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/engine.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/sql.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/argus/storage/sqlite.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/pyproject.toml +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/schema/018_widen_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/conftest.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/golden/review_response.schema.json +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_backend_contract.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_http.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_sql.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/storage/test_sqlite_backend.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_argus_review_local.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_bench.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_bench_config_guard.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_cli_preflight.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_gemini_runner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_fetch_diff.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_get_llm_temperature.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_preflight_image_bump.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_graph_timeout_surfacing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_llm_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_llm_models_override.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_multi_round.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_openai_client.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_output_models.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_packaging.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_precheck_engine.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_runners_context_usage.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_runners_helpers.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_runners_new.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_stage_costs.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.7 → argus_code_review-0.2.8}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,18 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.8] - 2026-10-02
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Raised the Claude-path turn budget from 30 to 50 and the default reviewer session
|
|
15
|
+
timeout from 900 to 1500 seconds globally across all runner platforms through
|
|
16
|
+
`ARGUS_SESSION_TIMEOUT` (TECH-7093), preserving headroom for longer sessions.
|
|
17
|
+
- Added Claude's budget disclosure and supported `PostToolUse`/`PostToolUseFailure`
|
|
18
|
+
hook-based convergence warnings (derived from budget constants; 37 and 47 tool calls
|
|
19
|
+
by default), using truthful top-level tool-call counts, including failed calls.
|
|
20
|
+
- Decoupled the Claude and OpenAI turn-budget constants; OpenAI remains at 30 turns.
|
|
21
|
+
|
|
10
22
|
## [0.2.7] - 2026-09-26
|
|
11
23
|
|
|
12
24
|
### Fixed
|
|
@@ -356,7 +368,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
356
368
|
packaged set.
|
|
357
369
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
358
370
|
|
|
359
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
371
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.8...HEAD
|
|
372
|
+
[0.2.8]: https://github.com/redesignhealth/argus-review/compare/v0.2.7...v0.2.8
|
|
360
373
|
[0.2.7]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...v0.2.7
|
|
361
374
|
[0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
|
|
362
375
|
[0.2.5]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...v0.2.5
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
A slim ``pydantic-settings`` implementation. All configuration comes from
|
|
4
4
|
environment variables (or a local ``.env`` file loaded via
|
|
5
|
-
``argus.dotenv_utils``)
|
|
5
|
+
``argus.dotenv_utils``) -- no AWS, no SSM.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
@@ -15,7 +15,7 @@ from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
|
15
15
|
# ``argus.runners._SUBPROCESS_TIMEOUT_S`` (the fallback used when no Settings
|
|
16
16
|
# instance is available, mostly tests) imports this same constant rather than
|
|
17
17
|
# hardcoding its own copy, so the two can never drift out of sync.
|
|
18
|
-
DEFAULT_ARGUS_SESSION_TIMEOUT_S =
|
|
18
|
+
DEFAULT_ARGUS_SESSION_TIMEOUT_S = 1500
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
class Settings(BaseSettings):
|
|
@@ -62,7 +62,7 @@ class Settings(BaseSettings):
|
|
|
62
62
|
``argus.precheck``.
|
|
63
63
|
ARGUS_STOCK_SEMGREP_PACKS: Comma-separated semgrep registry pack IDs
|
|
64
64
|
(e.g. ``"p/secrets"``) to run alongside (or instead of) a custom
|
|
65
|
-
``ARGUS_RULES_DIR``
|
|
65
|
+
``ARGUS_RULES_DIR`` -- unlike a local rules directory, each pack
|
|
66
66
|
is fetched over the network by semgrep itself on first use
|
|
67
67
|
(cached locally after). Unset by default: this is an opt-in
|
|
68
68
|
addition of vetted, community-maintained rules, not a silent
|
|
@@ -74,8 +74,8 @@ class Settings(BaseSettings):
|
|
|
74
74
|
(``PrecheckResult.failed_scanners``/``missing_scanners``
|
|
75
75
|
non-empty) instead of the default fail-open
|
|
76
76
|
behavior (surface it in the review comment's degraded-coverage
|
|
77
|
-
section
|
|
78
|
-
|
|
77
|
+
section -- see ``argus.helpers.build_degraded_coverage_labels``
|
|
78
|
+
-- but let the review's own verdict stand on its own merits).
|
|
79
79
|
False by default: every other part of this module's design is
|
|
80
80
|
deliberately fail-open (a broken scanner should never be the
|
|
81
81
|
reason a review can't complete), and this flag exists for
|
|
@@ -117,14 +117,21 @@ class Settings(BaseSettings):
|
|
|
117
117
|
Context7 call.
|
|
118
118
|
ARGUS_SESSION_TIMEOUT: Wall-clock seconds a reviewer subprocess is
|
|
119
119
|
allowed to run before it is killed and reported as a failure.
|
|
120
|
-
Defaults to
|
|
120
|
+
Defaults to 1500 (25 minutes) -- first raised from 300 to 420 after
|
|
121
121
|
production logs showed legitimate (non-runaway) specialist
|
|
122
122
|
reviewers finishing as late as 294s, right at the old timeout's
|
|
123
123
|
edge; raised again to 600 to match rh-data-platform's
|
|
124
124
|
production-proven value ahead of this package taking over as the
|
|
125
125
|
actual production reviewer (rh-data-platform's review_service is
|
|
126
|
-
being retired in its favor); raised
|
|
127
|
-
|
|
126
|
+
being retired in its favor); raised to 900 for additional headroom
|
|
127
|
+
across all three reviewer platforms; raised to 1500 in TECH-7093.
|
|
128
|
+
This timeout is deliberately global to all runner platforms through
|
|
129
|
+
ARGUS_SESSION_TIMEOUT, rather than scoped only to Claude, to preserve
|
|
130
|
+
a single operator control across runners. In production, an observed
|
|
131
|
+
cross-cutting session ran 533.8s under the 30-turn cap; linear
|
|
132
|
+
scaling 30 to 50 projects ~890s, meaning 900s leaves virtually zero
|
|
133
|
+
headroom. 1500s preserves roughly 1.69x margin, critical because a
|
|
134
|
+
timeout kill discards all reviewer output.
|
|
128
135
|
GOOGLE_API_KEY: Gemini platform credential, consumed via the
|
|
129
136
|
``google_credential`` property by ``argus.gemini_runner``
|
|
130
137
|
whenever a role's bench entry resolves to
|
|
@@ -206,7 +213,7 @@ class Settings(BaseSettings):
|
|
|
206
213
|
|
|
207
214
|
Callers that need to hand this credential to something that itself
|
|
208
215
|
reads an env var (the spawned ``claude`` CLI subprocess) must set
|
|
209
|
-
the SAME variable name the caller configured
|
|
216
|
+
the SAME variable name the caller configured -- forcing everything
|
|
210
217
|
to ``ANTHROPIC_API_KEY`` would send a proxy/gateway bearer token as
|
|
211
218
|
an ``x-api-key``, which not every gateway accepts. ``ANTHROPIC_API_KEY``
|
|
212
219
|
wins when both are set, matching the Anthropic SDK's own precedence.
|
|
@@ -41,8 +41,9 @@ function call the model requests each turn (there can be more than
|
|
|
41
41
|
one), feeding all of their results back as a single follow-up turn --
|
|
42
42
|
until the model stops requesting function calls, calls
|
|
43
43
|
``finish_review``, or the turn budget (``_MAX_TURNS_GEMINI``, this
|
|
44
|
-
module's own budget, independent of the Claude
|
|
45
|
-
``argus.runners.
|
|
44
|
+
module's own budget, independent of the Claude path's
|
|
45
|
+
``argus.runners._MAX_TURNS_CLAUDE`` and the OpenAI path's
|
|
46
|
+
``argus.openai_runner._MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
46
47
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
47
48
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
48
49
|
task's blast radius contained to this file plus ``argus.bench``'s
|
|
@@ -40,7 +40,7 @@ the model requests each turn (there can be more than one), chaining the
|
|
|
40
40
|
conversation state forward via ``previous_response_id`` and feeding back
|
|
41
41
|
all tool outputs as ``function_call_output`` items -- until the model stops
|
|
42
42
|
requesting function calls, calls ``finish_review``, or the turn budget
|
|
43
|
-
(``
|
|
43
|
+
(``_MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
44
44
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
45
45
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
46
46
|
task's blast radius contained, the final ``SessionResult.result_text`` is
|
|
@@ -110,7 +110,6 @@ from argus.bench import BenchEntry
|
|
|
110
110
|
from argus.llm.models import estimate_cost_usd
|
|
111
111
|
from argus.llm.models import resolve as resolve_model_alias
|
|
112
112
|
from argus.runners import (
|
|
113
|
-
_MAX_TURNS,
|
|
114
113
|
_MID_BUDGET_NUDGE,
|
|
115
114
|
_NUDGE_TURNS_BEFORE_BUDGET,
|
|
116
115
|
_SUBPROCESS_TIMEOUT_S,
|
|
@@ -123,6 +122,8 @@ from argus.runners import (
|
|
|
123
122
|
|
|
124
123
|
logger = logging.getLogger(__name__)
|
|
125
124
|
|
|
125
|
+
_MAX_TURNS_OPENAI = 30
|
|
126
|
+
|
|
126
127
|
_DEFAULT_READ_LIMIT = 2000 # mirrors argus.review_tools._DEFAULT_READ_LIMIT
|
|
127
128
|
|
|
128
129
|
# Matches the fenced-json extraction argus.helpers.parse_review_result
|
|
@@ -417,7 +418,7 @@ async def _run_turns(
|
|
|
417
418
|
whatever those completed turns actually produced and billed.
|
|
418
419
|
"""
|
|
419
420
|
system_prompt = (
|
|
420
|
-
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=
|
|
421
|
+
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_OPENAI)
|
|
421
422
|
)
|
|
422
423
|
api_key = getattr(settings, "OPENAI_API_KEY", None)
|
|
423
424
|
base_url = getattr(settings, "OPENAI_BASE_URL", None)
|
|
@@ -504,13 +505,13 @@ async def _run_turns(
|
|
|
504
505
|
)
|
|
505
506
|
|
|
506
507
|
exhausted = False
|
|
507
|
-
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(
|
|
508
|
+
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_OPENAI)
|
|
508
509
|
try:
|
|
509
510
|
async with asyncio.timeout(timeout_s):
|
|
510
511
|
with review_tools.review_session(repo_root) as findings_sink:
|
|
511
512
|
previous_response_id: str | None = None
|
|
512
513
|
tool_outputs: list[dict[str, Any]] = []
|
|
513
|
-
for _turn in range(
|
|
514
|
+
for _turn in range(_MAX_TURNS_OPENAI):
|
|
514
515
|
create_kwargs: dict[str, Any] = {
|
|
515
516
|
"model": model,
|
|
516
517
|
"instructions": system_prompt,
|
|
@@ -625,7 +626,7 @@ async def _run_turns(
|
|
|
625
626
|
if finished:
|
|
626
627
|
break
|
|
627
628
|
|
|
628
|
-
if _turn + 1 ==
|
|
629
|
+
if _turn + 1 == _MAX_TURNS_OPENAI - _NUDGE_TURNS_BEFORE_BUDGET:
|
|
629
630
|
tool_outputs.append({"role": "user", "content": _TURN_BUDGET_NUDGE})
|
|
630
631
|
|
|
631
632
|
# Not an `elif` -- these are two independent checkpoints
|
|
@@ -643,7 +644,7 @@ async def _run_turns(
|
|
|
643
644
|
logger.warning(
|
|
644
645
|
"OpenAI session [%s] exhausted its %d-turn budget without a finish_review call",
|
|
645
646
|
label or "unlabeled",
|
|
646
|
-
|
|
647
|
+
_MAX_TURNS_OPENAI,
|
|
647
648
|
)
|
|
648
649
|
exhausted = True
|
|
649
650
|
|
|
@@ -30,6 +30,7 @@ from claude_agent_sdk import (
|
|
|
30
30
|
AssistantMessage,
|
|
31
31
|
ClaudeAgentOptions,
|
|
32
32
|
ClaudeSDKClient,
|
|
33
|
+
HookMatcher,
|
|
33
34
|
ResultMessage,
|
|
34
35
|
TaskStartedMessage,
|
|
35
36
|
TextBlock,
|
|
@@ -38,6 +39,13 @@ from claude_agent_sdk import (
|
|
|
38
39
|
ToolUseBlock,
|
|
39
40
|
UserMessage,
|
|
40
41
|
)
|
|
42
|
+
from claude_agent_sdk.types import (
|
|
43
|
+
HookContext,
|
|
44
|
+
HookEvent,
|
|
45
|
+
PostToolUseFailureHookSpecificOutput,
|
|
46
|
+
PostToolUseHookSpecificOutput,
|
|
47
|
+
SyncHookJSONOutput,
|
|
48
|
+
)
|
|
41
49
|
from langsmith import traceable
|
|
42
50
|
from langsmith.run_helpers import LangSmithExtra, get_current_run_tree
|
|
43
51
|
|
|
@@ -141,9 +149,10 @@ if _CROSS_CUTTING_MODEL != ALIAS_MAP["claude-opus"]:
|
|
|
141
149
|
_CROSS_CUTTING_MODEL,
|
|
142
150
|
)
|
|
143
151
|
|
|
144
|
-
#
|
|
145
|
-
# reviewers, feedback verifier, blocking validator)
|
|
146
|
-
|
|
152
|
+
# Turn budget for Claude-path sessions (system/specialist/cross-cutting/tests-and-docs
|
|
153
|
+
# reviewers, feedback verifier, blocking validator) -- raised from 30 to 50
|
|
154
|
+
# in TECH-7093.
|
|
155
|
+
_MAX_TURNS_CLAUDE = 50
|
|
147
156
|
|
|
148
157
|
# Nudges the Gemini/OpenAI turn loops (see their own _run_turns) to call
|
|
149
158
|
# finish_review a few turns before the budget runs out.
|
|
@@ -155,8 +164,8 @@ _TURN_BUDGET_NUDGE = (
|
|
|
155
164
|
)
|
|
156
165
|
|
|
157
166
|
# One-sentence turn-budget disclosure appended to the Gemini/OpenAI reviewer
|
|
158
|
-
# system prompts.
|
|
159
|
-
#
|
|
167
|
+
# system prompts. See _TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE below for the
|
|
168
|
+
# Claude path equivalent.
|
|
160
169
|
_TURN_BUDGET_SYSTEM_PROMPT_LINE = (
|
|
161
170
|
"You have at most {max_turns} turns. Call `finish_review` before you run out -- "
|
|
162
171
|
"a review that never calls it is discarded entirely."
|
|
@@ -187,7 +196,7 @@ def _compute_mid_budget_nudge_turn(max_turns: int) -> int | None:
|
|
|
187
196
|
colliding the two nudges onto the same turn -- which would either
|
|
188
197
|
double-post one message, or silently drop the mid-budget one. Not
|
|
189
198
|
reachable at today's values (75 vs. 97 for Gemini's 100-turn budget, 22
|
|
190
|
-
vs. 27 for
|
|
199
|
+
vs. 27 for OpenAI's 30-turn budget) -- only with a much smaller
|
|
191
200
|
``max_turns``.
|
|
192
201
|
"""
|
|
193
202
|
checkpoint_turn = int(max_turns * _MID_BUDGET_NUDGE_FRACTION)
|
|
@@ -197,6 +206,159 @@ def _compute_mid_budget_nudge_turn(max_turns: int) -> int | None:
|
|
|
197
206
|
return checkpoint_turn
|
|
198
207
|
|
|
199
208
|
|
|
209
|
+
# Claude-path equivalent of the disclosure line above. Claude-path reviewers do
|
|
210
|
+
# not have a `finish_review` tool; they finish by emitting their role's final
|
|
211
|
+
# JSON output block.
|
|
212
|
+
_TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE = (
|
|
213
|
+
"You have at most {max_turns} turns. Emit your final JSON output block before you "
|
|
214
|
+
"run out -- a review that never emits it is discarded entirely."
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
# Tool-call budget warning thresholds and truthful messages for Claude-path
|
|
219
|
+
# sessions.
|
|
220
|
+
#
|
|
221
|
+
# In claude-agent-sdk 0.1.81, supported tool-lifecycle hooks that can inject
|
|
222
|
+
# feedback via hookSpecificOutput.additionalContext are PostToolUse and
|
|
223
|
+
# PostToolUseFailure. PostToolBatch is the exact-per-turn future option in
|
|
224
|
+
# Claude Code, but is untyped and unsupported in this SDK version.
|
|
225
|
+
# We attach to PostToolUse and PostToolUseFailure to track combined tool
|
|
226
|
+
# invocations and inject non-blocking convergence/stop nudges.
|
|
227
|
+
#
|
|
228
|
+
# Thresholds are derived from the turn budget constants: the final threshold
|
|
229
|
+
# fires at the emergency checkpoint (_MAX_TURNS_CLAUDE - _NUDGE_TURNS_BEFORE_BUDGET,
|
|
230
|
+
# or 47 under max_turns=50), and the mid threshold fires at the 75% checkpoint
|
|
231
|
+
# (_compute_mid_budget_nudge_turn(_MAX_TURNS_CLAUDE), or 37 under max_turns=50).
|
|
232
|
+
# Because every continuing agent turn has at least one tool call, these thresholds
|
|
233
|
+
# are reachable no later than the corresponding tool-using turns under max_turns.
|
|
234
|
+
# Multiple or parallel tool calls per turn can make the truthful tool-count advisory
|
|
235
|
+
# fire earlier. The copy makes no turns-left claim, stating only the exact number of
|
|
236
|
+
# tool calls made so far.
|
|
237
|
+
def _compute_claude_tool_budget_thresholds(max_turns: int) -> tuple[int, int]:
|
|
238
|
+
"""Compute and validate Claude tool-budget nudge thresholds (mid, final).
|
|
239
|
+
|
|
240
|
+
Returns:
|
|
241
|
+
tuple[int, int]: (mid_threshold, final_threshold) where mid_threshold
|
|
242
|
+
strictly precedes final_threshold.
|
|
243
|
+
|
|
244
|
+
Raises:
|
|
245
|
+
ValueError: If max_turns produces a mid-budget checkpoint that does not
|
|
246
|
+
strictly precede the final threshold (e.g. colliding or inverted thresholds
|
|
247
|
+
from a small budget).
|
|
248
|
+
"""
|
|
249
|
+
final_threshold = max_turns - _NUDGE_TURNS_BEFORE_BUDGET
|
|
250
|
+
mid_threshold = _compute_mid_budget_nudge_turn(max_turns)
|
|
251
|
+
if mid_threshold is None:
|
|
252
|
+
raise ValueError(
|
|
253
|
+
f"Invalid max_turns={max_turns}: mid-budget checkpoint must precede "
|
|
254
|
+
f"the final budget nudge threshold ({final_threshold})"
|
|
255
|
+
)
|
|
256
|
+
return mid_threshold, final_threshold
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
_TOOL_BUDGET_MID_THRESHOLD: int
|
|
260
|
+
_TOOL_BUDGET_FINAL_THRESHOLD: int
|
|
261
|
+
_TOOL_BUDGET_MID_THRESHOLD, _TOOL_BUDGET_FINAL_THRESHOLD = _compute_claude_tool_budget_thresholds(
|
|
262
|
+
_MAX_TURNS_CLAUDE
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
_TOOL_BUDGET_MID_NUDGE = (
|
|
266
|
+
"You have now made {n} tool calls. If you have gathered enough "
|
|
267
|
+
"context to identify findings, begin converging toward your final JSON output block "
|
|
268
|
+
"now rather than continuing to explore."
|
|
269
|
+
)
|
|
270
|
+
_TOOL_BUDGET_FINAL_NUDGE = (
|
|
271
|
+
"You have now made {n} tool calls. Stop reading and searching now, and emit "
|
|
272
|
+
"your final JSON output block with whatever you have found so far. If you have "
|
|
273
|
+
"found nothing, emit your final JSON output block with an empty findings list."
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _make_tool_budget_nudge_hooks(
|
|
278
|
+
label: str | None = None,
|
|
279
|
+
*,
|
|
280
|
+
mid_threshold: int = _TOOL_BUDGET_MID_THRESHOLD,
|
|
281
|
+
final_threshold: int = _TOOL_BUDGET_FINAL_THRESHOLD,
|
|
282
|
+
) -> dict[HookEvent, list[HookMatcher]]:
|
|
283
|
+
"""Create PostToolUse and PostToolUseFailure hooks that inject non-blocking
|
|
284
|
+
nudge warnings into Claude-path sessions when tool-call counts cross budget thresholds.
|
|
285
|
+
|
|
286
|
+
In claude-agent-sdk 0.1.81, supported tool-lifecycle hooks that can inject
|
|
287
|
+
feedback via hookSpecificOutput.additionalContext are PostToolUse and
|
|
288
|
+
PostToolUseFailure. PostToolBatch is the exact-per-turn future option in
|
|
289
|
+
Claude Code, but is untyped and unsupported in this SDK version.
|
|
290
|
+
|
|
291
|
+
One callback is shared by both events, closing over combined tool-call count
|
|
292
|
+
and one-shot flags. Subagent tool calls (bearing agent_id) are skipped.
|
|
293
|
+
"""
|
|
294
|
+
tool_call_count = 0
|
|
295
|
+
mid_nudge_sent = False
|
|
296
|
+
final_nudge_sent = False
|
|
297
|
+
|
|
298
|
+
async def _hook_callback(
|
|
299
|
+
input_data: Any,
|
|
300
|
+
tool_use_id: str | None,
|
|
301
|
+
context: HookContext,
|
|
302
|
+
) -> SyncHookJSONOutput:
|
|
303
|
+
nonlocal tool_call_count, mid_nudge_sent, final_nudge_sent
|
|
304
|
+
try:
|
|
305
|
+
if not isinstance(input_data, dict):
|
|
306
|
+
return {}
|
|
307
|
+
# Sub-agent attribution: skip Task-spawned subagents
|
|
308
|
+
if input_data.get("agent_id"):
|
|
309
|
+
return {}
|
|
310
|
+
|
|
311
|
+
tool_call_count += 1
|
|
312
|
+
nudge_message: str | None = None
|
|
313
|
+
|
|
314
|
+
if tool_call_count >= final_threshold and not final_nudge_sent:
|
|
315
|
+
final_nudge_sent = True
|
|
316
|
+
mid_nudge_sent = True
|
|
317
|
+
nudge_message = _TOOL_BUDGET_FINAL_NUDGE.format(n=tool_call_count)
|
|
318
|
+
elif tool_call_count >= mid_threshold and not mid_nudge_sent:
|
|
319
|
+
mid_nudge_sent = True
|
|
320
|
+
nudge_message = _TOOL_BUDGET_MID_NUDGE.format(n=tool_call_count)
|
|
321
|
+
|
|
322
|
+
if nudge_message is None:
|
|
323
|
+
return {}
|
|
324
|
+
|
|
325
|
+
event_name = input_data.get("hook_event_name")
|
|
326
|
+
logger.info(
|
|
327
|
+
"Injected Claude tool-budget nudge [%s] at tool call %d (event=%s)",
|
|
328
|
+
label or "unlabeled",
|
|
329
|
+
tool_call_count,
|
|
330
|
+
event_name,
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
# Mirror hook_event_name in hookSpecificOutput
|
|
334
|
+
if event_name == "PostToolUseFailure":
|
|
335
|
+
failure_output: PostToolUseFailureHookSpecificOutput = {
|
|
336
|
+
"hookEventName": "PostToolUseFailure",
|
|
337
|
+
"additionalContext": nudge_message,
|
|
338
|
+
}
|
|
339
|
+
return {"hookSpecificOutput": failure_output}
|
|
340
|
+
else:
|
|
341
|
+
success_output: PostToolUseHookSpecificOutput = {
|
|
342
|
+
"hookEventName": "PostToolUse",
|
|
343
|
+
"additionalContext": nudge_message,
|
|
344
|
+
}
|
|
345
|
+
return {"hookSpecificOutput": success_output}
|
|
346
|
+
|
|
347
|
+
except Exception:
|
|
348
|
+
logger.warning(
|
|
349
|
+
"Tool budget hook callback failed for [%s]",
|
|
350
|
+
label or "unlabeled",
|
|
351
|
+
exc_info=True,
|
|
352
|
+
)
|
|
353
|
+
return {}
|
|
354
|
+
|
|
355
|
+
matcher = HookMatcher(hooks=[_hook_callback])
|
|
356
|
+
return {
|
|
357
|
+
"PostToolUse": [matcher],
|
|
358
|
+
"PostToolUseFailure": [matcher],
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
|
|
200
362
|
# Fallback repo root for ClaudeSDKClient cwd — used when no SHA-pinned
|
|
201
363
|
# worktree has been provisioned (e.g. local dev runs, tests, subprocess
|
|
202
364
|
# worker). In production, callers pass an explicit repo_root provisioned
|
|
@@ -1697,15 +1859,22 @@ async def _run_claude_session(
|
|
|
1697
1859
|
betas: list[Literal["context-1m-2025-08-07"]] = (
|
|
1698
1860
|
["context-1m-2025-08-07"] if _attach_1m_context_beta else []
|
|
1699
1861
|
)
|
|
1862
|
+
system_prompt = (
|
|
1863
|
+
system_prompt
|
|
1864
|
+
+ "\n\n"
|
|
1865
|
+
+ _TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE.format(max_turns=_MAX_TURNS_CLAUDE)
|
|
1866
|
+
)
|
|
1867
|
+
hooks = _make_tool_budget_nudge_hooks(label)
|
|
1700
1868
|
options = ClaudeAgentOptions(
|
|
1701
1869
|
cwd=effective_root,
|
|
1702
1870
|
allowed_tools=["Read", "Glob", "Grep"] + context7_tools,
|
|
1703
1871
|
mcp_servers=mcp_servers,
|
|
1704
1872
|
strict_mcp_config=True,
|
|
1705
1873
|
permission_mode="default",
|
|
1874
|
+
hooks=hooks,
|
|
1706
1875
|
model=model,
|
|
1707
1876
|
system_prompt=system_prompt,
|
|
1708
|
-
max_turns=
|
|
1877
|
+
max_turns=_MAX_TURNS_CLAUDE,
|
|
1709
1878
|
env=dict([settings.anthropic_credential]),
|
|
1710
1879
|
stderr=_stderr_handler,
|
|
1711
1880
|
betas=betas,
|
|
@@ -1743,6 +1912,11 @@ async def _run_claude_session(
|
|
|
1743
1912
|
result_text = ""
|
|
1744
1913
|
cost_usd = 0.0
|
|
1745
1914
|
tool_calls: list[str] = []
|
|
1915
|
+
# Note: message_index counts AssistantMessage objects received from the SDK
|
|
1916
|
+
# stream, NOT CLI turns. Multiple AssistantMessage objects can occur within
|
|
1917
|
+
# a single turn or sub-step. It must never be used as a turn counter.
|
|
1918
|
+
# Key and log names ('msg_index', 'msg=%d') are preserved for log
|
|
1919
|
+
# compatibility.
|
|
1746
1920
|
message_index = 0
|
|
1747
1921
|
failure_reason: Literal["turn_budget_exhausted"] | None = None
|
|
1748
1922
|
async for message in client.receive_response():
|
|
@@ -1842,7 +2016,7 @@ async def _run_claude_session(
|
|
|
1842
2016
|
logger.warning(
|
|
1843
2017
|
"Agent [%s] exhausted its %d-turn budget (subtype=%s)",
|
|
1844
2018
|
label or "unlabeled",
|
|
1845
|
-
|
|
2019
|
+
_MAX_TURNS_CLAUDE,
|
|
1846
2020
|
message.subtype,
|
|
1847
2021
|
)
|
|
1848
2022
|
failure_reason = "turn_budget_exhausted"
|
|
@@ -214,11 +214,11 @@ def test_gemini_cache_ttl_override(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
|
214
214
|
assert settings.ARGUS_GEMINI_CACHE_TTL == 60
|
|
215
215
|
|
|
216
216
|
|
|
217
|
-
def
|
|
217
|
+
def test_session_timeout_defaults_to_1500(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
218
218
|
_set_required(monkeypatch)
|
|
219
219
|
monkeypatch.delenv("ARGUS_SESSION_TIMEOUT", raising=False)
|
|
220
220
|
settings = get_settings()
|
|
221
|
-
assert settings.ARGUS_SESSION_TIMEOUT ==
|
|
221
|
+
assert settings.ARGUS_SESSION_TIMEOUT == 1500
|
|
222
222
|
|
|
223
223
|
|
|
224
224
|
def test_session_timeout_override(monkeypatch: pytest.MonkeyPatch) -> None:
|
|
@@ -34,8 +34,12 @@ from openai.types.responses.response_usage import InputTokensDetails, OutputToke
|
|
|
34
34
|
|
|
35
35
|
from argus.bench import BenchEntry
|
|
36
36
|
from argus.llm.models import resolve as resolve_alias
|
|
37
|
-
from argus.openai_runner import
|
|
38
|
-
|
|
37
|
+
from argus.openai_runner import (
|
|
38
|
+
_MAX_TURNS_OPENAI,
|
|
39
|
+
_redact_openai_inputs,
|
|
40
|
+
run_session_openai,
|
|
41
|
+
)
|
|
42
|
+
from argus.runners import _TURN_BUDGET_SYSTEM_PROMPT_LINE
|
|
39
43
|
|
|
40
44
|
pytestmark = pytest.mark.asyncio
|
|
41
45
|
|
|
@@ -44,7 +48,9 @@ def _with_turn_budget_line(system_prompt: str) -> str:
|
|
|
44
48
|
"""The runner appends the shared turn-budget disclosure to every system
|
|
45
49
|
prompt it's given -- tests that assert on the exact `instructions` sent
|
|
46
50
|
to the API must account for it."""
|
|
47
|
-
return
|
|
51
|
+
return (
|
|
52
|
+
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_OPENAI)
|
|
53
|
+
)
|
|
48
54
|
|
|
49
55
|
|
|
50
56
|
# ---------------------------------------------------------------------------
|
|
@@ -292,15 +298,16 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
292
298
|
assert outputs[1]["call_id"] == "call_2"
|
|
293
299
|
|
|
294
300
|
async def test_exhausting_max_turns_still_returns_a_result(self, tmp_path: Any) -> None:
|
|
295
|
-
"""Exhausting
|
|
301
|
+
"""Exhausting _MAX_TURNS_OPENAI stops and builds whatever findings were
|
|
296
302
|
reported, with failure_reason="turn_budget_exhausted" -- not None,
|
|
297
303
|
which would be indistinguishable from a clean 0-finding completion.
|
|
298
304
|
"""
|
|
299
|
-
from argus.
|
|
305
|
+
from argus.openai_runner import _MAX_TURNS_OPENAI
|
|
300
306
|
|
|
301
307
|
finding_call = ("report_finding", {"file": "f.py", "line": 1, "description": "d"})
|
|
302
308
|
responses = [
|
|
303
|
-
_make_response(calls=[finding_call], resp_id=f"r_{i}")
|
|
309
|
+
_make_response(calls=[finding_call], resp_id=f"r_{i}")
|
|
310
|
+
for i in range(_MAX_TURNS_OPENAI + 5)
|
|
304
311
|
]
|
|
305
312
|
client = _make_fake_client(responses)
|
|
306
313
|
entry = _make_entry()
|
|
@@ -315,29 +322,30 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
315
322
|
)
|
|
316
323
|
|
|
317
324
|
assert result.failure_reason == "turn_budget_exhausted"
|
|
318
|
-
assert client.responses.create.call_count ==
|
|
325
|
+
assert client.responses.create.call_count == _MAX_TURNS_OPENAI
|
|
319
326
|
from argus.helpers import parse_review_result
|
|
320
327
|
|
|
321
328
|
parsed = parse_review_result(result.result_text, "test-group")
|
|
322
|
-
assert len(parsed.findings) ==
|
|
329
|
+
assert len(parsed.findings) == _MAX_TURNS_OPENAI
|
|
323
330
|
|
|
324
331
|
async def test_finish_review_on_final_turn_is_clean_completion_not_exhaustion(
|
|
325
332
|
self, tmp_path: Any
|
|
326
333
|
) -> None:
|
|
327
|
-
"""A successful finish_review on the LAST allowed turn (
|
|
334
|
+
"""A successful finish_review on the LAST allowed turn (_MAX_TURNS_OPENAI) must
|
|
328
335
|
still be treated as a clean completion with files_explored populated.
|
|
329
336
|
The turn loop's `for...else` only runs its exhaustion branch when the loop
|
|
330
337
|
completes without `break`."""
|
|
331
|
-
from argus.
|
|
338
|
+
from argus.openai_runner import _MAX_TURNS_OPENAI
|
|
332
339
|
|
|
333
340
|
finding_call = ("report_finding", {"file": "f.py", "line": 1, "description": "d"})
|
|
334
341
|
responses = [
|
|
335
|
-
_make_response(calls=[finding_call], resp_id=f"r_{i}")
|
|
342
|
+
_make_response(calls=[finding_call], resp_id=f"r_{i}")
|
|
343
|
+
for i in range(_MAX_TURNS_OPENAI - 1)
|
|
336
344
|
]
|
|
337
345
|
responses.append(
|
|
338
346
|
_make_response(
|
|
339
347
|
calls=[("finish_review", {"files_explored": ["f.py"]})],
|
|
340
|
-
resp_id=f"r_{
|
|
348
|
+
resp_id=f"r_{_MAX_TURNS_OPENAI - 1}",
|
|
341
349
|
)
|
|
342
350
|
)
|
|
343
351
|
client = _make_fake_client(responses)
|
|
@@ -353,18 +361,19 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
353
361
|
)
|
|
354
362
|
|
|
355
363
|
assert result.failure_reason is None
|
|
356
|
-
assert client.responses.create.call_count ==
|
|
364
|
+
assert client.responses.create.call_count == _MAX_TURNS_OPENAI
|
|
357
365
|
assert "finish_review" in result.tool_names
|
|
358
366
|
|
|
359
367
|
async def test_nudge_injected_once_at_turn_budget_minus_three(self, tmp_path: Any) -> None:
|
|
360
368
|
"""A session that never calls finish_review gets nudged exactly once,
|
|
361
369
|
_NUDGE_TURNS_BEFORE_BUDGET turns before exhaustion, and still preserves
|
|
362
370
|
findings reported both before and after the nudge."""
|
|
363
|
-
from argus.
|
|
371
|
+
from argus.openai_runner import _MAX_TURNS_OPENAI
|
|
372
|
+
from argus.runners import _NUDGE_TURNS_BEFORE_BUDGET, _TURN_BUDGET_NUDGE
|
|
364
373
|
|
|
365
374
|
finding_call = ("report_finding", {"file": "f.py", "line": 1, "description": "d"})
|
|
366
375
|
responses = [
|
|
367
|
-
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(
|
|
376
|
+
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(_MAX_TURNS_OPENAI)
|
|
368
377
|
]
|
|
369
378
|
client = _make_fake_client(responses)
|
|
370
379
|
entry = _make_entry()
|
|
@@ -382,9 +391,9 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
382
391
|
from argus.helpers import parse_review_result
|
|
383
392
|
|
|
384
393
|
parsed = parse_review_result(result.result_text, "test-group")
|
|
385
|
-
assert len(parsed.findings) ==
|
|
394
|
+
assert len(parsed.findings) == _MAX_TURNS_OPENAI
|
|
386
395
|
|
|
387
|
-
threshold =
|
|
396
|
+
threshold = _MAX_TURNS_OPENAI - _NUDGE_TURNS_BEFORE_BUDGET
|
|
388
397
|
nudge_item = {"role": "user", "content": _TURN_BUDGET_NUDGE}
|
|
389
398
|
for i, call in enumerate(client.responses.create.call_args_list):
|
|
390
399
|
call_input = call.kwargs["input"]
|
|
@@ -421,13 +430,14 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
421
430
|
|
|
422
431
|
async def test_mid_budget_nudge_injected_once_at_75_percent(self, tmp_path: Any) -> None:
|
|
423
432
|
"""A session that never calls finish_review gets the mid-budget
|
|
424
|
-
checkpoint nudge exactly once, at 75% of
|
|
433
|
+
checkpoint nudge exactly once, at 75% of _MAX_TURNS_OPENAI -- in addition
|
|
425
434
|
to (not instead of) the end-of-budget _TURN_BUDGET_NUDGE."""
|
|
426
|
-
from argus.
|
|
435
|
+
from argus.openai_runner import _MAX_TURNS_OPENAI
|
|
436
|
+
from argus.runners import _MID_BUDGET_NUDGE, _compute_mid_budget_nudge_turn
|
|
427
437
|
|
|
428
438
|
finding_call = ("report_finding", {"file": "f.py", "line": 1, "description": "d"})
|
|
429
439
|
responses = [
|
|
430
|
-
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(
|
|
440
|
+
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(_MAX_TURNS_OPENAI)
|
|
431
441
|
]
|
|
432
442
|
client = _make_fake_client(responses)
|
|
433
443
|
entry = _make_entry()
|
|
@@ -443,7 +453,7 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
443
453
|
|
|
444
454
|
assert result.failure_reason == "turn_budget_exhausted"
|
|
445
455
|
|
|
446
|
-
mid_budget_turn = _compute_mid_budget_nudge_turn(
|
|
456
|
+
mid_budget_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_OPENAI)
|
|
447
457
|
assert mid_budget_turn is not None
|
|
448
458
|
nudge_item = {"role": "user", "content": _MID_BUDGET_NUDGE}
|
|
449
459
|
for i, call in enumerate(client.responses.create.call_args_list):
|
|
@@ -485,8 +495,8 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
485
495
|
"""An exhausted session must receive BOTH the mid-budget checkpoint
|
|
486
496
|
nudge and the end-of-budget emergency nudge, at their own distinct
|
|
487
497
|
turns -- neither should suppress or overwrite the other."""
|
|
498
|
+
from argus.openai_runner import _MAX_TURNS_OPENAI
|
|
488
499
|
from argus.runners import (
|
|
489
|
-
_MAX_TURNS,
|
|
490
500
|
_MID_BUDGET_NUDGE,
|
|
491
501
|
_NUDGE_TURNS_BEFORE_BUDGET,
|
|
492
502
|
_TURN_BUDGET_NUDGE,
|
|
@@ -495,7 +505,7 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
495
505
|
|
|
496
506
|
finding_call = ("report_finding", {"file": "f.py", "line": 1, "description": "d"})
|
|
497
507
|
responses = [
|
|
498
|
-
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(
|
|
508
|
+
_make_response(calls=[finding_call], resp_id=f"r_{i}") for i in range(_MAX_TURNS_OPENAI)
|
|
499
509
|
]
|
|
500
510
|
client = _make_fake_client(responses)
|
|
501
511
|
entry = _make_entry()
|
|
@@ -511,8 +521,8 @@ class TestOpenAIRunnerBasicLoop:
|
|
|
511
521
|
|
|
512
522
|
assert result.failure_reason == "turn_budget_exhausted"
|
|
513
523
|
|
|
514
|
-
mid_budget_turn = _compute_mid_budget_nudge_turn(
|
|
515
|
-
emergency_turn =
|
|
524
|
+
mid_budget_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_OPENAI)
|
|
525
|
+
emergency_turn = _MAX_TURNS_OPENAI - _NUDGE_TURNS_BEFORE_BUDGET
|
|
516
526
|
assert mid_budget_turn is not None
|
|
517
527
|
assert mid_budget_turn != emergency_turn
|
|
518
528
|
|
|
@@ -957,8 +967,8 @@ class TestOpenAIRunnerTimeoutsAndFailures:
|
|
|
957
967
|
mock_init.assert_called_once()
|
|
958
968
|
assert mock_init.call_args.kwargs["timeout"] == 42.0
|
|
959
969
|
|
|
960
|
-
async def
|
|
961
|
-
"""When neither timeout_s nor ARGUS_SESSION_TIMEOUT is passed, defaults to
|
|
970
|
+
async def test_default_timeout_is_1500_seconds(self, tmp_path: Any) -> None:
|
|
971
|
+
"""When neither timeout_s nor ARGUS_SESSION_TIMEOUT is passed, defaults to 1500s."""
|
|
962
972
|
settings = MagicMock(spec=[])
|
|
963
973
|
settings.OPENAI_API_KEY = "key"
|
|
964
974
|
client = _make_fake_client([_make_response(calls=[])])
|
|
@@ -974,7 +984,7 @@ class TestOpenAIRunnerTimeoutsAndFailures:
|
|
|
974
984
|
)
|
|
975
985
|
|
|
976
986
|
mock_init.assert_called_once()
|
|
977
|
-
assert mock_init.call_args.kwargs["timeout"] ==
|
|
987
|
+
assert mock_init.call_args.kwargs["timeout"] == 1500
|
|
978
988
|
|
|
979
989
|
async def test_client_closed_on_normal_completion(self, tmp_path: Any) -> None:
|
|
980
990
|
"""client.close is called when session completes normally."""
|