argus-code-review 0.2.6__tar.gz → 0.2.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/CHANGELOG.md +33 -1
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/PKG-INFO +1 -1
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/config.py +16 -9
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/gemini_runner.py +3 -2
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/graph.py +5 -1
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/openai_runner.py +8 -7
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/runners.py +182 -8
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_backend_contract.py +98 -16
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_http.py +68 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_config.py +2 -2
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_openai_runner.py +38 -28
- argus_code_review-0.2.8/tests/test_runners_turn_budget.py +750 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/.gitignore +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/LICENSE +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/README.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/bench.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/bench_default.toml +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/cli.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/coverage.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/github_client.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/helpers.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/llm/models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/llm/output_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/llm/usage.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/pipeline_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/engine.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/sql.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/argus/storage/sqlite.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/pyproject.toml +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/schema/018_widen_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/conftest.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/golden/review_response.schema.json +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_sql.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/storage/test_sqlite_backend.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_argus_review_local.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_bench.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_bench_config_guard.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_cli_preflight.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_gemini_runner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_fetch_diff.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_get_llm_temperature.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_precheck.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_preflight_image_bump.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_graph_timeout_surfacing.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_llm_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_llm_models_override.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_multi_round.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_openai_client.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_output_models.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_packaging.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_precheck_engine.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_runners_context_usage.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_runners_helpers.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_runners_new.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_stage_costs.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.6 → argus_code_review-0.2.8}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,36 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.8] - 2026-10-02
|
|
11
|
+
|
|
12
|
+
### Changed
|
|
13
|
+
|
|
14
|
+
- Raised the Claude-path turn budget from 30 to 50 and the default reviewer session
|
|
15
|
+
timeout from 900 to 1500 seconds globally across all runner platforms through
|
|
16
|
+
`ARGUS_SESSION_TIMEOUT` (TECH-7093), preserving headroom for longer sessions.
|
|
17
|
+
- Added Claude's budget disclosure and supported `PostToolUse`/`PostToolUseFailure`
|
|
18
|
+
hook-based convergence warnings (derived from budget constants; 37 and 47 tool calls
|
|
19
|
+
by default), using truthful top-level tool-call counts, including failed calls.
|
|
20
|
+
- Decoupled the Claude and OpenAI turn-budget constants; OpenAI remains at 30 turns.
|
|
21
|
+
|
|
22
|
+
## [0.2.7] - 2026-09-26
|
|
23
|
+
|
|
24
|
+
### Fixed
|
|
25
|
+
|
|
26
|
+
- Reworded a misleading log message in `argus/graph.py`'s `_node_lite_review`
|
|
27
|
+
(TECH-6878, #29). The log previously stated "Lite round history skipped on HTTP
|
|
28
|
+
storage path", which was misinterpreted as indicating that review persistence
|
|
29
|
+
was skipped altogether and caused an operator/agent to file a false bug
|
|
30
|
+
report. The message now clarifies that only the comment markdown section for
|
|
31
|
+
prior lite history is omitted (because `select_recent_lite_rounds` is not
|
|
32
|
+
implemented in the HTTP storage shim), while the round's own finalize write is
|
|
33
|
+
not skipped and is attempted normally.
|
|
34
|
+
- Added test coverage for lite-round persistence round-tripping through both the
|
|
35
|
+
SQLite and HTTP storage backends (previously untested for HTTP), verifying that
|
|
36
|
+
`reviewer_version="v3-lite"` is preserved on write and read (TECH-6878, #29).
|
|
37
|
+
Also marked `test_insert_agent_runs_batch[http]` as an explicit xfail for the
|
|
38
|
+
documented HTTP shim analytics gap rather than letting it pass vacuously.
|
|
39
|
+
|
|
10
40
|
## [0.2.6] - 2026-09-19
|
|
11
41
|
|
|
12
42
|
### Fixed
|
|
@@ -338,7 +368,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
338
368
|
packaged set.
|
|
339
369
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
340
370
|
|
|
341
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
371
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.8...HEAD
|
|
372
|
+
[0.2.8]: https://github.com/redesignhealth/argus-review/compare/v0.2.7...v0.2.8
|
|
373
|
+
[0.2.7]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...v0.2.7
|
|
342
374
|
[0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
|
|
343
375
|
[0.2.5]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...v0.2.5
|
|
344
376
|
[0.2.4]: https://github.com/redesignhealth/argus-review/compare/v0.2.3...v0.2.4
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.8
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
A slim ``pydantic-settings`` implementation. All configuration comes from
|
|
4
4
|
environment variables (or a local ``.env`` file loaded via
|
|
5
|
-
``argus.dotenv_utils``)
|
|
5
|
+
``argus.dotenv_utils``) -- no AWS, no SSM.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
@@ -15,7 +15,7 @@ from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
|
15
15
|
# ``argus.runners._SUBPROCESS_TIMEOUT_S`` (the fallback used when no Settings
|
|
16
16
|
# instance is available, mostly tests) imports this same constant rather than
|
|
17
17
|
# hardcoding its own copy, so the two can never drift out of sync.
|
|
18
|
-
DEFAULT_ARGUS_SESSION_TIMEOUT_S =
|
|
18
|
+
DEFAULT_ARGUS_SESSION_TIMEOUT_S = 1500
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
class Settings(BaseSettings):
|
|
@@ -62,7 +62,7 @@ class Settings(BaseSettings):
|
|
|
62
62
|
``argus.precheck``.
|
|
63
63
|
ARGUS_STOCK_SEMGREP_PACKS: Comma-separated semgrep registry pack IDs
|
|
64
64
|
(e.g. ``"p/secrets"``) to run alongside (or instead of) a custom
|
|
65
|
-
``ARGUS_RULES_DIR``
|
|
65
|
+
``ARGUS_RULES_DIR`` -- unlike a local rules directory, each pack
|
|
66
66
|
is fetched over the network by semgrep itself on first use
|
|
67
67
|
(cached locally after). Unset by default: this is an opt-in
|
|
68
68
|
addition of vetted, community-maintained rules, not a silent
|
|
@@ -74,8 +74,8 @@ class Settings(BaseSettings):
|
|
|
74
74
|
(``PrecheckResult.failed_scanners``/``missing_scanners``
|
|
75
75
|
non-empty) instead of the default fail-open
|
|
76
76
|
behavior (surface it in the review comment's degraded-coverage
|
|
77
|
-
section
|
|
78
|
-
|
|
77
|
+
section -- see ``argus.helpers.build_degraded_coverage_labels``
|
|
78
|
+
-- but let the review's own verdict stand on its own merits).
|
|
79
79
|
False by default: every other part of this module's design is
|
|
80
80
|
deliberately fail-open (a broken scanner should never be the
|
|
81
81
|
reason a review can't complete), and this flag exists for
|
|
@@ -117,14 +117,21 @@ class Settings(BaseSettings):
|
|
|
117
117
|
Context7 call.
|
|
118
118
|
ARGUS_SESSION_TIMEOUT: Wall-clock seconds a reviewer subprocess is
|
|
119
119
|
allowed to run before it is killed and reported as a failure.
|
|
120
|
-
Defaults to
|
|
120
|
+
Defaults to 1500 (25 minutes) -- first raised from 300 to 420 after
|
|
121
121
|
production logs showed legitimate (non-runaway) specialist
|
|
122
122
|
reviewers finishing as late as 294s, right at the old timeout's
|
|
123
123
|
edge; raised again to 600 to match rh-data-platform's
|
|
124
124
|
production-proven value ahead of this package taking over as the
|
|
125
125
|
actual production reviewer (rh-data-platform's review_service is
|
|
126
|
-
being retired in its favor); raised
|
|
127
|
-
|
|
126
|
+
being retired in its favor); raised to 900 for additional headroom
|
|
127
|
+
across all three reviewer platforms; raised to 1500 in TECH-7093.
|
|
128
|
+
This timeout is deliberately global to all runner platforms through
|
|
129
|
+
ARGUS_SESSION_TIMEOUT, rather than scoped only to Claude, to preserve
|
|
130
|
+
a single operator control across runners. In production, an observed
|
|
131
|
+
cross-cutting session ran 533.8s under the 30-turn cap; linear
|
|
132
|
+
scaling 30 to 50 projects ~890s, meaning 900s leaves virtually zero
|
|
133
|
+
headroom. 1500s preserves roughly 1.69x margin, critical because a
|
|
134
|
+
timeout kill discards all reviewer output.
|
|
128
135
|
GOOGLE_API_KEY: Gemini platform credential, consumed via the
|
|
129
136
|
``google_credential`` property by ``argus.gemini_runner``
|
|
130
137
|
whenever a role's bench entry resolves to
|
|
@@ -206,7 +213,7 @@ class Settings(BaseSettings):
|
|
|
206
213
|
|
|
207
214
|
Callers that need to hand this credential to something that itself
|
|
208
215
|
reads an env var (the spawned ``claude`` CLI subprocess) must set
|
|
209
|
-
the SAME variable name the caller configured
|
|
216
|
+
the SAME variable name the caller configured -- forcing everything
|
|
210
217
|
to ``ANTHROPIC_API_KEY`` would send a proxy/gateway bearer token as
|
|
211
218
|
an ``x-api-key``, which not every gateway accepts. ``ANTHROPIC_API_KEY``
|
|
212
219
|
wins when both are set, matching the Anthropic SDK's own precedence.
|
|
@@ -41,8 +41,9 @@ function call the model requests each turn (there can be more than
|
|
|
41
41
|
one), feeding all of their results back as a single follow-up turn --
|
|
42
42
|
until the model stops requesting function calls, calls
|
|
43
43
|
``finish_review``, or the turn budget (``_MAX_TURNS_GEMINI``, this
|
|
44
|
-
module's own budget, independent of the Claude
|
|
45
|
-
``argus.runners.
|
|
44
|
+
module's own budget, independent of the Claude path's
|
|
45
|
+
``argus.runners._MAX_TURNS_CLAUDE`` and the OpenAI path's
|
|
46
|
+
``argus.openai_runner._MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
46
47
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
47
48
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
48
49
|
task's blast radius contained to this file plus ``argus.bench``'s
|
|
@@ -2228,7 +2228,11 @@ async def _node_lite_review(state: ReviewState) -> dict[str, Any]:
|
|
|
2228
2228
|
except Exception: # noqa: BLE001
|
|
2229
2229
|
logger.error("Failed to build lite round history — skipping section", exc_info=True)
|
|
2230
2230
|
elif req.pr_number and _lite_history_backend_kind == "http":
|
|
2231
|
-
logger.info(
|
|
2231
|
+
logger.info(
|
|
2232
|
+
"Lite round history markdown section omitted on HTTP storage path "
|
|
2233
|
+
"(select_recent_lite_rounds not implemented in HTTP shim; "
|
|
2234
|
+
"the round's own finalize write is not skipped and will be attempted normally)"
|
|
2235
|
+
)
|
|
2232
2236
|
|
|
2233
2237
|
response.review_comment = (
|
|
2234
2238
|
f"## Code Review — Round {round_num} (Lite Mode)\n\n"
|
|
@@ -40,7 +40,7 @@ the model requests each turn (there can be more than one), chaining the
|
|
|
40
40
|
conversation state forward via ``previous_response_id`` and feeding back
|
|
41
41
|
all tool outputs as ``function_call_output`` items -- until the model stops
|
|
42
42
|
requesting function calls, calls ``finish_review``, or the turn budget
|
|
43
|
-
(``
|
|
43
|
+
(``_MAX_TURNS_OPENAI``) is exhausted. Findings arrive via
|
|
44
44
|
``report_finding`` tool calls into ``review_tools``' per-session sink,
|
|
45
45
|
not as a JSON blob embedded in the model's own text -- so, to keep this
|
|
46
46
|
task's blast radius contained, the final ``SessionResult.result_text`` is
|
|
@@ -110,7 +110,6 @@ from argus.bench import BenchEntry
|
|
|
110
110
|
from argus.llm.models import estimate_cost_usd
|
|
111
111
|
from argus.llm.models import resolve as resolve_model_alias
|
|
112
112
|
from argus.runners import (
|
|
113
|
-
_MAX_TURNS,
|
|
114
113
|
_MID_BUDGET_NUDGE,
|
|
115
114
|
_NUDGE_TURNS_BEFORE_BUDGET,
|
|
116
115
|
_SUBPROCESS_TIMEOUT_S,
|
|
@@ -123,6 +122,8 @@ from argus.runners import (
|
|
|
123
122
|
|
|
124
123
|
logger = logging.getLogger(__name__)
|
|
125
124
|
|
|
125
|
+
_MAX_TURNS_OPENAI = 30
|
|
126
|
+
|
|
126
127
|
_DEFAULT_READ_LIMIT = 2000 # mirrors argus.review_tools._DEFAULT_READ_LIMIT
|
|
127
128
|
|
|
128
129
|
# Matches the fenced-json extraction argus.helpers.parse_review_result
|
|
@@ -417,7 +418,7 @@ async def _run_turns(
|
|
|
417
418
|
whatever those completed turns actually produced and billed.
|
|
418
419
|
"""
|
|
419
420
|
system_prompt = (
|
|
420
|
-
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=
|
|
421
|
+
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_OPENAI)
|
|
421
422
|
)
|
|
422
423
|
api_key = getattr(settings, "OPENAI_API_KEY", None)
|
|
423
424
|
base_url = getattr(settings, "OPENAI_BASE_URL", None)
|
|
@@ -504,13 +505,13 @@ async def _run_turns(
|
|
|
504
505
|
)
|
|
505
506
|
|
|
506
507
|
exhausted = False
|
|
507
|
-
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(
|
|
508
|
+
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_OPENAI)
|
|
508
509
|
try:
|
|
509
510
|
async with asyncio.timeout(timeout_s):
|
|
510
511
|
with review_tools.review_session(repo_root) as findings_sink:
|
|
511
512
|
previous_response_id: str | None = None
|
|
512
513
|
tool_outputs: list[dict[str, Any]] = []
|
|
513
|
-
for _turn in range(
|
|
514
|
+
for _turn in range(_MAX_TURNS_OPENAI):
|
|
514
515
|
create_kwargs: dict[str, Any] = {
|
|
515
516
|
"model": model,
|
|
516
517
|
"instructions": system_prompt,
|
|
@@ -625,7 +626,7 @@ async def _run_turns(
|
|
|
625
626
|
if finished:
|
|
626
627
|
break
|
|
627
628
|
|
|
628
|
-
if _turn + 1 ==
|
|
629
|
+
if _turn + 1 == _MAX_TURNS_OPENAI - _NUDGE_TURNS_BEFORE_BUDGET:
|
|
629
630
|
tool_outputs.append({"role": "user", "content": _TURN_BUDGET_NUDGE})
|
|
630
631
|
|
|
631
632
|
# Not an `elif` -- these are two independent checkpoints
|
|
@@ -643,7 +644,7 @@ async def _run_turns(
|
|
|
643
644
|
logger.warning(
|
|
644
645
|
"OpenAI session [%s] exhausted its %d-turn budget without a finish_review call",
|
|
645
646
|
label or "unlabeled",
|
|
646
|
-
|
|
647
|
+
_MAX_TURNS_OPENAI,
|
|
647
648
|
)
|
|
648
649
|
exhausted = True
|
|
649
650
|
|
|
@@ -30,6 +30,7 @@ from claude_agent_sdk import (
|
|
|
30
30
|
AssistantMessage,
|
|
31
31
|
ClaudeAgentOptions,
|
|
32
32
|
ClaudeSDKClient,
|
|
33
|
+
HookMatcher,
|
|
33
34
|
ResultMessage,
|
|
34
35
|
TaskStartedMessage,
|
|
35
36
|
TextBlock,
|
|
@@ -38,6 +39,13 @@ from claude_agent_sdk import (
|
|
|
38
39
|
ToolUseBlock,
|
|
39
40
|
UserMessage,
|
|
40
41
|
)
|
|
42
|
+
from claude_agent_sdk.types import (
|
|
43
|
+
HookContext,
|
|
44
|
+
HookEvent,
|
|
45
|
+
PostToolUseFailureHookSpecificOutput,
|
|
46
|
+
PostToolUseHookSpecificOutput,
|
|
47
|
+
SyncHookJSONOutput,
|
|
48
|
+
)
|
|
41
49
|
from langsmith import traceable
|
|
42
50
|
from langsmith.run_helpers import LangSmithExtra, get_current_run_tree
|
|
43
51
|
|
|
@@ -141,9 +149,10 @@ if _CROSS_CUTTING_MODEL != ALIAS_MAP["claude-opus"]:
|
|
|
141
149
|
_CROSS_CUTTING_MODEL,
|
|
142
150
|
)
|
|
143
151
|
|
|
144
|
-
#
|
|
145
|
-
# reviewers, feedback verifier, blocking validator)
|
|
146
|
-
|
|
152
|
+
# Turn budget for Claude-path sessions (system/specialist/cross-cutting/tests-and-docs
|
|
153
|
+
# reviewers, feedback verifier, blocking validator) -- raised from 30 to 50
|
|
154
|
+
# in TECH-7093.
|
|
155
|
+
_MAX_TURNS_CLAUDE = 50
|
|
147
156
|
|
|
148
157
|
# Nudges the Gemini/OpenAI turn loops (see their own _run_turns) to call
|
|
149
158
|
# finish_review a few turns before the budget runs out.
|
|
@@ -155,8 +164,8 @@ _TURN_BUDGET_NUDGE = (
|
|
|
155
164
|
)
|
|
156
165
|
|
|
157
166
|
# One-sentence turn-budget disclosure appended to the Gemini/OpenAI reviewer
|
|
158
|
-
# system prompts.
|
|
159
|
-
#
|
|
167
|
+
# system prompts. See _TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE below for the
|
|
168
|
+
# Claude path equivalent.
|
|
160
169
|
_TURN_BUDGET_SYSTEM_PROMPT_LINE = (
|
|
161
170
|
"You have at most {max_turns} turns. Call `finish_review` before you run out -- "
|
|
162
171
|
"a review that never calls it is discarded entirely."
|
|
@@ -187,7 +196,7 @@ def _compute_mid_budget_nudge_turn(max_turns: int) -> int | None:
|
|
|
187
196
|
colliding the two nudges onto the same turn -- which would either
|
|
188
197
|
double-post one message, or silently drop the mid-budget one. Not
|
|
189
198
|
reachable at today's values (75 vs. 97 for Gemini's 100-turn budget, 22
|
|
190
|
-
vs. 27 for
|
|
199
|
+
vs. 27 for OpenAI's 30-turn budget) -- only with a much smaller
|
|
191
200
|
``max_turns``.
|
|
192
201
|
"""
|
|
193
202
|
checkpoint_turn = int(max_turns * _MID_BUDGET_NUDGE_FRACTION)
|
|
@@ -197,6 +206,159 @@ def _compute_mid_budget_nudge_turn(max_turns: int) -> int | None:
|
|
|
197
206
|
return checkpoint_turn
|
|
198
207
|
|
|
199
208
|
|
|
209
|
+
# Claude-path equivalent of the disclosure line above. Claude-path reviewers do
|
|
210
|
+
# not have a `finish_review` tool; they finish by emitting their role's final
|
|
211
|
+
# JSON output block.
|
|
212
|
+
_TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE = (
|
|
213
|
+
"You have at most {max_turns} turns. Emit your final JSON output block before you "
|
|
214
|
+
"run out -- a review that never emits it is discarded entirely."
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
# Tool-call budget warning thresholds and truthful messages for Claude-path
|
|
219
|
+
# sessions.
|
|
220
|
+
#
|
|
221
|
+
# In claude-agent-sdk 0.1.81, supported tool-lifecycle hooks that can inject
|
|
222
|
+
# feedback via hookSpecificOutput.additionalContext are PostToolUse and
|
|
223
|
+
# PostToolUseFailure. PostToolBatch is the exact-per-turn future option in
|
|
224
|
+
# Claude Code, but is untyped and unsupported in this SDK version.
|
|
225
|
+
# We attach to PostToolUse and PostToolUseFailure to track combined tool
|
|
226
|
+
# invocations and inject non-blocking convergence/stop nudges.
|
|
227
|
+
#
|
|
228
|
+
# Thresholds are derived from the turn budget constants: the final threshold
|
|
229
|
+
# fires at the emergency checkpoint (_MAX_TURNS_CLAUDE - _NUDGE_TURNS_BEFORE_BUDGET,
|
|
230
|
+
# or 47 under max_turns=50), and the mid threshold fires at the 75% checkpoint
|
|
231
|
+
# (_compute_mid_budget_nudge_turn(_MAX_TURNS_CLAUDE), or 37 under max_turns=50).
|
|
232
|
+
# Because every continuing agent turn has at least one tool call, these thresholds
|
|
233
|
+
# are reachable no later than the corresponding tool-using turns under max_turns.
|
|
234
|
+
# Multiple or parallel tool calls per turn can make the truthful tool-count advisory
|
|
235
|
+
# fire earlier. The copy makes no turns-left claim, stating only the exact number of
|
|
236
|
+
# tool calls made so far.
|
|
237
|
+
def _compute_claude_tool_budget_thresholds(max_turns: int) -> tuple[int, int]:
|
|
238
|
+
"""Compute and validate Claude tool-budget nudge thresholds (mid, final).
|
|
239
|
+
|
|
240
|
+
Returns:
|
|
241
|
+
tuple[int, int]: (mid_threshold, final_threshold) where mid_threshold
|
|
242
|
+
strictly precedes final_threshold.
|
|
243
|
+
|
|
244
|
+
Raises:
|
|
245
|
+
ValueError: If max_turns produces a mid-budget checkpoint that does not
|
|
246
|
+
strictly precede the final threshold (e.g. colliding or inverted thresholds
|
|
247
|
+
from a small budget).
|
|
248
|
+
"""
|
|
249
|
+
final_threshold = max_turns - _NUDGE_TURNS_BEFORE_BUDGET
|
|
250
|
+
mid_threshold = _compute_mid_budget_nudge_turn(max_turns)
|
|
251
|
+
if mid_threshold is None:
|
|
252
|
+
raise ValueError(
|
|
253
|
+
f"Invalid max_turns={max_turns}: mid-budget checkpoint must precede "
|
|
254
|
+
f"the final budget nudge threshold ({final_threshold})"
|
|
255
|
+
)
|
|
256
|
+
return mid_threshold, final_threshold
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
_TOOL_BUDGET_MID_THRESHOLD: int
|
|
260
|
+
_TOOL_BUDGET_FINAL_THRESHOLD: int
|
|
261
|
+
_TOOL_BUDGET_MID_THRESHOLD, _TOOL_BUDGET_FINAL_THRESHOLD = _compute_claude_tool_budget_thresholds(
|
|
262
|
+
_MAX_TURNS_CLAUDE
|
|
263
|
+
)
|
|
264
|
+
|
|
265
|
+
_TOOL_BUDGET_MID_NUDGE = (
|
|
266
|
+
"You have now made {n} tool calls. If you have gathered enough "
|
|
267
|
+
"context to identify findings, begin converging toward your final JSON output block "
|
|
268
|
+
"now rather than continuing to explore."
|
|
269
|
+
)
|
|
270
|
+
_TOOL_BUDGET_FINAL_NUDGE = (
|
|
271
|
+
"You have now made {n} tool calls. Stop reading and searching now, and emit "
|
|
272
|
+
"your final JSON output block with whatever you have found so far. If you have "
|
|
273
|
+
"found nothing, emit your final JSON output block with an empty findings list."
|
|
274
|
+
)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def _make_tool_budget_nudge_hooks(
|
|
278
|
+
label: str | None = None,
|
|
279
|
+
*,
|
|
280
|
+
mid_threshold: int = _TOOL_BUDGET_MID_THRESHOLD,
|
|
281
|
+
final_threshold: int = _TOOL_BUDGET_FINAL_THRESHOLD,
|
|
282
|
+
) -> dict[HookEvent, list[HookMatcher]]:
|
|
283
|
+
"""Create PostToolUse and PostToolUseFailure hooks that inject non-blocking
|
|
284
|
+
nudge warnings into Claude-path sessions when tool-call counts cross budget thresholds.
|
|
285
|
+
|
|
286
|
+
In claude-agent-sdk 0.1.81, supported tool-lifecycle hooks that can inject
|
|
287
|
+
feedback via hookSpecificOutput.additionalContext are PostToolUse and
|
|
288
|
+
PostToolUseFailure. PostToolBatch is the exact-per-turn future option in
|
|
289
|
+
Claude Code, but is untyped and unsupported in this SDK version.
|
|
290
|
+
|
|
291
|
+
One callback is shared by both events, closing over combined tool-call count
|
|
292
|
+
and one-shot flags. Subagent tool calls (bearing agent_id) are skipped.
|
|
293
|
+
"""
|
|
294
|
+
tool_call_count = 0
|
|
295
|
+
mid_nudge_sent = False
|
|
296
|
+
final_nudge_sent = False
|
|
297
|
+
|
|
298
|
+
async def _hook_callback(
|
|
299
|
+
input_data: Any,
|
|
300
|
+
tool_use_id: str | None,
|
|
301
|
+
context: HookContext,
|
|
302
|
+
) -> SyncHookJSONOutput:
|
|
303
|
+
nonlocal tool_call_count, mid_nudge_sent, final_nudge_sent
|
|
304
|
+
try:
|
|
305
|
+
if not isinstance(input_data, dict):
|
|
306
|
+
return {}
|
|
307
|
+
# Sub-agent attribution: skip Task-spawned subagents
|
|
308
|
+
if input_data.get("agent_id"):
|
|
309
|
+
return {}
|
|
310
|
+
|
|
311
|
+
tool_call_count += 1
|
|
312
|
+
nudge_message: str | None = None
|
|
313
|
+
|
|
314
|
+
if tool_call_count >= final_threshold and not final_nudge_sent:
|
|
315
|
+
final_nudge_sent = True
|
|
316
|
+
mid_nudge_sent = True
|
|
317
|
+
nudge_message = _TOOL_BUDGET_FINAL_NUDGE.format(n=tool_call_count)
|
|
318
|
+
elif tool_call_count >= mid_threshold and not mid_nudge_sent:
|
|
319
|
+
mid_nudge_sent = True
|
|
320
|
+
nudge_message = _TOOL_BUDGET_MID_NUDGE.format(n=tool_call_count)
|
|
321
|
+
|
|
322
|
+
if nudge_message is None:
|
|
323
|
+
return {}
|
|
324
|
+
|
|
325
|
+
event_name = input_data.get("hook_event_name")
|
|
326
|
+
logger.info(
|
|
327
|
+
"Injected Claude tool-budget nudge [%s] at tool call %d (event=%s)",
|
|
328
|
+
label or "unlabeled",
|
|
329
|
+
tool_call_count,
|
|
330
|
+
event_name,
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
# Mirror hook_event_name in hookSpecificOutput
|
|
334
|
+
if event_name == "PostToolUseFailure":
|
|
335
|
+
failure_output: PostToolUseFailureHookSpecificOutput = {
|
|
336
|
+
"hookEventName": "PostToolUseFailure",
|
|
337
|
+
"additionalContext": nudge_message,
|
|
338
|
+
}
|
|
339
|
+
return {"hookSpecificOutput": failure_output}
|
|
340
|
+
else:
|
|
341
|
+
success_output: PostToolUseHookSpecificOutput = {
|
|
342
|
+
"hookEventName": "PostToolUse",
|
|
343
|
+
"additionalContext": nudge_message,
|
|
344
|
+
}
|
|
345
|
+
return {"hookSpecificOutput": success_output}
|
|
346
|
+
|
|
347
|
+
except Exception:
|
|
348
|
+
logger.warning(
|
|
349
|
+
"Tool budget hook callback failed for [%s]",
|
|
350
|
+
label or "unlabeled",
|
|
351
|
+
exc_info=True,
|
|
352
|
+
)
|
|
353
|
+
return {}
|
|
354
|
+
|
|
355
|
+
matcher = HookMatcher(hooks=[_hook_callback])
|
|
356
|
+
return {
|
|
357
|
+
"PostToolUse": [matcher],
|
|
358
|
+
"PostToolUseFailure": [matcher],
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
|
|
200
362
|
# Fallback repo root for ClaudeSDKClient cwd — used when no SHA-pinned
|
|
201
363
|
# worktree has been provisioned (e.g. local dev runs, tests, subprocess
|
|
202
364
|
# worker). In production, callers pass an explicit repo_root provisioned
|
|
@@ -1697,15 +1859,22 @@ async def _run_claude_session(
|
|
|
1697
1859
|
betas: list[Literal["context-1m-2025-08-07"]] = (
|
|
1698
1860
|
["context-1m-2025-08-07"] if _attach_1m_context_beta else []
|
|
1699
1861
|
)
|
|
1862
|
+
system_prompt = (
|
|
1863
|
+
system_prompt
|
|
1864
|
+
+ "\n\n"
|
|
1865
|
+
+ _TURN_BUDGET_SYSTEM_PROMPT_LINE_CLAUDE.format(max_turns=_MAX_TURNS_CLAUDE)
|
|
1866
|
+
)
|
|
1867
|
+
hooks = _make_tool_budget_nudge_hooks(label)
|
|
1700
1868
|
options = ClaudeAgentOptions(
|
|
1701
1869
|
cwd=effective_root,
|
|
1702
1870
|
allowed_tools=["Read", "Glob", "Grep"] + context7_tools,
|
|
1703
1871
|
mcp_servers=mcp_servers,
|
|
1704
1872
|
strict_mcp_config=True,
|
|
1705
1873
|
permission_mode="default",
|
|
1874
|
+
hooks=hooks,
|
|
1706
1875
|
model=model,
|
|
1707
1876
|
system_prompt=system_prompt,
|
|
1708
|
-
max_turns=
|
|
1877
|
+
max_turns=_MAX_TURNS_CLAUDE,
|
|
1709
1878
|
env=dict([settings.anthropic_credential]),
|
|
1710
1879
|
stderr=_stderr_handler,
|
|
1711
1880
|
betas=betas,
|
|
@@ -1743,6 +1912,11 @@ async def _run_claude_session(
|
|
|
1743
1912
|
result_text = ""
|
|
1744
1913
|
cost_usd = 0.0
|
|
1745
1914
|
tool_calls: list[str] = []
|
|
1915
|
+
# Note: message_index counts AssistantMessage objects received from the SDK
|
|
1916
|
+
# stream, NOT CLI turns. Multiple AssistantMessage objects can occur within
|
|
1917
|
+
# a single turn or sub-step. It must never be used as a turn counter.
|
|
1918
|
+
# Key and log names ('msg_index', 'msg=%d') are preserved for log
|
|
1919
|
+
# compatibility.
|
|
1746
1920
|
message_index = 0
|
|
1747
1921
|
failure_reason: Literal["turn_budget_exhausted"] | None = None
|
|
1748
1922
|
async for message in client.receive_response():
|
|
@@ -1842,7 +2016,7 @@ async def _run_claude_session(
|
|
|
1842
2016
|
logger.warning(
|
|
1843
2017
|
"Agent [%s] exhausted its %d-turn budget (subtype=%s)",
|
|
1844
2018
|
label or "unlabeled",
|
|
1845
|
-
|
|
2019
|
+
_MAX_TURNS_CLAUDE,
|
|
1846
2020
|
message.subtype,
|
|
1847
2021
|
)
|
|
1848
2022
|
failure_reason = "turn_budget_exhausted"
|
|
@@ -3,22 +3,20 @@
|
|
|
3
3
|
Every scenario in this file exercises the seven logical operations defined
|
|
4
4
|
by ``argus/storage/sql.py`` (the Postgres canonical writer) purely through
|
|
5
5
|
their Pydantic input/output shapes — no backend-specific assertions. The
|
|
6
|
-
goal is a single suite that
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
6
|
+
goal is a single suite that runs against any backend implementing the same
|
|
7
|
+
operations:
|
|
8
|
+
|
|
9
|
+
- **sqlite**: fully supported locally (:class:`argus.storage.sqlite.SqliteHistoryBackend`).
|
|
10
|
+
- **http**: wired via ``pytest-httpx`` standing in for the two-endpoint HTTP
|
|
11
|
+
storage contract (see ``docs/STORAGE.md``). Operations not implemented by
|
|
12
|
+
the minimal HTTP shim (e.g. status polling, recent-rounds listing,
|
|
13
|
+
select_recent_lite_rounds, insert_agent_runs) are documented and marked xfail.
|
|
14
|
+
Completed-round persistence and latest-round reads (including for lite-mode
|
|
15
|
+
reviews) are fully supported and verified.
|
|
14
16
|
- **postgres**: needs a live database and ``argus.storage.session`` wired
|
|
15
|
-
up.
|
|
16
|
-
the ``integration`` marker
|
|
17
|
+
up. Guarded with ``ARGUS_DB_URL`` / ``SUPABASE_DB_URL`` env-var presence and
|
|
18
|
+
the ``integration`` marker, mirroring the pattern in
|
|
17
19
|
``tests/storage/test_sql.py``.
|
|
18
|
-
- **http**: needs a ``pytest-httpx`` fixture standing in for your configured
|
|
19
|
-
HTTP backend's endpoints (see ``docs/STORAGE.md`` for the contract). Once
|
|
20
|
-
wired, ``HttpStorageClient`` would sit behind the same seven-operation
|
|
21
|
-
surface as a thin adapter.
|
|
22
20
|
|
|
23
21
|
Scenario coverage:
|
|
24
22
|
|
|
@@ -103,6 +101,9 @@ _HTTP_UNSUPPORTED_TESTS = {
|
|
|
103
101
|
# select_recent_rounds -- not exposed over HTTP.
|
|
104
102
|
"test_select_recent_rounds_orders_desc_and_respects_limit",
|
|
105
103
|
"test_repeated_running_upsert_same_flow_run_id_does_not_duplicate",
|
|
104
|
+
# insert_agent_runs -- documented no-op on HTTP path (analytics inserts
|
|
105
|
+
# not covered by minimal HTTP shim, TECH-3487 follow-up).
|
|
106
|
+
"test_insert_agent_runs_batch",
|
|
106
107
|
# select_recent_lite_rounds -- documented no-op (always returns []).
|
|
107
108
|
"test_select_recent_lite_rounds_filters_reviewer_version",
|
|
108
109
|
"test_select_recent_lite_rounds_default_limit_is_200",
|
|
@@ -112,6 +113,14 @@ _HTTP_UNSUPPORTED_TESTS = {
|
|
|
112
113
|
def _mark_known_http_gaps(request: pytest.FixtureRequest) -> None:
|
|
113
114
|
base_name = request.node.name.split("[")[0]
|
|
114
115
|
if base_name in _HTTP_UNSUPPORTED_TESTS:
|
|
116
|
+
# insert_agent_runs is a documented no-op on the HTTP shim (analytics
|
|
117
|
+
# inserts not covered, TECH-3487 follow-up); the call does not assert
|
|
118
|
+
# rows or raise, so xfail directly rather than letting it pass vacuously.
|
|
119
|
+
if base_name == "test_insert_agent_runs_batch":
|
|
120
|
+
pytest.xfail(
|
|
121
|
+
"HTTP storage shim does not support agent_runs analytics inserts "
|
|
122
|
+
"(documented Phase-1 gap, TECH-3487)"
|
|
123
|
+
)
|
|
115
124
|
request.node.add_marker(
|
|
116
125
|
pytest.mark.xfail(
|
|
117
126
|
reason=(
|
|
@@ -202,10 +211,10 @@ async def backend(
|
|
|
202
211
|
return
|
|
203
212
|
|
|
204
213
|
if backend_kind == "http":
|
|
214
|
+
_mark_known_http_gaps(request)
|
|
205
215
|
httpx_mock = request.getfixturevalue("httpx_mock")
|
|
206
216
|
httpx_mock.add_callback(_FakeHttpStorageBackend().handle, is_reusable=True)
|
|
207
217
|
install_http_storage(read_url=_HTTP_READ_URL, write_url=_HTTP_WRITE_URL)
|
|
208
|
-
_mark_known_http_gaps(request)
|
|
209
218
|
try:
|
|
210
219
|
yield HttpHistoryBackend()
|
|
211
220
|
finally:
|
|
@@ -317,6 +326,79 @@ async def test_upsert_completed_row_defaults_reviewer_version_and_stage(
|
|
|
317
326
|
assert written.verdict is None
|
|
318
327
|
|
|
319
328
|
|
|
329
|
+
async def test_lite_round_completed_upsert_round_trips_through_latest(
|
|
330
|
+
backend: HistoryBackend,
|
|
331
|
+
) -> None:
|
|
332
|
+
"""Lite-mode rounds (reviewer_version='v3-lite') must persist via
|
|
333
|
+
upsert_completed_row and round-trip through select_latest_completed_round
|
|
334
|
+
on all backends (including the HTTP storage shim).
|
|
335
|
+
"""
|
|
336
|
+
payload = CodeReviewRoundIn(
|
|
337
|
+
flow_run_id="fr-lite-rt-1",
|
|
338
|
+
repo="org/repo-lite",
|
|
339
|
+
pr_number=10,
|
|
340
|
+
verdict="approve",
|
|
341
|
+
risk_level="low",
|
|
342
|
+
blocking_count=0,
|
|
343
|
+
suggestion_count=0,
|
|
344
|
+
review_comment="lgtm lite review",
|
|
345
|
+
reviewer_version="v3-lite",
|
|
346
|
+
result_json={"review_round": 2, "preflight_reason": "test-only diff"},
|
|
347
|
+
sha="deadbeef123",
|
|
348
|
+
base_ref="main",
|
|
349
|
+
)
|
|
350
|
+
|
|
351
|
+
written = await backend.upsert_completed_row(row=payload)
|
|
352
|
+
assert written.reviewer_version == "v3-lite"
|
|
353
|
+
|
|
354
|
+
latest = await backend.select_latest_completed_round(repo="org/repo-lite", pr_number=10)
|
|
355
|
+
assert latest is not None
|
|
356
|
+
assert latest.id == written.id
|
|
357
|
+
assert latest.reviewer_version == "v3-lite"
|
|
358
|
+
assert latest.verdict == "approve"
|
|
359
|
+
assert latest.sha == "deadbeef123"
|
|
360
|
+
assert latest.result_json == {"review_round": 2, "preflight_reason": "test-only diff"}
|
|
361
|
+
assert latest.prior_count == 1
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
async def test_lite_round_becomes_latest_completed_round(
|
|
365
|
+
backend: HistoryBackend,
|
|
366
|
+
) -> None:
|
|
367
|
+
"""When a lite review follows a full review on the same PR, the lite
|
|
368
|
+
round becomes the latest completed round and preserves its version.
|
|
369
|
+
"""
|
|
370
|
+
first_full = await backend.upsert_completed_row(
|
|
371
|
+
row=CodeReviewRoundIn(
|
|
372
|
+
flow_run_id="fr-seq-1",
|
|
373
|
+
repo="org/repo-seq",
|
|
374
|
+
pr_number=20,
|
|
375
|
+
verdict="block",
|
|
376
|
+
reviewer_version="v3",
|
|
377
|
+
sha="sha-round-1",
|
|
378
|
+
)
|
|
379
|
+
)
|
|
380
|
+
second_lite = await backend.upsert_completed_row(
|
|
381
|
+
row=CodeReviewRoundIn(
|
|
382
|
+
flow_run_id="fr-seq-2",
|
|
383
|
+
repo="org/repo-seq",
|
|
384
|
+
pr_number=20,
|
|
385
|
+
verdict="approve",
|
|
386
|
+
reviewer_version="v3-lite",
|
|
387
|
+
sha="sha-round-2",
|
|
388
|
+
result_json={"review_round": 2, "preflight_reason": "catchup on green CI"},
|
|
389
|
+
)
|
|
390
|
+
)
|
|
391
|
+
|
|
392
|
+
latest = await backend.select_latest_completed_round(repo="org/repo-seq", pr_number=20)
|
|
393
|
+
assert latest is not None
|
|
394
|
+
assert latest.id == second_lite.id
|
|
395
|
+
assert latest.id != first_full.id
|
|
396
|
+
assert latest.reviewer_version == "v3-lite"
|
|
397
|
+
assert latest.verdict == "approve"
|
|
398
|
+
assert latest.sha == "sha-round-2"
|
|
399
|
+
assert latest.prior_count == 2
|
|
400
|
+
|
|
401
|
+
|
|
320
402
|
# ---------------------------------------------------------------------------
|
|
321
403
|
# 3. Two completed rounds -> latest wins; select_recent_rounds ordering + limit
|
|
322
404
|
# ---------------------------------------------------------------------------
|
|
@@ -471,7 +553,7 @@ async def test_completed_upsert_same_flow_run_id_preserves_sha_via_coalesce(
|
|
|
471
553
|
# ---------------------------------------------------------------------------
|
|
472
554
|
|
|
473
555
|
|
|
474
|
-
async def test_insert_agent_runs_batch(backend:
|
|
556
|
+
async def test_insert_agent_runs_batch(backend: HistoryBackend) -> None:
|
|
475
557
|
review = await backend.upsert_completed_row(
|
|
476
558
|
row=CodeReviewRoundIn(repo="org/repo6", pr_number=2, verdict="approve")
|
|
477
559
|
)
|