argus-code-review 0.2.4__tar.gz → 0.2.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/CHANGELOG.md +82 -1
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/PKG-INFO +2 -2
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/config.py +3 -2
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/gemini_runner.py +36 -8
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/graph.py +112 -79
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/helpers.py +102 -16
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/output_models.py +6 -0
- argus_code_review-0.2.6/argus/llm/usage.py +117 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/models.py +8 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/openai_runner.py +26 -7
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/pipeline_models.py +14 -13
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/engine.py +41 -7
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/runners.py +70 -2
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/sql.py +3 -3
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/sqlite.py +76 -28
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/pyproject.toml +1 -1
- argus_code_review-0.2.6/schema/018_widen_agent_runs_failure_reason.sql +36 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/golden/review_response.schema.json +16 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_sql.py +9 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_sqlite_backend.py +198 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_gemini_runner.py +205 -7
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_get_llm_temperature.py +7 -7
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_precheck.py +25 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_timeout_surfacing.py +80 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_openai_client.py +38 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_openai_runner.py +190 -6
- argus_code_review-0.2.6/tests/test_output_models.py +188 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_engine.py +55 -6
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_context_usage.py +47 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_helpers.py +113 -2
- argus_code_review-0.2.6/tests/test_stage_costs.py +141 -0
- argus_code_review-0.2.4/tests/test_lite_review_cost.py +0 -56
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/.gitignore +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/LICENSE +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/README.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/bench.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/bench_default.toml +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/cli.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/coverage.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/dotenv_utils.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/gemini_cache.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/github_client.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/models.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/pricing.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/openai_client.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/actions_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/.gitignore +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/README.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/package-lock.json +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/package.json +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/js_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/migration_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/rules/README.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/sarif.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/scanner_utils.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/secrets_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/shadow.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/terraform_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/workflow_lint_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-blocking-validator.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-coverage-check.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-cross-cutting.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-feedback-verifier.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-lite.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-planner.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-preflight-router.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-prior-art.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-deployment.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-frontend.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-infra.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-observability.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-security.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-sql.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-subagent.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-tests-and-docs.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-writer.md +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts_runtime.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/repo_provision.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/review_tools.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/http.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/models.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/precheck.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/resolver.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/session.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/008_add_code_reviews.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/009_add_reviewer_version.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/010_add_review_patterns.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/011_add_review_progress_columns.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/015_create_agent_runs.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/016_add_agent_runs_failure_reason.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/017_add_precheck_rules.sql +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/conftest.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/__init__.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_backend_contract.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_http.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_models.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_resolver.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_session.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_actions_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_argus_review_local.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_bench.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_bench_config_guard.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_catchup_gate.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_args.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_output_contract.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_post_review.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_preflight.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_prompts.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_config.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_conftest.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_gemini_cache.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_github_client_checks_signal.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_github_client_write.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_fetch_diff.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_http_guards.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_preflight_image_bump.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_progress.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_storage_resolution.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_js_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_models.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_models_override.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_pricing.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_migration_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_models.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_multi_round.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_packaging.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_plan_review.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_engine_integration.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_sarif.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_shadow.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_shadow_integration.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_prompts.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_repo_provision.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_review_patterns_integration.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_review_tools.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_context7.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_new.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_scanner_utils.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_secrets_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_specialist_validation.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_storage_precheck.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_terraform_scanner.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_terraform_scanner_integration.py +0 -0
- {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_workflow_lint_scanner.py +0 -0
|
@@ -7,6 +7,85 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.2.6] - 2026-09-19
|
|
11
|
+
|
|
12
|
+
### Fixed
|
|
13
|
+
|
|
14
|
+
- Bounded `openai` dependency to `>=1.66.0,<2` in `pyproject.toml` (TECH-6590) to
|
|
15
|
+
prevent unconstrained fresh installs from resolving to a future breaking major
|
|
16
|
+
release. The floor was also tightened from `>=1.50.0` to `>=1.66.0` where the
|
|
17
|
+
OpenAI Responses API (`client.responses.create()`) used by this codebase was
|
|
18
|
+
introduced. (Note: this is a general hygiene fix; it does not address the root
|
|
19
|
+
cause of the TECH-6590-reported crash, which was a separate shared-venv version-skew
|
|
20
|
+
issue tracked in TECH-6592.)
|
|
21
|
+
|
|
22
|
+
## [0.2.5] - 2026-09-18
|
|
23
|
+
|
|
24
|
+
### Added
|
|
25
|
+
|
|
26
|
+
- `failure_reason="turn_budget_exhausted"` (schema 018) on the Gemini and OpenAI
|
|
27
|
+
runner paths (#24): a reviewer session that runs out of turns without ever
|
|
28
|
+
calling `finish_review` is now flagged as a failure instead of returning
|
|
29
|
+
silently as a clean, zero-findings review. Across a six-PR replay, 17 of 52
|
|
30
|
+
reviewer sessions exhausted their budget this way, costing ~$18.69 (38% of
|
|
31
|
+
total spend) for no usable output — none of it previously visible.
|
|
32
|
+
- A finish-now nudge injected 3 turns before the turn-budget ceiling on the
|
|
33
|
+
Gemini and OpenAI loops, plus a one-line budget disclosure added to their
|
|
34
|
+
system prompts (#24), so a session that's about to exhaust is told to
|
|
35
|
+
converge instead of continuing to explore.
|
|
36
|
+
- A 75% mid-budget checkpoint nudge on the same two loops (TECH-6560),
|
|
37
|
+
independent of and in addition to the existing end-of-budget nudge above: a
|
|
38
|
+
softer "you're roughly 75% through your turn budget, start converging if
|
|
39
|
+
you have enough" check-in, fired early enough not to discourage legitimate
|
|
40
|
+
exploration on large diffs, with a full turn gap left before the emergency
|
|
41
|
+
nudge (turn 75 vs. turn 97 on Gemini's 100-turn budget; turn 22 vs. turn 27
|
|
42
|
+
on OpenAI's 30-turn budget). Not applied to the Claude Agent SDK path,
|
|
43
|
+
which has no documented mid-stream prompt-injection hook.
|
|
44
|
+
- Per-stage cost and duration ledger (#24), priced through the existing
|
|
45
|
+
litellm pricing table and surfaced on `ReviewResponse.stage_costs` /
|
|
46
|
+
`stage_seconds`. Previously only reviewer-agent cost was tracked; the
|
|
47
|
+
planner, preflight, coverage check, and writer stages each reported $0.00,
|
|
48
|
+
together ~6% of real spend (the planner alone 4.9%).
|
|
49
|
+
- Missing/uninstalled deterministic precheck scanners are now surfaced as
|
|
50
|
+
degraded coverage (#24), distinct from a scanner that ran and found
|
|
51
|
+
nothing. Four of six configured scanner binaries were absent in every
|
|
52
|
+
measured production run, so a review could previously report clean having
|
|
53
|
+
never actually scanned for secrets or destructive migrations.
|
|
54
|
+
|
|
55
|
+
### Changed
|
|
56
|
+
|
|
57
|
+
- Raised the Gemini reviewer's turn budget from 45 to 100 (TECH-6558),
|
|
58
|
+
`argus/gemini_runner.py`'s `_MAX_TURNS_GEMINI`. Gemini sessions were still
|
|
59
|
+
exhausting the 45-turn budget introduced in 0.2.4 even with the new
|
|
60
|
+
finish-now nudge above; raising the ceiling is a cheap mitigation to try
|
|
61
|
+
independent of the nudge, not a replacement for it — watch exhaustion
|
|
62
|
+
rates on real rounds to confirm it helps before assuming it does.
|
|
63
|
+
- A failed reviewer's coverage-gap finding is now promoted to BLOCKING when
|
|
64
|
+
nothing else in the round already blocks (#24), so a round can no longer
|
|
65
|
+
APPROVE on coverage it knows is degraded (a failed reviewer contributes
|
|
66
|
+
zero findings, which otherwise looks identical to a clean review).
|
|
67
|
+
- The planner is now instructed to cap system-group size (#24): turn-budget
|
|
68
|
+
exhaustion clustered on the largest planner groups in the replayed sample.
|
|
69
|
+
|
|
70
|
+
### Fixed
|
|
71
|
+
|
|
72
|
+
- A table rebuild in the SQLite storage backend that widens
|
|
73
|
+
`agent_runs.failure_reason` copied rows with `SELECT *`, pairing columns
|
|
74
|
+
positionally; a database that gained the column via `ALTER TABLE` has it
|
|
75
|
+
last, while the DDL declares it mid-table, silently shifting a datetime
|
|
76
|
+
into `failure_reason` and tripping its CHECK constraint on startup (#24).
|
|
77
|
+
Copies by explicit column name now, with a regression test.
|
|
78
|
+
- semgrep was scheduled in the precheck engine without an availability
|
|
79
|
+
check, so an absent binary never reached the new `missing_scanners`
|
|
80
|
+
reporting above — the exact gap that feature exists to close, for every
|
|
81
|
+
other scanner (#24).
|
|
82
|
+
- `run_lite_review`'s extraction cost went unrecorded once the new per-stage
|
|
83
|
+
ledger became the only cost source, underreporting every lite review (#24).
|
|
84
|
+
- Per-stage durations were measured from handler construction rather than
|
|
85
|
+
the call itself, unpriced models were silently costed as $0, and a
|
|
86
|
+
reviewer failure skipped its risk-level bump when a precheck failure had
|
|
87
|
+
already forced BLOCKING (#24).
|
|
88
|
+
|
|
10
89
|
## [0.2.4] - 2026-09-17
|
|
11
90
|
|
|
12
91
|
### Changed
|
|
@@ -259,7 +338,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
259
338
|
packaged set.
|
|
260
339
|
- `argus --version`, `argus prompts list`, and `argus prompts export`.
|
|
261
340
|
|
|
262
|
-
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.
|
|
341
|
+
[Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...HEAD
|
|
342
|
+
[0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
|
|
343
|
+
[0.2.5]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...v0.2.5
|
|
263
344
|
[0.2.4]: https://github.com/redesignhealth/argus-review/compare/v0.2.3...v0.2.4
|
|
264
345
|
[0.2.3]: https://github.com/redesignhealth/argus-review/compare/v0.2.2...v0.2.3
|
|
265
346
|
[0.2.2]: https://github.com/redesignhealth/argus-review/compare/v0.2.1...v0.2.2
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: argus-code-review
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.6
|
|
4
4
|
Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
|
|
5
5
|
Project-URL: Repository, https://github.com/redesignhealth/argus-review
|
|
6
6
|
Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
|
|
@@ -26,7 +26,7 @@ Requires-Dist: langgraph-checkpoint-sqlite>=2.0.0
|
|
|
26
26
|
Requires-Dist: langgraph<2,>=1.0.10
|
|
27
27
|
Requires-Dist: langsmith<1,>=0.2.0
|
|
28
28
|
Requires-Dist: litellm<2.0.0,>=1.50.0
|
|
29
|
-
Requires-Dist: openai
|
|
29
|
+
Requires-Dist: openai<2,>=1.66.0
|
|
30
30
|
Requires-Dist: psycopg-pool>=3.2.0
|
|
31
31
|
Requires-Dist: psycopg[binary]>=3.2.0
|
|
32
32
|
Requires-Dist: pydantic-settings>=2.0.0
|
|
@@ -70,8 +70,9 @@ class Settings(BaseSettings):
|
|
|
70
70
|
See ``docs/PRECHECKS.md``'s "stock rule sources" section.
|
|
71
71
|
ARGUS_PRECHECK_BLOCK_ON_SCANNER_FAILURE: Set truthy to force the
|
|
72
72
|
verdict to BLOCKING whenever a deterministic precheck scanner
|
|
73
|
-
crashed/timed out/errored this round
|
|
74
|
-
failed_scanners``
|
|
73
|
+
crashed/timed out/errored, OR was never installed, this round
|
|
74
|
+
(``PrecheckResult.failed_scanners``/``missing_scanners``
|
|
75
|
+
non-empty) instead of the default fail-open
|
|
75
76
|
behavior (surface it in the review comment's degraded-coverage
|
|
76
77
|
section — see ``argus.helpers.build_degraded_coverage_labels``
|
|
77
78
|
— but let the review's own verdict stand on its own merits).
|
|
@@ -159,14 +159,19 @@ from argus.gemini_cache import GeminiCacheKeeper
|
|
|
159
159
|
from argus.llm.models import estimate_cost_usd
|
|
160
160
|
from argus.llm.models import resolve as resolve_model_alias
|
|
161
161
|
from argus.runners import (
|
|
162
|
+
_MID_BUDGET_NUDGE,
|
|
163
|
+
_NUDGE_TURNS_BEFORE_BUDGET,
|
|
162
164
|
_SUBPROCESS_TIMEOUT_S,
|
|
165
|
+
_TURN_BUDGET_NUDGE,
|
|
166
|
+
_TURN_BUDGET_SYSTEM_PROMPT_LINE,
|
|
163
167
|
SessionResult,
|
|
168
|
+
_compute_mid_budget_nudge_turn,
|
|
164
169
|
_resolve_repo_root,
|
|
165
170
|
)
|
|
166
171
|
|
|
167
172
|
logger = logging.getLogger(__name__)
|
|
168
173
|
|
|
169
|
-
_MAX_TURNS_GEMINI =
|
|
174
|
+
_MAX_TURNS_GEMINI = 100 # raised from 45 -- see TECH-6558
|
|
170
175
|
|
|
171
176
|
# Both "auto" and "on" attempt explicit caching today -- there is no
|
|
172
177
|
# separate heuristic distinguishing them yet (Track 1 defined the
|
|
@@ -594,10 +599,15 @@ async def _run_turns(
|
|
|
594
599
|
timeout_s: float,
|
|
595
600
|
started_at: datetime,
|
|
596
601
|
) -> SessionResult:
|
|
597
|
-
"""The actual tool-calling loop.
|
|
598
|
-
|
|
599
|
-
|
|
602
|
+
"""The actual tool-calling loop. Returns ``failure_reason="turn_budget_exhausted"``
|
|
603
|
+
when the loop ran out of turns without a ``finish_review`` call, else
|
|
604
|
+
``None`` -- the enclosing ``run_session_gemini`` is what applies the
|
|
605
|
+
timeout and turns a cancellation into a ``failure_reason="timeout"``
|
|
606
|
+
result instead.
|
|
600
607
|
"""
|
|
608
|
+
system_prompt = (
|
|
609
|
+
system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_GEMINI)
|
|
610
|
+
)
|
|
601
611
|
_, api_key = settings.google_credential
|
|
602
612
|
# Second layer of defense: `asyncio.wait_for` in `run_session_gemini`
|
|
603
613
|
# cannot interrupt an in-flight blocking SDK call once it has started
|
|
@@ -682,14 +692,32 @@ async def _run_turns(
|
|
|
682
692
|
|
|
683
693
|
tool_calls: list[str] = []
|
|
684
694
|
files_explored: list[str] = []
|
|
695
|
+
exhausted = False
|
|
685
696
|
usage_prompt_total = 0
|
|
686
697
|
usage_candidates_total = 0
|
|
687
698
|
usage_cached_total = 0
|
|
688
699
|
usage_thoughts_total = 0
|
|
689
700
|
usage_tool_use_prompt_total = 0
|
|
690
701
|
|
|
702
|
+
_mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_GEMINI)
|
|
703
|
+
|
|
691
704
|
with review_tools.review_session(repo_root) as findings_sink:
|
|
692
705
|
for _turn in range(_MAX_TURNS_GEMINI):
|
|
706
|
+
if (
|
|
707
|
+
_turn == _MAX_TURNS_GEMINI - _NUDGE_TURNS_BEFORE_BUDGET
|
|
708
|
+
and contents[-1].parts is not None
|
|
709
|
+
):
|
|
710
|
+
contents[-1].parts.append(types.Part.from_text(text=_TURN_BUDGET_NUDGE))
|
|
711
|
+
# Not an `elif` -- these are two independent checkpoints at
|
|
712
|
+
# different turns and must both be able to fire in the same
|
|
713
|
+
# session (see _compute_mid_budget_nudge_turn's docstring
|
|
714
|
+
# for why they can never land on the same turn today).
|
|
715
|
+
if (
|
|
716
|
+
_mid_budget_nudge_turn is not None
|
|
717
|
+
and _turn == _mid_budget_nudge_turn
|
|
718
|
+
and contents[-1].parts is not None
|
|
719
|
+
):
|
|
720
|
+
contents[-1].parts.append(types.Part.from_text(text=_MID_BUDGET_NUDGE))
|
|
693
721
|
try:
|
|
694
722
|
response = await client.aio.models.generate_content(
|
|
695
723
|
model=model, contents=contents, config=config
|
|
@@ -819,14 +847,14 @@ async def _run_turns(
|
|
|
819
847
|
if finished:
|
|
820
848
|
break
|
|
821
849
|
else:
|
|
822
|
-
# `for...else`: only reached if every
|
|
823
|
-
#
|
|
824
|
-
# finish_review -- i.e. the turn budget was exhausted.
|
|
850
|
+
# `for...else`: only reached if every iteration executed a
|
|
851
|
+
# function call and none was finish_review, i.e. exhausted.
|
|
825
852
|
logger.warning(
|
|
826
853
|
"Gemini session [%s] exhausted its %d-turn budget without a finish_review call",
|
|
827
854
|
label or "unlabeled",
|
|
828
855
|
_MAX_TURNS_GEMINI,
|
|
829
856
|
)
|
|
857
|
+
exhausted = True
|
|
830
858
|
|
|
831
859
|
result_text = _build_result_text(findings_sink, files_explored)
|
|
832
860
|
|
|
@@ -874,7 +902,7 @@ async def _run_turns(
|
|
|
874
902
|
# docstring) -- always 0, never derived from tool_calls.
|
|
875
903
|
context7_call_count=0,
|
|
876
904
|
model=model,
|
|
877
|
-
failure_reason=None,
|
|
905
|
+
failure_reason="turn_budget_exhausted" if exhausted else None,
|
|
878
906
|
)
|
|
879
907
|
finally:
|
|
880
908
|
# Every session creates both a sync and async HTTP client (the sync
|
|
@@ -44,7 +44,14 @@ from argus.helpers import (
|
|
|
44
44
|
compute_persisted_finding_counts,
|
|
45
45
|
)
|
|
46
46
|
from argus.llm.models import ALIAS_MAP, CLAUDE_DEFAULT, CLAUDE_FRONTIER, CLAUDE_MINI
|
|
47
|
-
from argus.llm.
|
|
47
|
+
from argus.llm.usage import (
|
|
48
|
+
StageCostCallbackHandler,
|
|
49
|
+
price_openai_usage,
|
|
50
|
+
record_stage_cost,
|
|
51
|
+
stage_costs,
|
|
52
|
+
stage_ledger,
|
|
53
|
+
stage_seconds,
|
|
54
|
+
)
|
|
48
55
|
|
|
49
56
|
from langchain.chat_models import init_chat_model
|
|
50
57
|
from anthropic import APIConnectionError, APITimeoutError
|
|
@@ -71,7 +78,6 @@ from argus.models import (
|
|
|
71
78
|
ReviewResponse,
|
|
72
79
|
RiskLevel,
|
|
73
80
|
Severity,
|
|
74
|
-
TokenUsage,
|
|
75
81
|
Verdict,
|
|
76
82
|
)
|
|
77
83
|
from argus.pipeline_models import (
|
|
@@ -80,7 +86,6 @@ from argus.pipeline_models import (
|
|
|
80
86
|
CoverageResult,
|
|
81
87
|
DismissedFinding,
|
|
82
88
|
FeedbackVerificationResult,
|
|
83
|
-
FindingValidationResult,
|
|
84
89
|
PriorFinding,
|
|
85
90
|
PriorReviewContext,
|
|
86
91
|
RawFinding,
|
|
@@ -125,33 +130,9 @@ _checkpoint_tables_created = False
|
|
|
125
130
|
_PLANNER_MODEL = f"anthropic:{CLAUDE_FRONTIER}"
|
|
126
131
|
_WRITER_MODEL = f"anthropic:{CLAUDE_DEFAULT}"
|
|
127
132
|
_COVERAGE_MODEL = f"anthropic:{CLAUDE_FRONTIER}"
|
|
128
|
-
# run_lite_review() and _estimate_lite_review_cost() must agree on which
|
|
129
|
-
# model actually ran -- a single constant instead of two independent
|
|
130
|
-
# f"anthropic:{CLAUDE_DEFAULT}"/CLAUDE_DEFAULT references means a future
|
|
131
|
-
# change to one call site can't silently leave the other mispriced.
|
|
132
133
|
_LITE_REVIEW_MODEL = CLAUDE_DEFAULT
|
|
133
134
|
|
|
134
135
|
|
|
135
|
-
def _estimate_lite_review_cost(usage: TokenUsage, model: str) -> float:
|
|
136
|
-
"""Approximate USD cost for the lite-review path from raw token counts.
|
|
137
|
-
|
|
138
|
-
The lite path bypasses ``agent_runs`` cost tracking entirely, so this is
|
|
139
|
-
the only place its cost is ever computed. Pricing comes from
|
|
140
|
-
``argus.llm.pricing`` (litellm-backed); if the model has no pricing
|
|
141
|
-
entry, cost for this call is silently omitted (logged, not raised) --
|
|
142
|
-
cost tracking is observability, not a correctness gate.
|
|
143
|
-
"""
|
|
144
|
-
token_cost = get_token_cost(model)
|
|
145
|
-
if token_cost is None:
|
|
146
|
-
return 0.0
|
|
147
|
-
return (
|
|
148
|
-
usage.input_tokens * token_cost.input_cost_per_token
|
|
149
|
-
+ usage.output_tokens * token_cost.output_cost_per_token
|
|
150
|
-
+ usage.cache_read_tokens * token_cost.cache_read_cost_per_token
|
|
151
|
-
+ usage.cache_creation_tokens * token_cost.cache_write_cost_per_token
|
|
152
|
-
)
|
|
153
|
-
|
|
154
|
-
|
|
155
136
|
# Git SHA validation — accept short (7+) through full (40) hex digests. Used
|
|
156
137
|
# for defense-in-depth before interpolating SHAs into GitHub compare URLs.
|
|
157
138
|
_SHA_RE = re.compile(r"[0-9a-fA-F]{7,40}")
|
|
@@ -183,6 +164,15 @@ _HIGH_BLAST_RADIUS_SUFFIXES = (
|
|
|
183
164
|
# unbounded number of system groups (e.g. a very large monorepo PR).
|
|
184
165
|
_MAX_REVIEWER_FANOUT = 50
|
|
185
166
|
|
|
167
|
+
# Per-group file cap instructed to the planner (see _build_planner_messages).
|
|
168
|
+
# Measured: groups over this size reliably exhausted the reviewer's turn budget.
|
|
169
|
+
_MAX_GROUP_FILES = 8
|
|
170
|
+
|
|
171
|
+
# ReviewResponse fields the writer/lite-review OpenAI extraction call must
|
|
172
|
+
# never be asked to produce: free-form dicts (not expressible in OpenAI
|
|
173
|
+
# strict-mode schemas) and flags/data the pipeline sets itself, never the LLM.
|
|
174
|
+
_PIPELINE_ONLY_RESPONSE_FIELDS = frozenset({"stage_costs", "stage_seconds"})
|
|
175
|
+
|
|
186
176
|
# LangGraph max_concurrency cap passed to graph.ainvoke. Limits how many
|
|
187
177
|
# reviewer nodes run concurrently so the connection pool and Claude API
|
|
188
178
|
# rate limits are not overwhelmed on large fan-outs.
|
|
@@ -253,6 +243,9 @@ class ReviewState(TypedDict, total=False):
|
|
|
253
243
|
precheck_scanner_failures: list[
|
|
254
244
|
str
|
|
255
245
|
] # scanner names that returned None this round (crashed/timed out) -- observability only
|
|
246
|
+
precheck_missing_scanners: list[
|
|
247
|
+
str
|
|
248
|
+
] # scanner names never installed this round (standing config gap, not transient)
|
|
256
249
|
bench_config_changes: list[
|
|
257
250
|
str
|
|
258
251
|
] # bench config paths or routing lines modified this PR (TECH-6282)
|
|
@@ -520,7 +513,7 @@ async def _apply_dismissals(
|
|
|
520
513
|
)
|
|
521
514
|
|
|
522
515
|
llm = _get_llm(
|
|
523
|
-
f"anthropic:{CLAUDE_MINI}", max_tokens=1024, temperature=0
|
|
516
|
+
f"anthropic:{CLAUDE_MINI}", "dismiss_match", max_tokens=1024, temperature=0
|
|
524
517
|
).with_structured_output(DismissMatches)
|
|
525
518
|
result = await llm.ainvoke([{"role": "user", "content": prompt}])
|
|
526
519
|
|
|
@@ -833,7 +826,9 @@ def _build_planner_messages(prompt: str, diff: str, description: str) -> list[di
|
|
|
833
826
|
f"## PR Diff\n\n```diff\n{diff}\n```\n\n"
|
|
834
827
|
"Analyze this PR and produce a ReviewPlan. Group the changed files into "
|
|
835
828
|
"logical system groups for parallel review. Identify any cross-cutting concerns. "
|
|
836
|
-
"For each group, assign specialists_needed based on file patterns."
|
|
829
|
+
"For each group, assign specialists_needed based on file patterns. "
|
|
830
|
+
f"Keep each system group to at most {_MAX_GROUP_FILES} files -- split a group that "
|
|
831
|
+
"would exceed this into multiple narrower groups rather than emitting one oversized group."
|
|
837
832
|
)
|
|
838
833
|
return [
|
|
839
834
|
{"role": "system", "content": prompt},
|
|
@@ -977,7 +972,7 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
|
|
|
977
972
|
# fixed pins, on the theory that the empirical verification below
|
|
978
973
|
# was never run against an arbitrary override value. That created
|
|
979
974
|
# a real regression: `run_preflight_check` calls
|
|
980
|
-
# `_get_llm(f"anthropic:{CLAUDE_DEFAULT}", temperature=0)`, so
|
|
975
|
+
# `_get_llm(f"anthropic:{CLAUDE_DEFAULT}", "preflight", temperature=0)`, so
|
|
981
976
|
# setting ARGUS_SPECIALIST_MODEL=claude-sonnet-5 (the *previous*
|
|
982
977
|
# default, and a highly plausible rollback choice -- it's also
|
|
983
978
|
# used as an override value in this suite's own tests) resolved
|
|
@@ -1030,7 +1025,7 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
|
|
|
1030
1025
|
# claude-haiku-4-5`, used directly by this suite's own tests) would pull
|
|
1031
1026
|
# that value into the union above via the CLAUDE_DEFAULT/CLAUDE_FRONTIER
|
|
1032
1027
|
# terms, and then _apply_dismissals's `_get_llm(f"anthropic:{CLAUDE_MINI}",
|
|
1033
|
-
# temperature=0)` call would have ITS temperature silently stripped too
|
|
1028
|
+
# "dismiss_match", temperature=0)` call would have ITS temperature silently stripped too
|
|
1034
1029
|
# -- a real, silent determinism regression in a call site that has
|
|
1035
1030
|
# nothing to do with the override, caught in Argus round 3 review of
|
|
1036
1031
|
# this PR. `- {CLAUDE_MINI}` guarantees CLAUDE_MINI's resolved value can
|
|
@@ -1040,7 +1035,9 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
|
|
|
1040
1035
|
)
|
|
1041
1036
|
|
|
1042
1037
|
|
|
1043
|
-
def _get_llm(
|
|
1038
|
+
def _get_llm(
|
|
1039
|
+
model_id: str, stage: str, max_tokens: int = 16384, temperature: float | None = None
|
|
1040
|
+
) -> Any:
|
|
1044
1041
|
"""Create a LangChain chat model with explicit API key from settings.
|
|
1045
1042
|
|
|
1046
1043
|
``langchain_anthropic.ChatAnthropic`` has no ``auth_token``/bearer-style
|
|
@@ -1050,10 +1047,18 @@ def _get_llm(model_id: str, max_tokens: int = 16384, temperature: float | None =
|
|
|
1050
1047
|
only ANTHROPIC_AUTH_TOKEN is configured, pass its value through as the
|
|
1051
1048
|
api_key kwarg anyway: this is a real limitation for gateways that reject
|
|
1052
1049
|
x-api-key, but works for any gateway that accepts either header.
|
|
1050
|
+
|
|
1051
|
+
``stage`` attaches a :class:`~argus.llm.usage.StageCostCallbackHandler`
|
|
1052
|
+
so every call this model instance makes prices itself into the active
|
|
1053
|
+
:func:`~argus.llm.usage.stage_ledger` under that name.
|
|
1053
1054
|
"""
|
|
1054
1055
|
settings = get_settings()
|
|
1055
1056
|
_, anthropic_credential = settings.anthropic_credential
|
|
1056
|
-
kwargs: dict[str, Any] = {
|
|
1057
|
+
kwargs: dict[str, Any] = {
|
|
1058
|
+
"api_key": anthropic_credential,
|
|
1059
|
+
"max_tokens": max_tokens,
|
|
1060
|
+
"callbacks": [StageCostCallbackHandler(stage)],
|
|
1061
|
+
}
|
|
1057
1062
|
resolved_model = model_id.rsplit(":", 1)[-1]
|
|
1058
1063
|
if temperature is not None and resolved_model not in _TEMPERATURE_UNSUPPORTED_MODELS:
|
|
1059
1064
|
kwargs["temperature"] = temperature
|
|
@@ -1089,7 +1094,7 @@ async def plan_review(diff: str, description: str) -> ReviewPlan:
|
|
|
1089
1094
|
prompt = await fetch_prompt("pr-review-planner")
|
|
1090
1095
|
messages = _build_planner_messages(prompt, diff, description)
|
|
1091
1096
|
model = (
|
|
1092
|
-
_get_llm(_PLANNER_MODEL)
|
|
1097
|
+
_get_llm(_PLANNER_MODEL, "planner")
|
|
1093
1098
|
.bind_tools([ReviewPlan], tool_choice="ReviewPlan")
|
|
1094
1099
|
.with_config(
|
|
1095
1100
|
run_name="planner-phase1-stream",
|
|
@@ -1312,8 +1317,11 @@ async def verify_prior_feedback(
|
|
|
1312
1317
|
return await run_feedback_verifier_session(prior_context, diff, settings, repo_root=repo_root)
|
|
1313
1318
|
|
|
1314
1319
|
|
|
1315
|
-
async def check_coverage(
|
|
1316
|
-
|
|
1320
|
+
async def check_coverage(
|
|
1321
|
+
plan: ReviewPlan,
|
|
1322
|
+
findings: list[SystemReviewResult],
|
|
1323
|
+
) -> CoverageResult:
|
|
1324
|
+
"""Coverage check: mechanical set-difference, then LLM triage for ambiguous gaps."""
|
|
1317
1325
|
from argus.helpers import collect_reviewed_files as _collect_reviewed_files
|
|
1318
1326
|
|
|
1319
1327
|
manifest_files = {fe.path for fe in plan.file_manifest}
|
|
@@ -1331,7 +1339,7 @@ async def check_coverage(plan: ReviewPlan, findings: list[SystemReviewResult]) -
|
|
|
1331
1339
|
)
|
|
1332
1340
|
|
|
1333
1341
|
prompt = await fetch_prompt("pr-review-coverage-check")
|
|
1334
|
-
model = _get_llm(_COVERAGE_MODEL).with_structured_output(CoverageResult)
|
|
1342
|
+
model = _get_llm(_COVERAGE_MODEL, "coverage").with_structured_output(CoverageResult)
|
|
1335
1343
|
messages = [
|
|
1336
1344
|
{"role": "system", "content": prompt},
|
|
1337
1345
|
*_build_coverage_messages(plan, findings, sorted(uncovered)),
|
|
@@ -1360,7 +1368,7 @@ async def write_review(
|
|
|
1360
1368
|
|
|
1361
1369
|
settings = get_settings()
|
|
1362
1370
|
prompt = await fetch_prompt("pr-review-writer")
|
|
1363
|
-
model = _get_llm(_WRITER_MODEL)
|
|
1371
|
+
model = _get_llm(_WRITER_MODEL, "writer")
|
|
1364
1372
|
|
|
1365
1373
|
messages = [
|
|
1366
1374
|
{
|
|
@@ -1377,7 +1385,9 @@ async def write_review(
|
|
|
1377
1385
|
# Phase 2: GPT-5.4-mini extracts structured ReviewResponse
|
|
1378
1386
|
def _extract() -> ReviewResponse:
|
|
1379
1387
|
oai = OpenAIClientSync(api_key=settings.OPENAI_API_KEY)
|
|
1380
|
-
response_format = pydantic_to_response_format(
|
|
1388
|
+
response_format = pydantic_to_response_format(
|
|
1389
|
+
ReviewResponse, "review_response", exclude=_PIPELINE_ONLY_RESPONSE_FIELDS
|
|
1390
|
+
)
|
|
1381
1391
|
extraction_prompt = (
|
|
1382
1392
|
"Extract the structured review response from the following code review text. "
|
|
1383
1393
|
"Parse out ALL fields: verdict, risk_level, findings, prior_feedback, "
|
|
@@ -1397,6 +1407,8 @@ async def write_review(
|
|
|
1397
1407
|
instructions="You are a JSON extraction assistant. Parse the review text into the schema.",
|
|
1398
1408
|
text_format=response_format,
|
|
1399
1409
|
)
|
|
1410
|
+
if resp.usage is not None:
|
|
1411
|
+
record_stage_cost("writer_extract", price_openai_usage(GPT_MINI, resp.usage))
|
|
1400
1412
|
return ReviewResponse.model_validate_json(resp.output_text)
|
|
1401
1413
|
|
|
1402
1414
|
return await asyncio.to_thread(_extract)
|
|
@@ -1432,7 +1444,7 @@ async def run_preflight_check(
|
|
|
1432
1444
|
"""
|
|
1433
1445
|
prompt = await fetch_prompt("pr-review-preflight-router")
|
|
1434
1446
|
llm = _get_llm(
|
|
1435
|
-
f"anthropic:{CLAUDE_DEFAULT}", max_tokens=256, temperature=0
|
|
1447
|
+
f"anthropic:{CLAUDE_DEFAULT}", "preflight", max_tokens=256, temperature=0
|
|
1436
1448
|
).with_structured_output(PreflightResult)
|
|
1437
1449
|
prior_context = (
|
|
1438
1450
|
f"Prior round verdict: {prior_verdict}" if prior_verdict else "No prior review (round 1)"
|
|
@@ -1463,7 +1475,7 @@ async def run_lite_review(
|
|
|
1463
1475
|
|
|
1464
1476
|
settings = get_settings()
|
|
1465
1477
|
prompt = await fetch_prompt("pr-review-lite")
|
|
1466
|
-
model = _get_llm(f"anthropic:{_LITE_REVIEW_MODEL}")
|
|
1478
|
+
model = _get_llm(f"anthropic:{_LITE_REVIEW_MODEL}", "lite_review")
|
|
1467
1479
|
|
|
1468
1480
|
messages = [
|
|
1469
1481
|
{"role": "system", "content": prompt},
|
|
@@ -1482,7 +1494,9 @@ async def run_lite_review(
|
|
|
1482
1494
|
|
|
1483
1495
|
def _extract() -> ReviewResponse:
|
|
1484
1496
|
oai = OpenAIClientSync(api_key=settings.OPENAI_API_KEY)
|
|
1485
|
-
response_format = pydantic_to_response_format(
|
|
1497
|
+
response_format = pydantic_to_response_format(
|
|
1498
|
+
ReviewResponse, "review_response", exclude=_PIPELINE_ONLY_RESPONSE_FIELDS
|
|
1499
|
+
)
|
|
1486
1500
|
extraction_prompt = (
|
|
1487
1501
|
"Extract the structured review response from the following lite code review text. "
|
|
1488
1502
|
"Parse out: verdict, risk_level, review_comment (the full markdown text as-is), "
|
|
@@ -1497,6 +1511,8 @@ async def run_lite_review(
|
|
|
1497
1511
|
instructions="You are a JSON extraction assistant. Parse the review text into the schema.",
|
|
1498
1512
|
text_format=response_format,
|
|
1499
1513
|
)
|
|
1514
|
+
if resp.usage is not None:
|
|
1515
|
+
record_stage_cost("lite_extract", price_openai_usage(GPT_MINI, resp.usage))
|
|
1500
1516
|
return ReviewResponse.model_validate_json(resp.output_text)
|
|
1501
1517
|
|
|
1502
1518
|
response = await asyncio.to_thread(_extract)
|
|
@@ -1641,6 +1657,10 @@ async def _node_precheck_rules(state: ReviewState, config: RunnableConfig) -> di
|
|
|
1641
1657
|
# same degraded-coverage pattern already used for killed/timed-out
|
|
1642
1658
|
# LLM reviewer sessions.
|
|
1643
1659
|
update["precheck_scanner_failures"] = result.failed_scanners
|
|
1660
|
+
if result.missing_scanners:
|
|
1661
|
+
# Own key rather than merged into precheck_scanner_failures so it
|
|
1662
|
+
# reads as a standing config gap, not a crash.
|
|
1663
|
+
update["precheck_missing_scanners"] = result.missing_scanners
|
|
1644
1664
|
|
|
1645
1665
|
if result.candidate_findings:
|
|
1646
1666
|
try:
|
|
@@ -1777,6 +1797,7 @@ async def _node_early_verifier(state: ReviewState, config: RunnableConfig) -> di
|
|
|
1777
1797
|
sum(1 for i in result.items if i.status.value == "REGRESSED"),
|
|
1778
1798
|
)
|
|
1779
1799
|
|
|
1800
|
+
record_stage_cost("verify", result.cost_usd, agent_run.duration_seconds if agent_run else 0.0)
|
|
1780
1801
|
state_update: dict[str, Any] = {"verification": result.model_dump()}
|
|
1781
1802
|
if agent_run is not None:
|
|
1782
1803
|
state_update["agent_runs"] = [agent_run.model_dump(mode="json")]
|
|
@@ -1845,7 +1866,11 @@ async def _node_preflight(state: ReviewState) -> dict[str, Any]:
|
|
|
1845
1866
|
logger.warning("Preflight check failed — falling back to full review", exc_info=True)
|
|
1846
1867
|
result = PreflightResult(route="full", reason="preflight failed, defaulting to full review")
|
|
1847
1868
|
is_lite = False
|
|
1848
|
-
return {
|
|
1869
|
+
return {
|
|
1870
|
+
"preflight_result": result.model_dump(),
|
|
1871
|
+
"is_lite": is_lite,
|
|
1872
|
+
"is_catchup_merge": False,
|
|
1873
|
+
}
|
|
1849
1874
|
|
|
1850
1875
|
|
|
1851
1876
|
def _is_catchup_merge_only(repo: str, prior_sha: str, head_sha: str) -> bool:
|
|
@@ -2337,6 +2362,14 @@ async def _node_plan(state: ReviewState) -> dict[str, Any]:
|
|
|
2337
2362
|
len(plan.system_groups),
|
|
2338
2363
|
len(plan.cross_cutting_concerns),
|
|
2339
2364
|
)
|
|
2365
|
+
oversized = [g.name for g in plan.system_groups if len(g.files) > _MAX_GROUP_FILES]
|
|
2366
|
+
if oversized:
|
|
2367
|
+
logger.warning(
|
|
2368
|
+
"Planner emitted %d group(s) over the %d-file cap despite prompt instruction: %s",
|
|
2369
|
+
len(oversized),
|
|
2370
|
+
_MAX_GROUP_FILES,
|
|
2371
|
+
", ".join(oversized),
|
|
2372
|
+
)
|
|
2340
2373
|
|
|
2341
2374
|
return {"plan": plan.model_dump()}
|
|
2342
2375
|
|
|
@@ -2633,6 +2666,7 @@ async def _node_run_reviewer(inputs: ReviewerInput, config: RunnableConfig) -> d
|
|
|
2633
2666
|
_files = result.files_explored[:5]
|
|
2634
2667
|
_files_str = ", ".join(_files) + (" ..." if len(result.files_explored) > 5 else "")
|
|
2635
2668
|
_dur = agent_run.duration_seconds if agent_run else 0.0
|
|
2669
|
+
record_stage_cost(f"reviewer:{_label}", result.cost_usd, _dur)
|
|
2636
2670
|
if result.failure_reason is not None:
|
|
2637
2671
|
logger.warning(
|
|
2638
2672
|
"Reviewer [%s] FAILED (%s) after %.1fs — treated as 0 findings, "
|
|
@@ -2879,12 +2913,21 @@ async def _node_validate_blockings(state: ReviewState, config: RunnableConfig) -
|
|
|
2879
2913
|
|
|
2880
2914
|
settings = get_settings()
|
|
2881
2915
|
worktree_path: str | None = config.get("configurable", {}).get("worktree_path")
|
|
2916
|
+
|
|
2882
2917
|
validation, validator_agent_run = await run_blocking_validator_session(
|
|
2883
2918
|
[f.model_dump(mode="json") for f in blocking_findings],
|
|
2884
2919
|
state["diff"],
|
|
2885
2920
|
settings,
|
|
2886
2921
|
repo_root=worktree_path,
|
|
2887
2922
|
)
|
|
2923
|
+
record_stage_cost(
|
|
2924
|
+
"validate",
|
|
2925
|
+
validation.cost_usd,
|
|
2926
|
+
validator_agent_run.duration_seconds if validator_agent_run else 0.0,
|
|
2927
|
+
)
|
|
2928
|
+
validator_runs = (
|
|
2929
|
+
[validator_agent_run.model_dump(mode="json")] if validator_agent_run is not None else []
|
|
2930
|
+
)
|
|
2888
2931
|
|
|
2889
2932
|
# Partition findings into confirmed and rejected
|
|
2890
2933
|
rejected_indices: set[int] = set()
|
|
@@ -2892,10 +2935,6 @@ async def _node_validate_blockings(state: ReviewState, config: RunnableConfig) -
|
|
|
2892
2935
|
if item.verdict == ValidationVerdict.REJECTED:
|
|
2893
2936
|
rejected_indices.add(item.index)
|
|
2894
2937
|
|
|
2895
|
-
validator_runs: list[dict[str, Any]] = (
|
|
2896
|
-
[validator_agent_run.model_dump(mode="json")] if validator_agent_run is not None else []
|
|
2897
|
-
)
|
|
2898
|
-
|
|
2899
2938
|
if not rejected_indices:
|
|
2900
2939
|
logger.info("All %d BLOCKING findings confirmed", len(blocking_findings))
|
|
2901
2940
|
return {"validation": validation.model_dump(), "agent_runs": validator_runs}
|
|
@@ -3414,44 +3453,38 @@ async def run_review(request: ReviewRequest, flow_run_id: str | None = None) ->
|
|
|
3414
3453
|
),
|
|
3415
3454
|
)
|
|
3416
3455
|
|
|
3417
|
-
|
|
3418
|
-
|
|
3419
|
-
|
|
3420
|
-
|
|
3421
|
-
|
|
3422
|
-
|
|
3423
|
-
|
|
3424
|
-
|
|
3425
|
-
|
|
3426
|
-
|
|
3427
|
-
|
|
3428
|
-
|
|
3429
|
-
|
|
3430
|
-
|
|
3431
|
-
|
|
3432
|
-
|
|
3456
|
+
# Wraps the whole graph invocation, including fanned-out reviewer tasks,
|
|
3457
|
+
# so every LLM call records into the same per-review ledger.
|
|
3458
|
+
with stage_ledger():
|
|
3459
|
+
if head_sha_for_worktree is not None:
|
|
3460
|
+
# Fail-closed by design: if provisioning raises (e.g. SHA mismatch, or
|
|
3461
|
+
# a transient clone/fetch failure), we let it propagate and abort the
|
|
3462
|
+
# review rather than silently falling back to _invoke_graph(None).
|
|
3463
|
+
# Reviewing the wrong tree (or quietly degrading to diff-only when a
|
|
3464
|
+
# SHA was explicitly requested) is worse than failing the run, which
|
|
3465
|
+
# an external orchestrator can retry. Do not soften this to a
|
|
3466
|
+
# try/except fallback.
|
|
3467
|
+
async with provisioned_worktree(
|
|
3468
|
+
repo=request.repo,
|
|
3469
|
+
head_sha=head_sha_for_worktree,
|
|
3470
|
+
token=settings.GITHUB_TOKEN_RO,
|
|
3471
|
+
) as worktree_path:
|
|
3472
|
+
result = await _invoke_graph(worktree_path, head_sha=head_sha_for_worktree)
|
|
3473
|
+
else:
|
|
3474
|
+
result = await _invoke_graph(None)
|
|
3475
|
+
|
|
3476
|
+
response = ReviewResponse.model_validate(result["response"])
|
|
3477
|
+
response.stage_costs = stage_costs()
|
|
3478
|
+
response.stage_seconds = stage_seconds()
|
|
3433
3479
|
|
|
3434
|
-
response = ReviewResponse.model_validate(result["response"])
|
|
3435
3480
|
elapsed = time.monotonic() - pipeline_start
|
|
3436
3481
|
is_lite = result.get("is_lite", False)
|
|
3437
3482
|
reviewer_version = "v3-lite" if is_lite else "v3"
|
|
3438
3483
|
|
|
3439
|
-
#
|
|
3484
|
+
# stage_costs is the single source of truth for cost -- do not also sum
|
|
3485
|
+
# the per-component costs that fed it (that would double-count).
|
|
3440
3486
|
findings_models = [SystemReviewResult.model_validate(f) for f in result.get("findings", [])]
|
|
3441
|
-
total_cost_usd = sum(
|
|
3442
|
-
verification_data = result.get("verification", {})
|
|
3443
|
-
if verification_data:
|
|
3444
|
-
total_cost_usd += FeedbackVerificationResult.model_validate(verification_data).cost_usd
|
|
3445
|
-
validation_data = result.get("validation", {})
|
|
3446
|
-
if validation_data:
|
|
3447
|
-
total_cost_usd += FindingValidationResult.model_validate(validation_data).cost_usd
|
|
3448
|
-
|
|
3449
|
-
if is_lite:
|
|
3450
|
-
# Lite path bypasses agent_runs cost tracking; approximate from token counts
|
|
3451
|
-
# captured in run_lite_review. Use += to preserve early_verifier cost
|
|
3452
|
-
# (round 2+) already accumulated above.
|
|
3453
|
-
total_cost_usd += _estimate_lite_review_cost(response.usage, _LITE_REVIEW_MODEL)
|
|
3454
|
-
|
|
3487
|
+
total_cost_usd = sum(response.stage_costs.values())
|
|
3455
3488
|
response.usage.cost_usd = total_cost_usd
|
|
3456
3489
|
|
|
3457
3490
|
# Opt-in, off by default -- see ARGUS_PRECHECK_BLOCK_ON_SCANNER_FAILURE's
|