argus-code-review 0.2.4__tar.gz → 0.2.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (153) hide show
  1. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/CHANGELOG.md +82 -1
  2. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/PKG-INFO +2 -2
  3. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/config.py +3 -2
  4. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/gemini_runner.py +36 -8
  5. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/graph.py +112 -79
  6. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/helpers.py +102 -16
  7. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/output_models.py +6 -0
  8. argus_code_review-0.2.6/argus/llm/usage.py +117 -0
  9. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/models.py +8 -0
  10. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/openai_runner.py +26 -7
  11. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/pipeline_models.py +14 -13
  12. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/engine.py +41 -7
  13. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/runners.py +70 -2
  14. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/sql.py +3 -3
  15. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/sqlite.py +76 -28
  16. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/pyproject.toml +1 -1
  17. argus_code_review-0.2.6/schema/018_widen_agent_runs_failure_reason.sql +36 -0
  18. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/golden/review_response.schema.json +16 -0
  19. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_sql.py +9 -0
  20. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_sqlite_backend.py +198 -0
  21. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_gemini_runner.py +205 -7
  22. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_get_llm_temperature.py +7 -7
  23. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_precheck.py +25 -0
  24. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_timeout_surfacing.py +80 -0
  25. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_openai_client.py +38 -0
  26. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_openai_runner.py +190 -6
  27. argus_code_review-0.2.6/tests/test_output_models.py +188 -0
  28. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_engine.py +55 -6
  29. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_context_usage.py +47 -0
  30. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_helpers.py +113 -2
  31. argus_code_review-0.2.6/tests/test_stage_costs.py +141 -0
  32. argus_code_review-0.2.4/tests/test_lite_review_cost.py +0 -56
  33. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/.gitignore +0 -0
  34. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/LICENSE +0 -0
  35. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/README.md +0 -0
  36. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/__init__.py +0 -0
  37. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/bench.py +0 -0
  38. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/bench_default.toml +0 -0
  39. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/cli.py +0 -0
  40. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/coverage.py +0 -0
  41. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/dotenv_utils.py +0 -0
  42. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/gemini_cache.py +0 -0
  43. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/github_client.py +0 -0
  44. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/models.py +0 -0
  45. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/llm/pricing.py +0 -0
  46. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/openai_client.py +0 -0
  47. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/__init__.py +0 -0
  48. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/actions_scanner.py +0 -0
  49. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/.gitignore +0 -0
  50. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/README.md +0 -0
  51. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/eslint.config.js +0 -0
  52. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/package-lock.json +0 -0
  53. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/eslint_bundle/package.json +0 -0
  54. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/js_scanner.py +0 -0
  55. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/migration_scanner.py +0 -0
  56. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/rules/README.md +0 -0
  57. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/sarif.py +0 -0
  58. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/scanner_utils.py +0 -0
  59. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/secrets_scanner.py +0 -0
  60. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/shadow.py +0 -0
  61. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/terraform_scanner.py +0 -0
  62. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/precheck/workflow_lint_scanner.py +0 -0
  63. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/__init__.py +0 -0
  64. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-blocking-validator.md +0 -0
  65. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-coverage-check.md +0 -0
  66. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-cross-cutting.md +0 -0
  67. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-feedback-verifier.md +0 -0
  68. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-lite.md +0 -0
  69. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-planner.md +0 -0
  70. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-preflight-router.md +0 -0
  71. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-prior-art.md +0 -0
  72. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-deployment.md +0 -0
  73. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-frontend.md +0 -0
  74. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-infra.md +0 -0
  75. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-llm-patterns.md +0 -0
  76. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-observability.md +0 -0
  77. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-orchestration.md +0 -0
  78. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-security.md +0 -0
  79. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-slackbot.md +0 -0
  80. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-specialist-sql.md +0 -0
  81. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-subagent.md +0 -0
  82. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-tests-and-docs.md +0 -0
  83. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts/pr-review-writer.md +0 -0
  84. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/prompts_runtime.py +0 -0
  85. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/repo_provision.py +0 -0
  86. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/review_tools.py +0 -0
  87. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/__init__.py +0 -0
  88. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/http.py +0 -0
  89. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/models.py +0 -0
  90. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/precheck.py +0 -0
  91. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/resolver.py +0 -0
  92. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/argus/storage/session.py +0 -0
  93. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/008_add_code_reviews.sql +0 -0
  94. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/009_add_reviewer_version.sql +0 -0
  95. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/010_add_review_patterns.sql +0 -0
  96. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/011_add_review_progress_columns.sql +0 -0
  97. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/015_create_agent_runs.sql +0 -0
  98. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/016_add_agent_runs_failure_reason.sql +0 -0
  99. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/schema/017_add_precheck_rules.sql +0 -0
  100. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/__init__.py +0 -0
  101. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/conftest.py +0 -0
  102. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/__init__.py +0 -0
  103. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_backend_contract.py +0 -0
  104. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_http.py +0 -0
  105. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_models.py +0 -0
  106. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_resolver.py +0 -0
  107. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/storage/test_session.py +0 -0
  108. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_actions_scanner.py +0 -0
  109. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_argus_review_local.py +0 -0
  110. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_bench.py +0 -0
  111. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_bench_config_guard.py +0 -0
  112. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_catchup_gate.py +0 -0
  113. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_args.py +0 -0
  114. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_output_contract.py +0 -0
  115. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_post_review.py +0 -0
  116. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_preflight.py +0 -0
  117. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_cli_prompts.py +0 -0
  118. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_config.py +0 -0
  119. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_conftest.py +0 -0
  120. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_gemini_cache.py +0 -0
  121. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_github_client_checks_signal.py +0 -0
  122. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_github_client_write.py +0 -0
  123. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_fetch_diff.py +0 -0
  124. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_http_guards.py +0 -0
  125. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_preflight_image_bump.py +0 -0
  126. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_progress.py +0 -0
  127. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_graph_storage_resolution.py +0 -0
  128. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_js_scanner.py +0 -0
  129. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_models.py +0 -0
  130. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_models_override.py +0 -0
  131. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_llm_pricing.py +0 -0
  132. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_migration_scanner.py +0 -0
  133. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_models.py +0 -0
  134. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_multi_round.py +0 -0
  135. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_packaging.py +0 -0
  136. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_plan_review.py +0 -0
  137. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_engine_integration.py +0 -0
  138. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_sarif.py +0 -0
  139. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_shadow.py +0 -0
  140. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_precheck_shadow_integration.py +0 -0
  141. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_prompts.py +0 -0
  142. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_repo_provision.py +0 -0
  143. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_review_patterns_integration.py +0 -0
  144. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_review_tools.py +0 -0
  145. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_context7.py +0 -0
  146. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_runners_new.py +0 -0
  147. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_scanner_utils.py +0 -0
  148. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_secrets_scanner.py +0 -0
  149. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_specialist_validation.py +0 -0
  150. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_storage_precheck.py +0 -0
  151. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_terraform_scanner.py +0 -0
  152. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_terraform_scanner_integration.py +0 -0
  153. {argus_code_review-0.2.4 → argus_code_review-0.2.6}/tests/test_workflow_lint_scanner.py +0 -0
@@ -7,6 +7,85 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.2.6] - 2026-09-19
11
+
12
+ ### Fixed
13
+
14
+ - Bounded `openai` dependency to `>=1.66.0,<2` in `pyproject.toml` (TECH-6590) to
15
+ prevent unconstrained fresh installs from resolving to a future breaking major
16
+ release. The floor was also tightened from `>=1.50.0` to `>=1.66.0` where the
17
+ OpenAI Responses API (`client.responses.create()`) used by this codebase was
18
+ introduced. (Note: this is a general hygiene fix; it does not address the root
19
+ cause of the TECH-6590-reported crash, which was a separate shared-venv version-skew
20
+ issue tracked in TECH-6592.)
21
+
22
+ ## [0.2.5] - 2026-09-18
23
+
24
+ ### Added
25
+
26
+ - `failure_reason="turn_budget_exhausted"` (schema 018) on the Gemini and OpenAI
27
+ runner paths (#24): a reviewer session that runs out of turns without ever
28
+ calling `finish_review` is now flagged as a failure instead of returning
29
+ silently as a clean, zero-findings review. Across a six-PR replay, 17 of 52
30
+ reviewer sessions exhausted their budget this way, costing ~$18.69 (38% of
31
+ total spend) for no usable output — none of it previously visible.
32
+ - A finish-now nudge injected 3 turns before the turn-budget ceiling on the
33
+ Gemini and OpenAI loops, plus a one-line budget disclosure added to their
34
+ system prompts (#24), so a session that's about to exhaust is told to
35
+ converge instead of continuing to explore.
36
+ - A 75% mid-budget checkpoint nudge on the same two loops (TECH-6560),
37
+ independent of and in addition to the existing end-of-budget nudge above: a
38
+ softer "you're roughly 75% through your turn budget, start converging if
39
+ you have enough" check-in, fired early enough not to discourage legitimate
40
+ exploration on large diffs, with a full turn gap left before the emergency
41
+ nudge (turn 75 vs. turn 97 on Gemini's 100-turn budget; turn 22 vs. turn 27
42
+ on OpenAI's 30-turn budget). Not applied to the Claude Agent SDK path,
43
+ which has no documented mid-stream prompt-injection hook.
44
+ - Per-stage cost and duration ledger (#24), priced through the existing
45
+ litellm pricing table and surfaced on `ReviewResponse.stage_costs` /
46
+ `stage_seconds`. Previously only reviewer-agent cost was tracked; the
47
+ planner, preflight, coverage check, and writer stages each reported $0.00,
48
+ together ~6% of real spend (the planner alone 4.9%).
49
+ - Missing/uninstalled deterministic precheck scanners are now surfaced as
50
+ degraded coverage (#24), distinct from a scanner that ran and found
51
+ nothing. Four of six configured scanner binaries were absent in every
52
+ measured production run, so a review could previously report clean having
53
+ never actually scanned for secrets or destructive migrations.
54
+
55
+ ### Changed
56
+
57
+ - Raised the Gemini reviewer's turn budget from 45 to 100 (TECH-6558),
58
+ `argus/gemini_runner.py`'s `_MAX_TURNS_GEMINI`. Gemini sessions were still
59
+ exhausting the 45-turn budget introduced in 0.2.4 even with the new
60
+ finish-now nudge above; raising the ceiling is a cheap mitigation to try
61
+ independent of the nudge, not a replacement for it — watch exhaustion
62
+ rates on real rounds to confirm it helps before assuming it does.
63
+ - A failed reviewer's coverage-gap finding is now promoted to BLOCKING when
64
+ nothing else in the round already blocks (#24), so a round can no longer
65
+ APPROVE on coverage it knows is degraded (a failed reviewer contributes
66
+ zero findings, which otherwise looks identical to a clean review).
67
+ - The planner is now instructed to cap system-group size (#24): turn-budget
68
+ exhaustion clustered on the largest planner groups in the replayed sample.
69
+
70
+ ### Fixed
71
+
72
+ - A table rebuild in the SQLite storage backend that widens
73
+ `agent_runs.failure_reason` copied rows with `SELECT *`, pairing columns
74
+ positionally; a database that gained the column via `ALTER TABLE` has it
75
+ last, while the DDL declares it mid-table, silently shifting a datetime
76
+ into `failure_reason` and tripping its CHECK constraint on startup (#24).
77
+ Copies by explicit column name now, with a regression test.
78
+ - semgrep was scheduled in the precheck engine without an availability
79
+ check, so an absent binary never reached the new `missing_scanners`
80
+ reporting above — the exact gap that feature exists to close, for every
81
+ other scanner (#24).
82
+ - `run_lite_review`'s extraction cost went unrecorded once the new per-stage
83
+ ledger became the only cost source, underreporting every lite review (#24).
84
+ - Per-stage durations were measured from handler construction rather than
85
+ the call itself, unpriced models were silently costed as $0, and a
86
+ reviewer failure skipped its risk-level bump when a precheck failure had
87
+ already forced BLOCKING (#24).
88
+
10
89
  ## [0.2.4] - 2026-09-17
11
90
 
12
91
  ### Changed
@@ -259,7 +338,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
259
338
  packaged set.
260
339
  - `argus --version`, `argus prompts list`, and `argus prompts export`.
261
340
 
262
- [Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...HEAD
341
+ [Unreleased]: https://github.com/redesignhealth/argus-review/compare/v0.2.6...HEAD
342
+ [0.2.6]: https://github.com/redesignhealth/argus-review/compare/v0.2.5...v0.2.6
343
+ [0.2.5]: https://github.com/redesignhealth/argus-review/compare/v0.2.4...v0.2.5
263
344
  [0.2.4]: https://github.com/redesignhealth/argus-review/compare/v0.2.3...v0.2.4
264
345
  [0.2.3]: https://github.com/redesignhealth/argus-review/compare/v0.2.2...v0.2.3
265
346
  [0.2.2]: https://github.com/redesignhealth/argus-review/compare/v0.2.1...v0.2.2
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: argus-code-review
3
- Version: 0.2.4
3
+ Version: 0.2.6
4
4
  Summary: Self-orchestrated PR review agent using LangGraph + Claude Agent SDK
5
5
  Project-URL: Repository, https://github.com/redesignhealth/argus-review
6
6
  Project-URL: Issues, https://github.com/redesignhealth/argus-review/issues
@@ -26,7 +26,7 @@ Requires-Dist: langgraph-checkpoint-sqlite>=2.0.0
26
26
  Requires-Dist: langgraph<2,>=1.0.10
27
27
  Requires-Dist: langsmith<1,>=0.2.0
28
28
  Requires-Dist: litellm<2.0.0,>=1.50.0
29
- Requires-Dist: openai>=1.50.0
29
+ Requires-Dist: openai<2,>=1.66.0
30
30
  Requires-Dist: psycopg-pool>=3.2.0
31
31
  Requires-Dist: psycopg[binary]>=3.2.0
32
32
  Requires-Dist: pydantic-settings>=2.0.0
@@ -70,8 +70,9 @@ class Settings(BaseSettings):
70
70
  See ``docs/PRECHECKS.md``'s "stock rule sources" section.
71
71
  ARGUS_PRECHECK_BLOCK_ON_SCANNER_FAILURE: Set truthy to force the
72
72
  verdict to BLOCKING whenever a deterministic precheck scanner
73
- crashed/timed out/errored this round (``PrecheckResult.
74
- failed_scanners`` non-empty) instead of the default fail-open
73
+ crashed/timed out/errored, OR was never installed, this round
74
+ (``PrecheckResult.failed_scanners``/``missing_scanners``
75
+ non-empty) instead of the default fail-open
75
76
  behavior (surface it in the review comment's degraded-coverage
76
77
  section — see ``argus.helpers.build_degraded_coverage_labels``
77
78
  — but let the review's own verdict stand on its own merits).
@@ -159,14 +159,19 @@ from argus.gemini_cache import GeminiCacheKeeper
159
159
  from argus.llm.models import estimate_cost_usd
160
160
  from argus.llm.models import resolve as resolve_model_alias
161
161
  from argus.runners import (
162
+ _MID_BUDGET_NUDGE,
163
+ _NUDGE_TURNS_BEFORE_BUDGET,
162
164
  _SUBPROCESS_TIMEOUT_S,
165
+ _TURN_BUDGET_NUDGE,
166
+ _TURN_BUDGET_SYSTEM_PROMPT_LINE,
163
167
  SessionResult,
168
+ _compute_mid_budget_nudge_turn,
164
169
  _resolve_repo_root,
165
170
  )
166
171
 
167
172
  logger = logging.getLogger(__name__)
168
173
 
169
- _MAX_TURNS_GEMINI = 45 # 30 * 1.5 — see TECH-6453
174
+ _MAX_TURNS_GEMINI = 100 # raised from 45 -- see TECH-6558
170
175
 
171
176
  # Both "auto" and "on" attempt explicit caching today -- there is no
172
177
  # separate heuristic distinguishing them yet (Track 1 defined the
@@ -594,10 +599,15 @@ async def _run_turns(
594
599
  timeout_s: float,
595
600
  started_at: datetime,
596
601
  ) -> SessionResult:
597
- """The actual tool-calling loop. Always returns with ``failure_reason=None``
598
- -- the enclosing ``run_session_gemini`` is what applies the timeout and
599
- turns a cancellation into a ``failure_reason="timeout"`` result instead.
602
+ """The actual tool-calling loop. Returns ``failure_reason="turn_budget_exhausted"``
603
+ when the loop ran out of turns without a ``finish_review`` call, else
604
+ ``None`` -- the enclosing ``run_session_gemini`` is what applies the
605
+ timeout and turns a cancellation into a ``failure_reason="timeout"``
606
+ result instead.
600
607
  """
608
+ system_prompt = (
609
+ system_prompt + "\n\n" + _TURN_BUDGET_SYSTEM_PROMPT_LINE.format(max_turns=_MAX_TURNS_GEMINI)
610
+ )
601
611
  _, api_key = settings.google_credential
602
612
  # Second layer of defense: `asyncio.wait_for` in `run_session_gemini`
603
613
  # cannot interrupt an in-flight blocking SDK call once it has started
@@ -682,14 +692,32 @@ async def _run_turns(
682
692
 
683
693
  tool_calls: list[str] = []
684
694
  files_explored: list[str] = []
695
+ exhausted = False
685
696
  usage_prompt_total = 0
686
697
  usage_candidates_total = 0
687
698
  usage_cached_total = 0
688
699
  usage_thoughts_total = 0
689
700
  usage_tool_use_prompt_total = 0
690
701
 
702
+ _mid_budget_nudge_turn = _compute_mid_budget_nudge_turn(_MAX_TURNS_GEMINI)
703
+
691
704
  with review_tools.review_session(repo_root) as findings_sink:
692
705
  for _turn in range(_MAX_TURNS_GEMINI):
706
+ if (
707
+ _turn == _MAX_TURNS_GEMINI - _NUDGE_TURNS_BEFORE_BUDGET
708
+ and contents[-1].parts is not None
709
+ ):
710
+ contents[-1].parts.append(types.Part.from_text(text=_TURN_BUDGET_NUDGE))
711
+ # Not an `elif` -- these are two independent checkpoints at
712
+ # different turns and must both be able to fire in the same
713
+ # session (see _compute_mid_budget_nudge_turn's docstring
714
+ # for why they can never land on the same turn today).
715
+ if (
716
+ _mid_budget_nudge_turn is not None
717
+ and _turn == _mid_budget_nudge_turn
718
+ and contents[-1].parts is not None
719
+ ):
720
+ contents[-1].parts.append(types.Part.from_text(text=_MID_BUDGET_NUDGE))
693
721
  try:
694
722
  response = await client.aio.models.generate_content(
695
723
  model=model, contents=contents, config=config
@@ -819,14 +847,14 @@ async def _run_turns(
819
847
  if finished:
820
848
  break
821
849
  else:
822
- # `for...else`: only reached if every one of _MAX_TURNS_GEMINI
823
- # iterations executed a function call and none of them was
824
- # finish_review -- i.e. the turn budget was exhausted.
850
+ # `for...else`: only reached if every iteration executed a
851
+ # function call and none was finish_review, i.e. exhausted.
825
852
  logger.warning(
826
853
  "Gemini session [%s] exhausted its %d-turn budget without a finish_review call",
827
854
  label or "unlabeled",
828
855
  _MAX_TURNS_GEMINI,
829
856
  )
857
+ exhausted = True
830
858
 
831
859
  result_text = _build_result_text(findings_sink, files_explored)
832
860
 
@@ -874,7 +902,7 @@ async def _run_turns(
874
902
  # docstring) -- always 0, never derived from tool_calls.
875
903
  context7_call_count=0,
876
904
  model=model,
877
- failure_reason=None,
905
+ failure_reason="turn_budget_exhausted" if exhausted else None,
878
906
  )
879
907
  finally:
880
908
  # Every session creates both a sync and async HTTP client (the sync
@@ -44,7 +44,14 @@ from argus.helpers import (
44
44
  compute_persisted_finding_counts,
45
45
  )
46
46
  from argus.llm.models import ALIAS_MAP, CLAUDE_DEFAULT, CLAUDE_FRONTIER, CLAUDE_MINI
47
- from argus.llm.pricing import get_token_cost
47
+ from argus.llm.usage import (
48
+ StageCostCallbackHandler,
49
+ price_openai_usage,
50
+ record_stage_cost,
51
+ stage_costs,
52
+ stage_ledger,
53
+ stage_seconds,
54
+ )
48
55
 
49
56
  from langchain.chat_models import init_chat_model
50
57
  from anthropic import APIConnectionError, APITimeoutError
@@ -71,7 +78,6 @@ from argus.models import (
71
78
  ReviewResponse,
72
79
  RiskLevel,
73
80
  Severity,
74
- TokenUsage,
75
81
  Verdict,
76
82
  )
77
83
  from argus.pipeline_models import (
@@ -80,7 +86,6 @@ from argus.pipeline_models import (
80
86
  CoverageResult,
81
87
  DismissedFinding,
82
88
  FeedbackVerificationResult,
83
- FindingValidationResult,
84
89
  PriorFinding,
85
90
  PriorReviewContext,
86
91
  RawFinding,
@@ -125,33 +130,9 @@ _checkpoint_tables_created = False
125
130
  _PLANNER_MODEL = f"anthropic:{CLAUDE_FRONTIER}"
126
131
  _WRITER_MODEL = f"anthropic:{CLAUDE_DEFAULT}"
127
132
  _COVERAGE_MODEL = f"anthropic:{CLAUDE_FRONTIER}"
128
- # run_lite_review() and _estimate_lite_review_cost() must agree on which
129
- # model actually ran -- a single constant instead of two independent
130
- # f"anthropic:{CLAUDE_DEFAULT}"/CLAUDE_DEFAULT references means a future
131
- # change to one call site can't silently leave the other mispriced.
132
133
  _LITE_REVIEW_MODEL = CLAUDE_DEFAULT
133
134
 
134
135
 
135
- def _estimate_lite_review_cost(usage: TokenUsage, model: str) -> float:
136
- """Approximate USD cost for the lite-review path from raw token counts.
137
-
138
- The lite path bypasses ``agent_runs`` cost tracking entirely, so this is
139
- the only place its cost is ever computed. Pricing comes from
140
- ``argus.llm.pricing`` (litellm-backed); if the model has no pricing
141
- entry, cost for this call is silently omitted (logged, not raised) --
142
- cost tracking is observability, not a correctness gate.
143
- """
144
- token_cost = get_token_cost(model)
145
- if token_cost is None:
146
- return 0.0
147
- return (
148
- usage.input_tokens * token_cost.input_cost_per_token
149
- + usage.output_tokens * token_cost.output_cost_per_token
150
- + usage.cache_read_tokens * token_cost.cache_read_cost_per_token
151
- + usage.cache_creation_tokens * token_cost.cache_write_cost_per_token
152
- )
153
-
154
-
155
136
  # Git SHA validation — accept short (7+) through full (40) hex digests. Used
156
137
  # for defense-in-depth before interpolating SHAs into GitHub compare URLs.
157
138
  _SHA_RE = re.compile(r"[0-9a-fA-F]{7,40}")
@@ -183,6 +164,15 @@ _HIGH_BLAST_RADIUS_SUFFIXES = (
183
164
  # unbounded number of system groups (e.g. a very large monorepo PR).
184
165
  _MAX_REVIEWER_FANOUT = 50
185
166
 
167
+ # Per-group file cap instructed to the planner (see _build_planner_messages).
168
+ # Measured: groups over this size reliably exhausted the reviewer's turn budget.
169
+ _MAX_GROUP_FILES = 8
170
+
171
+ # ReviewResponse fields the writer/lite-review OpenAI extraction call must
172
+ # never be asked to produce: free-form dicts (not expressible in OpenAI
173
+ # strict-mode schemas) and flags/data the pipeline sets itself, never the LLM.
174
+ _PIPELINE_ONLY_RESPONSE_FIELDS = frozenset({"stage_costs", "stage_seconds"})
175
+
186
176
  # LangGraph max_concurrency cap passed to graph.ainvoke. Limits how many
187
177
  # reviewer nodes run concurrently so the connection pool and Claude API
188
178
  # rate limits are not overwhelmed on large fan-outs.
@@ -253,6 +243,9 @@ class ReviewState(TypedDict, total=False):
253
243
  precheck_scanner_failures: list[
254
244
  str
255
245
  ] # scanner names that returned None this round (crashed/timed out) -- observability only
246
+ precheck_missing_scanners: list[
247
+ str
248
+ ] # scanner names never installed this round (standing config gap, not transient)
256
249
  bench_config_changes: list[
257
250
  str
258
251
  ] # bench config paths or routing lines modified this PR (TECH-6282)
@@ -520,7 +513,7 @@ async def _apply_dismissals(
520
513
  )
521
514
 
522
515
  llm = _get_llm(
523
- f"anthropic:{CLAUDE_MINI}", max_tokens=1024, temperature=0
516
+ f"anthropic:{CLAUDE_MINI}", "dismiss_match", max_tokens=1024, temperature=0
524
517
  ).with_structured_output(DismissMatches)
525
518
  result = await llm.ainvoke([{"role": "user", "content": prompt}])
526
519
 
@@ -833,7 +826,9 @@ def _build_planner_messages(prompt: str, diff: str, description: str) -> list[di
833
826
  f"## PR Diff\n\n```diff\n{diff}\n```\n\n"
834
827
  "Analyze this PR and produce a ReviewPlan. Group the changed files into "
835
828
  "logical system groups for parallel review. Identify any cross-cutting concerns. "
836
- "For each group, assign specialists_needed based on file patterns."
829
+ "For each group, assign specialists_needed based on file patterns. "
830
+ f"Keep each system group to at most {_MAX_GROUP_FILES} files -- split a group that "
831
+ "would exceed this into multiple narrower groups rather than emitting one oversized group."
837
832
  )
838
833
  return [
839
834
  {"role": "system", "content": prompt},
@@ -977,7 +972,7 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
977
972
  # fixed pins, on the theory that the empirical verification below
978
973
  # was never run against an arbitrary override value. That created
979
974
  # a real regression: `run_preflight_check` calls
980
- # `_get_llm(f"anthropic:{CLAUDE_DEFAULT}", temperature=0)`, so
975
+ # `_get_llm(f"anthropic:{CLAUDE_DEFAULT}", "preflight", temperature=0)`, so
981
976
  # setting ARGUS_SPECIALIST_MODEL=claude-sonnet-5 (the *previous*
982
977
  # default, and a highly plausible rollback choice -- it's also
983
978
  # used as an override value in this suite's own tests) resolved
@@ -1030,7 +1025,7 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
1030
1025
  # claude-haiku-4-5`, used directly by this suite's own tests) would pull
1031
1026
  # that value into the union above via the CLAUDE_DEFAULT/CLAUDE_FRONTIER
1032
1027
  # terms, and then _apply_dismissals's `_get_llm(f"anthropic:{CLAUDE_MINI}",
1033
- # temperature=0)` call would have ITS temperature silently stripped too
1028
+ # "dismiss_match", temperature=0)` call would have ITS temperature silently stripped too
1034
1029
  # -- a real, silent determinism regression in a call site that has
1035
1030
  # nothing to do with the override, caught in Argus round 3 review of
1036
1031
  # this PR. `- {CLAUDE_MINI}` guarantees CLAUDE_MINI's resolved value can
@@ -1040,7 +1035,9 @@ _TEMPERATURE_UNSUPPORTED_MODELS: frozenset[str] = (
1040
1035
  )
1041
1036
 
1042
1037
 
1043
- def _get_llm(model_id: str, max_tokens: int = 16384, temperature: float | None = None) -> Any:
1038
+ def _get_llm(
1039
+ model_id: str, stage: str, max_tokens: int = 16384, temperature: float | None = None
1040
+ ) -> Any:
1044
1041
  """Create a LangChain chat model with explicit API key from settings.
1045
1042
 
1046
1043
  ``langchain_anthropic.ChatAnthropic`` has no ``auth_token``/bearer-style
@@ -1050,10 +1047,18 @@ def _get_llm(model_id: str, max_tokens: int = 16384, temperature: float | None =
1050
1047
  only ANTHROPIC_AUTH_TOKEN is configured, pass its value through as the
1051
1048
  api_key kwarg anyway: this is a real limitation for gateways that reject
1052
1049
  x-api-key, but works for any gateway that accepts either header.
1050
+
1051
+ ``stage`` attaches a :class:`~argus.llm.usage.StageCostCallbackHandler`
1052
+ so every call this model instance makes prices itself into the active
1053
+ :func:`~argus.llm.usage.stage_ledger` under that name.
1053
1054
  """
1054
1055
  settings = get_settings()
1055
1056
  _, anthropic_credential = settings.anthropic_credential
1056
- kwargs: dict[str, Any] = {"api_key": anthropic_credential, "max_tokens": max_tokens}
1057
+ kwargs: dict[str, Any] = {
1058
+ "api_key": anthropic_credential,
1059
+ "max_tokens": max_tokens,
1060
+ "callbacks": [StageCostCallbackHandler(stage)],
1061
+ }
1057
1062
  resolved_model = model_id.rsplit(":", 1)[-1]
1058
1063
  if temperature is not None and resolved_model not in _TEMPERATURE_UNSUPPORTED_MODELS:
1059
1064
  kwargs["temperature"] = temperature
@@ -1089,7 +1094,7 @@ async def plan_review(diff: str, description: str) -> ReviewPlan:
1089
1094
  prompt = await fetch_prompt("pr-review-planner")
1090
1095
  messages = _build_planner_messages(prompt, diff, description)
1091
1096
  model = (
1092
- _get_llm(_PLANNER_MODEL)
1097
+ _get_llm(_PLANNER_MODEL, "planner")
1093
1098
  .bind_tools([ReviewPlan], tool_choice="ReviewPlan")
1094
1099
  .with_config(
1095
1100
  run_name="planner-phase1-stream",
@@ -1312,8 +1317,11 @@ async def verify_prior_feedback(
1312
1317
  return await run_feedback_verifier_session(prior_context, diff, settings, repo_root=repo_root)
1313
1318
 
1314
1319
 
1315
- async def check_coverage(plan: ReviewPlan, findings: list[SystemReviewResult]) -> CoverageResult:
1316
- """Coverage check: mechanical set-difference + LLM triage for ambiguous gaps."""
1320
+ async def check_coverage(
1321
+ plan: ReviewPlan,
1322
+ findings: list[SystemReviewResult],
1323
+ ) -> CoverageResult:
1324
+ """Coverage check: mechanical set-difference, then LLM triage for ambiguous gaps."""
1317
1325
  from argus.helpers import collect_reviewed_files as _collect_reviewed_files
1318
1326
 
1319
1327
  manifest_files = {fe.path for fe in plan.file_manifest}
@@ -1331,7 +1339,7 @@ async def check_coverage(plan: ReviewPlan, findings: list[SystemReviewResult]) -
1331
1339
  )
1332
1340
 
1333
1341
  prompt = await fetch_prompt("pr-review-coverage-check")
1334
- model = _get_llm(_COVERAGE_MODEL).with_structured_output(CoverageResult)
1342
+ model = _get_llm(_COVERAGE_MODEL, "coverage").with_structured_output(CoverageResult)
1335
1343
  messages = [
1336
1344
  {"role": "system", "content": prompt},
1337
1345
  *_build_coverage_messages(plan, findings, sorted(uncovered)),
@@ -1360,7 +1368,7 @@ async def write_review(
1360
1368
 
1361
1369
  settings = get_settings()
1362
1370
  prompt = await fetch_prompt("pr-review-writer")
1363
- model = _get_llm(_WRITER_MODEL)
1371
+ model = _get_llm(_WRITER_MODEL, "writer")
1364
1372
 
1365
1373
  messages = [
1366
1374
  {
@@ -1377,7 +1385,9 @@ async def write_review(
1377
1385
  # Phase 2: GPT-5.4-mini extracts structured ReviewResponse
1378
1386
  def _extract() -> ReviewResponse:
1379
1387
  oai = OpenAIClientSync(api_key=settings.OPENAI_API_KEY)
1380
- response_format = pydantic_to_response_format(ReviewResponse, "review_response")
1388
+ response_format = pydantic_to_response_format(
1389
+ ReviewResponse, "review_response", exclude=_PIPELINE_ONLY_RESPONSE_FIELDS
1390
+ )
1381
1391
  extraction_prompt = (
1382
1392
  "Extract the structured review response from the following code review text. "
1383
1393
  "Parse out ALL fields: verdict, risk_level, findings, prior_feedback, "
@@ -1397,6 +1407,8 @@ async def write_review(
1397
1407
  instructions="You are a JSON extraction assistant. Parse the review text into the schema.",
1398
1408
  text_format=response_format,
1399
1409
  )
1410
+ if resp.usage is not None:
1411
+ record_stage_cost("writer_extract", price_openai_usage(GPT_MINI, resp.usage))
1400
1412
  return ReviewResponse.model_validate_json(resp.output_text)
1401
1413
 
1402
1414
  return await asyncio.to_thread(_extract)
@@ -1432,7 +1444,7 @@ async def run_preflight_check(
1432
1444
  """
1433
1445
  prompt = await fetch_prompt("pr-review-preflight-router")
1434
1446
  llm = _get_llm(
1435
- f"anthropic:{CLAUDE_DEFAULT}", max_tokens=256, temperature=0
1447
+ f"anthropic:{CLAUDE_DEFAULT}", "preflight", max_tokens=256, temperature=0
1436
1448
  ).with_structured_output(PreflightResult)
1437
1449
  prior_context = (
1438
1450
  f"Prior round verdict: {prior_verdict}" if prior_verdict else "No prior review (round 1)"
@@ -1463,7 +1475,7 @@ async def run_lite_review(
1463
1475
 
1464
1476
  settings = get_settings()
1465
1477
  prompt = await fetch_prompt("pr-review-lite")
1466
- model = _get_llm(f"anthropic:{_LITE_REVIEW_MODEL}")
1478
+ model = _get_llm(f"anthropic:{_LITE_REVIEW_MODEL}", "lite_review")
1467
1479
 
1468
1480
  messages = [
1469
1481
  {"role": "system", "content": prompt},
@@ -1482,7 +1494,9 @@ async def run_lite_review(
1482
1494
 
1483
1495
  def _extract() -> ReviewResponse:
1484
1496
  oai = OpenAIClientSync(api_key=settings.OPENAI_API_KEY)
1485
- response_format = pydantic_to_response_format(ReviewResponse, "review_response")
1497
+ response_format = pydantic_to_response_format(
1498
+ ReviewResponse, "review_response", exclude=_PIPELINE_ONLY_RESPONSE_FIELDS
1499
+ )
1486
1500
  extraction_prompt = (
1487
1501
  "Extract the structured review response from the following lite code review text. "
1488
1502
  "Parse out: verdict, risk_level, review_comment (the full markdown text as-is), "
@@ -1497,6 +1511,8 @@ async def run_lite_review(
1497
1511
  instructions="You are a JSON extraction assistant. Parse the review text into the schema.",
1498
1512
  text_format=response_format,
1499
1513
  )
1514
+ if resp.usage is not None:
1515
+ record_stage_cost("lite_extract", price_openai_usage(GPT_MINI, resp.usage))
1500
1516
  return ReviewResponse.model_validate_json(resp.output_text)
1501
1517
 
1502
1518
  response = await asyncio.to_thread(_extract)
@@ -1641,6 +1657,10 @@ async def _node_precheck_rules(state: ReviewState, config: RunnableConfig) -> di
1641
1657
  # same degraded-coverage pattern already used for killed/timed-out
1642
1658
  # LLM reviewer sessions.
1643
1659
  update["precheck_scanner_failures"] = result.failed_scanners
1660
+ if result.missing_scanners:
1661
+ # Own key rather than merged into precheck_scanner_failures so it
1662
+ # reads as a standing config gap, not a crash.
1663
+ update["precheck_missing_scanners"] = result.missing_scanners
1644
1664
 
1645
1665
  if result.candidate_findings:
1646
1666
  try:
@@ -1777,6 +1797,7 @@ async def _node_early_verifier(state: ReviewState, config: RunnableConfig) -> di
1777
1797
  sum(1 for i in result.items if i.status.value == "REGRESSED"),
1778
1798
  )
1779
1799
 
1800
+ record_stage_cost("verify", result.cost_usd, agent_run.duration_seconds if agent_run else 0.0)
1780
1801
  state_update: dict[str, Any] = {"verification": result.model_dump()}
1781
1802
  if agent_run is not None:
1782
1803
  state_update["agent_runs"] = [agent_run.model_dump(mode="json")]
@@ -1845,7 +1866,11 @@ async def _node_preflight(state: ReviewState) -> dict[str, Any]:
1845
1866
  logger.warning("Preflight check failed — falling back to full review", exc_info=True)
1846
1867
  result = PreflightResult(route="full", reason="preflight failed, defaulting to full review")
1847
1868
  is_lite = False
1848
- return {"preflight_result": result.model_dump(), "is_lite": is_lite, "is_catchup_merge": False}
1869
+ return {
1870
+ "preflight_result": result.model_dump(),
1871
+ "is_lite": is_lite,
1872
+ "is_catchup_merge": False,
1873
+ }
1849
1874
 
1850
1875
 
1851
1876
  def _is_catchup_merge_only(repo: str, prior_sha: str, head_sha: str) -> bool:
@@ -2337,6 +2362,14 @@ async def _node_plan(state: ReviewState) -> dict[str, Any]:
2337
2362
  len(plan.system_groups),
2338
2363
  len(plan.cross_cutting_concerns),
2339
2364
  )
2365
+ oversized = [g.name for g in plan.system_groups if len(g.files) > _MAX_GROUP_FILES]
2366
+ if oversized:
2367
+ logger.warning(
2368
+ "Planner emitted %d group(s) over the %d-file cap despite prompt instruction: %s",
2369
+ len(oversized),
2370
+ _MAX_GROUP_FILES,
2371
+ ", ".join(oversized),
2372
+ )
2340
2373
 
2341
2374
  return {"plan": plan.model_dump()}
2342
2375
 
@@ -2633,6 +2666,7 @@ async def _node_run_reviewer(inputs: ReviewerInput, config: RunnableConfig) -> d
2633
2666
  _files = result.files_explored[:5]
2634
2667
  _files_str = ", ".join(_files) + (" ..." if len(result.files_explored) > 5 else "")
2635
2668
  _dur = agent_run.duration_seconds if agent_run else 0.0
2669
+ record_stage_cost(f"reviewer:{_label}", result.cost_usd, _dur)
2636
2670
  if result.failure_reason is not None:
2637
2671
  logger.warning(
2638
2672
  "Reviewer [%s] FAILED (%s) after %.1fs — treated as 0 findings, "
@@ -2879,12 +2913,21 @@ async def _node_validate_blockings(state: ReviewState, config: RunnableConfig) -
2879
2913
 
2880
2914
  settings = get_settings()
2881
2915
  worktree_path: str | None = config.get("configurable", {}).get("worktree_path")
2916
+
2882
2917
  validation, validator_agent_run = await run_blocking_validator_session(
2883
2918
  [f.model_dump(mode="json") for f in blocking_findings],
2884
2919
  state["diff"],
2885
2920
  settings,
2886
2921
  repo_root=worktree_path,
2887
2922
  )
2923
+ record_stage_cost(
2924
+ "validate",
2925
+ validation.cost_usd,
2926
+ validator_agent_run.duration_seconds if validator_agent_run else 0.0,
2927
+ )
2928
+ validator_runs = (
2929
+ [validator_agent_run.model_dump(mode="json")] if validator_agent_run is not None else []
2930
+ )
2888
2931
 
2889
2932
  # Partition findings into confirmed and rejected
2890
2933
  rejected_indices: set[int] = set()
@@ -2892,10 +2935,6 @@ async def _node_validate_blockings(state: ReviewState, config: RunnableConfig) -
2892
2935
  if item.verdict == ValidationVerdict.REJECTED:
2893
2936
  rejected_indices.add(item.index)
2894
2937
 
2895
- validator_runs: list[dict[str, Any]] = (
2896
- [validator_agent_run.model_dump(mode="json")] if validator_agent_run is not None else []
2897
- )
2898
-
2899
2938
  if not rejected_indices:
2900
2939
  logger.info("All %d BLOCKING findings confirmed", len(blocking_findings))
2901
2940
  return {"validation": validation.model_dump(), "agent_runs": validator_runs}
@@ -3414,44 +3453,38 @@ async def run_review(request: ReviewRequest, flow_run_id: str | None = None) ->
3414
3453
  ),
3415
3454
  )
3416
3455
 
3417
- if head_sha_for_worktree is not None:
3418
- # Fail-closed by design: if provisioning raises (e.g. SHA mismatch, or
3419
- # a transient clone/fetch failure), we let it propagate and abort the
3420
- # review rather than silently falling back to _invoke_graph(None).
3421
- # Reviewing the wrong tree (or quietly degrading to diff-only when a
3422
- # SHA was explicitly requested) is worse than failing the run, which
3423
- # an external orchestrator can retry. Do not soften this to a
3424
- # try/except fallback.
3425
- async with provisioned_worktree(
3426
- repo=request.repo,
3427
- head_sha=head_sha_for_worktree,
3428
- token=settings.GITHUB_TOKEN_RO,
3429
- ) as worktree_path:
3430
- result = await _invoke_graph(worktree_path, head_sha=head_sha_for_worktree)
3431
- else:
3432
- result = await _invoke_graph(None)
3456
+ # Wraps the whole graph invocation, including fanned-out reviewer tasks,
3457
+ # so every LLM call records into the same per-review ledger.
3458
+ with stage_ledger():
3459
+ if head_sha_for_worktree is not None:
3460
+ # Fail-closed by design: if provisioning raises (e.g. SHA mismatch, or
3461
+ # a transient clone/fetch failure), we let it propagate and abort the
3462
+ # review rather than silently falling back to _invoke_graph(None).
3463
+ # Reviewing the wrong tree (or quietly degrading to diff-only when a
3464
+ # SHA was explicitly requested) is worse than failing the run, which
3465
+ # an external orchestrator can retry. Do not soften this to a
3466
+ # try/except fallback.
3467
+ async with provisioned_worktree(
3468
+ repo=request.repo,
3469
+ head_sha=head_sha_for_worktree,
3470
+ token=settings.GITHUB_TOKEN_RO,
3471
+ ) as worktree_path:
3472
+ result = await _invoke_graph(worktree_path, head_sha=head_sha_for_worktree)
3473
+ else:
3474
+ result = await _invoke_graph(None)
3475
+
3476
+ response = ReviewResponse.model_validate(result["response"])
3477
+ response.stage_costs = stage_costs()
3478
+ response.stage_seconds = stage_seconds()
3433
3479
 
3434
- response = ReviewResponse.model_validate(result["response"])
3435
3480
  elapsed = time.monotonic() - pipeline_start
3436
3481
  is_lite = result.get("is_lite", False)
3437
3482
  reviewer_version = "v3-lite" if is_lite else "v3"
3438
3483
 
3439
- # Aggregate cost from all subagent findings + verification + validation
3484
+ # stage_costs is the single source of truth for cost -- do not also sum
3485
+ # the per-component costs that fed it (that would double-count).
3440
3486
  findings_models = [SystemReviewResult.model_validate(f) for f in result.get("findings", [])]
3441
- total_cost_usd = sum(f.cost_usd for f in findings_models)
3442
- verification_data = result.get("verification", {})
3443
- if verification_data:
3444
- total_cost_usd += FeedbackVerificationResult.model_validate(verification_data).cost_usd
3445
- validation_data = result.get("validation", {})
3446
- if validation_data:
3447
- total_cost_usd += FindingValidationResult.model_validate(validation_data).cost_usd
3448
-
3449
- if is_lite:
3450
- # Lite path bypasses agent_runs cost tracking; approximate from token counts
3451
- # captured in run_lite_review. Use += to preserve early_verifier cost
3452
- # (round 2+) already accumulated above.
3453
- total_cost_usd += _estimate_lite_review_cost(response.usage, _LITE_REVIEW_MODEL)
3454
-
3487
+ total_cost_usd = sum(response.stage_costs.values())
3455
3488
  response.usage.cost_usd = total_cost_usd
3456
3489
 
3457
3490
  # Opt-in, off by default -- see ARGUS_PRECHECK_BLOCK_ON_SCANNER_FAILURE's