revenium-python-sdk 0.7.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/PKG-INFO +389 -17
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/README.md +387 -15
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/pyproject.toml +8 -2
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/__init__.py +20 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/__init__.py +9 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/cache_tokens.py +63 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/config.py +49 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/context.py +37 -1
- revenium_python_sdk-0.9.0/revenium_middleware/_core/enforcement.py +1524 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/exceptions.py +22 -5
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/fields.py +55 -0
- revenium_python_sdk-0.9.0/revenium_middleware/_core/outcomes.py +731 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/trace_fields.py +82 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/resources/ai.py +76 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/ai_create_audio_params.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/ai_create_completion_params.py +23 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/ai_create_image_params.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/ai_create_video_params.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/agentic_outcomes.py +110 -7
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/bedrock_adapter.py +9 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/bedrock_transport.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/middleware.py +445 -180
- revenium_python_sdk-0.9.0/revenium_middleware/anthropic/provider.py +262 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/fal/_metering.py +12 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/fal/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/utils.py +14 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/job_context.py +272 -37
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/job_history.py +8 -10
- revenium_python_sdk-0.9.0/revenium_middleware/job_type_economics.py +214 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/middleware.py +6 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/trace_fields.py +3 -0
- revenium_python_sdk-0.9.0/revenium_middleware/litellm/proxy/__init__.py +61 -0
- revenium_python_sdk-0.9.0/revenium_middleware/litellm/proxy/_metering_owner.py +50 -0
- revenium_python_sdk-0.9.0/revenium_middleware/litellm/proxy/guardrail.py +1426 -0
- revenium_python_sdk-0.9.0/revenium_middleware/litellm/proxy/middleware.py +675 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/ollama/middleware.py +8 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/ollama/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/middleware.py +15 -4
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/perplexity/middleware.py +6 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/perplexity/perplexity_sdk.py +6 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/perplexity/trace_fields.py +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_python_sdk.egg-info/PKG-INFO +389 -17
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_python_sdk.egg-info/SOURCES.txt +3 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_python_sdk.egg-info/requires.txt +1 -1
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/tests/test_metering.py +79 -0
- revenium_python_sdk-0.7.0/revenium_middleware/_core/enforcement.py +0 -817
- revenium_python_sdk-0.7.0/revenium_middleware/_core/outcomes.py +0 -424
- revenium_python_sdk-0.7.0/revenium_middleware/anthropic/provider.py +0 -141
- revenium_python_sdk-0.7.0/revenium_middleware/litellm/proxy/__init__.py +0 -26
- revenium_python_sdk-0.7.0/revenium_middleware/litellm/proxy/middleware.py +0 -263
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/LICENSE +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/decorators.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/load_diagnostics.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/log_sanitize.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/metering.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/metering_buffer.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/metering_status.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/metering_submission.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/patch_registry.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/prompt_extraction.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_core/subscriber.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/LICENSE +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_base_client.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_client.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_compat.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_constants.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_exceptions.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_files.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_models.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_qs.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_resource.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_response.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_streaming.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_types.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_logs.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_proxy.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_reflection.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_resources_proxy.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_streams.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_sync.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_transform.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_typing.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_utils/_utils.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/_version.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/context.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/decorator.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/py.typed +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/resources/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/resources/apis.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/resources/events.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/api_meter_request_params.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/api_meter_response_params.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/event_create_params.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/_metering/types/metering_response_resource.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/prompt_extractor.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/anthropic/stream_create.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/fal/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/fal/config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/fal/middleware.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/exceptions.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/protocols.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/common/types.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/google_ai/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/google_ai/middleware.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/google_ai/provider.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/prompt_extractor.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/vertex_ai/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/vertex_ai/middleware.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/google/vertex_ai/provider.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/_metadata.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/anthropic_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/litellm_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/ollama_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/openai_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/openai_embedding_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/griptape/universal_driver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/context.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/decorators.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/hooks.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/integrations/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/integrations/crewai.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/litellm/client/validation.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/ollama/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/azure_config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/azure_model_resolver.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/config.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/exceptions.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/langchain/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/langchain/_utils.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/langchain/unified_handler.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/prompt_extractor.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/openai/provider.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/perplexity/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/perplexity/provider.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/webhooks/__init__.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_middleware/webhooks/_verify.py +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_python_sdk.egg-info/dependency_links.txt +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/revenium_python_sdk.egg-info/top_level.txt +0 -0
- {revenium_python_sdk-0.7.0 → revenium_python_sdk-0.9.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: revenium-python-sdk
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: The official Revenium Python SDK — unified AI metering middleware for OpenAI, Anthropic, Google, Ollama, LiteLLM, Perplexity, and fal.ai.
|
|
5
5
|
Author-email: Revenium <support@revenium.io>
|
|
6
6
|
License: MIT
|
|
@@ -76,7 +76,7 @@ Requires-Dist: wrapt>=1.14.0; extra == "litellm"
|
|
|
76
76
|
Requires-Dist: litellm>=1.40.0; extra == "litellm"
|
|
77
77
|
Provides-Extra: litellm-proxy
|
|
78
78
|
Requires-Dist: wrapt>=1.14.0; extra == "litellm-proxy"
|
|
79
|
-
Requires-Dist: litellm[proxy]>=1.
|
|
79
|
+
Requires-Dist: litellm[proxy]>=1.93.0; extra == "litellm-proxy"
|
|
80
80
|
Provides-Extra: dev
|
|
81
81
|
Requires-Dist: pytest>=7.0.0; extra == "dev"
|
|
82
82
|
Requires-Dist: pytest-asyncio; extra == "dev"
|
|
@@ -264,7 +264,70 @@ when the provider's SDK is installed.
|
|
|
264
264
|
|
|
265
265
|
Emit per-agent terminal outcomes (`CONVERTED`, `DEFLECTED`, `ESCALATED`) alongside completion and tool-event records, so dashboards show business value next to AI cost.
|
|
266
266
|
|
|
267
|
-
> **You need a write-scope key (`rev_sk_`) to use the agentic outcomes API.** Metering keys (`rev_mk_`) can only meter completions and tool events — they cannot report, amend, or read job outcomes, and the SDK rejects them client-side before any HTTP request is made. Key resolution: explicit `api_key=` > `REVENIUM_OUTCOME_API_KEY` > `REVENIUM_METERING_API_KEY`.
|
|
267
|
+
> **You need a write-scope key (`rev_sk_`) to use the agentic outcomes API.** Metering keys (`rev_mk_`) can only meter completions and tool events — they cannot report, amend, or read job outcomes, and the SDK rejects them client-side before any HTTP request is made. Key resolution: explicit `api_key=` > `REVENIUM_WRITE_API_KEY` > `REVENIUM_OUTCOME_API_KEY` (deprecated fallback) > `REVENIUM_METERING_API_KEY`.
|
|
268
|
+
|
|
269
|
+
### Job-Type Economics and Outcome Facts
|
|
270
|
+
|
|
271
|
+
Keep a metering key for AI telemetry and a separate write key for outcomes and
|
|
272
|
+
job-type configuration. A registered `valuePerUnit` rule takes precedence over
|
|
273
|
+
an outcome's `outcome_value`; the backend never sums the two value sources.
|
|
274
|
+
|
|
275
|
+
```python
|
|
276
|
+
from revenium_middleware import (
|
|
277
|
+
Baseline, JobTypeEconomics, PeriodFactEntry, create_baseline,
|
|
278
|
+
report_period_facts, upsert_job_type_economics,
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
upsert_job_type_economics("claim", JobTypeEconomics(
|
|
282
|
+
unit_metric_key="completed_claims", unit_label="claim",
|
|
283
|
+
metrics=[
|
|
284
|
+
{
|
|
285
|
+
"key": "completed_claims", "type": "COUNT",
|
|
286
|
+
"direction": "HIGHER_IS_BETTER", "aggregation": "SUM",
|
|
287
|
+
"resolution": "PER_JOB",
|
|
288
|
+
},
|
|
289
|
+
{
|
|
290
|
+
"key": "claims_processed", "type": "COUNT",
|
|
291
|
+
"direction": "HIGHER_IS_BETTER", "aggregation": "SUM",
|
|
292
|
+
"resolution": "PERIOD",
|
|
293
|
+
},
|
|
294
|
+
],
|
|
295
|
+
dimensions=[{"key": "region", "allowedValues": ["us", "ca"]}],
|
|
296
|
+
monetization={
|
|
297
|
+
"metricKey": "completed_claims", "valuePerUnit": 4.25,
|
|
298
|
+
"currency": "USD", "category": "COST_AVOIDED", "basis": "REALIZED",
|
|
299
|
+
},
|
|
300
|
+
))
|
|
301
|
+
create_baseline("claim", Baseline(
|
|
302
|
+
effective_from="2026-08-01T00:00:00Z", cost_per_unit=4.25, currency="USD",
|
|
303
|
+
))
|
|
304
|
+
report_period_facts("claim", [PeriodFactEntry(
|
|
305
|
+
period_start="2026-08-01T00:00:00Z", period_end="2026-09-01T00:00:00Z",
|
|
306
|
+
dimension_key="region", dimension_value="us",
|
|
307
|
+
key="claims_processed", value=1280,
|
|
308
|
+
)])
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
`effective_from` is the only required field on a baseline; every other field
|
|
312
|
+
is optional, and a baseline without it is rejected. A job type must be
|
|
313
|
+
declared with `upsert_job_type_economics` before it accepts baselines or
|
|
314
|
+
facts, and `report_period_facts` accepts only metrics declared with
|
|
315
|
+
`"resolution": "PERIOD"`.
|
|
316
|
+
|
|
317
|
+
Job economics currency values must be USD. Baselines and period facts use
|
|
318
|
+
server-supplied attribution when their provenance,
|
|
319
|
+
reporter, and source fields are omitted. Set those fields only when you need an
|
|
320
|
+
explicit override. Economics metric directions are
|
|
321
|
+
`HIGHER_IS_BETTER` or `LOWER_IS_BETTER`; monetization categories are
|
|
322
|
+
`REVENUE`, `COST_AVOIDED`, `TIME_SAVED`, and `LEADING_VALUE`, with a
|
|
323
|
+
`REALIZED` or `EXPECTED` basis.
|
|
324
|
+
|
|
325
|
+
Use `CUSTOMER_DECLARED` or `MEASURED` for a baseline override. Use `MEASURED`,
|
|
326
|
+
`SELF_REPORTED`, or `DERIVED` for a period fact override.
|
|
327
|
+
|
|
328
|
+
Facts are append-only and keyed on the period, dimension and metric key
|
|
329
|
+
together. Re-appending that tuple supersedes the active fact, and the server
|
|
330
|
+
requires `reason=` on the entry when it does.
|
|
268
331
|
|
|
269
332
|
### JobContext
|
|
270
333
|
|
|
@@ -311,10 +374,70 @@ history = get_outcome_history("sales-lead-8842")
|
|
|
311
374
|
# List[JobOutcomeAmendment], ordered by amendment_sequence (1 = the initial report)
|
|
312
375
|
```
|
|
313
376
|
|
|
314
|
-
`amend_outcome()` takes
|
|
377
|
+
`amend_outcome()` takes `reason` — the amendment's audit justification, still the first positional argument — plus the same optional fields as `report_outcome()` (`execution_status`, `outcome_type`, `outcome_value`, `outcome_currency`, `metadata`, `reported_by`, `outcome_reason`, `metrics`), and returns the updated job as a dict.
|
|
378
|
+
|
|
379
|
+
- **Detecting a lost update:** `report_outcome()` and `amend_outcome()` record the job's `entityVersion` from the response on the handle (readable as `job.entity_version`). The next `amend_outcome()` on that same handle sends it as `expectedEntityVersion`, so an amendment that would overwrite a change made by another writer in the meantime raises `OutcomeAmendConflictError` instead of silently winning. Pass `expected_entity_version=` to lock against a version you fetched yourself; use a fresh `JobContext.attach()` handle — which has recorded nothing — for the old last-write-wins behavior.
|
|
380
|
+
|
|
381
|
+
```python
|
|
382
|
+
from revenium_middleware import OutcomeAmendConflictError, get_outcome_history
|
|
383
|
+
|
|
384
|
+
try:
|
|
385
|
+
job.amend_outcome(reason="Chargeback", outcome_value=0.0)
|
|
386
|
+
except OutcomeAmendConflictError as conflict:
|
|
387
|
+
# The conflict reports the version the platform actually holds.
|
|
388
|
+
print(conflict.current_entity_version) # e.g. 9
|
|
389
|
+
|
|
390
|
+
# Look at what the other writer changed, and only re-issue the amendment
|
|
391
|
+
# if it still applies to what is recorded now.
|
|
392
|
+
history = get_outcome_history("sales-lead-8842")
|
|
393
|
+
if still_applies(history[-1]):
|
|
394
|
+
job.amend_outcome(reason="Chargeback, re-checked", outcome_value=0.0,
|
|
395
|
+
expected_entity_version=conflict.current_entity_version)
|
|
396
|
+
```
|
|
315
397
|
|
|
398
|
+
The handle also records that version, so the retry above works with or without passing `expected_entity_version=` explicitly. `current_entity_version` is `None` when the conflict body carries no version; the version then has to come from a job read (`GET /v2/api/jobs/{agenticJobId}`), which this SDK does not wrap yet, and a retry without it is unlocked (last-write-wins). `get_outcome_history()` rows carry an `amendment_sequence`, not an entity version — history tells you *what* changed, never which version to retry with.
|
|
399
|
+
|
|
400
|
+
Every outcome call replaces the recorded version with the one its response reports, including clearing it when a response carries none, so a completed call never leaves a token behind that the platform has already moved past.
|
|
401
|
+
|
|
402
|
+
- **Omitting `reason`:** an API-key caller may leave `reason` out and the platform records an automated correction reason derived from the source. A session caller must supply one; a blank string is rejected client-side either way.
|
|
316
403
|
- **`reason` vs `outcome_reason`:** `reason` is the amendment's own audit justification (why the record changed); `outcome_reason` is the business explanation of why the job failed or was cancelled. Use `outcome_reason` for failure explanations rather than burying them in `metadata` — it is a first-class field on the outcome and is returned on every `get_outcome_history()` row.
|
|
317
404
|
- **Clearing `outcome_reason`:** omit the argument to leave the stored value untouched; pass an empty string (`outcome_reason=""`) to clear it.
|
|
405
|
+
- **`metrics`:** both `report_outcome()` and `amend_outcome()` accept a `metrics` argument for recording the measurable facts behind an outcome.
|
|
406
|
+
|
|
407
|
+
### Recording Metric Facts
|
|
408
|
+
|
|
409
|
+
Beyond the single `outcome_value`, a job can carry the measurable facts its job type declares — `quality_rate` and its siblings — either with the outcome or later, once they are measurable.
|
|
410
|
+
|
|
411
|
+
```python
|
|
412
|
+
from revenium_middleware import JobContext
|
|
413
|
+
|
|
414
|
+
with JobContext("claim-8842", type="claims_triage") as job:
|
|
415
|
+
...
|
|
416
|
+
job.report_outcome(
|
|
417
|
+
execution_status="SUCCESS",
|
|
418
|
+
outcome_type="CONVERTED",
|
|
419
|
+
metrics=[
|
|
420
|
+
{"key": "quality_rate", "value": 0.93, "provenance": "MEASURED"},
|
|
421
|
+
{"key": "cases_closed", "value": 12},
|
|
422
|
+
],
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
# Two days later a human grades a sample of that same job's output.
|
|
426
|
+
handle = JobContext.attach("claim-8842")
|
|
427
|
+
handle.append_outcome_metrics([
|
|
428
|
+
{"key": "quality_rate", "value": 0.87, "provenance": "ATTESTED",
|
|
429
|
+
"reason": "graded sample of 200 cases"},
|
|
430
|
+
])
|
|
431
|
+
handle.close()
|
|
432
|
+
```
|
|
433
|
+
|
|
434
|
+
- **Declare the metric first:** a fact only lands if the job type's economics contract declares that key as a `PER_JOB` metric; an undeclared key is rejected with a 400. `quality_rate` is a rate and the platform range-checks it to 0..1.
|
|
435
|
+
- **Entry shape:** `key` and `value` are required; `provenance` (`MEASURED` | `SELF_REPORTED` | `DERIVED` | `ATTESTED`), `recordedBy`, `source`, `reason` and `recordedAt` are optional. Entries are sent exactly as you write them, so the fields you omit take the platform's defaults (`SELF_REPORTED`, the calling principal, `api`) instead of being guessed by the SDK. A missing `key` or `value` — or no entries at all on `append_outcome_metrics()` — raises `ValueError` before any HTTP request, on `JobContext` and `AgenticOutcomeClient` alike.
|
|
436
|
+
- **Append-only:** facts accumulate; the SDK never dedupes or replaces one, because the platform owns fact identity. `metrics=` on `amend_outcome()` appends as part of the amendment.
|
|
437
|
+
- **Not part of outcome history:** `get_outcome_history()` returns the outcome revisions only — appended facts do not appear in those rows.
|
|
438
|
+
- **Retries:** an append is retried only on `429`, which proves the platform rejected the request before recording anything. A `502`/`503`/`504` is raised instead of retried: the facts may already be recorded, and a second append is a second fact, so the decision to resend is yours (check the recorded facts first).
|
|
439
|
+
- **Locking is unaffected:** appending facts does not change the job's `entityVersion`, so the handle keeps the version it recorded and a following `amend_outcome()` still locks against it. (`report_outcome()` and `amend_outcome()` clear the recorded version when their response carries none, because those calls advance it; an append does not.)
|
|
440
|
+
- **Why it matters:** AI Alerts evaluate `QUALITY_RATE` from these facts, so a job whose integration emits none is invisible to those rules.
|
|
318
441
|
|
|
319
442
|
### Outcome Exceptions
|
|
320
443
|
|
|
@@ -325,7 +448,7 @@ All outcome exceptions are importable from `revenium_middleware` and share the `
|
|
|
325
448
|
| `OutcomeReportingError` | Base class — configuration failures (no API key available, unresolvable `team_id`) | Fix the key / team configuration |
|
|
326
449
|
| `OutcomeAlreadyReportedError` | Re-reporting a job that already has an outcome (backend 409) | Amend with `amend_outcome()` instead; the exception carries `reported_at` and `amendment_count` |
|
|
327
450
|
| `OutcomeNotReportedError` | Amending a job that has no outcome yet (backend 422) | Call `report_outcome()` first |
|
|
328
|
-
| `OutcomeAmendConflictError` | A concurrent amendment changed the outcome (backend 409, optimistic lock) |
|
|
451
|
+
| `OutcomeAmendConflictError` | A concurrent amendment changed the outcome (backend 409, optimistic lock) | Re-check the outcome against `get_outcome_history()`, then retry with `expected_entity_version=conflict.current_entity_version` — the SDK does not auto-retry |
|
|
329
452
|
|
|
330
453
|
### Low-Level Client
|
|
331
454
|
|
|
@@ -340,10 +463,11 @@ client = AgenticOutcomeClient(settings)
|
|
|
340
463
|
client.emit_completion(...) # one per LLM call
|
|
341
464
|
client.emit_tool_event(...) # one per tool / step
|
|
342
465
|
client.report_outcome(job_id, {...}) # close the job with a terminal outcome
|
|
466
|
+
client.append_outcome_metrics(job_id, [...]) # append declared per-job facts later
|
|
343
467
|
client.close()
|
|
344
468
|
```
|
|
345
469
|
|
|
346
|
-
The job is created implicitly by the first metric ingested for `agenticJobId`. Call `client.create_job(job_id)` explicitly if you need to record an agent run before emitting any metrics
|
|
470
|
+
The job is created implicitly by the first metric ingested for `agenticJobId`. Call `client.create_job(job_id)` explicitly if you need to record an agent run before emitting any metrics; it returns the created job resource merged over the fields you supplied, including the `entityVersion` an outcome amendment sends back as `expectedEntityVersion`.
|
|
347
471
|
|
|
348
472
|
See [`examples/agentic_outcomes/`](examples/agentic_outcomes/) for runnable demos (sales / coding / support) with configurable failure rates and outcome distributions.
|
|
349
473
|
|
|
@@ -553,7 +677,31 @@ with client.messages.stream(
|
|
|
553
677
|
print(text, end="", flush=True)
|
|
554
678
|
```
|
|
555
679
|
|
|
556
|
-
|
|
680
|
+
Async streaming is metered the same way, with the same metadata:
|
|
681
|
+
|
|
682
|
+
```python
|
|
683
|
+
import asyncio
|
|
684
|
+
import anthropic
|
|
685
|
+
import revenium_middleware.anthropic
|
|
686
|
+
|
|
687
|
+
client = anthropic.AsyncAnthropic()
|
|
688
|
+
|
|
689
|
+
async def main():
|
|
690
|
+
async with client.messages.stream(
|
|
691
|
+
model="claude-opus-4-7",
|
|
692
|
+
max_tokens=200,
|
|
693
|
+
messages=[{"role": "user", "content": "Tell me a story"}],
|
|
694
|
+
usage_metadata={"task_type": "creative"}
|
|
695
|
+
) as stream:
|
|
696
|
+
async for text in stream.text_stream:
|
|
697
|
+
print(text, end="", flush=True)
|
|
698
|
+
# await stream.get_final_message() works too; either way the
|
|
699
|
+
# completion is metered once when the block exits.
|
|
700
|
+
|
|
701
|
+
asyncio.run(main())
|
|
702
|
+
```
|
|
703
|
+
|
|
704
|
+
**Note:** The middleware wraps the `messages.create` and `messages.stream` endpoints, sync and async alike (including `create(stream=True)`). Other Anthropic SDK features work normally but aren't metered.
|
|
557
705
|
|
|
558
706
|
#### AWS Bedrock
|
|
559
707
|
|
|
@@ -586,7 +734,7 @@ message = client.messages.create(
|
|
|
586
734
|
| Variable | Description | Default |
|
|
587
735
|
|----------|-------------|---------|
|
|
588
736
|
| `AWS_REGION` | AWS region for Bedrock | `us-east-1` |
|
|
589
|
-
| `REVENIUM_BEDROCK_DISABLE` | Set to `1` to disable Bedrock support | Not set |
|
|
737
|
+
| `REVENIUM_BEDROCK_DISABLE` | Set to `1` to disable Bedrock support (Bedrock detection only - Foundry detection is unaffected) | Not set |
|
|
590
738
|
|
|
591
739
|
**AWS authentication** uses the standard credential chain: environment variables, `~/.aws/credentials`, IAM roles, AWS SSO. Required permissions: `bedrock:InvokeModel` and `bedrock:InvokeModelWithResponseStream`.
|
|
592
740
|
|
|
@@ -608,6 +756,32 @@ message = client.messages.create(
|
|
|
608
756
|
|
|
609
757
|
For other models, the middleware uses the format `anthropic.{model_name}`.
|
|
610
758
|
|
|
759
|
+
#### Microsoft Foundry
|
|
760
|
+
|
|
761
|
+
Claude served through Microsoft Foundry is metered by the same patched endpoints as the
|
|
762
|
+
direct Anthropic API - the Anthropic SDK's Foundry clients need no extra setup.
|
|
763
|
+
|
|
764
|
+
```python
|
|
765
|
+
import anthropic
|
|
766
|
+
import revenium_middleware.anthropic
|
|
767
|
+
|
|
768
|
+
# Foundry is detected from the client class, so a custom base_url is fine too
|
|
769
|
+
client = anthropic.AnthropicFoundry(
|
|
770
|
+
resource="your-resource", # or ANTHROPIC_FOUNDRY_RESOURCE
|
|
771
|
+
)
|
|
772
|
+
|
|
773
|
+
message = client.messages.create(
|
|
774
|
+
model="claude-opus-4-7",
|
|
775
|
+
max_tokens=100,
|
|
776
|
+
messages=[{"role": "user", "content": "Hello from Foundry!"}]
|
|
777
|
+
)
|
|
778
|
+
```
|
|
779
|
+
|
|
780
|
+
Foundry usage is reported with provider `Foundry` and model source `ANTHROPIC`, so the spend
|
|
781
|
+
is separated from direct-Anthropic totals while still priced against the Anthropic rate card
|
|
782
|
+
that Foundry bills at. `AsyncAnthropicFoundry` is metered the same way, as is
|
|
783
|
+
`client.messages.stream()`.
|
|
784
|
+
|
|
611
785
|
**Examples:** `examples/anthropic/` - `anthropic-basic.py`, `anthropic-streaming.py`, `anthropic-bedrock.py`, `anthropic-advanced.py`
|
|
612
786
|
|
|
613
787
|
---
|
|
@@ -774,15 +948,187 @@ response = litellm.completion(
|
|
|
774
948
|
|
|
775
949
|
#### Proxy Mode
|
|
776
950
|
|
|
777
|
-
|
|
951
|
+
`ReveniumGuardrail` is the LiteLLM proxy integration. It is a LiteLLM
|
|
952
|
+
`CustomGuardrail` that **enforces the caller's budget before** the proxied call and
|
|
953
|
+
**meters usage after** it — successes, failures and streamed responses alike.
|
|
954
|
+
|
|
955
|
+
```bash
|
|
956
|
+
pip install "revenium-python-sdk[litellm-proxy]" # requires Python 3.10+
|
|
957
|
+
```
|
|
778
958
|
|
|
779
959
|
```yaml
|
|
780
|
-
|
|
781
|
-
|
|
960
|
+
guardrails:
|
|
961
|
+
- guardrail_name: "revenium"
|
|
962
|
+
litellm_params:
|
|
963
|
+
guardrail: revenium_middleware.litellm.proxy.guardrail.ReveniumGuardrail
|
|
964
|
+
mode:
|
|
965
|
+
- "pre_call" # budget enforcement
|
|
966
|
+
- "post_call" # usage metering
|
|
967
|
+
default_on: true
|
|
968
|
+
```
|
|
969
|
+
|
|
970
|
+
`guardrails` is a **top-level** key, not a member of `litellm_settings`. Nested
|
|
971
|
+
under `litellm_settings` it reaches LiteLLM's legacy v1 guardrail loader, which
|
|
972
|
+
expects a different shape and exits the proxy at startup with
|
|
973
|
+
`GuardrailItem() argument after ** must be a mapping, not str`.
|
|
974
|
+
|
|
975
|
+
`mode` must be a **list** to enable both hooks; a single string restricts the
|
|
976
|
+
guardrail to that one event type. `pre_call` alone enforces without metering;
|
|
977
|
+
`post_call` alone meters without enforcing.
|
|
978
|
+
|
|
979
|
+
**Budget enforcement** reuses the SDK's own circuit breaker, so a proxy enforces
|
|
980
|
+
exactly what every other Revenium integration enforces — including department
|
|
981
|
+
(org-unit) budgets. It is opt-in via `REVENIUM_CIRCUIT_BREAKER_ENABLED=true`; see
|
|
982
|
+
[Cost Controls](#cost-controls). A blocked call never reaches the provider and the
|
|
983
|
+
caller receives HTTP 429:
|
|
984
|
+
|
|
985
|
+
```json
|
|
986
|
+
{"error": {"message": "Request blocked by Revenium enforcement rule: Team Budget",
|
|
987
|
+
"type": "budget_exceeded", "guardrail": "revenium", "model": "gpt-4o",
|
|
988
|
+
"budgets": [{"name": "Team Budget", "ruleId": 7, "threshold": 10.0,
|
|
989
|
+
"currentValue": 11.5, "resetsAt": "2026-10-01T00:00:00Z"}]}}
|
|
990
|
+
```
|
|
991
|
+
|
|
992
|
+
Enforcement **fails open**: if the enforcement path is unreachable or misbehaves,
|
|
993
|
+
the call proceeds. Metering is likewise non-disruptive — nothing in the post-call
|
|
994
|
+
path can turn a successful LLM call into an error for the client.
|
|
995
|
+
|
|
996
|
+
**Attribution** travels as `x-revenium-*` request headers (subscriber, organization,
|
|
997
|
+
product, trace, task type, agent, subscription, quality score, and `x-revenium-effort`
|
|
998
|
+
for reasoning effort), with the calling virtual key's metadata as the fallback —
|
|
999
|
+
`revenium_user_id`, `revenium_organization_name`, `revenium_key_name`, and
|
|
1000
|
+
`revenium_agentic_job_*`. Agentic job tags (`x-revenium-agentic-job-id`, `-name`,
|
|
1001
|
+
`-type`, `-version`) ride along for cost/ROI correlation; the job id is required for
|
|
1002
|
+
the others to be recorded.
|
|
1003
|
+
|
|
1004
|
+
##### Counting a Claude Code call once (shared call id)
|
|
1005
|
+
|
|
1006
|
+
If you run Claude Code through your proxy **and** point Claude Code's own usage
|
|
1007
|
+
reporting at Revenium, every call is recorded twice: Claude Code files it under an
|
|
1008
|
+
identifier of its own making, the proxy files it under another, and the two can
|
|
1009
|
+
never match. `ReveniumGuardrail` makes them match, and does so by default.
|
|
1010
|
+
|
|
1011
|
+
The guardrail mints one identifier per proxied `/v1/messages` request, returns it to
|
|
1012
|
+
the client as the `request-id` and `x-revenium-transaction-id` response headers, and
|
|
1013
|
+
reports the same value as the call's transaction id. Claude Code copies `request-id`
|
|
1014
|
+
onto its own record, Revenium's duplicate check sees two records with one identifier,
|
|
1015
|
+
and one call becomes one record.
|
|
1016
|
+
|
|
1017
|
+
**It is on by default.** To opt out and keep LiteLLM's own response id as the
|
|
1018
|
+
transaction id, with no `request-id` header added:
|
|
1019
|
+
|
|
1020
|
+
```bash
|
|
1021
|
+
export REVENIUM_LITELLM_SHARED_CALL_ID=false
|
|
1022
|
+
```
|
|
1023
|
+
|
|
1024
|
+
New calls go straight back to two records when you do; records already merged stay
|
|
1025
|
+
merged. Four things have to be true for it to work:
|
|
1026
|
+
|
|
1027
|
+
* **LiteLLM 1.93.0 or newer.** That is the floor of the `litellm-proxy` extra, which
|
|
1028
|
+
is what installs the guardrail. On an older LiteLLM the guardrail logs one warning
|
|
1029
|
+
at startup and behaves exactly as it does when opted out: no header is added,
|
|
1030
|
+
the provider's response id is reported, and you get two records rather than none.
|
|
1031
|
+
* **`mode` includes `pre_call`.** The identifier is minted in the pre-call hook, and
|
|
1032
|
+
LiteLLM runs that hook only when the configured mode asks for it. A
|
|
1033
|
+
`mode: ["post_call"]` proxy mints nothing, and the guardrail says so at startup
|
|
1034
|
+
with one warning naming this variable. Add `pre_call` to the mode, or set the
|
|
1035
|
+
variable to `false` if the proxy only meters.
|
|
1036
|
+
* **The Anthropic messages route.** Other routes are untouched, so the provider's own
|
|
1037
|
+
`request-id` on the pass-through route is never overwritten.
|
|
1038
|
+
* **Both sides report to the same Revenium team.** The duplicate check is
|
|
1039
|
+
team-scoped.
|
|
1040
|
+
|
|
1041
|
+
**Point Claude Code at the proxy root, not at `/anthropic`.** Set `ANTHROPIC_BASE_URL`
|
|
1042
|
+
to the proxy's own base URL so Claude Code calls `<proxy>/v1/messages`:
|
|
1043
|
+
|
|
1044
|
+
```bash
|
|
1045
|
+
export ANTHROPIC_BASE_URL=https://proxy.example.com
|
|
1046
|
+
```
|
|
1047
|
+
|
|
1048
|
+
LiteLLM also offers an Anthropic pass-through at `<proxy>/anthropic/v1/messages`, and
|
|
1049
|
+
its own API reference recommends that route over `/v1/messages`. The shared identifier
|
|
1050
|
+
is minted only on `/v1/messages`. A proxy serving the pass-through route hands the
|
|
1051
|
+
provider's response straight back, so nothing is minted there, no `request-id` of ours
|
|
1052
|
+
is returned, and the counter described below stays silent on that route by design.
|
|
1053
|
+
The flag will appear to be on and the calls will keep being counted twice, with
|
|
1054
|
+
nothing in the proxy log to say why. If your Claude Code base URL ends in
|
|
1055
|
+
`/anthropic`, drop that suffix.
|
|
1056
|
+
|
|
1057
|
+
Only the guardrail mints. A proxy still on the deprecated callback alone gets no
|
|
1058
|
+
identifier and keeps reporting two records, which is one more reason to migrate.
|
|
1059
|
+
|
|
1060
|
+
One caution for a proxy running **both** the guardrail and the deprecated callback
|
|
1061
|
+
without `default_on: true`. That configuration meters every call twice already, and
|
|
1062
|
+
this flag hides the symptom rather than fixing it: both rows now carry the same
|
|
1063
|
+
identifier and Revenium's duplicate check keeps one. The configuration is still
|
|
1064
|
+
wrong. Delete the `litellm_settings.callbacks` entry.
|
|
1065
|
+
|
|
1066
|
+
Also do not register `ReveniumGuardrail` in `litellm_settings.callbacks` as well as
|
|
1067
|
+
in the `guardrails` block. LiteLLM keys registered callbacks on the class name plus
|
|
1068
|
+
its simple attributes, and the two instances differ, so both are registered and every
|
|
1069
|
+
logging hook runs twice.
|
|
1070
|
+
|
|
1071
|
+
If the flag is on and an Anthropic messages call is metered with no identifier on it,
|
|
1072
|
+
the guardrail counts it and logs a warning at most once a minute with the running
|
|
1073
|
+
total, so a mint that quietly stopped shows up in the proxy log rather than as a
|
|
1074
|
+
return of double counting. Your other routes never carry an identifier and are never
|
|
1075
|
+
counted or warned about, so a proxy that also serves chat completions or embeddings
|
|
1076
|
+
stays quiet.
|
|
1077
|
+
|
|
1078
|
+
##### Migrating from the callback
|
|
1079
|
+
|
|
1080
|
+
`revenium_middleware.litellm.proxy.middleware.MiddlewareHandler` — the
|
|
1081
|
+
`litellm_settings.callbacks` entry `proxy_handler_instance` — is **deprecated**. It
|
|
1082
|
+
meters but never enforces a budget. It keeps working in this release and emits a
|
|
1083
|
+
`DeprecationWarning` (and a log line) when the proxy builds it.
|
|
1084
|
+
|
|
1085
|
+
To migrate, delete the callbacks entry and add the `guardrails` block above:
|
|
1086
|
+
|
|
1087
|
+
```diff
|
|
1088
|
+
litellm_settings:
|
|
1089
|
+
- callbacks: ["revenium_middleware.litellm.proxy.middleware.proxy_handler_instance"]
|
|
1090
|
+
+
|
|
1091
|
+
+guardrails:
|
|
1092
|
+
+ - guardrail_name: "revenium"
|
|
1093
|
+
+ litellm_params:
|
|
1094
|
+
+ guardrail: revenium_middleware.litellm.proxy.guardrail.ReveniumGuardrail
|
|
1095
|
+
+ mode: ["pre_call", "post_call"]
|
|
1096
|
+
+ default_on: true
|
|
782
1097
|
```
|
|
783
1098
|
|
|
784
|
-
|
|
785
|
-
|
|
1099
|
+
Nothing else changes: the same headers, the same metered fields. Metered rows
|
|
1100
|
+
record `middleware_source: "GUARDRAIL"` instead of `"PROXY"`.
|
|
1101
|
+
|
|
1102
|
+
Leaving both enabled would meter every call twice. As a safety net for a proxy
|
|
1103
|
+
mid-migration, when the guardrail is configured to run on every request
|
|
1104
|
+
(`default_on: true` with `post_call` among its modes) it claims metering ownership
|
|
1105
|
+
and the deprecated callback stops submitting rows, logging once to say so. That net
|
|
1106
|
+
does **not** apply to a guardrail without `default_on`, without `post_call`, or
|
|
1107
|
+
configured with a per-tag `Mode` (which selects hooks per request): such a guardrail
|
|
1108
|
+
may not run on a given request, and suppressing the callback could drop metering
|
|
1109
|
+
entirely. A per-tag configuration logs, at info level, that it is not claiming
|
|
1110
|
+
ownership. Delete the callbacks entry rather than relying on the net.
|
|
1111
|
+
|
|
1112
|
+
##### Client API and guardrails
|
|
1113
|
+
|
|
1114
|
+
`CustomGuardrail`'s lifecycle hooks cannot be hosted by LiteLLM's client API.
|
|
1115
|
+
Verified against **litellm 1.100.1**: `async_pre_call_hook`,
|
|
1116
|
+
`async_post_call_success_hook` and `async_post_call_failure_hook` are dispatched
|
|
1117
|
+
only from `litellm/proxy/utils.py` (`ProxyLogging`) and
|
|
1118
|
+
`litellm/proxy/common_request_processing.py`. The client path
|
|
1119
|
+
(`litellm_core_utils/litellm_logging.py`) dispatches only the `CustomLogger`
|
|
1120
|
+
logging events, so a `CustomGuardrail` added to `litellm.callbacks` without the
|
|
1121
|
+
proxy running would be a logger with no pre-call hook and no ability to block a
|
|
1122
|
+
call — the enforcement half would silently not exist.
|
|
1123
|
+
|
|
1124
|
+
So the client integration keeps its own path, unchanged: use
|
|
1125
|
+
`revenium_middleware.litellm.client` as documented above. Note that the LiteLLM
|
|
1126
|
+
client wrapper meters but does not currently run the pre-call circuit breaker —
|
|
1127
|
+
enforcement in client mode is available today through the OpenAI middleware, and
|
|
1128
|
+
through this guardrail in proxy mode. Both the guardrail and the client wrapper
|
|
1129
|
+
already share their metering plumbing (`revenium_middleware._core`: field
|
|
1130
|
+
extraction, cache-token extraction and `submit_ai_event`), so the guardrail adds no
|
|
1131
|
+
second copy of it.
|
|
786
1132
|
|
|
787
1133
|
#### LiteLLM Decorators
|
|
788
1134
|
|
|
@@ -1142,6 +1488,7 @@ Enhanced observability fields for tracking AI operations across environments, re
|
|
|
1142
1488
|
| `transaction_name` | `REVENIUM_TRANSACTION_NAME` | Human-friendly operation name | Label operations (e.g., `"Generate Response"`, `"Analyze Sentiment"`) |
|
|
1143
1489
|
| `retry_number` | `REVENIUM_RETRY_NUMBER` | Retry attempt number (0 = first attempt) | Track retry attempts for failed operations |
|
|
1144
1490
|
| `ticket_id` | `REVENIUM_TICKET_ID` | External ticket or issue ID (e.g., Jira, Linear) (max 256 chars) | Attribute AI costs to individual tickets or issues |
|
|
1491
|
+
| `agent_version` | _(none — per call only)_ | Version of the AI agent that produced the call (max 64 chars) | Compare cost across agent releases; not `agentic_job_version`, which versions the job definition |
|
|
1145
1492
|
| `skill_name` | `REVENIUM_SKILL_NAME` | Name of the agent skill that produced the call (max 256 chars) | Attribute AI costs to the skill that generated them |
|
|
1146
1493
|
| `skill_source` | `REVENIUM_SKILL_SOURCE` | Where the skill was loaded from — accepted values: `bundled`, `projectSettings`, `userSettings`, `plugin` (case-sensitive) | Classify skill origin in the shared skill catalog |
|
|
1147
1494
|
| `skill_kind` | `REVENIUM_SKILL_KIND` | Kind of skill invoked — accepted value: `workflow` (omit otherwise) | Distinguish workflow skills in reporting |
|
|
@@ -1176,7 +1523,8 @@ response = client.chat.completions.create(
|
|
|
1176
1523
|
"trace_name": "Support Chat Session",
|
|
1177
1524
|
"transaction_name": "Generate Response",
|
|
1178
1525
|
"parent_transaction_id": "parent-txn-123",
|
|
1179
|
-
"ticket_id": "JIRA-123"
|
|
1526
|
+
"ticket_id": "JIRA-123",
|
|
1527
|
+
"agent_version": "1.4.2"
|
|
1180
1528
|
}
|
|
1181
1529
|
)
|
|
1182
1530
|
```
|
|
@@ -1520,6 +1868,29 @@ Set `REVENIUM_CB_FAIL_MODE=closed` to refuse calls until at least one rule fetch
|
|
|
1520
1868
|
|
|
1521
1869
|
Rules with `shadowMode: true` are observe-and-log: they are skipped by `check_enforcement`. Use shadow mode on the server side to audit a rule before flipping it to enforce.
|
|
1522
1870
|
|
|
1871
|
+
### Inspecting a Rule and Its Roster
|
|
1872
|
+
|
|
1873
|
+
Two read-only calls answer "why was this caller blocked, and who else does this rule cover?" without going anywhere near the pre-call path. Both talk to the server directly, neither is cached, and neither reads or writes the cache `check_enforcement` evaluates — so what they report is what the server holds right now.
|
|
1874
|
+
|
|
1875
|
+
```python
|
|
1876
|
+
from revenium_middleware._core import (
|
|
1877
|
+
fetch_enforcement_rule,
|
|
1878
|
+
fetch_enforcement_rule_roster,
|
|
1879
|
+
)
|
|
1880
|
+
|
|
1881
|
+
rule = fetch_enforcement_rule("mN3xpQz") # one rule, or None
|
|
1882
|
+
roster = fetch_enforcement_rule_roster("mN3xpQz") # who it measures, or None
|
|
1883
|
+
|
|
1884
|
+
if roster:
|
|
1885
|
+
print(f"{roster['blockedCount']} over the cap, {roster['warnedCount']} warned")
|
|
1886
|
+
for row in roster["rows"]:
|
|
1887
|
+
print(f"{row['label']}: ${row['spend']} / ${row['limit']} ({row['band']})")
|
|
1888
|
+
```
|
|
1889
|
+
|
|
1890
|
+
`fetch_enforcement_rule_roster` takes `page`, `size`, `search` and `band` (`BLOCKED`, `WARNED`, `UNDER`, `ALL`); the server does the filtering, sorting, banding and paging, and the three band counts always describe the whole roster rather than the page you asked for. Both calls return `None` rather than raising when the team has no such compiled rule, the rule has no reading yet, or the enforcement API cannot be reached — the same fail-open posture as the rest of the circuit breaker. An empty `rule_id` raises `ValueError`.
|
|
1891
|
+
|
|
1892
|
+
The background poller is unaffected: it keeps reading the **whole team's** rules every `REVENIUM_CB_POLL_INTERVAL_SECONDS`. That is deliberate. The server computes the department-budget maps team-wide and attaches them to the team-wide read, so a poll narrowed to a single rule would stop receiving them and department budgets would quietly stop blocking anyone. Narrowing is an explicit, opt-in inspection call and never the refresh.
|
|
1893
|
+
|
|
1523
1894
|
### End-to-End Example
|
|
1524
1895
|
|
|
1525
1896
|
See [`examples/openai/openai_blocking_demo.py`](examples/openai/openai_blocking_demo.py) for a runnable end-to-end demo using a seeded budget rule.
|
|
@@ -1593,9 +1964,10 @@ print(get_buffer_stats())
|
|
|
1593
1964
|
| `REVENIUM_AGENTIC_JOB_NAME` | - | Human-readable agentic job name |
|
|
1594
1965
|
| `REVENIUM_AGENTIC_JOB_TYPE` | - | Agentic job type category |
|
|
1595
1966
|
| `REVENIUM_AGENTIC_JOB_VERSION` | - | Agentic job version |
|
|
1596
|
-
| `
|
|
1967
|
+
| `REVENIUM_WRITE_API_KEY` | - | Primary write-scope key (`rev_sk_`) for the agentic outcomes API (report/amend/history); falls back to `REVENIUM_OUTCOME_API_KEY` (deprecated), then `REVENIUM_METERING_API_KEY` |
|
|
1968
|
+
| `REVENIUM_OUTCOME_API_KEY` | - | Deprecated fallback name for the write-scope key; used only when `REVENIUM_WRITE_API_KEY` is unset |
|
|
1597
1969
|
| `REVENIUM_PROFITSTREAM_BASE_URL` | `https://api.revenium.io` | Agentic outcomes API base URL |
|
|
1598
|
-
| `REVENIUM_BEDROCK_DISABLE` | - | Set to `1` to disable Bedrock auto-detection |
|
|
1970
|
+
| `REVENIUM_BEDROCK_DISABLE` | - | Set to `1` to disable Bedrock auto-detection; Foundry detection is unaffected |
|
|
1599
1971
|
| `REVENIUM_BUFFER_MAX_SIZE` | `1000` | Store-and-forward buffer capacity (oldest events evicted when full) |
|
|
1600
1972
|
| `REVENIUM_BUFFER_FLUSH_INTERVAL` | `30` | Seconds between automatic replay attempts for buffered events |
|
|
1601
1973
|
|
|
@@ -1640,7 +2012,7 @@ Per-call `usage_metadata` values take precedence over the `REVENIUM_AGENTIC_JOB_
|
|
|
1640
2012
|
|
|
1641
2013
|
**Debug mode:** Set `REVENIUM_LOG_LEVEL=DEBUG` to see detailed provider detection, routing decisions, and metering payloads.
|
|
1642
2014
|
|
|
1643
|
-
**Force direct Anthropic API:** Set `REVENIUM_BEDROCK_DISABLE=1` to disable Bedrock auto-detection.
|
|
2015
|
+
**Force direct Anthropic API (instead of Bedrock):** Set `REVENIUM_BEDROCK_DISABLE=1` to disable Bedrock auto-detection. Foundry detection is unaffected - a Foundry client is still labelled `Foundry`.
|
|
1644
2016
|
|
|
1645
2017
|
**Check initialization status (Anthropic):** Use `revenium_middleware.anthropic.is_initialized()` to verify setup.
|
|
1646
2018
|
|