wagent-framework 2.0.0a1__tar.gz → 2.0.0a2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/PKG-INFO +12 -5
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/README.md +11 -4
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/setup.py +1 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_cli.py +6 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_evaluation.py +27 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_local_runtime.py +75 -0
- wagent_framework-2.0.0a2/tests/test_pricing.py +71 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_react_agent.py +154 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_tui.py +4 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/__init__.py +31 -13
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/agents/__init__.py +2 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/agents/persistence.py +53 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/agents/react.py +192 -11
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/agents/types.py +59 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/cli.py +51 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/evaluation.py +44 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/local_runtime.py +95 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/__init__.py +16 -0
- wagent_framework-2.0.0a2/w_agent/models/pricing.py +198 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sessions.py +40 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/mcp_client.py +1 -1
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tui.py +49 -4
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/PKG-INFO +12 -5
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/SOURCES.txt +2 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/LICENSE +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/setup.cfg +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_agent_templates.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_aop.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_component_chain.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_compositions.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_config.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_container.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_deployment.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_distributed_lock.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_doctor.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_event_bus.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_http_provider_templates.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_kernel_events.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_kernel_loading.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_kernel_plugins.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_kernel_registry.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_legacy_compat.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_lifecycle.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_mcp_client.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_model_invocation.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_model_probing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_model_routing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_model_testing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_model_types.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_observability.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_openai_compatible_provider.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_package_integrity.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_sandbox.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_sandbox_runtime.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_scanner.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_sessions.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_timeout.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_tool_adapters.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_tool_loading.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_tool_runtime.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_workflow_adapters.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/tests/test_workflow_engine.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/__main__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/agents/templates.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/aop/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/aop/aspects.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/aop/joinpoint.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/aop/pointcut.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/aop/proxy_factory.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/compat.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/compositions.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/config/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/config/dynamic_config.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/container/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/container/bean_factory.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/container/reflection_cache.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/core/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/core/agent.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/core/decorators.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/core/doctor.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/core/event_bus.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/deployment/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/deployment/fastapi_depends.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/distributed/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/distributed/lock.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/distributed/lock_pool.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/exceptions/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/exceptions/framework_errors.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/events.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/exceptions.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/loading.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/plugins.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/registry.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/kernel/scope.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/lifecycle/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/lifecycle/graceful_shutdown.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/lifecycle/manager.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/lifecycle/order.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/migration/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/migration/migrate_previous.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/errors.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/http_provider.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/invocation.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/native_mappings.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/openai_compatible.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/probing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/provider.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/routing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/templates.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/models/types.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/observability/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/observability/health.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/observability/logging.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/observability/metrics.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/observability/tracing.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/resilience/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/resilience/bulkhead.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/resilience/timeout.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/docker.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/process.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/registry.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/types.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/sandbox/unsafe_local.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/scanner/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/scanner/cache.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/scanner/parallel_scanner.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/security/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/security/mcp_auth.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/sandbox/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/sandbox/nsjail_sandbox.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/sandbox/wasm_sandbox.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/signature.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/skills/skill.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/testing/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/testing/mock_utils.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/testing/model_provider.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/command_adapter.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/execution.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/http_adapter.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/langchain_adapter.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/loading.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/mcp_adapter.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/policies.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/python.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/registry.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/sandbox_adapter.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/tools/types.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/__init__.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/adapters.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/engine.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/persistence.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/registry.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/w_agent/workflows/types.py +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/dependency_links.txt +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/entry_points.txt +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/requires.txt +0 -0
- {wagent_framework-2.0.0a1 → wagent_framework-2.0.0a2}/wagent_framework.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: wagent-framework
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.0a2
|
|
4
4
|
Summary: Open, composable Python framework for building agents
|
|
5
5
|
Home-page: https://github.com/LuckyStar2456/W-Agent-FrameWork
|
|
6
6
|
Author: LuckyStar2456
|
|
@@ -73,7 +73,7 @@ Dynamic: summary
|
|
|
73
73
|
|
|
74
74
|
W-Agent 是一个面向本地开发者的 Python 开源 Agent 开发框架。它的目标不是提供托管平台或固定 Harness,而是提供稳定、可扩展的协议与可自由装配的模块,让开发者能够替换模型、路由、Agent Loop、Workflow、工具、状态、沙箱和界面实现。
|
|
75
75
|
|
|
76
|
-
当前稳定发布版本是 `1.5.2
|
|
76
|
+
当前稳定发布版本是 `1.5.2`,最新 Alpha 为 `2.0.0a2`。1.x 工程底座继续保留,微内核、模型、工具、单 Agent 与本地 Workflow 基础已经实现,其余下一代能力按路线图分阶段交付。文档使用明确状态标记,避免把规划能力描述为现有能力。
|
|
77
77
|
|
|
78
78
|
## 状态标记
|
|
79
79
|
|
|
@@ -109,7 +109,7 @@ W-Agent 遵循以下原则:
|
|
|
109
109
|
- Skill 加载、签名校验、MCP JWT 认证、Redis 分布式锁。
|
|
110
110
|
- LangChain 工具适配、FastAPI 示例集成和测试辅助设施。
|
|
111
111
|
|
|
112
|
-
以下 Phase 1 能力在当前 `2.0.
|
|
112
|
+
以下 Phase 1 能力在当前 `2.0.0a2` 源码中为 `Implemented`:
|
|
113
113
|
|
|
114
114
|
- `PluginSpec`、装饰器、YAML 引用与 Python entry point 发现。
|
|
115
115
|
- 统一、版本感知、分作用域的能力注册表与不可变快照。
|
|
@@ -131,12 +131,13 @@ W-Agent 遵循以下原则:
|
|
|
131
131
|
- 注册即安全探测的 `ModelRegistrationProbeService` 与外部路由健康桥接;直接调用 `ModelRegistry.register()` 仍保持无副作用。
|
|
132
132
|
- 工具 Definition/Binding/Registry 分层、Python/HTTP/无 Shell 命令模板、MCP 绑定与 2026-07-28 stdio/Streamable HTTP 客户端、参数校验、权限/逐调用审批、超时/取消和 Prompt-free 审计。
|
|
133
133
|
- 可替换 `AgentLoop` 协议与有界单 Agent `ReactAgentLoop`,覆盖模型→工具→结果→模型闭环、严格 Token 预算、JSONL RunEvent/尝试账本记录和审批断点恢复。
|
|
134
|
+
- 应用提供的版本化价格表、普通/缓存输入与输出费用计量、失败关闭的 Run 费用预算,以及 Session/Checkpoint/评测费用可见性。
|
|
134
135
|
- 统一 `WorkflowRegistry`、可替换 `WorkflowEngineProtocol` 与 `LocalWorkflowEngine`,支持静态 DAG、状态图、Python 入口、节点事件、取消,以及内存/JSONL 节点边界暂停恢复。
|
|
135
136
|
- 统一 `SandboxProvider`/`SandboxRegistry`、Docker/OCI 生命周期后端、显式运行时授权的 `UnsafeLocalSandboxProvider`,以及受工具策略保护的 `sandbox_command_tool()`。
|
|
136
137
|
- Agent/Workflow 双向适配器,以及可完全覆盖的客服/RAG 与编码 Agent 模板。
|
|
137
138
|
- `CompositionManifest` 的确定性编码、安全预览,以及冲突安全的本地版本和别名管理。
|
|
138
139
|
- 本地 Session 创建/列表/归档、JSON 持久化、跨 Run 文本上下文与审批恢复协调。
|
|
139
|
-
- 严格本地 JSON 装配的文本 Agent CLI/TUI
|
|
140
|
+
- 严格本地 JSON 装配的文本 Agent CLI/TUI 运行入口,使用环境变量凭据引用、逐次调用确认、可见 Token/费用预算与计量。
|
|
140
141
|
- 确定性脚本化 Model Provider、显式授权的 JSONL 录制/顺序回放,以及本地评测运行器与脱敏 JSON 报告。
|
|
141
142
|
|
|
142
143
|
当前 `BaseAgent` 仍是简单的 1.x 抽象;`LegacyAgentAdapter` 已能把它严格桥接为文本 Workflow 节点,新的 ReAct Runtime 独立提供。专用 OpenAI Responses 与 vLLM 差异适配、模型跨流恢复、多模态/工具事件通用回放、并行或嵌套 Workflow、装配依赖安装/加载确认和完整交互 TUI 仍为 `Planned`,不能当作现成功能使用。Docker 后端已有模拟 CLI 生命周期测试,但不代表当前机器已安装或启动 Docker;厂商模板经过模拟传输测试,也不代表所有远程型号已经在线验证。
|
|
@@ -159,6 +160,12 @@ W-Agent 遵循以下原则:
|
|
|
159
160
|
pip install wagent-framework
|
|
160
161
|
```
|
|
161
162
|
|
|
163
|
+
安装最新 Alpha:
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
pip install --pre wagent-framework==2.0.0a2
|
|
167
|
+
```
|
|
168
|
+
|
|
162
169
|
可选依赖:
|
|
163
170
|
|
|
164
171
|
```bash
|
|
@@ -168,7 +175,7 @@ pip install "wagent-framework[wasm]"
|
|
|
168
175
|
pip install "wagent-framework[tui]"
|
|
169
176
|
```
|
|
170
177
|
|
|
171
|
-
PyPI 1.x 支持 Python 3.9
|
|
178
|
+
PyPI 1.x 支持 Python 3.9+;`2.0.0a2` 与后续 2.x Alpha 要求 Python 3.11+。
|
|
172
179
|
|
|
173
180
|
## 1.x 最小示例
|
|
174
181
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
W-Agent 是一个面向本地开发者的 Python 开源 Agent 开发框架。它的目标不是提供托管平台或固定 Harness,而是提供稳定、可扩展的协议与可自由装配的模块,让开发者能够替换模型、路由、Agent Loop、Workflow、工具、状态、沙箱和界面实现。
|
|
6
6
|
|
|
7
|
-
当前稳定发布版本是 `1.5.2
|
|
7
|
+
当前稳定发布版本是 `1.5.2`,最新 Alpha 为 `2.0.0a2`。1.x 工程底座继续保留,微内核、模型、工具、单 Agent 与本地 Workflow 基础已经实现,其余下一代能力按路线图分阶段交付。文档使用明确状态标记,避免把规划能力描述为现有能力。
|
|
8
8
|
|
|
9
9
|
## 状态标记
|
|
10
10
|
|
|
@@ -40,7 +40,7 @@ W-Agent 遵循以下原则:
|
|
|
40
40
|
- Skill 加载、签名校验、MCP JWT 认证、Redis 分布式锁。
|
|
41
41
|
- LangChain 工具适配、FastAPI 示例集成和测试辅助设施。
|
|
42
42
|
|
|
43
|
-
以下 Phase 1 能力在当前 `2.0.
|
|
43
|
+
以下 Phase 1 能力在当前 `2.0.0a2` 源码中为 `Implemented`:
|
|
44
44
|
|
|
45
45
|
- `PluginSpec`、装饰器、YAML 引用与 Python entry point 发现。
|
|
46
46
|
- 统一、版本感知、分作用域的能力注册表与不可变快照。
|
|
@@ -62,12 +62,13 @@ W-Agent 遵循以下原则:
|
|
|
62
62
|
- 注册即安全探测的 `ModelRegistrationProbeService` 与外部路由健康桥接;直接调用 `ModelRegistry.register()` 仍保持无副作用。
|
|
63
63
|
- 工具 Definition/Binding/Registry 分层、Python/HTTP/无 Shell 命令模板、MCP 绑定与 2026-07-28 stdio/Streamable HTTP 客户端、参数校验、权限/逐调用审批、超时/取消和 Prompt-free 审计。
|
|
64
64
|
- 可替换 `AgentLoop` 协议与有界单 Agent `ReactAgentLoop`,覆盖模型→工具→结果→模型闭环、严格 Token 预算、JSONL RunEvent/尝试账本记录和审批断点恢复。
|
|
65
|
+
- 应用提供的版本化价格表、普通/缓存输入与输出费用计量、失败关闭的 Run 费用预算,以及 Session/Checkpoint/评测费用可见性。
|
|
65
66
|
- 统一 `WorkflowRegistry`、可替换 `WorkflowEngineProtocol` 与 `LocalWorkflowEngine`,支持静态 DAG、状态图、Python 入口、节点事件、取消,以及内存/JSONL 节点边界暂停恢复。
|
|
66
67
|
- 统一 `SandboxProvider`/`SandboxRegistry`、Docker/OCI 生命周期后端、显式运行时授权的 `UnsafeLocalSandboxProvider`,以及受工具策略保护的 `sandbox_command_tool()`。
|
|
67
68
|
- Agent/Workflow 双向适配器,以及可完全覆盖的客服/RAG 与编码 Agent 模板。
|
|
68
69
|
- `CompositionManifest` 的确定性编码、安全预览,以及冲突安全的本地版本和别名管理。
|
|
69
70
|
- 本地 Session 创建/列表/归档、JSON 持久化、跨 Run 文本上下文与审批恢复协调。
|
|
70
|
-
- 严格本地 JSON 装配的文本 Agent CLI/TUI
|
|
71
|
+
- 严格本地 JSON 装配的文本 Agent CLI/TUI 运行入口,使用环境变量凭据引用、逐次调用确认、可见 Token/费用预算与计量。
|
|
71
72
|
- 确定性脚本化 Model Provider、显式授权的 JSONL 录制/顺序回放,以及本地评测运行器与脱敏 JSON 报告。
|
|
72
73
|
|
|
73
74
|
当前 `BaseAgent` 仍是简单的 1.x 抽象;`LegacyAgentAdapter` 已能把它严格桥接为文本 Workflow 节点,新的 ReAct Runtime 独立提供。专用 OpenAI Responses 与 vLLM 差异适配、模型跨流恢复、多模态/工具事件通用回放、并行或嵌套 Workflow、装配依赖安装/加载确认和完整交互 TUI 仍为 `Planned`,不能当作现成功能使用。Docker 后端已有模拟 CLI 生命周期测试,但不代表当前机器已安装或启动 Docker;厂商模板经过模拟传输测试,也不代表所有远程型号已经在线验证。
|
|
@@ -90,6 +91,12 @@ W-Agent 遵循以下原则:
|
|
|
90
91
|
pip install wagent-framework
|
|
91
92
|
```
|
|
92
93
|
|
|
94
|
+
安装最新 Alpha:
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
pip install --pre wagent-framework==2.0.0a2
|
|
98
|
+
```
|
|
99
|
+
|
|
93
100
|
可选依赖:
|
|
94
101
|
|
|
95
102
|
```bash
|
|
@@ -99,7 +106,7 @@ pip install "wagent-framework[wasm]"
|
|
|
99
106
|
pip install "wagent-framework[tui]"
|
|
100
107
|
```
|
|
101
108
|
|
|
102
|
-
PyPI 1.x 支持 Python 3.9
|
|
109
|
+
PyPI 1.x 支持 Python 3.9+;`2.0.0a2` 与后续 2.x Alpha 要求 Python 3.11+。
|
|
103
110
|
|
|
104
111
|
## 1.x 最小示例
|
|
105
112
|
|
|
@@ -8,7 +8,7 @@ with open(os.path.join(here, "README.md"), "r", encoding="utf-8") as f:
|
|
|
8
8
|
|
|
9
9
|
setup(
|
|
10
10
|
name="wagent-framework",
|
|
11
|
-
version="2.0.
|
|
11
|
+
version="2.0.0a2",
|
|
12
12
|
description="Open, composable Python framework for building agents",
|
|
13
13
|
long_description=long_description,
|
|
14
14
|
long_description_content_type="text/markdown",
|
|
@@ -10,6 +10,7 @@ from w_agent import (
|
|
|
10
10
|
ModelCapability,
|
|
11
11
|
ModelDescriptor,
|
|
12
12
|
ModelMessage,
|
|
13
|
+
ModelCost,
|
|
13
14
|
RunResult,
|
|
14
15
|
StopReason,
|
|
15
16
|
TokenUsage,
|
|
@@ -24,7 +25,7 @@ def test_cli_version_and_profile_json_are_machine_readable():
|
|
|
24
25
|
profiles = runner.invoke(app, ["profile", "list", "--json"])
|
|
25
26
|
|
|
26
27
|
assert version.exit_code == 0
|
|
27
|
-
assert "2.0.
|
|
28
|
+
assert "2.0.0a2" in version.stdout
|
|
28
29
|
assert profiles.exit_code == 0
|
|
29
30
|
assert [item["key"] for item in json.loads(profiles.stdout)] == [
|
|
30
31
|
"customer-support",
|
|
@@ -332,6 +333,8 @@ def test_cli_evaluate_runs_dataset_and_writes_private_report(tmp_path, monkeypat
|
|
|
332
333
|
usage=TokenUsage(3, 2),
|
|
333
334
|
model_calls=1,
|
|
334
335
|
reported_usage_calls=1,
|
|
336
|
+
cost=ModelCost("USD", "prices-v1", "0.01", "0.02"),
|
|
337
|
+
priced_usage_calls=1,
|
|
335
338
|
)
|
|
336
339
|
return SimpleNamespace(result=result)
|
|
337
340
|
|
|
@@ -364,6 +367,8 @@ def test_cli_evaluate_runs_dataset_and_writes_private_report(tmp_path, monkeypat
|
|
|
364
367
|
assert payload["usage"]["input_tokens"] == 3
|
|
365
368
|
assert payload["usage"]["output_tokens"] == 2
|
|
366
369
|
assert payload["usage_complete"] is True
|
|
370
|
+
assert payload["cost"]["total"] == "0.03"
|
|
371
|
+
assert payload["cost_complete"] is True
|
|
367
372
|
assert "output" not in payload["cases"][0]
|
|
368
373
|
assert "secret prompt" not in persisted
|
|
369
374
|
assert "VISIBLE_OUTPUT" not in persisted
|
|
@@ -10,6 +10,7 @@ from w_agent import (
|
|
|
10
10
|
JsonEvaluationReporter,
|
|
11
11
|
LocalEvaluationRunner,
|
|
12
12
|
MessageRole,
|
|
13
|
+
ModelCost,
|
|
13
14
|
ModelMessage,
|
|
14
15
|
RunEvent,
|
|
15
16
|
RunEventType,
|
|
@@ -21,7 +22,7 @@ from w_agent import (
|
|
|
21
22
|
)
|
|
22
23
|
|
|
23
24
|
|
|
24
|
-
def _result(name, output, *, reason=StopReason.COMPLETED, events=()):
|
|
25
|
+
def _result(name, output, *, reason=StopReason.COMPLETED, events=(), cost=None):
|
|
25
26
|
return RunResult(
|
|
26
27
|
f"run-{name}",
|
|
27
28
|
reason,
|
|
@@ -33,6 +34,8 @@ def _result(name, output, *, reason=StopReason.COMPLETED, events=()):
|
|
|
33
34
|
usage=TokenUsage(4, 2),
|
|
34
35
|
model_calls=1,
|
|
35
36
|
reported_usage_calls=1,
|
|
37
|
+
cost=cost,
|
|
38
|
+
priced_usage_calls=1 if cost is not None else 0,
|
|
36
39
|
)
|
|
37
40
|
|
|
38
41
|
|
|
@@ -64,6 +67,29 @@ async def test_local_evaluation_scores_cases_and_aggregates_usage():
|
|
|
64
67
|
assert report.cases[1].passed is False
|
|
65
68
|
|
|
66
69
|
|
|
70
|
+
@pytest.mark.asyncio
|
|
71
|
+
async def test_local_evaluation_aggregates_versioned_cost_metrics():
|
|
72
|
+
async def target(case):
|
|
73
|
+
return _result(
|
|
74
|
+
case.name,
|
|
75
|
+
"ok",
|
|
76
|
+
cost=ModelCost("USD", "prices-v1", "0.01", "0.02", "0.003"),
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
report = await LocalEvaluationRunner().run(
|
|
80
|
+
(EvaluationCase("one", "a"), EvaluationCase("two", "b")),
|
|
81
|
+
target,
|
|
82
|
+
)
|
|
83
|
+
payload = evaluation_report_to_dict(report)
|
|
84
|
+
|
|
85
|
+
assert report.cost is not None
|
|
86
|
+
assert str(report.cost.total) == "0.066"
|
|
87
|
+
assert report.cost_complete is True
|
|
88
|
+
assert payload["cost"]["total"] == "0.066"
|
|
89
|
+
assert payload["cost_complete"] is True
|
|
90
|
+
assert payload["cases"][0]["cost"]["price_table_version"] == "prices-v1"
|
|
91
|
+
|
|
92
|
+
|
|
67
93
|
@pytest.mark.asyncio
|
|
68
94
|
async def test_evaluation_report_omits_prompts_outputs_and_exception_messages(tmp_path):
|
|
69
95
|
cases = (EvaluationCase("failure", "secret prompt", "unused"),)
|
|
@@ -125,6 +125,81 @@ async def test_local_runtime_assembles_full_text_run_and_persists_session(tmp_pa
|
|
|
125
125
|
assert "not-persisted" not in persisted
|
|
126
126
|
|
|
127
127
|
|
|
128
|
+
@pytest.mark.asyncio
|
|
129
|
+
async def test_local_runtime_loads_explicit_pricing_and_persists_cost(tmp_path):
|
|
130
|
+
config = local_runtime_config_from_mapping(
|
|
131
|
+
{
|
|
132
|
+
"provider": {
|
|
133
|
+
"template": "fake",
|
|
134
|
+
"name": "test-provider",
|
|
135
|
+
"model": "test-model",
|
|
136
|
+
},
|
|
137
|
+
"pricing": {
|
|
138
|
+
"version": "prices-v1",
|
|
139
|
+
"currency": "USD",
|
|
140
|
+
"max_cost": "10",
|
|
141
|
+
"prices": [
|
|
142
|
+
{
|
|
143
|
+
"provider": "test-provider",
|
|
144
|
+
"model": "test-model",
|
|
145
|
+
"input_per_million": "1000000",
|
|
146
|
+
"output_per_million": "1000000",
|
|
147
|
+
}
|
|
148
|
+
],
|
|
149
|
+
},
|
|
150
|
+
}
|
|
151
|
+
)
|
|
152
|
+
runtime = assemble_local_runtime(
|
|
153
|
+
config,
|
|
154
|
+
tmp_path / ".wagent",
|
|
155
|
+
templates=_templates({}),
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
run = await runtime.run("priced")
|
|
159
|
+
|
|
160
|
+
assert run.result.cost is not None
|
|
161
|
+
assert run.result.cost.total == 5
|
|
162
|
+
assert run.result.cost_complete is True
|
|
163
|
+
assert run.session.runs[0].cost is not None
|
|
164
|
+
assert run.session.runs[0].cost.total == 5
|
|
165
|
+
persisted = next((tmp_path / ".wagent" / "sessions").glob("*.json")).read_text(
|
|
166
|
+
encoding="utf-8"
|
|
167
|
+
)
|
|
168
|
+
assert '"price_table_version":"prices-v1"' in persisted
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def test_local_runtime_pricing_requires_configured_provider_model_rate(tmp_path):
|
|
172
|
+
config = local_runtime_config_from_mapping(
|
|
173
|
+
{
|
|
174
|
+
"provider": {
|
|
175
|
+
"template": "fake",
|
|
176
|
+
"name": "test-provider",
|
|
177
|
+
"model": "test-model",
|
|
178
|
+
},
|
|
179
|
+
"pricing": {
|
|
180
|
+
"version": "prices-v1",
|
|
181
|
+
"currency": "USD",
|
|
182
|
+
"max_cost": "1",
|
|
183
|
+
"prices": [
|
|
184
|
+
{
|
|
185
|
+
"provider": "other",
|
|
186
|
+
"model": "test-model",
|
|
187
|
+
"input_per_million": "1",
|
|
188
|
+
"output_per_million": "2",
|
|
189
|
+
}
|
|
190
|
+
],
|
|
191
|
+
},
|
|
192
|
+
}
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
with pytest.raises(LocalRuntimeConfigError, match="configured provider"):
|
|
196
|
+
assemble_local_runtime(
|
|
197
|
+
config,
|
|
198
|
+
tmp_path,
|
|
199
|
+
templates=_templates({}),
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
128
203
|
def test_local_runtime_requires_env_reference_and_rejects_secret_fields(tmp_path):
|
|
129
204
|
with pytest.raises(LocalRuntimeConfigError, match="environment variable"):
|
|
130
205
|
assemble_local_runtime(
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
from decimal import Decimal
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from w_agent import (
|
|
6
|
+
ModelPrice,
|
|
7
|
+
PriceNotFoundError,
|
|
8
|
+
PriceTable,
|
|
9
|
+
PricingCatalog,
|
|
10
|
+
PricingError,
|
|
11
|
+
TokenUsage,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def test_versioned_price_table_quotes_cached_and_uncached_tokens_exactly():
|
|
16
|
+
table = PriceTable(
|
|
17
|
+
"2026-09-18",
|
|
18
|
+
"usd",
|
|
19
|
+
(
|
|
20
|
+
ModelPrice(
|
|
21
|
+
"provider",
|
|
22
|
+
"model",
|
|
23
|
+
input_per_million="2",
|
|
24
|
+
output_per_million="4",
|
|
25
|
+
cached_input_per_million="1",
|
|
26
|
+
),
|
|
27
|
+
),
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
cost = table.quote("provider", "model", TokenUsage(1000, 200, 400))
|
|
31
|
+
|
|
32
|
+
assert cost.currency == "USD"
|
|
33
|
+
assert cost.price_table_version == "2026-09-18"
|
|
34
|
+
assert cost.input_cost == Decimal("0.0012")
|
|
35
|
+
assert cost.cached_input_cost == Decimal("0.0004")
|
|
36
|
+
assert cost.output_cost == Decimal("0.0008")
|
|
37
|
+
assert cost.total == Decimal("0.0024")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def test_pricing_catalog_requires_explicit_unique_versions_and_model_rates():
|
|
41
|
+
table = PriceTable(
|
|
42
|
+
"v1",
|
|
43
|
+
"USD",
|
|
44
|
+
(ModelPrice("provider", "model", "1", "2"),),
|
|
45
|
+
)
|
|
46
|
+
catalog = PricingCatalog((table,))
|
|
47
|
+
|
|
48
|
+
assert catalog.quote("v1", "provider", "model", TokenUsage(1, 1)).total > 0
|
|
49
|
+
with pytest.raises(PriceNotFoundError):
|
|
50
|
+
catalog.quote("v1", "provider", "other", TokenUsage(1, 1))
|
|
51
|
+
with pytest.raises(PricingError, match="already exists"):
|
|
52
|
+
catalog.register(table)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def test_price_table_rejects_invalid_cached_usage_and_duplicate_rates():
|
|
56
|
+
with pytest.raises(PricingError, match="duplicate"):
|
|
57
|
+
PriceTable(
|
|
58
|
+
"v1",
|
|
59
|
+
"USD",
|
|
60
|
+
(
|
|
61
|
+
ModelPrice("provider", "model", "1", "2"),
|
|
62
|
+
ModelPrice("provider", "model", "3", "4"),
|
|
63
|
+
),
|
|
64
|
+
)
|
|
65
|
+
table = PriceTable(
|
|
66
|
+
"v1",
|
|
67
|
+
"USD",
|
|
68
|
+
(ModelPrice("provider", "model", "1", "2"),),
|
|
69
|
+
)
|
|
70
|
+
with pytest.raises(PricingError, match="cannot exceed"):
|
|
71
|
+
table.quote("provider", "model", TokenUsage(1, 0, 2))
|
|
@@ -18,6 +18,10 @@ from w_agent import (
|
|
|
18
18
|
ModelRegistry,
|
|
19
19
|
ModelRouter,
|
|
20
20
|
JsonlRunStore,
|
|
21
|
+
CostBudget,
|
|
22
|
+
ModelPrice,
|
|
23
|
+
PriceTable,
|
|
24
|
+
PricingCatalog,
|
|
21
25
|
ReactAgentLoop,
|
|
22
26
|
RunResumeConflictError,
|
|
23
27
|
RunContext,
|
|
@@ -79,7 +83,7 @@ class AgentProvider:
|
|
|
79
83
|
yield FinishEvent(reason)
|
|
80
84
|
|
|
81
85
|
|
|
82
|
-
def _loop(responses, binding, *, store=None, policy=None):
|
|
86
|
+
def _loop(responses, binding, *, store=None, policy=None, pricing=None):
|
|
83
87
|
provider = AgentProvider(responses)
|
|
84
88
|
models = ModelRegistry()
|
|
85
89
|
models.register("fake", provider)
|
|
@@ -96,6 +100,7 @@ def _loop(responses, binding, *, store=None, policy=None):
|
|
|
96
100
|
tools,
|
|
97
101
|
tool_executor,
|
|
98
102
|
store=store,
|
|
103
|
+
pricing=pricing,
|
|
99
104
|
), provider
|
|
100
105
|
|
|
101
106
|
|
|
@@ -387,6 +392,90 @@ async def test_react_loop_can_fail_closed_when_provider_omits_usage():
|
|
|
387
392
|
assert result.usage_complete is False
|
|
388
393
|
|
|
389
394
|
|
|
395
|
+
@pytest.mark.asyncio
|
|
396
|
+
async def test_react_loop_enforces_versioned_cost_budget_before_tools():
|
|
397
|
+
calls = 0
|
|
398
|
+
|
|
399
|
+
def ping() -> str:
|
|
400
|
+
nonlocal calls
|
|
401
|
+
calls += 1
|
|
402
|
+
return "pong"
|
|
403
|
+
|
|
404
|
+
pricing = PricingCatalog(
|
|
405
|
+
(
|
|
406
|
+
PriceTable(
|
|
407
|
+
"prices-v1",
|
|
408
|
+
"USD",
|
|
409
|
+
(ModelPrice("fake", "chat", "1000000", "1000000"),),
|
|
410
|
+
),
|
|
411
|
+
)
|
|
412
|
+
)
|
|
413
|
+
loop, _ = _loop(
|
|
414
|
+
[
|
|
415
|
+
(
|
|
416
|
+
(ToolCallContent("ping-1", "ping", "{}"),),
|
|
417
|
+
FinishReason.TOOL_CALLS,
|
|
418
|
+
TokenUsage(8, 4),
|
|
419
|
+
)
|
|
420
|
+
],
|
|
421
|
+
python_tool(ping),
|
|
422
|
+
pricing=pricing,
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
result = await loop.run(
|
|
426
|
+
AgentDefinition(
|
|
427
|
+
"cost-bounded",
|
|
428
|
+
cost_budget=CostBudget("10", "prices-v1", "USD"),
|
|
429
|
+
),
|
|
430
|
+
_context(),
|
|
431
|
+
)
|
|
432
|
+
|
|
433
|
+
assert result.stop_reason is StopReason.COST_BUDGET
|
|
434
|
+
assert result.cost is not None
|
|
435
|
+
assert result.cost.total == 12
|
|
436
|
+
assert result.cost_complete is True
|
|
437
|
+
assert result.priced_usage_calls == 1
|
|
438
|
+
assert calls == 0
|
|
439
|
+
cost_event = next(
|
|
440
|
+
event for event in result.events if event.type is RunEventType.COST_USAGE
|
|
441
|
+
)
|
|
442
|
+
assert cost_event.data["cost"]["total"] == "12"
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
@pytest.mark.asyncio
|
|
446
|
+
async def test_react_cost_budget_fails_closed_without_usage_or_pricing():
|
|
447
|
+
pricing = PricingCatalog(
|
|
448
|
+
(
|
|
449
|
+
PriceTable(
|
|
450
|
+
"prices-v1",
|
|
451
|
+
"USD",
|
|
452
|
+
(ModelPrice("fake", "chat", "1", "2"),),
|
|
453
|
+
),
|
|
454
|
+
)
|
|
455
|
+
)
|
|
456
|
+
unmetered, _ = _loop(
|
|
457
|
+
[((TextContent("unmetered"),), FinishReason.STOP)],
|
|
458
|
+
python_tool(lambda: "unused", name="unused"),
|
|
459
|
+
pricing=pricing,
|
|
460
|
+
)
|
|
461
|
+
missing_resolver, provider = _loop(
|
|
462
|
+
[((TextContent("must not run"),), FinishReason.STOP, TokenUsage(1, 1))],
|
|
463
|
+
python_tool(lambda: "unused", name="unused"),
|
|
464
|
+
)
|
|
465
|
+
definition = AgentDefinition(
|
|
466
|
+
"strict-cost",
|
|
467
|
+
cost_budget=CostBudget("1", "prices-v1", "USD"),
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
unmetered_result = await unmetered.run(definition, _context())
|
|
471
|
+
missing_result = await missing_resolver.run(definition, _context())
|
|
472
|
+
|
|
473
|
+
assert unmetered_result.stop_reason is StopReason.COST_UNAVAILABLE
|
|
474
|
+
assert unmetered_result.cost_complete is False
|
|
475
|
+
assert missing_result.stop_reason is StopReason.COST_UNAVAILABLE
|
|
476
|
+
assert provider.requests == []
|
|
477
|
+
|
|
478
|
+
|
|
390
479
|
@pytest.mark.asyncio
|
|
391
480
|
async def test_react_attempt_ledger_marks_retry_usage_incomplete():
|
|
392
481
|
loop, _ = _loop(
|
|
@@ -471,6 +560,70 @@ async def test_attempt_checkpoint_persists_ledger_without_failure_message(tmp_pa
|
|
|
471
560
|
assert "secret provider response" not in persisted
|
|
472
561
|
|
|
473
562
|
|
|
563
|
+
@pytest.mark.asyncio
|
|
564
|
+
async def test_cost_ledger_survives_approval_checkpoint_resume(tmp_path):
|
|
565
|
+
saved = []
|
|
566
|
+
binding = python_tool(
|
|
567
|
+
lambda text: saved.append(text) or "saved",
|
|
568
|
+
name="save",
|
|
569
|
+
side_effect=ToolSideEffect.WRITE,
|
|
570
|
+
)
|
|
571
|
+
pricing = PricingCatalog(
|
|
572
|
+
(
|
|
573
|
+
PriceTable(
|
|
574
|
+
"prices-v1",
|
|
575
|
+
"USD",
|
|
576
|
+
(ModelPrice("fake", "chat", "1000000", "1000000"),),
|
|
577
|
+
),
|
|
578
|
+
)
|
|
579
|
+
)
|
|
580
|
+
definition = AgentDefinition(
|
|
581
|
+
"priced-writer",
|
|
582
|
+
cost_budget=CostBudget("100", "prices-v1", "USD"),
|
|
583
|
+
)
|
|
584
|
+
first_loop, _ = _loop(
|
|
585
|
+
[
|
|
586
|
+
(
|
|
587
|
+
(ToolCallContent("write-1", "save", '{"text":"value"}'),),
|
|
588
|
+
FinishReason.TOOL_CALLS,
|
|
589
|
+
TokenUsage(6, 2),
|
|
590
|
+
)
|
|
591
|
+
],
|
|
592
|
+
binding,
|
|
593
|
+
store=JsonlRunStore(tmp_path),
|
|
594
|
+
pricing=pricing,
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
pending = await first_loop.run(definition, _context())
|
|
598
|
+
checkpoint = await JsonlRunStore(tmp_path).load_checkpoint("run-1")
|
|
599
|
+
|
|
600
|
+
assert pending.stop_reason is StopReason.NEEDS_APPROVAL
|
|
601
|
+
assert checkpoint is not None
|
|
602
|
+
assert checkpoint.cost is not None
|
|
603
|
+
assert checkpoint.cost.total == 8
|
|
604
|
+
assert checkpoint.priced_usage_calls == 1
|
|
605
|
+
|
|
606
|
+
second_loop, _ = _loop(
|
|
607
|
+
[((TextContent("saved"),), FinishReason.STOP, TokenUsage(7, 3))],
|
|
608
|
+
binding,
|
|
609
|
+
store=JsonlRunStore(tmp_path),
|
|
610
|
+
pricing=pricing,
|
|
611
|
+
)
|
|
612
|
+
resumed = await second_loop.resume(
|
|
613
|
+
"run-1",
|
|
614
|
+
tool_context=ToolExecutionContext(
|
|
615
|
+
approved_call_ids=frozenset({"write-1"})
|
|
616
|
+
),
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
assert resumed.stop_reason is StopReason.COMPLETED
|
|
620
|
+
assert resumed.cost is not None
|
|
621
|
+
assert resumed.cost.total == 18
|
|
622
|
+
assert resumed.cost_complete is True
|
|
623
|
+
assert resumed.priced_usage_calls == 2
|
|
624
|
+
assert saved == ["value"]
|
|
625
|
+
|
|
626
|
+
|
|
474
627
|
@pytest.mark.asyncio
|
|
475
628
|
async def test_react_approval_resumes_from_jsonl_without_repeating_model(tmp_path):
|
|
476
629
|
saved = []
|
|
@@ -11,6 +11,7 @@ from w_agent import (
|
|
|
11
11
|
ModelCapability,
|
|
12
12
|
ModelDescriptor,
|
|
13
13
|
ModelMessage,
|
|
14
|
+
ModelCost,
|
|
14
15
|
RunResult,
|
|
15
16
|
StopReason,
|
|
16
17
|
TokenUsage,
|
|
@@ -227,6 +228,8 @@ async def test_tui_runs_private_evaluation_and_closes_runtime(tmp_path, monkeypa
|
|
|
227
228
|
usage=TokenUsage(4, 2),
|
|
228
229
|
model_calls=1,
|
|
229
230
|
reported_usage_calls=1,
|
|
231
|
+
cost=ModelCost("USD", "prices-v1", "0.01", "0.02"),
|
|
232
|
+
priced_usage_calls=1,
|
|
230
233
|
)
|
|
231
234
|
)
|
|
232
235
|
|
|
@@ -252,6 +255,7 @@ async def test_tui_runs_private_evaluation_and_closes_runtime(tmp_path, monkeypa
|
|
|
252
255
|
rendered = str(app.query_one("#evaluation-result").content)
|
|
253
256
|
assert "Passed: 1/1" in rendered
|
|
254
257
|
assert "in=4, out=2" in rendered
|
|
258
|
+
assert "Cost: 0.03 USD" in rendered
|
|
255
259
|
assert "SECRET_PROMPT" not in rendered
|
|
256
260
|
assert "VISIBLE_OUTPUT" not in rendered
|
|
257
261
|
assert app.query_one("#evaluation-confirm").value == ""
|