agentbyte 0.28.0__tar.gz → 0.30.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agentbyte-0.28.0 → agentbyte-0.30.0}/CHANGELOG.md +39 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/PKG-INFO +2 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/README.md +1 -1
- agentbyte-0.30.0/src/agentbyte/__about__.py +2 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent.py +20 -3
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/base.py +33 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/__init__.py +12 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/base.py +20 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/__init__.py +2 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/local.py +6 -0
- agentbyte-0.30.0/src/agentbyte/eval/checks/process.py +123 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/tool.py +2 -2
- agentbyte-0.30.0/src/agentbyte/eval/comparison.py +58 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/__init__.py +2 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/composite.py +28 -12
- agentbyte-0.30.0/src/agentbyte/eval/judges/llm.py +220 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/pairwise.py +32 -41
- agentbyte-0.30.0/src/agentbyte/eval/judges/trajectory.py +159 -0
- agentbyte-0.30.0/src/agentbyte/eval/judges/validation.py +93 -0
- agentbyte-0.30.0/src/agentbyte/eval/pairwise.py +170 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/report.py +35 -9
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/runner.py +47 -36
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/__init__.py +8 -1
- agentbyte-0.30.0/src/agentbyte/eval/targets/agent.py +102 -0
- agentbyte-0.30.0/src/agentbyte/eval/targets/multi_turn.py +12 -0
- agentbyte-0.30.0/src/agentbyte/eval/targets/orchestrator.py +90 -0
- agentbyte-0.30.0/src/agentbyte/eval/targets/runtime.py +53 -0
- agentbyte-0.30.0/src/agentbyte/eval/targets/workflow.py +104 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/types.py +120 -10
- agentbyte-0.30.0/src/agentbyte/execution_trace/__init__.py +22 -0
- agentbyte-0.30.0/src/agentbyte/execution_trace/collector.py +362 -0
- agentbyte-0.30.0/src/agentbyte/execution_trace/models.py +91 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/__init__.py +9 -1
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/base.py +70 -18
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/gepa.py +11 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/pareto.py +7 -3
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/reflective.py +2 -1
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/trace.py +6 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/base.py +94 -5
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/handoff.py +4 -1
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/policies.py +8 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/workflow.py +11 -11
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/types.py +5 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/execution.py +120 -50
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/models.py +6 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/server.py +27 -11
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/__init__.py +6 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/agent.py +10 -3
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/__init__.py +2 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/models.py +39 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/runner.py +90 -7
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/__init__.py +3 -1
- agentbyte-0.30.0/src/agentbyte/workflow/steps/agent.py +141 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/step.py +59 -24
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/subworkflow.py +18 -2
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_basic.py +135 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_context_providers.py +84 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_tool_approval.py +59 -0
- agentbyte-0.30.0/tests/eval/test_execution_trajectories.py +638 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_pairwise.py +89 -25
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_phase1_runner_and_targets.py +89 -4
- agentbyte-0.30.0/tests/eval/test_score_integrity.py +194 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_types_and_judges.py +162 -71
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_gepa.py +32 -1
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_mipro.py +1 -1
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_pareto.py +1 -1
- agentbyte-0.30.0/tests/optim/test_score_integrity.py +140 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_trace.py +2 -1
- agentbyte-0.30.0/tests/orchestration/test_orchestrator_finalization.py +333 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_workflow.py +4 -4
- agentbyte-0.30.0/tests/webui/test_workflow_streaming.py +351 -0
- agentbyte-0.30.0/tests/workflow/test_agent_step_imports.py +15 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_steps.py +8 -6
- agentbyte-0.28.0/src/agentbyte/__about__.py +0 -2
- agentbyte-0.28.0/src/agentbyte/eval/comparison.py +0 -38
- agentbyte-0.28.0/src/agentbyte/eval/judges/llm.py +0 -279
- agentbyte-0.28.0/src/agentbyte/eval/pairwise.py +0 -106
- agentbyte-0.28.0/src/agentbyte/eval/targets/agent.py +0 -78
- agentbyte-0.28.0/src/agentbyte/eval/targets/multi_turn.py +0 -142
- agentbyte-0.28.0/src/agentbyte/eval/targets/orchestrator.py +0 -89
- agentbyte-0.28.0/src/agentbyte/workflow/steps/agentbyte_agent.py +0 -81
- {agentbyte-0.28.0 → agentbyte-0.30.0}/.gitignore +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/LICENSE +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/pyproject.toml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent_as_tool.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/embedding_agent.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cancellation_token.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/catalog.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cli/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cli/main.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/component.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skill_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skills.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/config.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/importer.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/json.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/loader.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish_config.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publishers.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sources.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite_db.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/write_config.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/writers.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/entity.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/decorator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/keyword.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/eval_dataset.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/reference.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/splitting.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/model.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/_retry_observability.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/auth.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/auth.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/chat.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/embedding.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/settings.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai_embedding.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/embeddings_base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/chat.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/embedding.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/settings.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai_embedding.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/pricing.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/retry_policy.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/settings.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/logger.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/memory/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/memory/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/messages.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/otel.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/retry.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/sql_usage.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/usage_logger.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/notebook.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/config.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/mipro.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/spec.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/ai.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/plan.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/round_robin.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/agents.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/clients.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instruction_registry.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/orchestrator.yaml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/query_rewriter.yaml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/researcher.yaml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/reviewer.yaml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/writer.yaml +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/orchestration.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/contracts-analyst/SKILL.md +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/SKILL.md +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/departments.md +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/employees.md +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/payroll.md +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/streaming.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/session_store.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/resources.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/scripts.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/sources.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/validation.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/cancellation.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/composite.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/consecutive_agent.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/external.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/function_call.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/handoff.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/max_message.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/predicate.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/source.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/text_mention.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/timeout.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/token_usage.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/coding_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/core_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/decorator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/memory_tool.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/research_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/discovery.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/registry.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/session_store.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/sessions.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-BF3DwXaF.js +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-ar5tOeqt.css +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/index.html +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/vite.svg +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/_structure_hash.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/checkpoint.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/workflow.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/defaults.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/loader.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema_utils.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/echo.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/function.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/http.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/transform.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/visualizer.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_as_tool.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_error_response.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_event_types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_memory_integration.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_middleware_integration.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_response_accessors.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_retry_middleware.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_stream_events.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_embedding_agent.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/cli/test_registry_check.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/test_skill_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/test_skills_provider.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_loader.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_multi_table.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_publish.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_sqlite_db.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_eval_dataset.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_multi_turn.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_phase2_checks_and_reports.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_splitting.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_azure_client.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_azure_embedding_client.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_llm_types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_openai_client.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_openai_embedding_client.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_pricing.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retry_observability.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retry_policy_api.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retryable_error_substrings.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/memory/test_memory.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_deduplicate_tool_result.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_middleware_chain.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_otel.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_retry_middleware.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_sql_usage.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_usage_logger.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_base_integration.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_config.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_reflective.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_spec.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_ai_orchestrator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_base_orchestrator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_handoff_orchestrator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_plan_orchestrator.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_round_robin.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_agents.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_clients.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_instruction_registry.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_orchestration.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_streaming.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_resources.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_scripts.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_sources.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_base.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_cancellation.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_composite.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_consecutive_agent.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_external.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_function_call.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_handoff.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_max_message.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_predicate.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_source.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_text_mention.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_timeout.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_token_usage.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_cancellation_token.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_context.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_logger.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_messages.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_package_api.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_session_store.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_types.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_vanilla_chunker.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_coding_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_memory_tool.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_research_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_tools.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/__init__.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/helpers.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_execution.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_package_api.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_registry.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_server.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_sessions.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_checkpoint.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_subworkflow_step.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_agent.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_class.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_models.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_runner.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_schema.py +0 -0
- {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_visualizer.py +0 -0
|
@@ -4,6 +4,45 @@ All notable changes to Agentbyte are documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format follows Keep a Changelog principles and semantic versioning.
|
|
6
6
|
|
|
7
|
+
## [0.30.0] - 2026-09-30
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Workflow streaming parity (spec 0057): `WorkflowRunner.run_stream(..., stream_tokens=...)` forwards live child agent events as `StepEvent` envelopes (workflow, execution, invocation, step, attempt and sequence attribution) through a bounded queue, including across `workflow.as_agent()` and `SubWorkflowStep` boundaries. `WorkflowAgent.run_stream` honors `stream_tokens`. Closing the stream cancels and awaits outstanding steps.
|
|
12
|
+
- `AgentStep` streams its agent and turns tool approval into workflow suspension; resuming validates the exact pending decisions and never reruns completed tools.
|
|
13
|
+
- Checkpoint-backed serving: the WebUI/FastAPI run and SSE endpoints accept `workflow_responses` and `workflow_checkpoint_id` to resume a suspended workflow in the same session. `WorkflowExecution.checkpoint_id` exposes the suspended checkpoint, and `WorkflowRunner.validate_resume_responses` / `BaseStep.validate_resume_response` validate responses without consuming it. Stale, unknown or wrong-owner checkpoints and concurrent runs of the same workflow or session are rejected with HTTP 409.
|
|
14
|
+
- Judge score integrity (spec 0055): `EvalScore.scoring_status` (`scored` | `failed` | `cancelled`) and `failure_reason`, `JudgeScoringError` with safe reason codes, `mean_scored()`, `OptimizationEvidenceError`, and `PairwiseResult.comparison_status`.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- **Breaking:** Judges never invent scores. `LLMEvalJudge` and `LLMTrajectoryJudge` require every requested criterion exactly once (matched case- and whitespace-insensitively, reported with the requested spelling) with a finite 0–10 value, and raise `JudgeScoringError` otherwise instead of filling missing criteria with 5.0, clamping, or returning a neutral 5.0 after a failure. `overall` is always the mean of the validated dimensions; a model-supplied overall is ignored. Cancellation raises `asyncio.CancelledError`.
|
|
19
|
+
- **Breaking:** `EvalScore.overall` is `float | None`. `EvalRunner` records judge failures, target exceptions and cancellation as `failed`/`cancelled` outcomes with `overall=None`, empty dimensions and the trajectory kept, instead of a fabricated 0.0. Failure metadata keeps the exception type, never its text. A scored `EvalScore` must have a finite `overall` and dimensions in 0–10.
|
|
20
|
+
- **Breaking:** `PairwiseJudge.compare()` raises instead of returning a tie on failure; unknown winners and out-of-range margins are rejected, not coerced. `PairwiseRunner` records failed comparisons with `winner=None`. `compare_pairwise()` adds `compared`/`failed`/`cancelled` counts and computes win rates over valid comparisons only (`None` when nothing was compared).
|
|
21
|
+
- **Breaking:** `EvalReport.avg_score` and `compare_configurations()` averages cover scored outcomes only and are `None` when nothing was scored; an item with a failed or cancelled score fails the suite. `CompositeJudge` fails when any sub-judge fails rather than renormalizing the remaining weights.
|
|
22
|
+
- **Breaking:** Optimizers only rank candidates whose every task was scored. `Candidate.avg` is `None` for ineligible candidates, the minibatch gate rejects incomplete batches, a seed with an unscored task raises `OptimizationEvidenceError`, and the GEPA adapter raises rather than passing a number for an unscored example. Optimization trace `eval` events report `avg` over scored tasks (`None` if none) plus `n_scored`.
|
|
23
|
+
|
|
24
|
+
## [0.29.0] - 2026-09-30
|
|
25
|
+
|
|
26
|
+
### Added
|
|
27
|
+
|
|
28
|
+
- Unified trajectory evaluation (spec 0046): every eval target now records a runtime-neutral execution trace (`trajectory.trace`) with spans for agents, workflow steps, tools and model calls, plus normalized terminal status and usage. New `WorkflowEvalTarget` and `WorkflowEvalTask` evaluate workflows directly with typed input; workflows wrapped via `workflow.as_agent()` capture the same evidence.
|
|
29
|
+
- Process evaluation: `LLMTrajectoryJudge` scores `tool_use_correctness`, `coordination` and `execution_efficiency` with cited span/event evidence and reports failures as `metadata["judge_failed"]` instead of silently truncating. New deterministic checks `terminal_status`, `steps_completed`, `edge_activated` and `usage_limits`, and `select_span` to score one span in isolation.
|
|
30
|
+
- Orchestrator finalization (spec 0048): `FinalResultPolicy` gains `final_agent` and `finalize_on_exhaustion` (default `True`). When a final agent is configured (`final_agent`, or `prefer_agent` as fallback) and `max_iterations` is about to run out, the last iteration runs that agent with a finalization instruction, bypassing the pattern's selection, context and state hooks; the run stops with `stop_message.source == "MaxIterationsFinalized"`. An unknown `final_agent` raises `ValueError`.
|
|
31
|
+
- `OrchestrationResponse.final_structured_result` exposes the `structured_content` of the message selected as `final_result`, so a typed final agent's answer is available without parsing JSON.
|
|
32
|
+
- `HandoffOrchestrator` accepts `final_result_policy`.
|
|
33
|
+
|
|
34
|
+
### Changed
|
|
35
|
+
|
|
36
|
+
- **Breaking:** The agent workflow step module moved from `agentbyte.workflow.steps.agentbyte_agent` to `agentbyte.workflow.steps.agent`, and `AgentbyteAgentInput` / `AgentbyteAgentOutput` are renamed to `AgentStepInput` / `AgentStepOutput` (exported from `agentbyte.workflow` and `agentbyte.workflow.steps`). No compatibility aliases; update imports. `AgentStep` is unchanged.
|
|
37
|
+
- **Breaking:** Agents reject duplicate tool names across `tools` and context-provider tools (spec 0047). The run fails with `finish_reason="error"` and an `AgentConfigurationError` before any model call. Previously, attaching several `SkillsProvider`s silently made every provider but the first unreachable; attach one `SkillsProvider` with several paths (`from_paths([...])` or `AggregatingSkillsSource`) instead.
|
|
38
|
+
- Orchestrators configured with `FinalResultPolicy(prefer_agent=...)` now reserve the last iteration for that agent when the budget is exhausted. Opt out with `finalize_on_exhaustion=False`.
|
|
39
|
+
- `stream_tokens=True` is disabled with a warning when the agent has an `output_format`, because streamed chunks are not parsed into structured output. Typed turns now always carry `structured_content`.
|
|
40
|
+
|
|
41
|
+
### Fixed
|
|
42
|
+
|
|
43
|
+
- The agent finalization turn drops tool calls the model returns despite having no tools, so they are never executed or left unanswered in history, and the run reports `max_iterations_finalized` (spec 0042 hardening).
|
|
44
|
+
- Workflow examples that imported the removed `AgentbyteAgentStep` and `Context` names now use `AgentStep` and `WorkflowContext`.
|
|
45
|
+
|
|
7
46
|
## [0.28.0] - 2026-09-15
|
|
8
47
|
|
|
9
48
|
### Changed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: agentbyte
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30.0
|
|
4
4
|
Summary: A toolkit for designing multiagent systems
|
|
5
5
|
Author-email: MrDataPsycho <mr.data.psycho@gmail.com>
|
|
6
6
|
License-Expression: LicenseRef-Proprietary
|
|
@@ -86,7 +86,7 @@ Description-Content-Type: text/markdown
|
|
|
86
86
|
|
|
87
87
|
Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
|
|
88
88
|
|
|
89
|
-
Current release: **0.
|
|
89
|
+
Current release: **0.30.0**
|
|
90
90
|
|
|
91
91
|
## Building an Agent
|
|
92
92
|
|
|
@@ -12,6 +12,7 @@ from typing import List, Optional, Union
|
|
|
12
12
|
|
|
13
13
|
from agentbyte.cancellation_token import CancellationToken
|
|
14
14
|
from agentbyte.context import AgentContext
|
|
15
|
+
from agentbyte.execution_trace import trace_stream
|
|
15
16
|
from agentbyte.llm.types import ChatCompletionChunk
|
|
16
17
|
from agentbyte.messages import (
|
|
17
18
|
AssistantMessage,
|
|
@@ -338,6 +339,7 @@ class Agent(BaseAgent):
|
|
|
338
339
|
|
|
339
340
|
yield message
|
|
340
341
|
|
|
342
|
+
@trace_stream("agent")
|
|
341
343
|
async def run_stream(
|
|
342
344
|
self,
|
|
343
345
|
task: Optional[Union[str, UserMessage, List[Message]]] = None,
|
|
@@ -414,6 +416,15 @@ class Agent(BaseAgent):
|
|
|
414
416
|
stacklevel=2,
|
|
415
417
|
)
|
|
416
418
|
effective_stream_tokens = False
|
|
419
|
+
elif stream_tokens and self.output_format is not None:
|
|
420
|
+
# Streamed chunks are not schema-constrained or parsed, so typed
|
|
421
|
+
# answers would lose structured_content.
|
|
422
|
+
warnings.warn(
|
|
423
|
+
"stream_tokens=True is disabled when output_format is set; "
|
|
424
|
+
"falling back to non-token execution to keep structured output.",
|
|
425
|
+
stacklevel=2,
|
|
426
|
+
)
|
|
427
|
+
effective_stream_tokens = False
|
|
417
428
|
|
|
418
429
|
try:
|
|
419
430
|
task_messages = self._convert_task_to_messages(task) if task else []
|
|
@@ -434,6 +445,7 @@ class Agent(BaseAgent):
|
|
|
434
445
|
if instr:
|
|
435
446
|
provider_instructions.append(instr)
|
|
436
447
|
provider_tools.extend(cp_tools)
|
|
448
|
+
self._check_unique_tool_names(extra_tools=provider_tools)
|
|
437
449
|
extra_instructions = "\n\n".join(provider_instructions) if provider_instructions else None
|
|
438
450
|
|
|
439
451
|
iteration = 0
|
|
@@ -474,7 +486,8 @@ class Agent(BaseAgent):
|
|
|
474
486
|
tools = self._get_tools_for_llm(extra_tools=provider_tools) if (self.tools or provider_tools) else None
|
|
475
487
|
|
|
476
488
|
# Reserve the final iteration for a tool-free answer so an exhausted
|
|
477
|
-
# budget yields a complete response instead of nothing.
|
|
489
|
+
# budget yields a complete response instead of nothing. Needs
|
|
490
|
+
# max_iterations > 1, otherwise the only turn would lose its tools.
|
|
478
491
|
is_final_iteration = (
|
|
479
492
|
self.finalize_on_exhaustion
|
|
480
493
|
and self.max_iterations > 1
|
|
@@ -625,10 +638,14 @@ class Agent(BaseAgent):
|
|
|
625
638
|
if completion_result.usage.cost_estimate is not None:
|
|
626
639
|
cost_estimate += completion_result.usage.cost_estimate
|
|
627
640
|
|
|
641
|
+
# The finalization turn offers no tools; drop any tool calls the
|
|
642
|
+
# model returns anyway so they never execute or dangle in history.
|
|
628
643
|
assistant_message = AssistantMessage(
|
|
629
644
|
content=completion_result.message.content,
|
|
630
645
|
source=self.name,
|
|
631
|
-
tool_calls=
|
|
646
|
+
tool_calls=None
|
|
647
|
+
if is_final_iteration
|
|
648
|
+
else completion_result.message.tool_calls,
|
|
632
649
|
structured_content=completion_result.structured_output,
|
|
633
650
|
usage=completion_result.usage,
|
|
634
651
|
)
|
|
@@ -638,7 +655,7 @@ class Agent(BaseAgent):
|
|
|
638
655
|
source=self.name,
|
|
639
656
|
model=completion_result.model,
|
|
640
657
|
response=completion_result.message.content,
|
|
641
|
-
has_tool_calls=bool(
|
|
658
|
+
has_tool_calls=bool(assistant_message.tool_calls),
|
|
642
659
|
)
|
|
643
660
|
|
|
644
661
|
working_context.add_message(assistant_message)
|
|
@@ -61,6 +61,15 @@ class BaseAgent(ABC):
|
|
|
61
61
|
context_providers: Optional[List[ContextProvider]] = None,
|
|
62
62
|
**kwargs: Any,
|
|
63
63
|
) -> None:
|
|
64
|
+
"""Initialize the agent.
|
|
65
|
+
|
|
66
|
+
When ``finalize_on_exhaustion`` is enabled and the agent has tools, the
|
|
67
|
+
last of ``max_iterations`` is reserved for a tool-free final answer
|
|
68
|
+
(``finish_reason="max_iterations_finalized"``). This requires
|
|
69
|
+
``max_iterations >= 2``: with ``max_iterations=1`` the only turn keeps
|
|
70
|
+
its tools, so a run that calls a tool ends with
|
|
71
|
+
``finish_reason="max_iterations"`` and no final answer.
|
|
72
|
+
"""
|
|
64
73
|
self.name = name
|
|
65
74
|
self.description = description
|
|
66
75
|
self.instructions = instructions
|
|
@@ -124,6 +133,30 @@ class BaseAgent(ABC):
|
|
|
124
133
|
return tool
|
|
125
134
|
return None
|
|
126
135
|
|
|
136
|
+
def _check_unique_tool_names(
|
|
137
|
+
self,
|
|
138
|
+
extra_tools: Optional[List[BaseTool]] = None,
|
|
139
|
+
) -> None:
|
|
140
|
+
"""Raise if agent tools and context-provider tools share a name.
|
|
141
|
+
|
|
142
|
+
Duplicate names are ambiguous for the model and ``_find_tool`` always
|
|
143
|
+
resolves to the first match, silently shadowing later tools.
|
|
144
|
+
"""
|
|
145
|
+
seen: set[str] = set()
|
|
146
|
+
duplicates: list[str] = []
|
|
147
|
+
for tool in [*self.tools, *(extra_tools or [])]:
|
|
148
|
+
if tool.name in seen and tool.name not in duplicates:
|
|
149
|
+
duplicates.append(tool.name)
|
|
150
|
+
seen.add(tool.name)
|
|
151
|
+
if duplicates:
|
|
152
|
+
names = ", ".join(repr(name) for name in duplicates)
|
|
153
|
+
raise AgentConfigurationError(
|
|
154
|
+
f"Duplicate tool name(s) {names} across agent tools and context "
|
|
155
|
+
"providers; tool names must be unique. To use several skill "
|
|
156
|
+
"directories, attach one SkillsProvider with all paths, e.g. "
|
|
157
|
+
"SkillsProvider.from_paths([...]) or AggregatingSkillsSource."
|
|
158
|
+
)
|
|
159
|
+
|
|
127
160
|
def _get_tools_for_llm(
|
|
128
161
|
self,
|
|
129
162
|
extra_tools: Optional[List[BaseTool]] = None,
|
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
"""Evaluation framework for AgentByte."""
|
|
2
2
|
|
|
3
|
-
from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget
|
|
3
|
+
from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget, JudgeScoringError
|
|
4
|
+
from .checks.process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
|
|
5
|
+
from .judges.trajectory import LLMTrajectoryJudge
|
|
6
|
+
from .targets.workflow import WorkflowEvalTarget
|
|
4
7
|
from .checks import (
|
|
5
8
|
CheckResult,
|
|
6
9
|
EvalCheck,
|
|
@@ -16,7 +19,7 @@ from .checks import (
|
|
|
16
19
|
tool_calls_present,
|
|
17
20
|
tool_calls_present_check,
|
|
18
21
|
)
|
|
19
|
-
from .comparison import compare_configurations
|
|
22
|
+
from .comparison import compare_configurations, mean_scored
|
|
20
23
|
from .eval_dataset import (
|
|
21
24
|
EvalDatasetRecord,
|
|
22
25
|
save_eval_dataset,
|
|
@@ -49,11 +52,15 @@ from .types import (
|
|
|
49
52
|
EvalTask,
|
|
50
53
|
EvalTrajectory,
|
|
51
54
|
ExpectedToolCall,
|
|
55
|
+
ScoringStatus,
|
|
52
56
|
MultiTurnEvalTask,
|
|
57
|
+
WorkflowEvalTask,
|
|
53
58
|
PairwiseResult,
|
|
54
59
|
)
|
|
55
60
|
|
|
56
61
|
__all__ = [
|
|
62
|
+
"WorkflowEvalTask", "WorkflowEvalTarget", "LLMTrajectoryJudge",
|
|
63
|
+
"select_span", "terminal_status", "steps_completed", "edge_activated", "usage_limits",
|
|
57
64
|
"AnswerStrategy",
|
|
58
65
|
"ExpectedToolCall",
|
|
59
66
|
"EvalTask",
|
|
@@ -64,6 +71,8 @@ __all__ = [
|
|
|
64
71
|
"BaseEvalTarget",
|
|
65
72
|
"BaseEvalJudge",
|
|
66
73
|
"BaseEvalRunner",
|
|
74
|
+
"JudgeScoringError",
|
|
75
|
+
"ScoringStatus",
|
|
67
76
|
"ExactMatchJudge",
|
|
68
77
|
"ContainsJudge",
|
|
69
78
|
"FuzzyMatchJudge",
|
|
@@ -82,6 +91,7 @@ __all__ = [
|
|
|
82
91
|
"EvalSplitConfig",
|
|
83
92
|
"EvalSplitStrategy",
|
|
84
93
|
"compare_configurations",
|
|
94
|
+
"mean_scored",
|
|
85
95
|
"EvalDatasetRecord",
|
|
86
96
|
"save_eval_dataset",
|
|
87
97
|
"tasks_from_eval_dataset",
|
|
@@ -17,6 +17,19 @@ _VALID_ANSWER_STRATEGIES: tuple[AnswerStrategy, ...] = (
|
|
|
17
17
|
)
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
class JudgeScoringError(Exception):
|
|
21
|
+
"""A judge could not produce a valid judgment.
|
|
22
|
+
|
|
23
|
+
``reason`` is a safe, stable code (for example ``provider_error`` or
|
|
24
|
+
``missing_criterion``) suitable for persistence; the message may carry more
|
|
25
|
+
detail and is not persisted by default.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self, reason: str, message: str | None = None) -> None:
|
|
29
|
+
super().__init__(message or reason)
|
|
30
|
+
self.reason = reason
|
|
31
|
+
|
|
32
|
+
|
|
20
33
|
class BaseEvalTarget(ABC):
|
|
21
34
|
"""Anything that can execute an evaluation task into a trajectory."""
|
|
22
35
|
|
|
@@ -96,7 +109,12 @@ class BaseEvalJudge(ABC):
|
|
|
96
109
|
criteria: list[str] | None = None,
|
|
97
110
|
cancellation_token: CancellationToken | None = None,
|
|
98
111
|
) -> EvalScore:
|
|
99
|
-
"""Score an evaluation trajectory.
|
|
112
|
+
"""Score an evaluation trajectory.
|
|
113
|
+
|
|
114
|
+
Returns a scored ``EvalScore``. Raises ``JudgeScoringError`` when no
|
|
115
|
+
valid judgment is available and ``asyncio.CancelledError`` when
|
|
116
|
+
cancelled; judges never substitute a placeholder score.
|
|
117
|
+
"""
|
|
100
118
|
|
|
101
119
|
|
|
102
120
|
class BaseEvalRunner(ABC):
|
|
@@ -116,4 +134,4 @@ class BaseEvalRunner(ABC):
|
|
|
116
134
|
"""Evaluate a target on multiple tasks."""
|
|
117
135
|
|
|
118
136
|
|
|
119
|
-
__all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner"]
|
|
137
|
+
__all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner", "JudgeScoringError"]
|
|
@@ -4,6 +4,7 @@ from agentbyte.eval.report import EvalItemReport, EvalNotPassedError, EvalReport
|
|
|
4
4
|
|
|
5
5
|
from .decorator import evaluator
|
|
6
6
|
from .keyword import keyword_check
|
|
7
|
+
from .process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
|
|
7
8
|
from .local import CheckEvaluator, threshold_gate
|
|
8
9
|
from .tool import (
|
|
9
10
|
tool_call_args_match,
|
|
@@ -14,6 +15,7 @@ from .tool import (
|
|
|
14
15
|
from .types import CheckResult, EvalCheck, ExpectedToolCall
|
|
15
16
|
|
|
16
17
|
__all__ = [
|
|
18
|
+
"select_span", "terminal_status", "steps_completed", "edge_activated", "usage_limits",
|
|
17
19
|
"ExpectedToolCall",
|
|
18
20
|
"CheckResult",
|
|
19
21
|
"EvalCheck",
|
|
@@ -93,6 +93,12 @@ def threshold_gate(
|
|
|
93
93
|
|
|
94
94
|
async def _check(trajectory: EvalTrajectory) -> CheckResult:
|
|
95
95
|
score = await judge.score(trajectory)
|
|
96
|
+
if not score.is_scored or score.overall is None:
|
|
97
|
+
return CheckResult(
|
|
98
|
+
passed=False,
|
|
99
|
+
reason=f"no valid score ({score.scoring_status}: {score.failure_reason})",
|
|
100
|
+
check_name=f"threshold_gate[{judge.name}]",
|
|
101
|
+
)
|
|
96
102
|
passed = score.overall >= threshold
|
|
97
103
|
return CheckResult(
|
|
98
104
|
passed=passed,
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Deterministic checks over execution evidence."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from agentbyte.eval.types import EvalTrajectory
|
|
6
|
+
from .types import CheckResult, EvalCheck
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def select_span(trajectory: EvalTrajectory, span_id: str) -> EvalTrajectory:
|
|
10
|
+
"""Project one invocation for existing judges and checks without changing its task."""
|
|
11
|
+
if trajectory.trace is None:
|
|
12
|
+
raise ValueError("Trajectory has no execution trace")
|
|
13
|
+
trace = trajectory.trace.select(span_id)
|
|
14
|
+
root = trace.spans[0]
|
|
15
|
+
return trajectory.model_copy(
|
|
16
|
+
update={
|
|
17
|
+
"trace": trace,
|
|
18
|
+
"messages": trace.evidence_messages(),
|
|
19
|
+
"usage": trace.aggregate_usage(),
|
|
20
|
+
"success": root.status == "completed",
|
|
21
|
+
"error": root.error,
|
|
22
|
+
"metadata": {
|
|
23
|
+
**trajectory.metadata,
|
|
24
|
+
"execution_status": root.status,
|
|
25
|
+
"span_id": span_id,
|
|
26
|
+
},
|
|
27
|
+
}
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def terminal_status(expected: str = "completed") -> EvalCheck:
|
|
32
|
+
"""Require an explicit normalized terminal status."""
|
|
33
|
+
|
|
34
|
+
async def check(trajectory: EvalTrajectory) -> CheckResult:
|
|
35
|
+
actual = trajectory.metadata.get("execution_status")
|
|
36
|
+
return CheckResult(
|
|
37
|
+
passed=actual == expected,
|
|
38
|
+
reason=f"Expected {expected}; observed {actual}",
|
|
39
|
+
check_name="terminal_status",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
return check
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def steps_completed(*step_ids: str) -> EvalCheck:
|
|
46
|
+
"""Require at least one completed attempt of each step; select a span to scope nesting."""
|
|
47
|
+
|
|
48
|
+
async def check(trajectory: EvalTrajectory) -> CheckResult:
|
|
49
|
+
actual = (
|
|
50
|
+
{s.step_id for s in trajectory.trace.spans if s.status == "completed"}
|
|
51
|
+
if trajectory.trace
|
|
52
|
+
else set()
|
|
53
|
+
)
|
|
54
|
+
missing = set(step_ids) - actual
|
|
55
|
+
return CheckResult(
|
|
56
|
+
passed=trajectory.trace is not None and not missing,
|
|
57
|
+
reason=f"Missing completed steps: {sorted(missing)}",
|
|
58
|
+
check_name="steps_completed",
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
return check
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def edge_activated(from_step: str, to_step: str) -> EvalCheck:
|
|
65
|
+
"""Require an observed routing edge, not merely adjacent message order."""
|
|
66
|
+
|
|
67
|
+
async def check(trajectory: EvalTrajectory) -> CheckResult:
|
|
68
|
+
found = trajectory.trace is not None and any(
|
|
69
|
+
e.kind == "EdgeActivatedEvent"
|
|
70
|
+
and e.data.get("from_step") == from_step
|
|
71
|
+
and e.data.get("to_step") == to_step
|
|
72
|
+
for e in trajectory.trace.events
|
|
73
|
+
)
|
|
74
|
+
return CheckResult(
|
|
75
|
+
passed=found,
|
|
76
|
+
reason=f"Edge {from_step} -> {to_step} observed: {found}",
|
|
77
|
+
check_name="edge_activated",
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
return check
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def usage_limits(**limits: float) -> EvalCheck:
|
|
84
|
+
"""Require usage fields to remain within bounds; unknown cost/evidence fails the gate."""
|
|
85
|
+
valid = {
|
|
86
|
+
"duration_ms",
|
|
87
|
+
"llm_calls",
|
|
88
|
+
"tokens_input",
|
|
89
|
+
"tokens_output",
|
|
90
|
+
"tokens_cached",
|
|
91
|
+
"total_tokens",
|
|
92
|
+
"tool_calls",
|
|
93
|
+
"memory_operations",
|
|
94
|
+
"cost_estimate",
|
|
95
|
+
}
|
|
96
|
+
if not limits or set(limits) - valid or any(v < 0 for v in limits.values()):
|
|
97
|
+
raise ValueError("Supply non-negative limits for recognized Usage fields")
|
|
98
|
+
|
|
99
|
+
async def check(trajectory: EvalTrajectory) -> CheckResult:
|
|
100
|
+
known = trajectory.usage is not None and (
|
|
101
|
+
trajectory.trace is None or trajectory.trace.usage_complete
|
|
102
|
+
)
|
|
103
|
+
passed = known and all(
|
|
104
|
+
getattr(trajectory.usage, k) is not None
|
|
105
|
+
and getattr(trajectory.usage, k) <= v
|
|
106
|
+
for k, v in limits.items()
|
|
107
|
+
)
|
|
108
|
+
return CheckResult(
|
|
109
|
+
passed=passed,
|
|
110
|
+
reason=f"Usage limits {limits}; complete evidence: {known}",
|
|
111
|
+
check_name="usage_limits",
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
return check
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
__all__ = [
|
|
118
|
+
"select_span",
|
|
119
|
+
"terminal_status",
|
|
120
|
+
"steps_completed",
|
|
121
|
+
"edge_activated",
|
|
122
|
+
"usage_limits",
|
|
123
|
+
]
|
|
@@ -21,7 +21,7 @@ async def tool_calls_present(trajectory: EvalTrajectory) -> CheckResult:
|
|
|
21
21
|
|
|
22
22
|
actual_names = {
|
|
23
23
|
tool_call.tool_name
|
|
24
|
-
for message in trajectory.messages
|
|
24
|
+
for message in (trajectory.trace.evidence_messages() if trajectory.trace else trajectory.messages)
|
|
25
25
|
if isinstance(message, AssistantMessage) and message.tool_calls
|
|
26
26
|
for tool_call in message.tool_calls
|
|
27
27
|
}
|
|
@@ -69,7 +69,7 @@ async def tool_call_args_match(trajectory: EvalTrajectory) -> CheckResult:
|
|
|
69
69
|
)
|
|
70
70
|
|
|
71
71
|
actual_calls: list[tuple[str, dict[str, object]]] = []
|
|
72
|
-
for message in trajectory.messages:
|
|
72
|
+
for message in (trajectory.trace.evidence_messages() if trajectory.trace else trajectory.messages):
|
|
73
73
|
if not isinstance(message, AssistantMessage) or not message.tool_calls:
|
|
74
74
|
continue
|
|
75
75
|
for tool_call in message.tool_calls:
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Helpers for comparing evaluated configurations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from agentbyte.eval.types import EvalScore
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def compare_configurations(
|
|
9
|
+
scores_a: list[EvalScore],
|
|
10
|
+
scores_b: list[EvalScore],
|
|
11
|
+
label_a: str = "baseline",
|
|
12
|
+
label_b: str = "candidate",
|
|
13
|
+
) -> dict[str, float | str | int | None]:
|
|
14
|
+
"""Compare average score between two evaluated configurations.
|
|
15
|
+
|
|
16
|
+
Averages cover scored outcomes only; failed and cancelled outcomes are
|
|
17
|
+
counted per side and never enter a mean. ``avg_*`` is None when a side has
|
|
18
|
+
no scored outcome, and ``improvement_pct`` is None unless both sides have a
|
|
19
|
+
nonzero baseline to compare against.
|
|
20
|
+
"""
|
|
21
|
+
avg_a = mean_scored(scores_a)
|
|
22
|
+
avg_b = mean_scored(scores_b)
|
|
23
|
+
|
|
24
|
+
improvement_pct: float | None = None
|
|
25
|
+
if avg_a is not None and avg_b is not None:
|
|
26
|
+
if avg_a == 0:
|
|
27
|
+
improvement_pct = 0.0 if avg_b == 0 else 100.0
|
|
28
|
+
else:
|
|
29
|
+
improvement_pct = ((avg_b - avg_a) / avg_a) * 100.0
|
|
30
|
+
|
|
31
|
+
return {
|
|
32
|
+
"label_a": label_a,
|
|
33
|
+
"label_b": label_b,
|
|
34
|
+
"avg_a": avg_a,
|
|
35
|
+
"avg_b": avg_b,
|
|
36
|
+
"scored_a": _count(scores_a, "scored"),
|
|
37
|
+
"scored_b": _count(scores_b, "scored"),
|
|
38
|
+
"failed_a": _count(scores_a, "failed"),
|
|
39
|
+
"failed_b": _count(scores_b, "failed"),
|
|
40
|
+
"cancelled_a": _count(scores_a, "cancelled"),
|
|
41
|
+
"cancelled_b": _count(scores_b, "cancelled"),
|
|
42
|
+
"improvement_pct": improvement_pct,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def mean_scored(scores: list[EvalScore]) -> float | None:
|
|
47
|
+
"""Mean ``overall`` across scored outcomes, or None when none were scored."""
|
|
48
|
+
values = [score.overall for score in scores if score.is_scored and score.overall is not None]
|
|
49
|
+
if not values:
|
|
50
|
+
return None
|
|
51
|
+
return sum(values) / len(values)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _count(scores: list[EvalScore], status: str) -> int:
|
|
55
|
+
return sum(1 for score in scores if score.scoring_status == status)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
__all__ = ["compare_configurations", "mean_scored"]
|
|
@@ -4,10 +4,12 @@ from agentbyte.eval.base import BaseEvalJudge
|
|
|
4
4
|
|
|
5
5
|
from .composite import CompositeJudge
|
|
6
6
|
from .llm import LLMEvalJudge
|
|
7
|
+
from .trajectory import LLMTrajectoryJudge, ProcessJudgeResponse, ProcessCriterionScore
|
|
7
8
|
from .pairwise import PairwiseJudge
|
|
8
9
|
from .reference import ContainsJudge, ExactMatchJudge, FuzzyMatchJudge
|
|
9
10
|
|
|
10
11
|
__all__ = [
|
|
12
|
+
"LLMTrajectoryJudge", "ProcessJudgeResponse", "ProcessCriterionScore",
|
|
11
13
|
"BaseEvalJudge",
|
|
12
14
|
"ExactMatchJudge",
|
|
13
15
|
"ContainsJudge",
|
|
@@ -7,7 +7,7 @@ from collections.abc import Sequence
|
|
|
7
7
|
|
|
8
8
|
from agentbyte.cancellation_token import CancellationToken
|
|
9
9
|
|
|
10
|
-
from agentbyte.eval.base import BaseEvalJudge
|
|
10
|
+
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
11
11
|
from agentbyte.eval.types import EvalScore, EvalTrajectory
|
|
12
12
|
|
|
13
13
|
|
|
@@ -45,12 +45,30 @@ class CompositeJudge(BaseEvalJudge):
|
|
|
45
45
|
criteria: list[str] | None = None,
|
|
46
46
|
cancellation_token: CancellationToken | None = None,
|
|
47
47
|
) -> EvalScore:
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
48
|
+
"""Combine sub-judge scores; any sub-judge failure fails the composite.
|
|
49
|
+
|
|
50
|
+
Every sub-judge is required. A failed or cancelled component is never
|
|
51
|
+
dropped and the remaining weights are never renormalized into success.
|
|
52
|
+
"""
|
|
53
|
+
try:
|
|
54
|
+
async with asyncio.TaskGroup() as group:
|
|
55
|
+
tasks = [
|
|
56
|
+
group.create_task(judge.score(trajectory, criteria, cancellation_token))
|
|
57
|
+
for judge, _ in self.judges
|
|
58
|
+
]
|
|
59
|
+
except* JudgeScoringError as failures:
|
|
60
|
+
first = failures.exceptions[0]
|
|
61
|
+
raise JudgeScoringError(
|
|
62
|
+
"component_failed", f"Sub-judge failed: {first.reason}"
|
|
63
|
+
) from first
|
|
64
|
+
results = [task.result() for task in tasks]
|
|
65
|
+
overalls: list[float] = []
|
|
66
|
+
for (judge, _), result in zip(self.judges, results):
|
|
67
|
+
if not result.is_scored or result.overall is None:
|
|
68
|
+
raise JudgeScoringError(
|
|
69
|
+
"component_failed", f"Sub-judge {judge.name!r} returned no valid score"
|
|
70
|
+
)
|
|
71
|
+
overalls.append(result.overall)
|
|
54
72
|
|
|
55
73
|
dimensions: dict[str, float] = {}
|
|
56
74
|
dimension_weight_totals: dict[str, float] = {}
|
|
@@ -58,11 +76,9 @@ class CompositeJudge(BaseEvalJudge):
|
|
|
58
76
|
metadata_sub_judges: list[dict[str, float | str]] = []
|
|
59
77
|
weighted_overall = 0.0
|
|
60
78
|
|
|
61
|
-
for (judge, weight), result in zip(self.judges, results):
|
|
62
|
-
metadata_sub_judges.append(
|
|
63
|
-
|
|
64
|
-
)
|
|
65
|
-
weighted_overall += result.overall * weight
|
|
79
|
+
for (judge, weight), result, overall in zip(self.judges, results, overalls):
|
|
80
|
+
metadata_sub_judges.append({"name": judge.name, "weight": weight, "score": overall})
|
|
81
|
+
weighted_overall += overall * weight
|
|
66
82
|
for dimension, score in result.dimensions.items():
|
|
67
83
|
dimensions[dimension] = dimensions.get(dimension, 0.0) + (score * weight)
|
|
68
84
|
dimension_weight_totals[dimension] = (
|