agentbyte 0.29.0__tar.gz → 0.30.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agentbyte-0.29.0 → agentbyte-0.30.0}/CHANGELOG.md +17 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/PKG-INFO +2 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/README.md +1 -1
- agentbyte-0.30.0/src/agentbyte/__about__.py +2 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/__init__.py +6 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/base.py +20 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/local.py +6 -0
- agentbyte-0.30.0/src/agentbyte/eval/comparison.py +58 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/composite.py +28 -12
- agentbyte-0.30.0/src/agentbyte/eval/judges/llm.py +220 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/pairwise.py +32 -41
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/trajectory.py +66 -65
- agentbyte-0.30.0/src/agentbyte/eval/judges/validation.py +93 -0
- agentbyte-0.30.0/src/agentbyte/eval/pairwise.py +170 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/report.py +35 -9
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/runner.py +47 -36
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/types.py +99 -9
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/collector.py +5 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/__init__.py +9 -1
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/base.py +70 -18
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/gepa.py +11 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/pareto.py +7 -3
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/reflective.py +2 -1
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/trace.py +6 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/execution.py +120 -50
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/models.py +6 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/server.py +27 -11
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/__init__.py +2 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/agent.py +8 -3
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/__init__.py +2 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/models.py +39 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/runner.py +88 -7
- agentbyte-0.30.0/src/agentbyte/workflow/steps/agent.py +141 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/step.py +9 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/subworkflow.py +18 -2
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_tool_approval.py +59 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_execution_trajectories.py +17 -11
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_pairwise.py +89 -25
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_phase1_runner_and_targets.py +89 -4
- agentbyte-0.30.0/tests/eval/test_score_integrity.py +194 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_types_and_judges.py +162 -71
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_gepa.py +32 -1
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_mipro.py +1 -1
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_pareto.py +1 -1
- agentbyte-0.30.0/tests/optim/test_score_integrity.py +140 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_trace.py +2 -1
- agentbyte-0.30.0/tests/webui/test_workflow_streaming.py +351 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_steps.py +5 -3
- agentbyte-0.29.0/src/agentbyte/__about__.py +0 -2
- agentbyte-0.29.0/src/agentbyte/eval/comparison.py +0 -38
- agentbyte-0.29.0/src/agentbyte/eval/judges/llm.py +0 -279
- agentbyte-0.29.0/src/agentbyte/eval/pairwise.py +0 -106
- agentbyte-0.29.0/src/agentbyte/workflow/steps/agent.py +0 -87
- {agentbyte-0.29.0 → agentbyte-0.30.0}/.gitignore +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/LICENSE +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/pyproject.toml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent_as_tool.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/embedding_agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cancellation_token.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/catalog.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cli/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cli/main.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/component.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skill_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skills.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/config.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/importer.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/json.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/loader.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish_config.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publishers.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sources.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite_db.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/write_config.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/writers.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/entity.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/decorator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/keyword.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/process.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/tool.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/eval_dataset.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/reference.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/splitting.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/model.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/multi_turn.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/orchestrator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/runtime.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/workflow.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/models.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/_retry_observability.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/auth.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/auth.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/chat.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/embedding.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/settings.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai_embedding.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/embeddings_base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/chat.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/embedding.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/settings.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai_embedding.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/pricing.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/retry_policy.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/settings.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/logger.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/memory/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/memory/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/messages.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/otel.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/retry.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/sql_usage.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/usage_logger.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/notebook.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/config.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/mipro.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/spec.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/ai.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/handoff.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/plan.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/policies.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/round_robin.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/agents.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/clients.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instruction_registry.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/orchestrator.yaml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/query_rewriter.yaml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/researcher.yaml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/reviewer.yaml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/writer.yaml +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/orchestration.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/contracts-analyst/SKILL.md +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/SKILL.md +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/departments.md +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/employees.md +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/payroll.md +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/streaming.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/workflow.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/session_store.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/resources.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/scripts.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/sources.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/validation.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/cancellation.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/composite.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/consecutive_agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/external.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/function_call.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/handoff.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/max_message.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/predicate.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/source.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/text_mention.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/timeout.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/token_usage.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/coding_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/core_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/decorator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/memory_tool.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/research_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/discovery.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/registry.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/session_store.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/sessions.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-BF3DwXaF.js +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-ar5tOeqt.css +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/index.html +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/vite.svg +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/_structure_hash.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/checkpoint.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/workflow.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/defaults.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/loader.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema_utils.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/echo.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/function.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/http.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/transform.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/visualizer.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_as_tool.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_basic.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_context_providers.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_error_response.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_event_types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_memory_integration.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_middleware_integration.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_response_accessors.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_retry_middleware.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_stream_events.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_embedding_agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/cli/test_registry_check.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/test_skill_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/test_skills_provider.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_loader.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_multi_table.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_publish.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_sqlite_db.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_eval_dataset.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_multi_turn.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_phase2_checks_and_reports.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_splitting.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_azure_client.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_azure_embedding_client.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_llm_types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_openai_client.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_openai_embedding_client.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_pricing.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retry_observability.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retry_policy_api.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retryable_error_substrings.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/memory/test_memory.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_deduplicate_tool_result.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_middleware_chain.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_otel.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_retry_middleware.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_sql_usage.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_usage_logger.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_base_integration.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_config.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_reflective.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_spec.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_ai_orchestrator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_base_orchestrator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_handoff_orchestrator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_orchestrator_finalization.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_plan_orchestrator.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_round_robin.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_agents.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_clients.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_instruction_registry.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_orchestration.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_streaming.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_workflow.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_resources.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_scripts.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_sources.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_base.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_cancellation.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_composite.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_consecutive_agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_external.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_function_call.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_handoff.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_max_message.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_predicate.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_source.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_text_mention.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_timeout.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_token_usage.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_cancellation_token.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_context.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_logger.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_messages.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_package_api.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_session_store.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_types.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_vanilla_chunker.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_coding_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_memory_tool.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_research_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_tools.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/__init__.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/helpers.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_execution.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_package_api.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_registry.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_server.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_sessions.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_agent_step_imports.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_checkpoint.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_subworkflow_step.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_agent.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_class.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_models.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_runner.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_schema.py +0 -0
- {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_visualizer.py +0 -0
|
@@ -4,6 +4,23 @@ All notable changes to Agentbyte are documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format follows Keep a Changelog principles and semantic versioning.
|
|
6
6
|
|
|
7
|
+
## [0.30.0] - 2026-09-30
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- Workflow streaming parity (spec 0057): `WorkflowRunner.run_stream(..., stream_tokens=...)` forwards live child agent events as `StepEvent` envelopes (workflow, execution, invocation, step, attempt and sequence attribution) through a bounded queue, including across `workflow.as_agent()` and `SubWorkflowStep` boundaries. `WorkflowAgent.run_stream` honors `stream_tokens`. Closing the stream cancels and awaits outstanding steps.
|
|
12
|
+
- `AgentStep` streams its agent and turns tool approval into workflow suspension; resuming validates the exact pending decisions and never reruns completed tools.
|
|
13
|
+
- Checkpoint-backed serving: the WebUI/FastAPI run and SSE endpoints accept `workflow_responses` and `workflow_checkpoint_id` to resume a suspended workflow in the same session. `WorkflowExecution.checkpoint_id` exposes the suspended checkpoint, and `WorkflowRunner.validate_resume_responses` / `BaseStep.validate_resume_response` validate responses without consuming it. Stale, unknown or wrong-owner checkpoints and concurrent runs of the same workflow or session are rejected with HTTP 409.
|
|
14
|
+
- Judge score integrity (spec 0055): `EvalScore.scoring_status` (`scored` | `failed` | `cancelled`) and `failure_reason`, `JudgeScoringError` with safe reason codes, `mean_scored()`, `OptimizationEvidenceError`, and `PairwiseResult.comparison_status`.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- **Breaking:** Judges never invent scores. `LLMEvalJudge` and `LLMTrajectoryJudge` require every requested criterion exactly once (matched case- and whitespace-insensitively, reported with the requested spelling) with a finite 0–10 value, and raise `JudgeScoringError` otherwise instead of filling missing criteria with 5.0, clamping, or returning a neutral 5.0 after a failure. `overall` is always the mean of the validated dimensions; a model-supplied overall is ignored. Cancellation raises `asyncio.CancelledError`.
|
|
19
|
+
- **Breaking:** `EvalScore.overall` is `float | None`. `EvalRunner` records judge failures, target exceptions and cancellation as `failed`/`cancelled` outcomes with `overall=None`, empty dimensions and the trajectory kept, instead of a fabricated 0.0. Failure metadata keeps the exception type, never its text. A scored `EvalScore` must have a finite `overall` and dimensions in 0–10.
|
|
20
|
+
- **Breaking:** `PairwiseJudge.compare()` raises instead of returning a tie on failure; unknown winners and out-of-range margins are rejected, not coerced. `PairwiseRunner` records failed comparisons with `winner=None`. `compare_pairwise()` adds `compared`/`failed`/`cancelled` counts and computes win rates over valid comparisons only (`None` when nothing was compared).
|
|
21
|
+
- **Breaking:** `EvalReport.avg_score` and `compare_configurations()` averages cover scored outcomes only and are `None` when nothing was scored; an item with a failed or cancelled score fails the suite. `CompositeJudge` fails when any sub-judge fails rather than renormalizing the remaining weights.
|
|
22
|
+
- **Breaking:** Optimizers only rank candidates whose every task was scored. `Candidate.avg` is `None` for ineligible candidates, the minibatch gate rejects incomplete batches, a seed with an unscored task raises `OptimizationEvidenceError`, and the GEPA adapter raises rather than passing a number for an unscored example. Optimization trace `eval` events report `avg` over scored tasks (`None` if none) plus `n_scored`.
|
|
23
|
+
|
|
7
24
|
## [0.29.0] - 2026-09-30
|
|
8
25
|
|
|
9
26
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: agentbyte
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.30.0
|
|
4
4
|
Summary: A toolkit for designing multiagent systems
|
|
5
5
|
Author-email: MrDataPsycho <mr.data.psycho@gmail.com>
|
|
6
6
|
License-Expression: LicenseRef-Proprietary
|
|
@@ -86,7 +86,7 @@ Description-Content-Type: text/markdown
|
|
|
86
86
|
|
|
87
87
|
Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
|
|
88
88
|
|
|
89
|
-
Current release: **0.
|
|
89
|
+
Current release: **0.30.0**
|
|
90
90
|
|
|
91
91
|
## Building an Agent
|
|
92
92
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Evaluation framework for AgentByte."""
|
|
2
2
|
|
|
3
|
-
from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget
|
|
3
|
+
from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget, JudgeScoringError
|
|
4
4
|
from .checks.process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
|
|
5
5
|
from .judges.trajectory import LLMTrajectoryJudge
|
|
6
6
|
from .targets.workflow import WorkflowEvalTarget
|
|
@@ -19,7 +19,7 @@ from .checks import (
|
|
|
19
19
|
tool_calls_present,
|
|
20
20
|
tool_calls_present_check,
|
|
21
21
|
)
|
|
22
|
-
from .comparison import compare_configurations
|
|
22
|
+
from .comparison import compare_configurations, mean_scored
|
|
23
23
|
from .eval_dataset import (
|
|
24
24
|
EvalDatasetRecord,
|
|
25
25
|
save_eval_dataset,
|
|
@@ -52,6 +52,7 @@ from .types import (
|
|
|
52
52
|
EvalTask,
|
|
53
53
|
EvalTrajectory,
|
|
54
54
|
ExpectedToolCall,
|
|
55
|
+
ScoringStatus,
|
|
55
56
|
MultiTurnEvalTask,
|
|
56
57
|
WorkflowEvalTask,
|
|
57
58
|
PairwiseResult,
|
|
@@ -70,6 +71,8 @@ __all__ = [
|
|
|
70
71
|
"BaseEvalTarget",
|
|
71
72
|
"BaseEvalJudge",
|
|
72
73
|
"BaseEvalRunner",
|
|
74
|
+
"JudgeScoringError",
|
|
75
|
+
"ScoringStatus",
|
|
73
76
|
"ExactMatchJudge",
|
|
74
77
|
"ContainsJudge",
|
|
75
78
|
"FuzzyMatchJudge",
|
|
@@ -88,6 +91,7 @@ __all__ = [
|
|
|
88
91
|
"EvalSplitConfig",
|
|
89
92
|
"EvalSplitStrategy",
|
|
90
93
|
"compare_configurations",
|
|
94
|
+
"mean_scored",
|
|
91
95
|
"EvalDatasetRecord",
|
|
92
96
|
"save_eval_dataset",
|
|
93
97
|
"tasks_from_eval_dataset",
|
|
@@ -17,6 +17,19 @@ _VALID_ANSWER_STRATEGIES: tuple[AnswerStrategy, ...] = (
|
|
|
17
17
|
)
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
class JudgeScoringError(Exception):
|
|
21
|
+
"""A judge could not produce a valid judgment.
|
|
22
|
+
|
|
23
|
+
``reason`` is a safe, stable code (for example ``provider_error`` or
|
|
24
|
+
``missing_criterion``) suitable for persistence; the message may carry more
|
|
25
|
+
detail and is not persisted by default.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self, reason: str, message: str | None = None) -> None:
|
|
29
|
+
super().__init__(message or reason)
|
|
30
|
+
self.reason = reason
|
|
31
|
+
|
|
32
|
+
|
|
20
33
|
class BaseEvalTarget(ABC):
|
|
21
34
|
"""Anything that can execute an evaluation task into a trajectory."""
|
|
22
35
|
|
|
@@ -96,7 +109,12 @@ class BaseEvalJudge(ABC):
|
|
|
96
109
|
criteria: list[str] | None = None,
|
|
97
110
|
cancellation_token: CancellationToken | None = None,
|
|
98
111
|
) -> EvalScore:
|
|
99
|
-
"""Score an evaluation trajectory.
|
|
112
|
+
"""Score an evaluation trajectory.
|
|
113
|
+
|
|
114
|
+
Returns a scored ``EvalScore``. Raises ``JudgeScoringError`` when no
|
|
115
|
+
valid judgment is available and ``asyncio.CancelledError`` when
|
|
116
|
+
cancelled; judges never substitute a placeholder score.
|
|
117
|
+
"""
|
|
100
118
|
|
|
101
119
|
|
|
102
120
|
class BaseEvalRunner(ABC):
|
|
@@ -116,4 +134,4 @@ class BaseEvalRunner(ABC):
|
|
|
116
134
|
"""Evaluate a target on multiple tasks."""
|
|
117
135
|
|
|
118
136
|
|
|
119
|
-
__all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner"]
|
|
137
|
+
__all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner", "JudgeScoringError"]
|
|
@@ -93,6 +93,12 @@ def threshold_gate(
|
|
|
93
93
|
|
|
94
94
|
async def _check(trajectory: EvalTrajectory) -> CheckResult:
|
|
95
95
|
score = await judge.score(trajectory)
|
|
96
|
+
if not score.is_scored or score.overall is None:
|
|
97
|
+
return CheckResult(
|
|
98
|
+
passed=False,
|
|
99
|
+
reason=f"no valid score ({score.scoring_status}: {score.failure_reason})",
|
|
100
|
+
check_name=f"threshold_gate[{judge.name}]",
|
|
101
|
+
)
|
|
96
102
|
passed = score.overall >= threshold
|
|
97
103
|
return CheckResult(
|
|
98
104
|
passed=passed,
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Helpers for comparing evaluated configurations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from agentbyte.eval.types import EvalScore
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def compare_configurations(
|
|
9
|
+
scores_a: list[EvalScore],
|
|
10
|
+
scores_b: list[EvalScore],
|
|
11
|
+
label_a: str = "baseline",
|
|
12
|
+
label_b: str = "candidate",
|
|
13
|
+
) -> dict[str, float | str | int | None]:
|
|
14
|
+
"""Compare average score between two evaluated configurations.
|
|
15
|
+
|
|
16
|
+
Averages cover scored outcomes only; failed and cancelled outcomes are
|
|
17
|
+
counted per side and never enter a mean. ``avg_*`` is None when a side has
|
|
18
|
+
no scored outcome, and ``improvement_pct`` is None unless both sides have a
|
|
19
|
+
nonzero baseline to compare against.
|
|
20
|
+
"""
|
|
21
|
+
avg_a = mean_scored(scores_a)
|
|
22
|
+
avg_b = mean_scored(scores_b)
|
|
23
|
+
|
|
24
|
+
improvement_pct: float | None = None
|
|
25
|
+
if avg_a is not None and avg_b is not None:
|
|
26
|
+
if avg_a == 0:
|
|
27
|
+
improvement_pct = 0.0 if avg_b == 0 else 100.0
|
|
28
|
+
else:
|
|
29
|
+
improvement_pct = ((avg_b - avg_a) / avg_a) * 100.0
|
|
30
|
+
|
|
31
|
+
return {
|
|
32
|
+
"label_a": label_a,
|
|
33
|
+
"label_b": label_b,
|
|
34
|
+
"avg_a": avg_a,
|
|
35
|
+
"avg_b": avg_b,
|
|
36
|
+
"scored_a": _count(scores_a, "scored"),
|
|
37
|
+
"scored_b": _count(scores_b, "scored"),
|
|
38
|
+
"failed_a": _count(scores_a, "failed"),
|
|
39
|
+
"failed_b": _count(scores_b, "failed"),
|
|
40
|
+
"cancelled_a": _count(scores_a, "cancelled"),
|
|
41
|
+
"cancelled_b": _count(scores_b, "cancelled"),
|
|
42
|
+
"improvement_pct": improvement_pct,
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def mean_scored(scores: list[EvalScore]) -> float | None:
|
|
47
|
+
"""Mean ``overall`` across scored outcomes, or None when none were scored."""
|
|
48
|
+
values = [score.overall for score in scores if score.is_scored and score.overall is not None]
|
|
49
|
+
if not values:
|
|
50
|
+
return None
|
|
51
|
+
return sum(values) / len(values)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _count(scores: list[EvalScore], status: str) -> int:
|
|
55
|
+
return sum(1 for score in scores if score.scoring_status == status)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
__all__ = ["compare_configurations", "mean_scored"]
|
|
@@ -7,7 +7,7 @@ from collections.abc import Sequence
|
|
|
7
7
|
|
|
8
8
|
from agentbyte.cancellation_token import CancellationToken
|
|
9
9
|
|
|
10
|
-
from agentbyte.eval.base import BaseEvalJudge
|
|
10
|
+
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
11
11
|
from agentbyte.eval.types import EvalScore, EvalTrajectory
|
|
12
12
|
|
|
13
13
|
|
|
@@ -45,12 +45,30 @@ class CompositeJudge(BaseEvalJudge):
|
|
|
45
45
|
criteria: list[str] | None = None,
|
|
46
46
|
cancellation_token: CancellationToken | None = None,
|
|
47
47
|
) -> EvalScore:
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
48
|
+
"""Combine sub-judge scores; any sub-judge failure fails the composite.
|
|
49
|
+
|
|
50
|
+
Every sub-judge is required. A failed or cancelled component is never
|
|
51
|
+
dropped and the remaining weights are never renormalized into success.
|
|
52
|
+
"""
|
|
53
|
+
try:
|
|
54
|
+
async with asyncio.TaskGroup() as group:
|
|
55
|
+
tasks = [
|
|
56
|
+
group.create_task(judge.score(trajectory, criteria, cancellation_token))
|
|
57
|
+
for judge, _ in self.judges
|
|
58
|
+
]
|
|
59
|
+
except* JudgeScoringError as failures:
|
|
60
|
+
first = failures.exceptions[0]
|
|
61
|
+
raise JudgeScoringError(
|
|
62
|
+
"component_failed", f"Sub-judge failed: {first.reason}"
|
|
63
|
+
) from first
|
|
64
|
+
results = [task.result() for task in tasks]
|
|
65
|
+
overalls: list[float] = []
|
|
66
|
+
for (judge, _), result in zip(self.judges, results):
|
|
67
|
+
if not result.is_scored or result.overall is None:
|
|
68
|
+
raise JudgeScoringError(
|
|
69
|
+
"component_failed", f"Sub-judge {judge.name!r} returned no valid score"
|
|
70
|
+
)
|
|
71
|
+
overalls.append(result.overall)
|
|
54
72
|
|
|
55
73
|
dimensions: dict[str, float] = {}
|
|
56
74
|
dimension_weight_totals: dict[str, float] = {}
|
|
@@ -58,11 +76,9 @@ class CompositeJudge(BaseEvalJudge):
|
|
|
58
76
|
metadata_sub_judges: list[dict[str, float | str]] = []
|
|
59
77
|
weighted_overall = 0.0
|
|
60
78
|
|
|
61
|
-
for (judge, weight), result in zip(self.judges, results):
|
|
62
|
-
metadata_sub_judges.append(
|
|
63
|
-
|
|
64
|
-
)
|
|
65
|
-
weighted_overall += result.overall * weight
|
|
79
|
+
for (judge, weight), result, overall in zip(self.judges, results, overalls):
|
|
80
|
+
metadata_sub_judges.append({"name": judge.name, "weight": weight, "score": overall})
|
|
81
|
+
weighted_overall += overall * weight
|
|
66
82
|
for dimension, score in result.dimensions.items():
|
|
67
83
|
dimensions[dimension] = dimensions.get(dimension, 0.0) + (score * weight)
|
|
68
84
|
dimension_weight_totals[dimension] = (
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
"""LLM-powered evaluation judge."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictFloat, ValidationError
|
|
10
|
+
|
|
11
|
+
from agentbyte.cancellation_token import CancellationToken
|
|
12
|
+
from agentbyte.llm.base import BaseChatCompletionClient
|
|
13
|
+
from agentbyte.messages import SystemMessage, UserMessage
|
|
14
|
+
|
|
15
|
+
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
16
|
+
from agentbyte.eval.judges.validation import (
|
|
17
|
+
mean_score,
|
|
18
|
+
normalize_criterion,
|
|
19
|
+
validate_criterion_scores,
|
|
20
|
+
validate_requested_criteria,
|
|
21
|
+
)
|
|
22
|
+
from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class _JudgeCriterionScore(BaseModel):
|
|
26
|
+
criterion: str
|
|
27
|
+
score: StrictFloat = Field(ge=0.0, le=10.0)
|
|
28
|
+
reasoning: str
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class _JudgeStructuredResponse(BaseModel):
|
|
32
|
+
scores: list[_JudgeCriterionScore]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# Kept for the use_structured_output=False path. Values stay untyped so the
|
|
36
|
+
# shared validator (not Pydantic coercion) decides what counts as a score.
|
|
37
|
+
class _JudgeResponse(BaseModel):
|
|
38
|
+
model_config = ConfigDict(extra="allow")
|
|
39
|
+
|
|
40
|
+
dimensions: dict[str, Any] = Field(default_factory=dict)
|
|
41
|
+
reasoning: dict[str, Any] = Field(default_factory=dict)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class LLMEvalJudge(BaseEvalJudge):
|
|
45
|
+
"""Use an LLM to judge a trajectory on one or more criteria."""
|
|
46
|
+
|
|
47
|
+
def __init__(
|
|
48
|
+
self,
|
|
49
|
+
client: BaseChatCompletionClient,
|
|
50
|
+
*,
|
|
51
|
+
name: str | None = None,
|
|
52
|
+
answer_strategy: AnswerStrategy = "last_non_empty",
|
|
53
|
+
source_filter: str | None = None,
|
|
54
|
+
default_criteria: list[str] | None = None,
|
|
55
|
+
custom_instructions: str | None = None,
|
|
56
|
+
use_structured_output: bool = True,
|
|
57
|
+
) -> None:
|
|
58
|
+
super().__init__(
|
|
59
|
+
name=name or f"LLM-{client.model}",
|
|
60
|
+
answer_strategy=answer_strategy,
|
|
61
|
+
source_filter=source_filter,
|
|
62
|
+
)
|
|
63
|
+
self.client = client
|
|
64
|
+
self.default_criteria = default_criteria or [
|
|
65
|
+
"accuracy",
|
|
66
|
+
"completeness",
|
|
67
|
+
"helpfulness",
|
|
68
|
+
]
|
|
69
|
+
self.custom_instructions = custom_instructions
|
|
70
|
+
self.use_structured_output = use_structured_output
|
|
71
|
+
|
|
72
|
+
async def score(
|
|
73
|
+
self,
|
|
74
|
+
trajectory: EvalTrajectory,
|
|
75
|
+
criteria: list[str] | None = None,
|
|
76
|
+
cancellation_token: CancellationToken | None = None,
|
|
77
|
+
) -> EvalScore:
|
|
78
|
+
"""Score every requested criterion or raise ``JudgeScoringError``.
|
|
79
|
+
|
|
80
|
+
``overall`` is the mean of the validated dimensions. Cancellation raises
|
|
81
|
+
``asyncio.CancelledError``; no placeholder score is ever returned.
|
|
82
|
+
"""
|
|
83
|
+
# resolve: call-level arg > task metadata > judge default
|
|
84
|
+
if criteria is None:
|
|
85
|
+
raw = trajectory.task.metadata.get("_criteria")
|
|
86
|
+
if isinstance(raw, list):
|
|
87
|
+
criteria = raw
|
|
88
|
+
elif isinstance(raw, str):
|
|
89
|
+
criteria = [raw]
|
|
90
|
+
eval_criteria = validate_requested_criteria(criteria or self.default_criteria)
|
|
91
|
+
if cancellation_token and cancellation_token.is_cancelled():
|
|
92
|
+
raise asyncio.CancelledError()
|
|
93
|
+
|
|
94
|
+
answer = self.extract_answer(trajectory)
|
|
95
|
+
user_payload = {
|
|
96
|
+
"task_name": trajectory.task.name,
|
|
97
|
+
"task_input": trajectory.task.input,
|
|
98
|
+
"expected_output": trajectory.task.expected_output,
|
|
99
|
+
"actual_output": answer,
|
|
100
|
+
"success": trajectory.success,
|
|
101
|
+
"error": trajectory.error,
|
|
102
|
+
"criteria": eval_criteria,
|
|
103
|
+
"usage": trajectory.usage.model_dump(mode="json")
|
|
104
|
+
if trajectory.usage is not None
|
|
105
|
+
else None,
|
|
106
|
+
"metadata": trajectory.metadata,
|
|
107
|
+
}
|
|
108
|
+
user_message = UserMessage(
|
|
109
|
+
content=json.dumps(user_payload, ensure_ascii=True, indent=2),
|
|
110
|
+
source="user",
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
try:
|
|
114
|
+
if self.use_structured_output:
|
|
115
|
+
result = await self.client.create(
|
|
116
|
+
messages=[
|
|
117
|
+
SystemMessage(
|
|
118
|
+
content=self._build_structured_prompt(eval_criteria),
|
|
119
|
+
source="system",
|
|
120
|
+
),
|
|
121
|
+
user_message,
|
|
122
|
+
],
|
|
123
|
+
output_format=_JudgeStructuredResponse,
|
|
124
|
+
)
|
|
125
|
+
else:
|
|
126
|
+
result = await self.client.create(
|
|
127
|
+
messages=[
|
|
128
|
+
SystemMessage(content=self._build_raw_prompt(), source="system"),
|
|
129
|
+
user_message,
|
|
130
|
+
],
|
|
131
|
+
)
|
|
132
|
+
except Exception as exc:
|
|
133
|
+
raise JudgeScoringError("provider_error", "Judge model call failed") from exc
|
|
134
|
+
|
|
135
|
+
if self.use_structured_output:
|
|
136
|
+
dimensions, reasoning = self._validate_structured(
|
|
137
|
+
result.structured_output, eval_criteria
|
|
138
|
+
)
|
|
139
|
+
else:
|
|
140
|
+
dimensions, reasoning = self._validate_raw(result.message.content, eval_criteria)
|
|
141
|
+
|
|
142
|
+
return EvalScore(
|
|
143
|
+
overall=mean_score(dimensions),
|
|
144
|
+
dimensions=dimensions,
|
|
145
|
+
reasoning=reasoning,
|
|
146
|
+
trajectory=trajectory,
|
|
147
|
+
metadata={
|
|
148
|
+
"judge": self.name,
|
|
149
|
+
"model": result.model,
|
|
150
|
+
"finish_reason": result.finish_reason,
|
|
151
|
+
},
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
def _build_structured_prompt(self, criteria: list[str]) -> str:
|
|
155
|
+
criteria_list = ", ".join(f'"{c}"' for c in criteria)
|
|
156
|
+
prompt = (
|
|
157
|
+
"You are an evaluation judge. "
|
|
158
|
+
f"Score the following criteria: [{criteria_list}]. "
|
|
159
|
+
"Return exactly one scores[] entry per criterion, using the names as given. "
|
|
160
|
+
"Score each 0–10."
|
|
161
|
+
)
|
|
162
|
+
if self.custom_instructions:
|
|
163
|
+
prompt += f"\n\nAdditional instructions:\n{self.custom_instructions.strip()}"
|
|
164
|
+
return prompt
|
|
165
|
+
|
|
166
|
+
def _build_raw_prompt(self) -> str:
|
|
167
|
+
prompt = (
|
|
168
|
+
"You are an evaluation judge. "
|
|
169
|
+
"Return ONLY a strict JSON object with exactly these keys:\n"
|
|
170
|
+
' "dimensions": {"<criterion>": <number 0-10>, ...},\n'
|
|
171
|
+
' "reasoning": {"<criterion>": "<explanation string>", ...}\n'
|
|
172
|
+
"Score every requested criterion exactly once. "
|
|
173
|
+
"The `reasoning` value MUST be an object (dict) — one key per criterion, "
|
|
174
|
+
"never a plain string. No extra keys, no markdown, no prose outside the JSON."
|
|
175
|
+
)
|
|
176
|
+
if self.custom_instructions:
|
|
177
|
+
prompt += f"\n\nAdditional instructions:\n{self.custom_instructions.strip()}"
|
|
178
|
+
return prompt
|
|
179
|
+
|
|
180
|
+
@staticmethod
|
|
181
|
+
def _validate_structured(
|
|
182
|
+
structured: Any,
|
|
183
|
+
criteria: list[str],
|
|
184
|
+
) -> tuple[dict[str, float], dict[str, str]]:
|
|
185
|
+
if not isinstance(structured, _JudgeStructuredResponse):
|
|
186
|
+
raise JudgeScoringError(
|
|
187
|
+
"invalid_response",
|
|
188
|
+
f"Expected _JudgeStructuredResponse, got {type(structured).__name__}",
|
|
189
|
+
)
|
|
190
|
+
return validate_criterion_scores(
|
|
191
|
+
criteria,
|
|
192
|
+
((item.criterion, item.score, item.reasoning) for item in structured.scores),
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
@staticmethod
|
|
196
|
+
def _validate_raw(
|
|
197
|
+
raw_content: str,
|
|
198
|
+
criteria: list[str],
|
|
199
|
+
) -> tuple[dict[str, float], dict[str, str]]:
|
|
200
|
+
try:
|
|
201
|
+
parsed = _JudgeResponse.model_validate(json.loads(raw_content))
|
|
202
|
+
except (json.JSONDecodeError, ValidationError, TypeError) as exc:
|
|
203
|
+
raise JudgeScoringError("invalid_response", "Judge returned malformed JSON") from exc
|
|
204
|
+
dimensions, _ = validate_criterion_scores(
|
|
205
|
+
criteria,
|
|
206
|
+
((name, value, None) for name, value in parsed.dimensions.items()),
|
|
207
|
+
)
|
|
208
|
+
reasoning_lookup = {
|
|
209
|
+
normalize_criterion(str(name)): text
|
|
210
|
+
for name, text in parsed.reasoning.items()
|
|
211
|
+
if isinstance(text, str)
|
|
212
|
+
}
|
|
213
|
+
reasoning = {
|
|
214
|
+
criterion: reasoning_lookup.get(normalize_criterion(criterion), "")
|
|
215
|
+
for criterion in criteria
|
|
216
|
+
}
|
|
217
|
+
return dimensions, reasoning
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
__all__ = ["LLMEvalJudge"]
|
|
@@ -2,25 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
|
+
import asyncio
|
|
5
6
|
import json
|
|
7
|
+
import math
|
|
6
8
|
from typing import Any
|
|
7
9
|
|
|
8
|
-
from pydantic import BaseModel, ConfigDict, Field
|
|
10
|
+
from pydantic import BaseModel, ConfigDict, Field, StrictFloat, ValidationError
|
|
9
11
|
|
|
10
12
|
from agentbyte.cancellation_token import CancellationToken
|
|
11
13
|
from agentbyte.llm.base import BaseChatCompletionClient
|
|
12
14
|
from agentbyte.messages import SystemMessage, UserMessage
|
|
13
15
|
|
|
14
|
-
from agentbyte.eval.base import BaseEvalJudge
|
|
16
|
+
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
15
17
|
from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory, PairwiseResult
|
|
16
18
|
|
|
17
19
|
|
|
18
20
|
class _PairwiseResponse(BaseModel):
|
|
19
21
|
model_config = ConfigDict(extra="allow")
|
|
20
22
|
|
|
21
|
-
winner: str
|
|
22
|
-
margin:
|
|
23
|
-
reasoning: str = Field(default="
|
|
23
|
+
winner: str
|
|
24
|
+
margin: StrictFloat
|
|
25
|
+
reasoning: str = Field(default="")
|
|
24
26
|
|
|
25
27
|
|
|
26
28
|
class PairwiseJudge(BaseEvalJudge):
|
|
@@ -63,13 +65,14 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
63
65
|
trajectory_b: EvalTrajectory,
|
|
64
66
|
cancellation_token: CancellationToken | None = None,
|
|
65
67
|
) -> PairwiseResult:
|
|
66
|
-
"""Compare two trajectories and return a pairwise decision.
|
|
68
|
+
"""Compare two trajectories and return a measured pairwise decision.
|
|
67
69
|
|
|
68
|
-
|
|
69
|
-
|
|
70
|
+
Raises ``JudgeScoringError`` when no valid decision is available and
|
|
71
|
+
``asyncio.CancelledError`` when cancelled. A failure is never reported
|
|
72
|
+
as a tie.
|
|
70
73
|
"""
|
|
71
74
|
if cancellation_token and cancellation_token.is_cancelled():
|
|
72
|
-
|
|
75
|
+
raise asyncio.CancelledError()
|
|
73
76
|
|
|
74
77
|
answer_a = self.extract_answer(trajectory_a)
|
|
75
78
|
answer_b = self.extract_answer(trajectory_b)
|
|
@@ -106,22 +109,27 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
106
109
|
],
|
|
107
110
|
output_format=_PairwiseResponse,
|
|
108
111
|
)
|
|
109
|
-
parsed = self._parse_response(result.structured_output, result.message.content)
|
|
110
|
-
winner = parsed.winner if parsed.winner in ("a", "b", "tie") else "tie"
|
|
111
|
-
margin = max(0.0, min(1.0, float(parsed.margin)))
|
|
112
|
-
if winner == "tie":
|
|
113
|
-
margin = 0.0
|
|
114
|
-
return PairwiseResult(
|
|
115
|
-
task=trajectory_a.task,
|
|
116
|
-
winner=winner, # type: ignore[arg-type]
|
|
117
|
-
margin=margin,
|
|
118
|
-
reasoning=parsed.reasoning,
|
|
119
|
-
trajectory_a=trajectory_a,
|
|
120
|
-
trajectory_b=trajectory_b,
|
|
121
|
-
metadata={"judge": self.name, "model": result.model},
|
|
122
|
-
)
|
|
123
112
|
except Exception as exc:
|
|
124
|
-
|
|
113
|
+
raise JudgeScoringError("provider_error", "Pairwise judge model call failed") from exc
|
|
114
|
+
try:
|
|
115
|
+
parsed = self._parse_response(result.structured_output, result.message.content)
|
|
116
|
+
except ValidationError as exc:
|
|
117
|
+
raise JudgeScoringError("invalid_response", "Malformed pairwise response") from exc
|
|
118
|
+
winner = parsed.winner.strip().lower()
|
|
119
|
+
if winner not in ("a", "b", "tie"):
|
|
120
|
+
raise JudgeScoringError("invalid_response", f"Unknown winner: {parsed.winner!r}")
|
|
121
|
+
margin = float(parsed.margin)
|
|
122
|
+
if not math.isfinite(margin) or not 0.0 <= margin <= 1.0:
|
|
123
|
+
raise JudgeScoringError("invalid_score", f"Margin outside 0-1: {parsed.margin!r}")
|
|
124
|
+
return PairwiseResult(
|
|
125
|
+
task=trajectory_a.task,
|
|
126
|
+
winner=winner, # type: ignore[arg-type]
|
|
127
|
+
margin=0.0 if winner == "tie" else margin,
|
|
128
|
+
reasoning=parsed.reasoning,
|
|
129
|
+
trajectory_a=trajectory_a,
|
|
130
|
+
trajectory_b=trajectory_b,
|
|
131
|
+
metadata={"judge": self.name, "model": result.model},
|
|
132
|
+
)
|
|
125
133
|
|
|
126
134
|
async def score(self, *args: Any, **kwargs: Any) -> EvalScore:
|
|
127
135
|
"""Not supported on PairwiseJudge — use compare() instead."""
|
|
@@ -142,22 +150,5 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
142
150
|
return _PairwiseResponse.model_validate(structured_output)
|
|
143
151
|
return _PairwiseResponse.model_validate_json(raw_content)
|
|
144
152
|
|
|
145
|
-
def _tie(
|
|
146
|
-
self,
|
|
147
|
-
trajectory_a: EvalTrajectory,
|
|
148
|
-
trajectory_b: EvalTrajectory,
|
|
149
|
-
*,
|
|
150
|
-
fallback_reason: str,
|
|
151
|
-
) -> PairwiseResult:
|
|
152
|
-
return PairwiseResult(
|
|
153
|
-
task=trajectory_a.task,
|
|
154
|
-
winner="tie",
|
|
155
|
-
margin=0.0,
|
|
156
|
-
reasoning=fallback_reason,
|
|
157
|
-
trajectory_a=trajectory_a,
|
|
158
|
-
trajectory_b=trajectory_b,
|
|
159
|
-
metadata={"judge": self.name, "fallback": True, "fallback_reason": fallback_reason},
|
|
160
|
-
)
|
|
161
|
-
|
|
162
153
|
|
|
163
154
|
__all__ = ["PairwiseJudge"]
|