agentbyte 0.30.0__tar.gz → 0.30.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agentbyte-0.30.0 → agentbyte-0.30.1}/CHANGELOG.md +14 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/PKG-INFO +2 -2
- {agentbyte-0.30.0 → agentbyte-0.30.1}/README.md +1 -1
- agentbyte-0.30.1/src/agentbyte/__about__.py +2 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/llm.py +13 -8
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/pairwise.py +11 -3
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/trajectory.py +11 -7
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/validation.py +29 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/pairwise.py +7 -2
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/runner.py +15 -1
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/model.py +3 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/types.py +10 -0
- agentbyte-0.30.1/tests/eval/test_score_integrity.py +409 -0
- agentbyte-0.30.0/src/agentbyte/__about__.py +0 -2
- agentbyte-0.30.0/tests/eval/test_score_integrity.py +0 -194
- {agentbyte-0.30.0 → agentbyte-0.30.1}/.gitignore +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/LICENSE +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/pyproject.toml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/agent_as_tool.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/embedding_agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/agents/types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/cancellation_token.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/catalog.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/cli/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/cli/main.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/component.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/context.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/context_providers/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/context_providers/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/context_providers/skill_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/context_providers/skills.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/config.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/importer.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/json.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/loader.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/publish.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/publish_config.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/publishers.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/sources.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/sqlite.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/sqlite_db.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/write_config.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/dataset/writers.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/entity.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/decorator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/keyword.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/local.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/process.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/tool.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/checks/types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/comparison.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/eval_dataset.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/composite.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/judges/reference.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/report.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/splitting.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/multi_turn.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/orchestrator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/runtime.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/eval/targets/workflow.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/execution_trace/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/execution_trace/collector.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/execution_trace/models.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/_retry_observability.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/auth.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure/auth.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure/chat.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure/embedding.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure/settings.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure_openai.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/azure_openai_embedding.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/embeddings_base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/openai/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/openai/chat.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/openai/embedding.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/openai/settings.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/openai_embedding.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/pricing.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/retry_policy.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/settings.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/llm/types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/logger.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/memory/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/memory/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/messages.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/otel.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/retry.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/sql_usage.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/middleware/usage_logger.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/notebook.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/config.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/gepa.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/mipro.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/pareto.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/reflective.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/spec.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/optim/trace.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/ai.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/handoff.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/plan.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/policies.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/orchestration/round_robin.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/agents.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/clients.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instruction_registry.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instructions/orchestrator.yaml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instructions/query_rewriter.yaml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instructions/researcher.yaml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instructions/reviewer.yaml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/instructions/writer.yaml +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/orchestration.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/skills/contracts-analyst/SKILL.md +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/skills/hr-analyst/SKILL.md +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/skills/hr-analyst/resources/departments.md +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/skills/hr-analyst/resources/employees.md +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/skills/hr-analyst/resources/payroll.md +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/streaming.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/presets/workflow.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/session_store.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/resources.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/scripts.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/sources.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/skills/validation.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/cancellation.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/composite.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/consecutive_agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/external.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/function_call.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/handoff.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/max_message.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/predicate.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/source.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/text_mention.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/timeout.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/termination/token_usage.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/coding_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/core_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/decorator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/memory_tool.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/tools/research_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/discovery.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/execution.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/models.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/registry.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/server.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/session_store.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/sessions.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/ui/assets/index-BF3DwXaF.js +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/ui/assets/index-ar5tOeqt.css +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/ui/index.html +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/webui/ui/vite.svg +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/_structure_hash.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/checkpoint.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/models.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/runner.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/core/workflow.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/defaults.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/loader.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/schema.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/schema_utils.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/echo.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/function.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/http.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/step.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/subworkflow.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/steps/transform.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/src/agentbyte/workflow/visualizer.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_as_tool.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_basic.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_context_providers.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_error_response.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_event_types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_memory_integration.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_middleware_integration.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_response_accessors.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_retry_middleware.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_agent_stream_events.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_embedding_agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/agents/test_tool_approval.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/cli/test_registry_check.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/context_providers/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/context_providers/test_skill_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/context_providers/test_skills_provider.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/dataset/test_loader.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/dataset/test_multi_table.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/dataset/test_publish.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/dataset/test_sqlite_db.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_eval_dataset.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_execution_trajectories.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_multi_turn.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_pairwise.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_phase1_runner_and_targets.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_phase2_checks_and_reports.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_splitting.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/eval/test_types_and_judges.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_azure_client.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_azure_embedding_client.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_llm_types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_openai_client.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_openai_embedding_client.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_pricing.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_retry_observability.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_retry_policy_api.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/llm/test_retryable_error_substrings.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/memory/test_memory.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_deduplicate_tool_result.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_middleware_chain.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_otel.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_retry_middleware.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_sql_usage.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/middleware/test_usage_logger.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_base_integration.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_config.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_gepa.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_mipro.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_pareto.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_reflective.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_score_integrity.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_spec.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/optim/test_trace.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_ai_orchestrator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_base_orchestrator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_handoff_orchestrator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_orchestrator_finalization.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_plan_orchestrator.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/orchestration/test_round_robin.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_agents.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_clients.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_instruction_registry.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_orchestration.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_streaming.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/presets/test_workflow.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/skills/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/skills/test_base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/skills/test_resources.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/skills/test_scripts.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/skills/test_sources.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_base.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_cancellation.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_composite.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_consecutive_agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_external.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_function_call.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_handoff.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_max_message.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_predicate.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_source.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_text_mention.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_timeout.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/termination/test_token_usage.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_cancellation_token.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_context.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_logger.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_messages.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_package_api.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_session_store.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_types.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/test_vanilla_chunker.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/tools/test_coding_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/tools/test_memory_tool.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/tools/test_research_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/tools/test_tools.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/__init__.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/helpers.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_execution.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_package_api.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_registry.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_server.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_sessions.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/webui/test_workflow_streaming.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_agent_step_imports.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_checkpoint.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_subworkflow_step.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_agent.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_class.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_models.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_runner.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_schema.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_steps.py +0 -0
- {agentbyte-0.30.0 → agentbyte-0.30.1}/tests/workflow/test_workflow_visualizer.py +0 -0
|
@@ -4,6 +4,20 @@ All notable changes to Agentbyte are documented in this file.
|
|
|
4
4
|
|
|
5
5
|
The format follows Keep a Changelog principles and semantic versioning.
|
|
6
6
|
|
|
7
|
+
## [0.30.1] - 2026-09-30
|
|
8
|
+
|
|
9
|
+
### Fixed
|
|
10
|
+
|
|
11
|
+
- Judge score integrity follow-ups (spec 0055 v0.3.0): judges re-check cancellation after each model call, and `EvalRunner`/`PairwiseRunner` record a run as `cancelled` without judging it when the token is cancelled or the trajectory has `execution_status="cancelled"`.
|
|
12
|
+
- Raw-JSON judge responses with duplicate keys are rejected (`duplicate_criterion` for `LLMEvalJudge`, `invalid_response` for `PairwiseJudge`) instead of keeping the last value.
|
|
13
|
+
- When a target raises, `EvalRunner` and `PairwiseRunner` store only the exception type in `trajectory.error`, never the exception message.
|
|
14
|
+
- `ModelEvalTarget` records `execution_status` on failed and cancelled trajectories.
|
|
15
|
+
|
|
16
|
+
### Changed
|
|
17
|
+
|
|
18
|
+
- **Breaking:** `criteria=None` selects judge defaults; an explicit empty list raises `JudgeScoringError("invalid_criteria")`. An empty `default_criteria` (`LLMEvalJudge`, `LLMTrajectoryJudge`) or `criteria` (`PairwiseJudge`) raises `ValueError` at construction.
|
|
19
|
+
- **Breaking:** a scored `EvalScore` carrying legacy `fallback` or `judge_failed` metadata fails validation.
|
|
20
|
+
|
|
7
21
|
## [0.30.0] - 2026-09-30
|
|
8
22
|
|
|
9
23
|
### Added
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: agentbyte
|
|
3
|
-
Version: 0.30.
|
|
3
|
+
Version: 0.30.1
|
|
4
4
|
Summary: A toolkit for designing multiagent systems
|
|
5
5
|
Author-email: MrDataPsycho <mr.data.psycho@gmail.com>
|
|
6
6
|
License-Expression: LicenseRef-Proprietary
|
|
@@ -86,7 +86,7 @@ Description-Content-Type: text/markdown
|
|
|
86
86
|
|
|
87
87
|
Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
|
|
88
88
|
|
|
89
|
-
Current release: **0.30.
|
|
89
|
+
Current release: **0.30.1**
|
|
90
90
|
|
|
91
91
|
## Building an Agent
|
|
92
92
|
|
|
@@ -15,9 +15,11 @@ from agentbyte.messages import SystemMessage, UserMessage
|
|
|
15
15
|
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
16
16
|
from agentbyte.eval.judges.validation import (
|
|
17
17
|
mean_score,
|
|
18
|
+
loads_judge_json,
|
|
18
19
|
normalize_criterion,
|
|
20
|
+
require_non_empty_defaults,
|
|
21
|
+
resolve_criteria,
|
|
19
22
|
validate_criterion_scores,
|
|
20
|
-
validate_requested_criteria,
|
|
21
23
|
)
|
|
22
24
|
from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory
|
|
23
25
|
|
|
@@ -60,12 +62,13 @@ class LLMEvalJudge(BaseEvalJudge):
|
|
|
60
62
|
answer_strategy=answer_strategy,
|
|
61
63
|
source_filter=source_filter,
|
|
62
64
|
)
|
|
65
|
+
require_non_empty_defaults(default_criteria, "default_criteria")
|
|
63
66
|
self.client = client
|
|
64
|
-
self.default_criteria =
|
|
65
|
-
"accuracy",
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
67
|
+
self.default_criteria = (
|
|
68
|
+
["accuracy", "completeness", "helpfulness"]
|
|
69
|
+
if default_criteria is None
|
|
70
|
+
else list(default_criteria)
|
|
71
|
+
)
|
|
69
72
|
self.custom_instructions = custom_instructions
|
|
70
73
|
self.use_structured_output = use_structured_output
|
|
71
74
|
|
|
@@ -87,7 +90,7 @@ class LLMEvalJudge(BaseEvalJudge):
|
|
|
87
90
|
criteria = raw
|
|
88
91
|
elif isinstance(raw, str):
|
|
89
92
|
criteria = [raw]
|
|
90
|
-
eval_criteria =
|
|
93
|
+
eval_criteria = resolve_criteria(criteria, self.default_criteria)
|
|
91
94
|
if cancellation_token and cancellation_token.is_cancelled():
|
|
92
95
|
raise asyncio.CancelledError()
|
|
93
96
|
|
|
@@ -131,6 +134,8 @@ class LLMEvalJudge(BaseEvalJudge):
|
|
|
131
134
|
)
|
|
132
135
|
except Exception as exc:
|
|
133
136
|
raise JudgeScoringError("provider_error", "Judge model call failed") from exc
|
|
137
|
+
if cancellation_token and cancellation_token.is_cancelled():
|
|
138
|
+
raise asyncio.CancelledError()
|
|
134
139
|
|
|
135
140
|
if self.use_structured_output:
|
|
136
141
|
dimensions, reasoning = self._validate_structured(
|
|
@@ -198,7 +203,7 @@ class LLMEvalJudge(BaseEvalJudge):
|
|
|
198
203
|
criteria: list[str],
|
|
199
204
|
) -> tuple[dict[str, float], dict[str, str]]:
|
|
200
205
|
try:
|
|
201
|
-
parsed = _JudgeResponse.model_validate(
|
|
206
|
+
parsed = _JudgeResponse.model_validate(loads_judge_json(raw_content))
|
|
202
207
|
except (json.JSONDecodeError, ValidationError, TypeError) as exc:
|
|
203
208
|
raise JudgeScoringError("invalid_response", "Judge returned malformed JSON") from exc
|
|
204
209
|
dimensions, _ = validate_criterion_scores(
|
|
@@ -14,6 +14,7 @@ from agentbyte.llm.base import BaseChatCompletionClient
|
|
|
14
14
|
from agentbyte.messages import SystemMessage, UserMessage
|
|
15
15
|
|
|
16
16
|
from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
17
|
+
from agentbyte.eval.judges.validation import loads_judge_json, require_non_empty_defaults
|
|
17
18
|
from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory, PairwiseResult
|
|
18
19
|
|
|
19
20
|
|
|
@@ -55,8 +56,11 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
55
56
|
answer_strategy=answer_strategy,
|
|
56
57
|
source_filter=source_filter,
|
|
57
58
|
)
|
|
59
|
+
require_non_empty_defaults(criteria, "criteria")
|
|
58
60
|
self.client = client
|
|
59
|
-
self.criteria =
|
|
61
|
+
self.criteria = (
|
|
62
|
+
["accuracy", "helpfulness", "clarity"] if criteria is None else list(criteria)
|
|
63
|
+
)
|
|
60
64
|
self.custom_instructions = custom_instructions
|
|
61
65
|
|
|
62
66
|
async def compare(
|
|
@@ -111,9 +115,11 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
111
115
|
)
|
|
112
116
|
except Exception as exc:
|
|
113
117
|
raise JudgeScoringError("provider_error", "Pairwise judge model call failed") from exc
|
|
118
|
+
if cancellation_token and cancellation_token.is_cancelled():
|
|
119
|
+
raise asyncio.CancelledError()
|
|
114
120
|
try:
|
|
115
121
|
parsed = self._parse_response(result.structured_output, result.message.content)
|
|
116
|
-
except ValidationError as exc:
|
|
122
|
+
except (ValidationError, json.JSONDecodeError, TypeError) as exc:
|
|
117
123
|
raise JudgeScoringError("invalid_response", "Malformed pairwise response") from exc
|
|
118
124
|
winner = parsed.winner.strip().lower()
|
|
119
125
|
if winner not in ("a", "b", "tie"):
|
|
@@ -148,7 +154,9 @@ class PairwiseJudge(BaseEvalJudge):
|
|
|
148
154
|
return _PairwiseResponse.model_validate(structured_output.model_dump())
|
|
149
155
|
if structured_output is not None:
|
|
150
156
|
return _PairwiseResponse.model_validate(structured_output)
|
|
151
|
-
return _PairwiseResponse.
|
|
157
|
+
return _PairwiseResponse.model_validate(
|
|
158
|
+
loads_judge_json(raw_content, duplicate_reason="invalid_response")
|
|
159
|
+
)
|
|
152
160
|
|
|
153
161
|
|
|
154
162
|
__all__ = ["PairwiseJudge"]
|
|
@@ -14,8 +14,9 @@ from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
|
|
|
14
14
|
from agentbyte.eval.judges.validation import (
|
|
15
15
|
mean_score,
|
|
16
16
|
normalize_criterion,
|
|
17
|
+
require_non_empty_defaults,
|
|
18
|
+
resolve_criteria,
|
|
17
19
|
validate_criterion_scores,
|
|
18
|
-
validate_requested_criteria,
|
|
19
20
|
)
|
|
20
21
|
from agentbyte.eval.types import EvalScore, EvalTrajectory
|
|
21
22
|
|
|
@@ -52,12 +53,13 @@ class LLMTrajectoryJudge(BaseEvalJudge):
|
|
|
52
53
|
super().__init__(name, answer_strategy="last_assistant")
|
|
53
54
|
if max_payload_chars <= 0:
|
|
54
55
|
raise ValueError("max_payload_chars must be positive")
|
|
56
|
+
require_non_empty_defaults(default_criteria, "default_criteria")
|
|
55
57
|
self.client = client
|
|
56
|
-
self.default_criteria =
|
|
57
|
-
"tool_use_correctness",
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
58
|
+
self.default_criteria = (
|
|
59
|
+
["tool_use_correctness", "coordination", "execution_efficiency"]
|
|
60
|
+
if default_criteria is None
|
|
61
|
+
else list(default_criteria)
|
|
62
|
+
)
|
|
61
63
|
self.custom_instructions = custom_instructions or ""
|
|
62
64
|
self.max_payload_chars = max_payload_chars
|
|
63
65
|
|
|
@@ -76,7 +78,7 @@ class LLMTrajectoryJudge(BaseEvalJudge):
|
|
|
76
78
|
if isinstance(raw, str)
|
|
77
79
|
else None
|
|
78
80
|
)
|
|
79
|
-
criteria =
|
|
81
|
+
criteria = resolve_criteria(criteria, self.default_criteria)
|
|
80
82
|
if cancellation_token and cancellation_token.is_cancelled():
|
|
81
83
|
raise asyncio.CancelledError()
|
|
82
84
|
trace = trajectory.trace
|
|
@@ -126,6 +128,8 @@ class LLMTrajectoryJudge(BaseEvalJudge):
|
|
|
126
128
|
)
|
|
127
129
|
except Exception as exc:
|
|
128
130
|
raise JudgeScoringError("provider_error", "Process judge model call failed") from exc
|
|
131
|
+
if cancellation_token and cancellation_token.is_cancelled():
|
|
132
|
+
raise asyncio.CancelledError()
|
|
129
133
|
response = result.structured_output
|
|
130
134
|
if not isinstance(response, ProcessJudgeResponse):
|
|
131
135
|
raise JudgeScoringError(
|
|
@@ -7,6 +7,7 @@ requested criterion exactly once with a finite 0-10 number is rejected with a
|
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
9
9
|
|
|
10
|
+
import json
|
|
10
11
|
import math
|
|
11
12
|
from collections.abc import Iterable
|
|
12
13
|
from typing import Any
|
|
@@ -36,6 +37,31 @@ def validate_requested_criteria(criteria: list[str]) -> list[str]:
|
|
|
36
37
|
return list(criteria)
|
|
37
38
|
|
|
38
39
|
|
|
40
|
+
def resolve_criteria(criteria: list[str] | None, defaults: list[str]) -> list[str]:
|
|
41
|
+
"""Use ``defaults`` only when criteria is None; an explicit empty list is rejected."""
|
|
42
|
+
return validate_requested_criteria(defaults if criteria is None else criteria)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def require_non_empty_defaults(criteria: list[str] | None, name: str) -> None:
|
|
46
|
+
"""Constructor guard: an explicit empty criteria list never means "use defaults"."""
|
|
47
|
+
if criteria is not None and not criteria:
|
|
48
|
+
raise ValueError(f"{name} must not be empty; pass None to use the defaults")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def loads_judge_json(text: str, *, duplicate_reason: str = "duplicate_criterion") -> Any:
|
|
52
|
+
"""Decode judge JSON, rejecting duplicate keys that ``json.loads`` would silently drop."""
|
|
53
|
+
|
|
54
|
+
def reject_duplicates(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
55
|
+
result: dict[str, Any] = {}
|
|
56
|
+
for key, value in pairs:
|
|
57
|
+
if key in result:
|
|
58
|
+
raise JudgeScoringError(duplicate_reason, f"Duplicate JSON key: {key!r}")
|
|
59
|
+
result[key] = value
|
|
60
|
+
return result
|
|
61
|
+
|
|
62
|
+
return json.loads(text, object_pairs_hook=reject_duplicates)
|
|
63
|
+
|
|
64
|
+
|
|
39
65
|
def validate_score_value(value: Any) -> float:
|
|
40
66
|
"""Accept only finite real numbers in [0, 10]; booleans and strings are rejected."""
|
|
41
67
|
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
@@ -87,6 +113,9 @@ def mean_score(dimensions: dict[str, float]) -> float:
|
|
|
87
113
|
__all__ = [
|
|
88
114
|
"normalize_criterion",
|
|
89
115
|
"validate_requested_criteria",
|
|
116
|
+
"resolve_criteria",
|
|
117
|
+
"require_non_empty_defaults",
|
|
118
|
+
"loads_judge_json",
|
|
90
119
|
"validate_score_value",
|
|
91
120
|
"validate_criterion_scores",
|
|
92
121
|
"mean_score",
|
|
@@ -10,6 +10,7 @@ from agentbyte.messages import Usage
|
|
|
10
10
|
|
|
11
11
|
from agentbyte.eval.base import BaseEvalTarget, JudgeScoringError
|
|
12
12
|
from agentbyte.eval.judges.pairwise import PairwiseJudge
|
|
13
|
+
from agentbyte.eval.runner import is_cancelled_run
|
|
13
14
|
from agentbyte.eval.types import EvalTask, EvalTrajectory, PairwiseResult, ScoringStatus
|
|
14
15
|
|
|
15
16
|
|
|
@@ -100,14 +101,18 @@ class PairwiseRunner:
|
|
|
100
101
|
) -> tuple[EvalTrajectory, str | None]:
|
|
101
102
|
"""Run one side, returning its trajectory and a failure reason code if it raised."""
|
|
102
103
|
try:
|
|
103
|
-
|
|
104
|
+
trajectory = await target.run(task, cancellation_token)
|
|
104
105
|
except asyncio.CancelledError:
|
|
105
106
|
if cancellation_token and cancellation_token.is_cancelled():
|
|
106
107
|
reason, error = "cancelled", "Cancelled"
|
|
107
108
|
else:
|
|
108
109
|
raise
|
|
109
110
|
except Exception as exc:
|
|
110
|
-
reason, error = "target_error",
|
|
111
|
+
reason, error = "target_error", type(exc).__name__
|
|
112
|
+
else:
|
|
113
|
+
if is_cancelled_run(trajectory, cancellation_token):
|
|
114
|
+
return trajectory, "cancelled"
|
|
115
|
+
return trajectory, None
|
|
111
116
|
trajectory = EvalTrajectory(
|
|
112
117
|
task=task, messages=[], success=False, error=error, usage=Usage()
|
|
113
118
|
)
|
|
@@ -113,12 +113,17 @@ class EvalRunner(BaseEvalRunner):
|
|
|
113
113
|
raise
|
|
114
114
|
except Exception as exc:
|
|
115
115
|
return self._build_unscored(
|
|
116
|
-
self._build_failed_trajectory(task, error=
|
|
116
|
+
self._build_failed_trajectory(task, error=type(exc).__name__),
|
|
117
117
|
status="failed",
|
|
118
118
|
reason="target_error",
|
|
119
119
|
error_type=type(exc).__name__,
|
|
120
120
|
)
|
|
121
121
|
|
|
122
|
+
# A cancelled run is not an agent failure to be measured (e.g. a 0 from
|
|
123
|
+
# a reference judge); it carries no numeric evidence.
|
|
124
|
+
if is_cancelled_run(trajectory, cancellation_token):
|
|
125
|
+
return self._build_unscored(trajectory, status="cancelled", reason="cancelled")
|
|
126
|
+
|
|
122
127
|
# resolve: call-level arg > task metadata > judge default
|
|
123
128
|
effective_criteria = criteria
|
|
124
129
|
if effective_criteria is None:
|
|
@@ -175,4 +180,13 @@ class EvalRunner(BaseEvalRunner):
|
|
|
175
180
|
)
|
|
176
181
|
|
|
177
182
|
|
|
183
|
+
def is_cancelled_run(
|
|
184
|
+
trajectory: EvalTrajectory, cancellation_token: CancellationToken | None
|
|
185
|
+
) -> bool:
|
|
186
|
+
"""Whether a returned trajectory was cut short by cancellation rather than failing."""
|
|
187
|
+
if cancellation_token is not None and cancellation_token.is_cancelled():
|
|
188
|
+
return True
|
|
189
|
+
return trajectory.metadata.get("execution_status") == "cancelled"
|
|
190
|
+
|
|
191
|
+
|
|
178
192
|
__all__ = ["EvalRunner"]
|
|
@@ -36,6 +36,7 @@ class ModelEvalTarget(BaseEvalTarget):
|
|
|
36
36
|
task,
|
|
37
37
|
start_time,
|
|
38
38
|
error="Execution cancelled before model call",
|
|
39
|
+
status="cancelled",
|
|
39
40
|
)
|
|
40
41
|
|
|
41
42
|
messages = []
|
|
@@ -69,6 +70,7 @@ class ModelEvalTarget(BaseEvalTarget):
|
|
|
69
70
|
start_time: float,
|
|
70
71
|
*,
|
|
71
72
|
error: str,
|
|
73
|
+
status: str = "failed",
|
|
72
74
|
) -> EvalTrajectory:
|
|
73
75
|
elapsed_ms = int((time.perf_counter() - start_time) * 1000)
|
|
74
76
|
return EvalTrajectory(
|
|
@@ -80,6 +82,7 @@ class ModelEvalTarget(BaseEvalTarget):
|
|
|
80
82
|
metadata={
|
|
81
83
|
"target_type": "model",
|
|
82
84
|
"target_name": self.name,
|
|
85
|
+
"execution_status": status,
|
|
83
86
|
"execution_time_ms": elapsed_ms,
|
|
84
87
|
},
|
|
85
88
|
)
|
|
@@ -113,6 +113,11 @@ class EvalTrajectory(BaseModel):
|
|
|
113
113
|
ScoringStatus = Literal["scored", "failed", "cancelled"]
|
|
114
114
|
|
|
115
115
|
|
|
116
|
+
# Markers written by pre-0055 judges on placeholder scores. Such records are
|
|
117
|
+
# never valid numeric evidence.
|
|
118
|
+
_FALLBACK_FLAGS = ("fallback", "judge_failed")
|
|
119
|
+
|
|
120
|
+
|
|
116
121
|
def _is_valid_score_value(value: float) -> bool:
|
|
117
122
|
return math.isfinite(value) and 0.0 <= value <= 10.0
|
|
118
123
|
|
|
@@ -166,6 +171,11 @@ class EvalScore(BaseModel):
|
|
|
166
171
|
raise ValueError("scored EvalScore requires finite dimensions in [0, 10]")
|
|
167
172
|
if self.failure_reason is not None:
|
|
168
173
|
raise ValueError("scored EvalScore cannot carry a failure_reason")
|
|
174
|
+
if any(self.metadata.get(flag) for flag in _FALLBACK_FLAGS):
|
|
175
|
+
raise ValueError(
|
|
176
|
+
"EvalScore marked as a judge fallback is not a measurement; "
|
|
177
|
+
"record it as scoring_status='failed'"
|
|
178
|
+
)
|
|
169
179
|
return self
|
|
170
180
|
if self.overall is not None or self.dimensions:
|
|
171
181
|
raise ValueError(
|