agentbyte 0.29.0__tar.gz → 0.30.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (321) hide show
  1. {agentbyte-0.29.0 → agentbyte-0.30.0}/CHANGELOG.md +17 -0
  2. {agentbyte-0.29.0 → agentbyte-0.30.0}/PKG-INFO +2 -2
  3. {agentbyte-0.29.0 → agentbyte-0.30.0}/README.md +1 -1
  4. agentbyte-0.30.0/src/agentbyte/__about__.py +2 -0
  5. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/__init__.py +6 -2
  6. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/base.py +20 -2
  7. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/local.py +6 -0
  8. agentbyte-0.30.0/src/agentbyte/eval/comparison.py +58 -0
  9. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/composite.py +28 -12
  10. agentbyte-0.30.0/src/agentbyte/eval/judges/llm.py +220 -0
  11. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/pairwise.py +32 -41
  12. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/trajectory.py +66 -65
  13. agentbyte-0.30.0/src/agentbyte/eval/judges/validation.py +93 -0
  14. agentbyte-0.30.0/src/agentbyte/eval/pairwise.py +170 -0
  15. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/report.py +35 -9
  16. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/runner.py +47 -36
  17. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/types.py +99 -9
  18. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/collector.py +5 -0
  19. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/__init__.py +9 -1
  20. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/base.py +70 -18
  21. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/gepa.py +11 -2
  22. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/pareto.py +7 -3
  23. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/reflective.py +2 -1
  24. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/trace.py +6 -2
  25. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/execution.py +120 -50
  26. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/models.py +6 -0
  27. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/server.py +27 -11
  28. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/__init__.py +2 -0
  29. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/agent.py +8 -3
  30. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/__init__.py +2 -0
  31. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/models.py +39 -2
  32. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/runner.py +88 -7
  33. agentbyte-0.30.0/src/agentbyte/workflow/steps/agent.py +141 -0
  34. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/step.py +9 -0
  35. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/subworkflow.py +18 -2
  36. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_tool_approval.py +59 -0
  37. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_execution_trajectories.py +17 -11
  38. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_pairwise.py +89 -25
  39. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_phase1_runner_and_targets.py +89 -4
  40. agentbyte-0.30.0/tests/eval/test_score_integrity.py +194 -0
  41. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_types_and_judges.py +162 -71
  42. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_gepa.py +32 -1
  43. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_mipro.py +1 -1
  44. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_pareto.py +1 -1
  45. agentbyte-0.30.0/tests/optim/test_score_integrity.py +140 -0
  46. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_trace.py +2 -1
  47. agentbyte-0.30.0/tests/webui/test_workflow_streaming.py +351 -0
  48. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_steps.py +5 -3
  49. agentbyte-0.29.0/src/agentbyte/__about__.py +0 -2
  50. agentbyte-0.29.0/src/agentbyte/eval/comparison.py +0 -38
  51. agentbyte-0.29.0/src/agentbyte/eval/judges/llm.py +0 -279
  52. agentbyte-0.29.0/src/agentbyte/eval/pairwise.py +0 -106
  53. agentbyte-0.29.0/src/agentbyte/workflow/steps/agent.py +0 -87
  54. {agentbyte-0.29.0 → agentbyte-0.30.0}/.gitignore +0 -0
  55. {agentbyte-0.29.0 → agentbyte-0.30.0}/LICENSE +0 -0
  56. {agentbyte-0.29.0 → agentbyte-0.30.0}/pyproject.toml +0 -0
  57. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/__init__.py +0 -0
  58. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/__init__.py +0 -0
  59. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent.py +0 -0
  60. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent_as_tool.py +0 -0
  61. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/base.py +0 -0
  62. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/embedding_agent.py +0 -0
  63. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/agents/types.py +0 -0
  64. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cancellation_token.py +0 -0
  65. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/catalog.py +0 -0
  66. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cli/__init__.py +0 -0
  67. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/cli/main.py +0 -0
  68. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/component.py +0 -0
  69. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context.py +0 -0
  70. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/__init__.py +0 -0
  71. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/base.py +0 -0
  72. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skill_tools.py +0 -0
  73. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skills.py +0 -0
  74. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/__init__.py +0 -0
  75. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/base.py +0 -0
  76. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/config.py +0 -0
  77. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/importer.py +0 -0
  78. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/json.py +0 -0
  79. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/loader.py +0 -0
  80. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish.py +0 -0
  81. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish_config.py +0 -0
  82. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publishers.py +0 -0
  83. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sources.py +0 -0
  84. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite.py +0 -0
  85. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite_db.py +0 -0
  86. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/write_config.py +0 -0
  87. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/dataset/writers.py +0 -0
  88. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/entity.py +0 -0
  89. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/__init__.py +0 -0
  90. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/decorator.py +0 -0
  91. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/keyword.py +0 -0
  92. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/process.py +0 -0
  93. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/tool.py +0 -0
  94. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/types.py +0 -0
  95. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/eval_dataset.py +0 -0
  96. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/__init__.py +0 -0
  97. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/base.py +0 -0
  98. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/reference.py +0 -0
  99. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/splitting.py +0 -0
  100. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/__init__.py +0 -0
  101. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/agent.py +0 -0
  102. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/model.py +0 -0
  103. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/multi_turn.py +0 -0
  104. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/orchestrator.py +0 -0
  105. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/runtime.py +0 -0
  106. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/workflow.py +0 -0
  107. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/__init__.py +0 -0
  108. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/execution_trace/models.py +0 -0
  109. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/__init__.py +0 -0
  110. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/_retry_observability.py +0 -0
  111. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/auth.py +0 -0
  112. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/__init__.py +0 -0
  113. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/auth.py +0 -0
  114. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/chat.py +0 -0
  115. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/embedding.py +0 -0
  116. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/settings.py +0 -0
  117. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai.py +0 -0
  118. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai_embedding.py +0 -0
  119. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/base.py +0 -0
  120. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/embeddings_base.py +0 -0
  121. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/__init__.py +0 -0
  122. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/chat.py +0 -0
  123. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/embedding.py +0 -0
  124. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/settings.py +0 -0
  125. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai_embedding.py +0 -0
  126. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/pricing.py +0 -0
  127. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/retry_policy.py +0 -0
  128. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/settings.py +0 -0
  129. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/llm/types.py +0 -0
  130. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/logger.py +0 -0
  131. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/memory/__init__.py +0 -0
  132. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/memory/base.py +0 -0
  133. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/messages.py +0 -0
  134. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/__init__.py +0 -0
  135. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/base.py +0 -0
  136. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/otel.py +0 -0
  137. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/retry.py +0 -0
  138. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/sql_usage.py +0 -0
  139. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/middleware/usage_logger.py +0 -0
  140. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/notebook.py +0 -0
  141. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/config.py +0 -0
  142. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/mipro.py +0 -0
  143. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/optim/spec.py +0 -0
  144. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/__init__.py +0 -0
  145. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/ai.py +0 -0
  146. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/base.py +0 -0
  147. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/handoff.py +0 -0
  148. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/plan.py +0 -0
  149. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/policies.py +0 -0
  150. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/round_robin.py +0 -0
  151. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/__init__.py +0 -0
  152. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/agents.py +0 -0
  153. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/clients.py +0 -0
  154. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instruction_registry.py +0 -0
  155. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/orchestrator.yaml +0 -0
  156. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/query_rewriter.yaml +0 -0
  157. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/researcher.yaml +0 -0
  158. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/reviewer.yaml +0 -0
  159. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/writer.yaml +0 -0
  160. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/orchestration.py +0 -0
  161. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/contracts-analyst/SKILL.md +0 -0
  162. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/SKILL.md +0 -0
  163. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/departments.md +0 -0
  164. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/employees.md +0 -0
  165. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/payroll.md +0 -0
  166. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/streaming.py +0 -0
  167. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/presets/workflow.py +0 -0
  168. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/session_store.py +0 -0
  169. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/__init__.py +0 -0
  170. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/base.py +0 -0
  171. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/resources.py +0 -0
  172. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/scripts.py +0 -0
  173. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/sources.py +0 -0
  174. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/skills/validation.py +0 -0
  175. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/__init__.py +0 -0
  176. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/base.py +0 -0
  177. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/cancellation.py +0 -0
  178. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/composite.py +0 -0
  179. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/consecutive_agent.py +0 -0
  180. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/external.py +0 -0
  181. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/function_call.py +0 -0
  182. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/handoff.py +0 -0
  183. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/max_message.py +0 -0
  184. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/predicate.py +0 -0
  185. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/source.py +0 -0
  186. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/text_mention.py +0 -0
  187. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/timeout.py +0 -0
  188. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/termination/token_usage.py +0 -0
  189. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/__init__.py +0 -0
  190. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/base.py +0 -0
  191. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/coding_tools.py +0 -0
  192. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/core_tools.py +0 -0
  193. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/decorator.py +0 -0
  194. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/memory_tool.py +0 -0
  195. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/tools/research_tools.py +0 -0
  196. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/types.py +0 -0
  197. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/__init__.py +0 -0
  198. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/discovery.py +0 -0
  199. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/registry.py +0 -0
  200. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/session_store.py +0 -0
  201. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/sessions.py +0 -0
  202. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-BF3DwXaF.js +0 -0
  203. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-ar5tOeqt.css +0 -0
  204. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/index.html +0 -0
  205. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/vite.svg +0 -0
  206. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/_structure_hash.py +0 -0
  207. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/checkpoint.py +0 -0
  208. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/workflow.py +0 -0
  209. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/defaults.py +0 -0
  210. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/loader.py +0 -0
  211. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema.py +0 -0
  212. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema_utils.py +0 -0
  213. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/__init__.py +0 -0
  214. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/echo.py +0 -0
  215. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/function.py +0 -0
  216. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/http.py +0 -0
  217. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/transform.py +0 -0
  218. {agentbyte-0.29.0 → agentbyte-0.30.0}/src/agentbyte/workflow/visualizer.py +0 -0
  219. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_as_tool.py +0 -0
  220. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_basic.py +0 -0
  221. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_context_providers.py +0 -0
  222. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_error_response.py +0 -0
  223. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_event_types.py +0 -0
  224. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_memory_integration.py +0 -0
  225. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_middleware_integration.py +0 -0
  226. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_response_accessors.py +0 -0
  227. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_retry_middleware.py +0 -0
  228. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_agent_stream_events.py +0 -0
  229. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/agents/test_embedding_agent.py +0 -0
  230. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/cli/test_registry_check.py +0 -0
  231. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/__init__.py +0 -0
  232. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/test_skill_tools.py +0 -0
  233. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/context_providers/test_skills_provider.py +0 -0
  234. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_loader.py +0 -0
  235. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_multi_table.py +0 -0
  236. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_publish.py +0 -0
  237. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/dataset/test_sqlite_db.py +0 -0
  238. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_eval_dataset.py +0 -0
  239. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_multi_turn.py +0 -0
  240. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_phase2_checks_and_reports.py +0 -0
  241. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/eval/test_splitting.py +0 -0
  242. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_azure_client.py +0 -0
  243. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_azure_embedding_client.py +0 -0
  244. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_llm_types.py +0 -0
  245. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_openai_client.py +0 -0
  246. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_openai_embedding_client.py +0 -0
  247. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_pricing.py +0 -0
  248. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retry_observability.py +0 -0
  249. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retry_policy_api.py +0 -0
  250. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/llm/test_retryable_error_substrings.py +0 -0
  251. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/memory/test_memory.py +0 -0
  252. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_deduplicate_tool_result.py +0 -0
  253. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_middleware_chain.py +0 -0
  254. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_otel.py +0 -0
  255. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_retry_middleware.py +0 -0
  256. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_sql_usage.py +0 -0
  257. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/middleware/test_usage_logger.py +0 -0
  258. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/__init__.py +0 -0
  259. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_base.py +0 -0
  260. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_base_integration.py +0 -0
  261. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_config.py +0 -0
  262. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_reflective.py +0 -0
  263. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/optim/test_spec.py +0 -0
  264. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_ai_orchestrator.py +0 -0
  265. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_base_orchestrator.py +0 -0
  266. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_handoff_orchestrator.py +0 -0
  267. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_orchestrator_finalization.py +0 -0
  268. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_plan_orchestrator.py +0 -0
  269. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/orchestration/test_round_robin.py +0 -0
  270. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_agents.py +0 -0
  271. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_clients.py +0 -0
  272. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_instruction_registry.py +0 -0
  273. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_orchestration.py +0 -0
  274. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_streaming.py +0 -0
  275. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/presets/test_workflow.py +0 -0
  276. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/__init__.py +0 -0
  277. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_base.py +0 -0
  278. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_resources.py +0 -0
  279. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_scripts.py +0 -0
  280. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/skills/test_sources.py +0 -0
  281. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_base.py +0 -0
  282. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_cancellation.py +0 -0
  283. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_composite.py +0 -0
  284. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_consecutive_agent.py +0 -0
  285. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_external.py +0 -0
  286. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_function_call.py +0 -0
  287. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_handoff.py +0 -0
  288. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_max_message.py +0 -0
  289. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_predicate.py +0 -0
  290. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_source.py +0 -0
  291. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_text_mention.py +0 -0
  292. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_timeout.py +0 -0
  293. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/termination/test_token_usage.py +0 -0
  294. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_cancellation_token.py +0 -0
  295. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_context.py +0 -0
  296. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_logger.py +0 -0
  297. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_messages.py +0 -0
  298. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_package_api.py +0 -0
  299. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_session_store.py +0 -0
  300. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_types.py +0 -0
  301. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/test_vanilla_chunker.py +0 -0
  302. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_coding_tools.py +0 -0
  303. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_memory_tool.py +0 -0
  304. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_research_tools.py +0 -0
  305. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/tools/test_tools.py +0 -0
  306. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/__init__.py +0 -0
  307. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/helpers.py +0 -0
  308. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_execution.py +0 -0
  309. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_package_api.py +0 -0
  310. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_registry.py +0 -0
  311. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_server.py +0 -0
  312. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/webui/test_sessions.py +0 -0
  313. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_agent_step_imports.py +0 -0
  314. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_checkpoint.py +0 -0
  315. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_subworkflow_step.py +0 -0
  316. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_agent.py +0 -0
  317. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_class.py +0 -0
  318. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_models.py +0 -0
  319. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_runner.py +0 -0
  320. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_schema.py +0 -0
  321. {agentbyte-0.29.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_visualizer.py +0 -0
@@ -4,6 +4,23 @@ All notable changes to Agentbyte are documented in this file.
4
4
 
5
5
  The format follows Keep a Changelog principles and semantic versioning.
6
6
 
7
+ ## [0.30.0] - 2026-09-30
8
+
9
+ ### Added
10
+
11
+ - Workflow streaming parity (spec 0057): `WorkflowRunner.run_stream(..., stream_tokens=...)` forwards live child agent events as `StepEvent` envelopes (workflow, execution, invocation, step, attempt and sequence attribution) through a bounded queue, including across `workflow.as_agent()` and `SubWorkflowStep` boundaries. `WorkflowAgent.run_stream` honors `stream_tokens`. Closing the stream cancels and awaits outstanding steps.
12
+ - `AgentStep` streams its agent and turns tool approval into workflow suspension; resuming validates the exact pending decisions and never reruns completed tools.
13
+ - Checkpoint-backed serving: the WebUI/FastAPI run and SSE endpoints accept `workflow_responses` and `workflow_checkpoint_id` to resume a suspended workflow in the same session. `WorkflowExecution.checkpoint_id` exposes the suspended checkpoint, and `WorkflowRunner.validate_resume_responses` / `BaseStep.validate_resume_response` validate responses without consuming it. Stale, unknown or wrong-owner checkpoints and concurrent runs of the same workflow or session are rejected with HTTP 409.
14
+ - Judge score integrity (spec 0055): `EvalScore.scoring_status` (`scored` | `failed` | `cancelled`) and `failure_reason`, `JudgeScoringError` with safe reason codes, `mean_scored()`, `OptimizationEvidenceError`, and `PairwiseResult.comparison_status`.
15
+
16
+ ### Changed
17
+
18
+ - **Breaking:** Judges never invent scores. `LLMEvalJudge` and `LLMTrajectoryJudge` require every requested criterion exactly once (matched case- and whitespace-insensitively, reported with the requested spelling) with a finite 0–10 value, and raise `JudgeScoringError` otherwise instead of filling missing criteria with 5.0, clamping, or returning a neutral 5.0 after a failure. `overall` is always the mean of the validated dimensions; a model-supplied overall is ignored. Cancellation raises `asyncio.CancelledError`.
19
+ - **Breaking:** `EvalScore.overall` is `float | None`. `EvalRunner` records judge failures, target exceptions and cancellation as `failed`/`cancelled` outcomes with `overall=None`, empty dimensions and the trajectory kept, instead of a fabricated 0.0. Failure metadata keeps the exception type, never its text. A scored `EvalScore` must have a finite `overall` and dimensions in 0–10.
20
+ - **Breaking:** `PairwiseJudge.compare()` raises instead of returning a tie on failure; unknown winners and out-of-range margins are rejected, not coerced. `PairwiseRunner` records failed comparisons with `winner=None`. `compare_pairwise()` adds `compared`/`failed`/`cancelled` counts and computes win rates over valid comparisons only (`None` when nothing was compared).
21
+ - **Breaking:** `EvalReport.avg_score` and `compare_configurations()` averages cover scored outcomes only and are `None` when nothing was scored; an item with a failed or cancelled score fails the suite. `CompositeJudge` fails when any sub-judge fails rather than renormalizing the remaining weights.
22
+ - **Breaking:** Optimizers only rank candidates whose every task was scored. `Candidate.avg` is `None` for ineligible candidates, the minibatch gate rejects incomplete batches, a seed with an unscored task raises `OptimizationEvidenceError`, and the GEPA adapter raises rather than passing a number for an unscored example. Optimization trace `eval` events report `avg` over scored tasks (`None` if none) plus `n_scored`.
23
+
7
24
  ## [0.29.0] - 2026-09-30
8
25
 
9
26
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: agentbyte
3
- Version: 0.29.0
3
+ Version: 0.30.0
4
4
  Summary: A toolkit for designing multiagent systems
5
5
  Author-email: MrDataPsycho <mr.data.psycho@gmail.com>
6
6
  License-Expression: LicenseRef-Proprietary
@@ -86,7 +86,7 @@ Description-Content-Type: text/markdown
86
86
 
87
87
  Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
88
88
 
89
- Current release: **0.29.0**
89
+ Current release: **0.30.0**
90
90
 
91
91
  ## Building an Agent
92
92
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
4
4
 
5
- Current release: **0.29.0**
5
+ Current release: **0.30.0**
6
6
 
7
7
  ## Building an Agent
8
8
 
@@ -0,0 +1,2 @@
1
+ __version__ = "0.30.0"
2
+ VERSION = __version__
@@ -1,6 +1,6 @@
1
1
  """Evaluation framework for AgentByte."""
2
2
 
3
- from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget
3
+ from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget, JudgeScoringError
4
4
  from .checks.process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
5
5
  from .judges.trajectory import LLMTrajectoryJudge
6
6
  from .targets.workflow import WorkflowEvalTarget
@@ -19,7 +19,7 @@ from .checks import (
19
19
  tool_calls_present,
20
20
  tool_calls_present_check,
21
21
  )
22
- from .comparison import compare_configurations
22
+ from .comparison import compare_configurations, mean_scored
23
23
  from .eval_dataset import (
24
24
  EvalDatasetRecord,
25
25
  save_eval_dataset,
@@ -52,6 +52,7 @@ from .types import (
52
52
  EvalTask,
53
53
  EvalTrajectory,
54
54
  ExpectedToolCall,
55
+ ScoringStatus,
55
56
  MultiTurnEvalTask,
56
57
  WorkflowEvalTask,
57
58
  PairwiseResult,
@@ -70,6 +71,8 @@ __all__ = [
70
71
  "BaseEvalTarget",
71
72
  "BaseEvalJudge",
72
73
  "BaseEvalRunner",
74
+ "JudgeScoringError",
75
+ "ScoringStatus",
73
76
  "ExactMatchJudge",
74
77
  "ContainsJudge",
75
78
  "FuzzyMatchJudge",
@@ -88,6 +91,7 @@ __all__ = [
88
91
  "EvalSplitConfig",
89
92
  "EvalSplitStrategy",
90
93
  "compare_configurations",
94
+ "mean_scored",
91
95
  "EvalDatasetRecord",
92
96
  "save_eval_dataset",
93
97
  "tasks_from_eval_dataset",
@@ -17,6 +17,19 @@ _VALID_ANSWER_STRATEGIES: tuple[AnswerStrategy, ...] = (
17
17
  )
18
18
 
19
19
 
20
+ class JudgeScoringError(Exception):
21
+ """A judge could not produce a valid judgment.
22
+
23
+ ``reason`` is a safe, stable code (for example ``provider_error`` or
24
+ ``missing_criterion``) suitable for persistence; the message may carry more
25
+ detail and is not persisted by default.
26
+ """
27
+
28
+ def __init__(self, reason: str, message: str | None = None) -> None:
29
+ super().__init__(message or reason)
30
+ self.reason = reason
31
+
32
+
20
33
  class BaseEvalTarget(ABC):
21
34
  """Anything that can execute an evaluation task into a trajectory."""
22
35
 
@@ -96,7 +109,12 @@ class BaseEvalJudge(ABC):
96
109
  criteria: list[str] | None = None,
97
110
  cancellation_token: CancellationToken | None = None,
98
111
  ) -> EvalScore:
99
- """Score an evaluation trajectory."""
112
+ """Score an evaluation trajectory.
113
+
114
+ Returns a scored ``EvalScore``. Raises ``JudgeScoringError`` when no
115
+ valid judgment is available and ``asyncio.CancelledError`` when
116
+ cancelled; judges never substitute a placeholder score.
117
+ """
100
118
 
101
119
 
102
120
  class BaseEvalRunner(ABC):
@@ -116,4 +134,4 @@ class BaseEvalRunner(ABC):
116
134
  """Evaluate a target on multiple tasks."""
117
135
 
118
136
 
119
- __all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner"]
137
+ __all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner", "JudgeScoringError"]
@@ -93,6 +93,12 @@ def threshold_gate(
93
93
 
94
94
  async def _check(trajectory: EvalTrajectory) -> CheckResult:
95
95
  score = await judge.score(trajectory)
96
+ if not score.is_scored or score.overall is None:
97
+ return CheckResult(
98
+ passed=False,
99
+ reason=f"no valid score ({score.scoring_status}: {score.failure_reason})",
100
+ check_name=f"threshold_gate[{judge.name}]",
101
+ )
96
102
  passed = score.overall >= threshold
97
103
  return CheckResult(
98
104
  passed=passed,
@@ -0,0 +1,58 @@
1
+ """Helpers for comparing evaluated configurations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from agentbyte.eval.types import EvalScore
6
+
7
+
8
+ def compare_configurations(
9
+ scores_a: list[EvalScore],
10
+ scores_b: list[EvalScore],
11
+ label_a: str = "baseline",
12
+ label_b: str = "candidate",
13
+ ) -> dict[str, float | str | int | None]:
14
+ """Compare average score between two evaluated configurations.
15
+
16
+ Averages cover scored outcomes only; failed and cancelled outcomes are
17
+ counted per side and never enter a mean. ``avg_*`` is None when a side has
18
+ no scored outcome, and ``improvement_pct`` is None unless both sides have a
19
+ nonzero baseline to compare against.
20
+ """
21
+ avg_a = mean_scored(scores_a)
22
+ avg_b = mean_scored(scores_b)
23
+
24
+ improvement_pct: float | None = None
25
+ if avg_a is not None and avg_b is not None:
26
+ if avg_a == 0:
27
+ improvement_pct = 0.0 if avg_b == 0 else 100.0
28
+ else:
29
+ improvement_pct = ((avg_b - avg_a) / avg_a) * 100.0
30
+
31
+ return {
32
+ "label_a": label_a,
33
+ "label_b": label_b,
34
+ "avg_a": avg_a,
35
+ "avg_b": avg_b,
36
+ "scored_a": _count(scores_a, "scored"),
37
+ "scored_b": _count(scores_b, "scored"),
38
+ "failed_a": _count(scores_a, "failed"),
39
+ "failed_b": _count(scores_b, "failed"),
40
+ "cancelled_a": _count(scores_a, "cancelled"),
41
+ "cancelled_b": _count(scores_b, "cancelled"),
42
+ "improvement_pct": improvement_pct,
43
+ }
44
+
45
+
46
+ def mean_scored(scores: list[EvalScore]) -> float | None:
47
+ """Mean ``overall`` across scored outcomes, or None when none were scored."""
48
+ values = [score.overall for score in scores if score.is_scored and score.overall is not None]
49
+ if not values:
50
+ return None
51
+ return sum(values) / len(values)
52
+
53
+
54
+ def _count(scores: list[EvalScore], status: str) -> int:
55
+ return sum(1 for score in scores if score.scoring_status == status)
56
+
57
+
58
+ __all__ = ["compare_configurations", "mean_scored"]
@@ -7,7 +7,7 @@ from collections.abc import Sequence
7
7
 
8
8
  from agentbyte.cancellation_token import CancellationToken
9
9
 
10
- from agentbyte.eval.base import BaseEvalJudge
10
+ from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
11
11
  from agentbyte.eval.types import EvalScore, EvalTrajectory
12
12
 
13
13
 
@@ -45,12 +45,30 @@ class CompositeJudge(BaseEvalJudge):
45
45
  criteria: list[str] | None = None,
46
46
  cancellation_token: CancellationToken | None = None,
47
47
  ) -> EvalScore:
48
- results = await asyncio.gather(
49
- *[
50
- judge.score(trajectory, criteria, cancellation_token)
51
- for judge, _ in self.judges
52
- ]
53
- )
48
+ """Combine sub-judge scores; any sub-judge failure fails the composite.
49
+
50
+ Every sub-judge is required. A failed or cancelled component is never
51
+ dropped and the remaining weights are never renormalized into success.
52
+ """
53
+ try:
54
+ async with asyncio.TaskGroup() as group:
55
+ tasks = [
56
+ group.create_task(judge.score(trajectory, criteria, cancellation_token))
57
+ for judge, _ in self.judges
58
+ ]
59
+ except* JudgeScoringError as failures:
60
+ first = failures.exceptions[0]
61
+ raise JudgeScoringError(
62
+ "component_failed", f"Sub-judge failed: {first.reason}"
63
+ ) from first
64
+ results = [task.result() for task in tasks]
65
+ overalls: list[float] = []
66
+ for (judge, _), result in zip(self.judges, results):
67
+ if not result.is_scored or result.overall is None:
68
+ raise JudgeScoringError(
69
+ "component_failed", f"Sub-judge {judge.name!r} returned no valid score"
70
+ )
71
+ overalls.append(result.overall)
54
72
 
55
73
  dimensions: dict[str, float] = {}
56
74
  dimension_weight_totals: dict[str, float] = {}
@@ -58,11 +76,9 @@ class CompositeJudge(BaseEvalJudge):
58
76
  metadata_sub_judges: list[dict[str, float | str]] = []
59
77
  weighted_overall = 0.0
60
78
 
61
- for (judge, weight), result in zip(self.judges, results):
62
- metadata_sub_judges.append(
63
- {"name": judge.name, "weight": weight, "score": result.overall}
64
- )
65
- weighted_overall += result.overall * weight
79
+ for (judge, weight), result, overall in zip(self.judges, results, overalls):
80
+ metadata_sub_judges.append({"name": judge.name, "weight": weight, "score": overall})
81
+ weighted_overall += overall * weight
66
82
  for dimension, score in result.dimensions.items():
67
83
  dimensions[dimension] = dimensions.get(dimension, 0.0) + (score * weight)
68
84
  dimension_weight_totals[dimension] = (
@@ -0,0 +1,220 @@
1
+ """LLM-powered evaluation judge."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ from typing import Any
8
+
9
+ from pydantic import BaseModel, ConfigDict, Field, StrictFloat, ValidationError
10
+
11
+ from agentbyte.cancellation_token import CancellationToken
12
+ from agentbyte.llm.base import BaseChatCompletionClient
13
+ from agentbyte.messages import SystemMessage, UserMessage
14
+
15
+ from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
16
+ from agentbyte.eval.judges.validation import (
17
+ mean_score,
18
+ normalize_criterion,
19
+ validate_criterion_scores,
20
+ validate_requested_criteria,
21
+ )
22
+ from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory
23
+
24
+
25
+ class _JudgeCriterionScore(BaseModel):
26
+ criterion: str
27
+ score: StrictFloat = Field(ge=0.0, le=10.0)
28
+ reasoning: str
29
+
30
+
31
+ class _JudgeStructuredResponse(BaseModel):
32
+ scores: list[_JudgeCriterionScore]
33
+
34
+
35
+ # Kept for the use_structured_output=False path. Values stay untyped so the
36
+ # shared validator (not Pydantic coercion) decides what counts as a score.
37
+ class _JudgeResponse(BaseModel):
38
+ model_config = ConfigDict(extra="allow")
39
+
40
+ dimensions: dict[str, Any] = Field(default_factory=dict)
41
+ reasoning: dict[str, Any] = Field(default_factory=dict)
42
+
43
+
44
+ class LLMEvalJudge(BaseEvalJudge):
45
+ """Use an LLM to judge a trajectory on one or more criteria."""
46
+
47
+ def __init__(
48
+ self,
49
+ client: BaseChatCompletionClient,
50
+ *,
51
+ name: str | None = None,
52
+ answer_strategy: AnswerStrategy = "last_non_empty",
53
+ source_filter: str | None = None,
54
+ default_criteria: list[str] | None = None,
55
+ custom_instructions: str | None = None,
56
+ use_structured_output: bool = True,
57
+ ) -> None:
58
+ super().__init__(
59
+ name=name or f"LLM-{client.model}",
60
+ answer_strategy=answer_strategy,
61
+ source_filter=source_filter,
62
+ )
63
+ self.client = client
64
+ self.default_criteria = default_criteria or [
65
+ "accuracy",
66
+ "completeness",
67
+ "helpfulness",
68
+ ]
69
+ self.custom_instructions = custom_instructions
70
+ self.use_structured_output = use_structured_output
71
+
72
+ async def score(
73
+ self,
74
+ trajectory: EvalTrajectory,
75
+ criteria: list[str] | None = None,
76
+ cancellation_token: CancellationToken | None = None,
77
+ ) -> EvalScore:
78
+ """Score every requested criterion or raise ``JudgeScoringError``.
79
+
80
+ ``overall`` is the mean of the validated dimensions. Cancellation raises
81
+ ``asyncio.CancelledError``; no placeholder score is ever returned.
82
+ """
83
+ # resolve: call-level arg > task metadata > judge default
84
+ if criteria is None:
85
+ raw = trajectory.task.metadata.get("_criteria")
86
+ if isinstance(raw, list):
87
+ criteria = raw
88
+ elif isinstance(raw, str):
89
+ criteria = [raw]
90
+ eval_criteria = validate_requested_criteria(criteria or self.default_criteria)
91
+ if cancellation_token and cancellation_token.is_cancelled():
92
+ raise asyncio.CancelledError()
93
+
94
+ answer = self.extract_answer(trajectory)
95
+ user_payload = {
96
+ "task_name": trajectory.task.name,
97
+ "task_input": trajectory.task.input,
98
+ "expected_output": trajectory.task.expected_output,
99
+ "actual_output": answer,
100
+ "success": trajectory.success,
101
+ "error": trajectory.error,
102
+ "criteria": eval_criteria,
103
+ "usage": trajectory.usage.model_dump(mode="json")
104
+ if trajectory.usage is not None
105
+ else None,
106
+ "metadata": trajectory.metadata,
107
+ }
108
+ user_message = UserMessage(
109
+ content=json.dumps(user_payload, ensure_ascii=True, indent=2),
110
+ source="user",
111
+ )
112
+
113
+ try:
114
+ if self.use_structured_output:
115
+ result = await self.client.create(
116
+ messages=[
117
+ SystemMessage(
118
+ content=self._build_structured_prompt(eval_criteria),
119
+ source="system",
120
+ ),
121
+ user_message,
122
+ ],
123
+ output_format=_JudgeStructuredResponse,
124
+ )
125
+ else:
126
+ result = await self.client.create(
127
+ messages=[
128
+ SystemMessage(content=self._build_raw_prompt(), source="system"),
129
+ user_message,
130
+ ],
131
+ )
132
+ except Exception as exc:
133
+ raise JudgeScoringError("provider_error", "Judge model call failed") from exc
134
+
135
+ if self.use_structured_output:
136
+ dimensions, reasoning = self._validate_structured(
137
+ result.structured_output, eval_criteria
138
+ )
139
+ else:
140
+ dimensions, reasoning = self._validate_raw(result.message.content, eval_criteria)
141
+
142
+ return EvalScore(
143
+ overall=mean_score(dimensions),
144
+ dimensions=dimensions,
145
+ reasoning=reasoning,
146
+ trajectory=trajectory,
147
+ metadata={
148
+ "judge": self.name,
149
+ "model": result.model,
150
+ "finish_reason": result.finish_reason,
151
+ },
152
+ )
153
+
154
+ def _build_structured_prompt(self, criteria: list[str]) -> str:
155
+ criteria_list = ", ".join(f'"{c}"' for c in criteria)
156
+ prompt = (
157
+ "You are an evaluation judge. "
158
+ f"Score the following criteria: [{criteria_list}]. "
159
+ "Return exactly one scores[] entry per criterion, using the names as given. "
160
+ "Score each 0–10."
161
+ )
162
+ if self.custom_instructions:
163
+ prompt += f"\n\nAdditional instructions:\n{self.custom_instructions.strip()}"
164
+ return prompt
165
+
166
+ def _build_raw_prompt(self) -> str:
167
+ prompt = (
168
+ "You are an evaluation judge. "
169
+ "Return ONLY a strict JSON object with exactly these keys:\n"
170
+ ' "dimensions": {"<criterion>": <number 0-10>, ...},\n'
171
+ ' "reasoning": {"<criterion>": "<explanation string>", ...}\n'
172
+ "Score every requested criterion exactly once. "
173
+ "The `reasoning` value MUST be an object (dict) — one key per criterion, "
174
+ "never a plain string. No extra keys, no markdown, no prose outside the JSON."
175
+ )
176
+ if self.custom_instructions:
177
+ prompt += f"\n\nAdditional instructions:\n{self.custom_instructions.strip()}"
178
+ return prompt
179
+
180
+ @staticmethod
181
+ def _validate_structured(
182
+ structured: Any,
183
+ criteria: list[str],
184
+ ) -> tuple[dict[str, float], dict[str, str]]:
185
+ if not isinstance(structured, _JudgeStructuredResponse):
186
+ raise JudgeScoringError(
187
+ "invalid_response",
188
+ f"Expected _JudgeStructuredResponse, got {type(structured).__name__}",
189
+ )
190
+ return validate_criterion_scores(
191
+ criteria,
192
+ ((item.criterion, item.score, item.reasoning) for item in structured.scores),
193
+ )
194
+
195
+ @staticmethod
196
+ def _validate_raw(
197
+ raw_content: str,
198
+ criteria: list[str],
199
+ ) -> tuple[dict[str, float], dict[str, str]]:
200
+ try:
201
+ parsed = _JudgeResponse.model_validate(json.loads(raw_content))
202
+ except (json.JSONDecodeError, ValidationError, TypeError) as exc:
203
+ raise JudgeScoringError("invalid_response", "Judge returned malformed JSON") from exc
204
+ dimensions, _ = validate_criterion_scores(
205
+ criteria,
206
+ ((name, value, None) for name, value in parsed.dimensions.items()),
207
+ )
208
+ reasoning_lookup = {
209
+ normalize_criterion(str(name)): text
210
+ for name, text in parsed.reasoning.items()
211
+ if isinstance(text, str)
212
+ }
213
+ reasoning = {
214
+ criterion: reasoning_lookup.get(normalize_criterion(criterion), "")
215
+ for criterion in criteria
216
+ }
217
+ return dimensions, reasoning
218
+
219
+
220
+ __all__ = ["LLMEvalJudge"]
@@ -2,25 +2,27 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import asyncio
5
6
  import json
7
+ import math
6
8
  from typing import Any
7
9
 
8
- from pydantic import BaseModel, ConfigDict, Field
10
+ from pydantic import BaseModel, ConfigDict, Field, StrictFloat, ValidationError
9
11
 
10
12
  from agentbyte.cancellation_token import CancellationToken
11
13
  from agentbyte.llm.base import BaseChatCompletionClient
12
14
  from agentbyte.messages import SystemMessage, UserMessage
13
15
 
14
- from agentbyte.eval.base import BaseEvalJudge
16
+ from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
15
17
  from agentbyte.eval.types import AnswerStrategy, EvalScore, EvalTrajectory, PairwiseResult
16
18
 
17
19
 
18
20
  class _PairwiseResponse(BaseModel):
19
21
  model_config = ConfigDict(extra="allow")
20
22
 
21
- winner: str = Field(default="tie")
22
- margin: float = Field(default=0.0)
23
- reasoning: str = Field(default="No reasoning provided")
23
+ winner: str
24
+ margin: StrictFloat
25
+ reasoning: str = Field(default="")
24
26
 
25
27
 
26
28
  class PairwiseJudge(BaseEvalJudge):
@@ -63,13 +65,14 @@ class PairwiseJudge(BaseEvalJudge):
63
65
  trajectory_b: EvalTrajectory,
64
66
  cancellation_token: CancellationToken | None = None,
65
67
  ) -> PairwiseResult:
66
- """Compare two trajectories and return a pairwise decision.
68
+ """Compare two trajectories and return a measured pairwise decision.
67
69
 
68
- Returns a tie with ``margin=0.0`` on cancellation or LLM parse failure.
69
- The failure reason is preserved in ``metadata["fallback_reason"]``.
70
+ Raises ``JudgeScoringError`` when no valid decision is available and
71
+ ``asyncio.CancelledError`` when cancelled. A failure is never reported
72
+ as a tie.
70
73
  """
71
74
  if cancellation_token and cancellation_token.is_cancelled():
72
- return self._tie(trajectory_a, trajectory_b, fallback_reason="Cancelled")
75
+ raise asyncio.CancelledError()
73
76
 
74
77
  answer_a = self.extract_answer(trajectory_a)
75
78
  answer_b = self.extract_answer(trajectory_b)
@@ -106,22 +109,27 @@ class PairwiseJudge(BaseEvalJudge):
106
109
  ],
107
110
  output_format=_PairwiseResponse,
108
111
  )
109
- parsed = self._parse_response(result.structured_output, result.message.content)
110
- winner = parsed.winner if parsed.winner in ("a", "b", "tie") else "tie"
111
- margin = max(0.0, min(1.0, float(parsed.margin)))
112
- if winner == "tie":
113
- margin = 0.0
114
- return PairwiseResult(
115
- task=trajectory_a.task,
116
- winner=winner, # type: ignore[arg-type]
117
- margin=margin,
118
- reasoning=parsed.reasoning,
119
- trajectory_a=trajectory_a,
120
- trajectory_b=trajectory_b,
121
- metadata={"judge": self.name, "model": result.model},
122
- )
123
112
  except Exception as exc:
124
- return self._tie(trajectory_a, trajectory_b, fallback_reason=str(exc))
113
+ raise JudgeScoringError("provider_error", "Pairwise judge model call failed") from exc
114
+ try:
115
+ parsed = self._parse_response(result.structured_output, result.message.content)
116
+ except ValidationError as exc:
117
+ raise JudgeScoringError("invalid_response", "Malformed pairwise response") from exc
118
+ winner = parsed.winner.strip().lower()
119
+ if winner not in ("a", "b", "tie"):
120
+ raise JudgeScoringError("invalid_response", f"Unknown winner: {parsed.winner!r}")
121
+ margin = float(parsed.margin)
122
+ if not math.isfinite(margin) or not 0.0 <= margin <= 1.0:
123
+ raise JudgeScoringError("invalid_score", f"Margin outside 0-1: {parsed.margin!r}")
124
+ return PairwiseResult(
125
+ task=trajectory_a.task,
126
+ winner=winner, # type: ignore[arg-type]
127
+ margin=0.0 if winner == "tie" else margin,
128
+ reasoning=parsed.reasoning,
129
+ trajectory_a=trajectory_a,
130
+ trajectory_b=trajectory_b,
131
+ metadata={"judge": self.name, "model": result.model},
132
+ )
125
133
 
126
134
  async def score(self, *args: Any, **kwargs: Any) -> EvalScore:
127
135
  """Not supported on PairwiseJudge — use compare() instead."""
@@ -142,22 +150,5 @@ class PairwiseJudge(BaseEvalJudge):
142
150
  return _PairwiseResponse.model_validate(structured_output)
143
151
  return _PairwiseResponse.model_validate_json(raw_content)
144
152
 
145
- def _tie(
146
- self,
147
- trajectory_a: EvalTrajectory,
148
- trajectory_b: EvalTrajectory,
149
- *,
150
- fallback_reason: str,
151
- ) -> PairwiseResult:
152
- return PairwiseResult(
153
- task=trajectory_a.task,
154
- winner="tie",
155
- margin=0.0,
156
- reasoning=fallback_reason,
157
- trajectory_a=trajectory_a,
158
- trajectory_b=trajectory_b,
159
- metadata={"judge": self.name, "fallback": True, "fallback_reason": fallback_reason},
160
- )
161
-
162
153
 
163
154
  __all__ = ["PairwiseJudge"]