agentbyte 0.28.0__tar.gz → 0.30.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (324) hide show
  1. {agentbyte-0.28.0 → agentbyte-0.30.0}/CHANGELOG.md +39 -0
  2. {agentbyte-0.28.0 → agentbyte-0.30.0}/PKG-INFO +2 -2
  3. {agentbyte-0.28.0 → agentbyte-0.30.0}/README.md +1 -1
  4. agentbyte-0.30.0/src/agentbyte/__about__.py +2 -0
  5. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent.py +20 -3
  6. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/base.py +33 -0
  7. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/__init__.py +12 -2
  8. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/base.py +20 -2
  9. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/__init__.py +2 -0
  10. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/local.py +6 -0
  11. agentbyte-0.30.0/src/agentbyte/eval/checks/process.py +123 -0
  12. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/tool.py +2 -2
  13. agentbyte-0.30.0/src/agentbyte/eval/comparison.py +58 -0
  14. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/__init__.py +2 -0
  15. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/composite.py +28 -12
  16. agentbyte-0.30.0/src/agentbyte/eval/judges/llm.py +220 -0
  17. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/pairwise.py +32 -41
  18. agentbyte-0.30.0/src/agentbyte/eval/judges/trajectory.py +159 -0
  19. agentbyte-0.30.0/src/agentbyte/eval/judges/validation.py +93 -0
  20. agentbyte-0.30.0/src/agentbyte/eval/pairwise.py +170 -0
  21. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/report.py +35 -9
  22. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/runner.py +47 -36
  23. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/__init__.py +8 -1
  24. agentbyte-0.30.0/src/agentbyte/eval/targets/agent.py +102 -0
  25. agentbyte-0.30.0/src/agentbyte/eval/targets/multi_turn.py +12 -0
  26. agentbyte-0.30.0/src/agentbyte/eval/targets/orchestrator.py +90 -0
  27. agentbyte-0.30.0/src/agentbyte/eval/targets/runtime.py +53 -0
  28. agentbyte-0.30.0/src/agentbyte/eval/targets/workflow.py +104 -0
  29. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/types.py +120 -10
  30. agentbyte-0.30.0/src/agentbyte/execution_trace/__init__.py +22 -0
  31. agentbyte-0.30.0/src/agentbyte/execution_trace/collector.py +362 -0
  32. agentbyte-0.30.0/src/agentbyte/execution_trace/models.py +91 -0
  33. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/__init__.py +9 -1
  34. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/base.py +70 -18
  35. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/gepa.py +11 -2
  36. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/pareto.py +7 -3
  37. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/reflective.py +2 -1
  38. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/trace.py +6 -2
  39. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/base.py +94 -5
  40. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/handoff.py +4 -1
  41. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/policies.py +8 -0
  42. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/workflow.py +11 -11
  43. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/types.py +5 -0
  44. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/execution.py +120 -50
  45. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/models.py +6 -0
  46. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/server.py +27 -11
  47. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/__init__.py +6 -0
  48. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/agent.py +10 -3
  49. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/__init__.py +2 -0
  50. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/models.py +39 -2
  51. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/runner.py +90 -7
  52. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/__init__.py +3 -1
  53. agentbyte-0.30.0/src/agentbyte/workflow/steps/agent.py +141 -0
  54. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/step.py +59 -24
  55. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/subworkflow.py +18 -2
  56. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_basic.py +135 -0
  57. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_context_providers.py +84 -0
  58. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_tool_approval.py +59 -0
  59. agentbyte-0.30.0/tests/eval/test_execution_trajectories.py +638 -0
  60. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_pairwise.py +89 -25
  61. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_phase1_runner_and_targets.py +89 -4
  62. agentbyte-0.30.0/tests/eval/test_score_integrity.py +194 -0
  63. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_types_and_judges.py +162 -71
  64. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_gepa.py +32 -1
  65. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_mipro.py +1 -1
  66. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_pareto.py +1 -1
  67. agentbyte-0.30.0/tests/optim/test_score_integrity.py +140 -0
  68. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_trace.py +2 -1
  69. agentbyte-0.30.0/tests/orchestration/test_orchestrator_finalization.py +333 -0
  70. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_workflow.py +4 -4
  71. agentbyte-0.30.0/tests/webui/test_workflow_streaming.py +351 -0
  72. agentbyte-0.30.0/tests/workflow/test_agent_step_imports.py +15 -0
  73. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_steps.py +8 -6
  74. agentbyte-0.28.0/src/agentbyte/__about__.py +0 -2
  75. agentbyte-0.28.0/src/agentbyte/eval/comparison.py +0 -38
  76. agentbyte-0.28.0/src/agentbyte/eval/judges/llm.py +0 -279
  77. agentbyte-0.28.0/src/agentbyte/eval/pairwise.py +0 -106
  78. agentbyte-0.28.0/src/agentbyte/eval/targets/agent.py +0 -78
  79. agentbyte-0.28.0/src/agentbyte/eval/targets/multi_turn.py +0 -142
  80. agentbyte-0.28.0/src/agentbyte/eval/targets/orchestrator.py +0 -89
  81. agentbyte-0.28.0/src/agentbyte/workflow/steps/agentbyte_agent.py +0 -81
  82. {agentbyte-0.28.0 → agentbyte-0.30.0}/.gitignore +0 -0
  83. {agentbyte-0.28.0 → agentbyte-0.30.0}/LICENSE +0 -0
  84. {agentbyte-0.28.0 → agentbyte-0.30.0}/pyproject.toml +0 -0
  85. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/__init__.py +0 -0
  86. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/__init__.py +0 -0
  87. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/agent_as_tool.py +0 -0
  88. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/embedding_agent.py +0 -0
  89. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/agents/types.py +0 -0
  90. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cancellation_token.py +0 -0
  91. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/catalog.py +0 -0
  92. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cli/__init__.py +0 -0
  93. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/cli/main.py +0 -0
  94. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/component.py +0 -0
  95. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context.py +0 -0
  96. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/__init__.py +0 -0
  97. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/base.py +0 -0
  98. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skill_tools.py +0 -0
  99. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/context_providers/skills.py +0 -0
  100. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/__init__.py +0 -0
  101. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/base.py +0 -0
  102. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/config.py +0 -0
  103. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/importer.py +0 -0
  104. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/json.py +0 -0
  105. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/loader.py +0 -0
  106. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish.py +0 -0
  107. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publish_config.py +0 -0
  108. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/publishers.py +0 -0
  109. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sources.py +0 -0
  110. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite.py +0 -0
  111. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/sqlite_db.py +0 -0
  112. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/write_config.py +0 -0
  113. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/dataset/writers.py +0 -0
  114. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/entity.py +0 -0
  115. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/decorator.py +0 -0
  116. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/keyword.py +0 -0
  117. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/checks/types.py +0 -0
  118. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/eval_dataset.py +0 -0
  119. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/base.py +0 -0
  120. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/judges/reference.py +0 -0
  121. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/splitting.py +0 -0
  122. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/eval/targets/model.py +0 -0
  123. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/__init__.py +0 -0
  124. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/_retry_observability.py +0 -0
  125. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/auth.py +0 -0
  126. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/__init__.py +0 -0
  127. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/auth.py +0 -0
  128. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/chat.py +0 -0
  129. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/embedding.py +0 -0
  130. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure/settings.py +0 -0
  131. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai.py +0 -0
  132. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/azure_openai_embedding.py +0 -0
  133. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/base.py +0 -0
  134. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/embeddings_base.py +0 -0
  135. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/__init__.py +0 -0
  136. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/chat.py +0 -0
  137. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/embedding.py +0 -0
  138. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai/settings.py +0 -0
  139. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/openai_embedding.py +0 -0
  140. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/pricing.py +0 -0
  141. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/retry_policy.py +0 -0
  142. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/settings.py +0 -0
  143. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/llm/types.py +0 -0
  144. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/logger.py +0 -0
  145. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/memory/__init__.py +0 -0
  146. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/memory/base.py +0 -0
  147. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/messages.py +0 -0
  148. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/__init__.py +0 -0
  149. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/base.py +0 -0
  150. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/otel.py +0 -0
  151. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/retry.py +0 -0
  152. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/sql_usage.py +0 -0
  153. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/middleware/usage_logger.py +0 -0
  154. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/notebook.py +0 -0
  155. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/config.py +0 -0
  156. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/mipro.py +0 -0
  157. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/optim/spec.py +0 -0
  158. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/__init__.py +0 -0
  159. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/ai.py +0 -0
  160. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/plan.py +0 -0
  161. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/orchestration/round_robin.py +0 -0
  162. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/__init__.py +0 -0
  163. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/agents.py +0 -0
  164. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/clients.py +0 -0
  165. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instruction_registry.py +0 -0
  166. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/orchestrator.yaml +0 -0
  167. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/query_rewriter.yaml +0 -0
  168. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/researcher.yaml +0 -0
  169. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/reviewer.yaml +0 -0
  170. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/instructions/writer.yaml +0 -0
  171. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/orchestration.py +0 -0
  172. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/contracts-analyst/SKILL.md +0 -0
  173. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/SKILL.md +0 -0
  174. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/departments.md +0 -0
  175. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/employees.md +0 -0
  176. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/skills/hr-analyst/resources/payroll.md +0 -0
  177. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/presets/streaming.py +0 -0
  178. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/session_store.py +0 -0
  179. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/__init__.py +0 -0
  180. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/base.py +0 -0
  181. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/resources.py +0 -0
  182. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/scripts.py +0 -0
  183. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/sources.py +0 -0
  184. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/skills/validation.py +0 -0
  185. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/__init__.py +0 -0
  186. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/base.py +0 -0
  187. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/cancellation.py +0 -0
  188. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/composite.py +0 -0
  189. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/consecutive_agent.py +0 -0
  190. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/external.py +0 -0
  191. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/function_call.py +0 -0
  192. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/handoff.py +0 -0
  193. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/max_message.py +0 -0
  194. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/predicate.py +0 -0
  195. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/source.py +0 -0
  196. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/text_mention.py +0 -0
  197. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/timeout.py +0 -0
  198. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/termination/token_usage.py +0 -0
  199. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/__init__.py +0 -0
  200. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/base.py +0 -0
  201. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/coding_tools.py +0 -0
  202. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/core_tools.py +0 -0
  203. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/decorator.py +0 -0
  204. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/memory_tool.py +0 -0
  205. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/tools/research_tools.py +0 -0
  206. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/__init__.py +0 -0
  207. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/discovery.py +0 -0
  208. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/registry.py +0 -0
  209. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/session_store.py +0 -0
  210. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/sessions.py +0 -0
  211. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-BF3DwXaF.js +0 -0
  212. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/assets/index-ar5tOeqt.css +0 -0
  213. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/index.html +0 -0
  214. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/webui/ui/vite.svg +0 -0
  215. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/_structure_hash.py +0 -0
  216. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/checkpoint.py +0 -0
  217. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/core/workflow.py +0 -0
  218. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/defaults.py +0 -0
  219. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/loader.py +0 -0
  220. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema.py +0 -0
  221. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/schema_utils.py +0 -0
  222. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/echo.py +0 -0
  223. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/function.py +0 -0
  224. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/http.py +0 -0
  225. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/steps/transform.py +0 -0
  226. {agentbyte-0.28.0 → agentbyte-0.30.0}/src/agentbyte/workflow/visualizer.py +0 -0
  227. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_as_tool.py +0 -0
  228. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_error_response.py +0 -0
  229. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_event_types.py +0 -0
  230. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_memory_integration.py +0 -0
  231. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_middleware_integration.py +0 -0
  232. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_response_accessors.py +0 -0
  233. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_retry_middleware.py +0 -0
  234. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_agent_stream_events.py +0 -0
  235. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/agents/test_embedding_agent.py +0 -0
  236. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/cli/test_registry_check.py +0 -0
  237. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/__init__.py +0 -0
  238. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/test_skill_tools.py +0 -0
  239. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/context_providers/test_skills_provider.py +0 -0
  240. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_loader.py +0 -0
  241. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_multi_table.py +0 -0
  242. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_publish.py +0 -0
  243. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/dataset/test_sqlite_db.py +0 -0
  244. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_eval_dataset.py +0 -0
  245. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_multi_turn.py +0 -0
  246. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_phase2_checks_and_reports.py +0 -0
  247. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/eval/test_splitting.py +0 -0
  248. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_azure_client.py +0 -0
  249. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_azure_embedding_client.py +0 -0
  250. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_llm_types.py +0 -0
  251. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_openai_client.py +0 -0
  252. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_openai_embedding_client.py +0 -0
  253. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_pricing.py +0 -0
  254. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retry_observability.py +0 -0
  255. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retry_policy_api.py +0 -0
  256. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/llm/test_retryable_error_substrings.py +0 -0
  257. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/memory/test_memory.py +0 -0
  258. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_deduplicate_tool_result.py +0 -0
  259. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_middleware_chain.py +0 -0
  260. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_otel.py +0 -0
  261. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_retry_middleware.py +0 -0
  262. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_sql_usage.py +0 -0
  263. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/middleware/test_usage_logger.py +0 -0
  264. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/__init__.py +0 -0
  265. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_base.py +0 -0
  266. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_base_integration.py +0 -0
  267. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_config.py +0 -0
  268. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_reflective.py +0 -0
  269. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/optim/test_spec.py +0 -0
  270. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_ai_orchestrator.py +0 -0
  271. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_base_orchestrator.py +0 -0
  272. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_handoff_orchestrator.py +0 -0
  273. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_plan_orchestrator.py +0 -0
  274. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/orchestration/test_round_robin.py +0 -0
  275. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_agents.py +0 -0
  276. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_clients.py +0 -0
  277. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_instruction_registry.py +0 -0
  278. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_orchestration.py +0 -0
  279. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/presets/test_streaming.py +0 -0
  280. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/__init__.py +0 -0
  281. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_base.py +0 -0
  282. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_resources.py +0 -0
  283. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_scripts.py +0 -0
  284. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/skills/test_sources.py +0 -0
  285. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_base.py +0 -0
  286. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_cancellation.py +0 -0
  287. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_composite.py +0 -0
  288. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_consecutive_agent.py +0 -0
  289. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_external.py +0 -0
  290. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_function_call.py +0 -0
  291. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_handoff.py +0 -0
  292. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_max_message.py +0 -0
  293. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_predicate.py +0 -0
  294. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_source.py +0 -0
  295. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_text_mention.py +0 -0
  296. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_timeout.py +0 -0
  297. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/termination/test_token_usage.py +0 -0
  298. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_cancellation_token.py +0 -0
  299. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_context.py +0 -0
  300. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_logger.py +0 -0
  301. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_messages.py +0 -0
  302. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_package_api.py +0 -0
  303. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_session_store.py +0 -0
  304. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_types.py +0 -0
  305. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/test_vanilla_chunker.py +0 -0
  306. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_coding_tools.py +0 -0
  307. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_memory_tool.py +0 -0
  308. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_research_tools.py +0 -0
  309. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/tools/test_tools.py +0 -0
  310. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/__init__.py +0 -0
  311. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/helpers.py +0 -0
  312. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_execution.py +0 -0
  313. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_package_api.py +0 -0
  314. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_registry.py +0 -0
  315. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_server.py +0 -0
  316. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/webui/test_sessions.py +0 -0
  317. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_checkpoint.py +0 -0
  318. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_subworkflow_step.py +0 -0
  319. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_agent.py +0 -0
  320. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_class.py +0 -0
  321. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_models.py +0 -0
  322. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_runner.py +0 -0
  323. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_schema.py +0 -0
  324. {agentbyte-0.28.0 → agentbyte-0.30.0}/tests/workflow/test_workflow_visualizer.py +0 -0
@@ -4,6 +4,45 @@ All notable changes to Agentbyte are documented in this file.
4
4
 
5
5
  The format follows Keep a Changelog principles and semantic versioning.
6
6
 
7
+ ## [0.30.0] - 2026-09-30
8
+
9
+ ### Added
10
+
11
+ - Workflow streaming parity (spec 0057): `WorkflowRunner.run_stream(..., stream_tokens=...)` forwards live child agent events as `StepEvent` envelopes (workflow, execution, invocation, step, attempt and sequence attribution) through a bounded queue, including across `workflow.as_agent()` and `SubWorkflowStep` boundaries. `WorkflowAgent.run_stream` honors `stream_tokens`. Closing the stream cancels and awaits outstanding steps.
12
+ - `AgentStep` streams its agent and turns tool approval into workflow suspension; resuming validates the exact pending decisions and never reruns completed tools.
13
+ - Checkpoint-backed serving: the WebUI/FastAPI run and SSE endpoints accept `workflow_responses` and `workflow_checkpoint_id` to resume a suspended workflow in the same session. `WorkflowExecution.checkpoint_id` exposes the suspended checkpoint, and `WorkflowRunner.validate_resume_responses` / `BaseStep.validate_resume_response` validate responses without consuming it. Stale, unknown or wrong-owner checkpoints and concurrent runs of the same workflow or session are rejected with HTTP 409.
14
+ - Judge score integrity (spec 0055): `EvalScore.scoring_status` (`scored` | `failed` | `cancelled`) and `failure_reason`, `JudgeScoringError` with safe reason codes, `mean_scored()`, `OptimizationEvidenceError`, and `PairwiseResult.comparison_status`.
15
+
16
+ ### Changed
17
+
18
+ - **Breaking:** Judges never invent scores. `LLMEvalJudge` and `LLMTrajectoryJudge` require every requested criterion exactly once (matched case- and whitespace-insensitively, reported with the requested spelling) with a finite 0–10 value, and raise `JudgeScoringError` otherwise instead of filling missing criteria with 5.0, clamping, or returning a neutral 5.0 after a failure. `overall` is always the mean of the validated dimensions; a model-supplied overall is ignored. Cancellation raises `asyncio.CancelledError`.
19
+ - **Breaking:** `EvalScore.overall` is `float | None`. `EvalRunner` records judge failures, target exceptions and cancellation as `failed`/`cancelled` outcomes with `overall=None`, empty dimensions and the trajectory kept, instead of a fabricated 0.0. Failure metadata keeps the exception type, never its text. A scored `EvalScore` must have a finite `overall` and dimensions in 0–10.
20
+ - **Breaking:** `PairwiseJudge.compare()` raises instead of returning a tie on failure; unknown winners and out-of-range margins are rejected, not coerced. `PairwiseRunner` records failed comparisons with `winner=None`. `compare_pairwise()` adds `compared`/`failed`/`cancelled` counts and computes win rates over valid comparisons only (`None` when nothing was compared).
21
+ - **Breaking:** `EvalReport.avg_score` and `compare_configurations()` averages cover scored outcomes only and are `None` when nothing was scored; an item with a failed or cancelled score fails the suite. `CompositeJudge` fails when any sub-judge fails rather than renormalizing the remaining weights.
22
+ - **Breaking:** Optimizers only rank candidates whose every task was scored. `Candidate.avg` is `None` for ineligible candidates, the minibatch gate rejects incomplete batches, a seed with an unscored task raises `OptimizationEvidenceError`, and the GEPA adapter raises rather than passing a number for an unscored example. Optimization trace `eval` events report `avg` over scored tasks (`None` if none) plus `n_scored`.
23
+
24
+ ## [0.29.0] - 2026-09-30
25
+
26
+ ### Added
27
+
28
+ - Unified trajectory evaluation (spec 0046): every eval target now records a runtime-neutral execution trace (`trajectory.trace`) with spans for agents, workflow steps, tools and model calls, plus normalized terminal status and usage. New `WorkflowEvalTarget` and `WorkflowEvalTask` evaluate workflows directly with typed input; workflows wrapped via `workflow.as_agent()` capture the same evidence.
29
+ - Process evaluation: `LLMTrajectoryJudge` scores `tool_use_correctness`, `coordination` and `execution_efficiency` with cited span/event evidence and reports failures as `metadata["judge_failed"]` instead of silently truncating. New deterministic checks `terminal_status`, `steps_completed`, `edge_activated` and `usage_limits`, and `select_span` to score one span in isolation.
30
+ - Orchestrator finalization (spec 0048): `FinalResultPolicy` gains `final_agent` and `finalize_on_exhaustion` (default `True`). When a final agent is configured (`final_agent`, or `prefer_agent` as fallback) and `max_iterations` is about to run out, the last iteration runs that agent with a finalization instruction, bypassing the pattern's selection, context and state hooks; the run stops with `stop_message.source == "MaxIterationsFinalized"`. An unknown `final_agent` raises `ValueError`.
31
+ - `OrchestrationResponse.final_structured_result` exposes the `structured_content` of the message selected as `final_result`, so a typed final agent's answer is available without parsing JSON.
32
+ - `HandoffOrchestrator` accepts `final_result_policy`.
33
+
34
+ ### Changed
35
+
36
+ - **Breaking:** The agent workflow step module moved from `agentbyte.workflow.steps.agentbyte_agent` to `agentbyte.workflow.steps.agent`, and `AgentbyteAgentInput` / `AgentbyteAgentOutput` are renamed to `AgentStepInput` / `AgentStepOutput` (exported from `agentbyte.workflow` and `agentbyte.workflow.steps`). No compatibility aliases; update imports. `AgentStep` is unchanged.
37
+ - **Breaking:** Agents reject duplicate tool names across `tools` and context-provider tools (spec 0047). The run fails with `finish_reason="error"` and an `AgentConfigurationError` before any model call. Previously, attaching several `SkillsProvider`s silently made every provider but the first unreachable; attach one `SkillsProvider` with several paths (`from_paths([...])` or `AggregatingSkillsSource`) instead.
38
+ - Orchestrators configured with `FinalResultPolicy(prefer_agent=...)` now reserve the last iteration for that agent when the budget is exhausted. Opt out with `finalize_on_exhaustion=False`.
39
+ - `stream_tokens=True` is disabled with a warning when the agent has an `output_format`, because streamed chunks are not parsed into structured output. Typed turns now always carry `structured_content`.
40
+
41
+ ### Fixed
42
+
43
+ - The agent finalization turn drops tool calls the model returns despite having no tools, so they are never executed or left unanswered in history, and the run reports `max_iterations_finalized` (spec 0042 hardening).
44
+ - Workflow examples that imported the removed `AgentbyteAgentStep` and `Context` names now use `AgentStep` and `WorkflowContext`.
45
+
7
46
  ## [0.28.0] - 2026-09-15
8
47
 
9
48
  ### Changed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: agentbyte
3
- Version: 0.28.0
3
+ Version: 0.30.0
4
4
  Summary: A toolkit for designing multiagent systems
5
5
  Author-email: MrDataPsycho <mr.data.psycho@gmail.com>
6
6
  License-Expression: LicenseRef-Proprietary
@@ -86,7 +86,7 @@ Description-Content-Type: text/markdown
86
86
 
87
87
  Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
88
88
 
89
- Current release: **0.28.0**
89
+ Current release: **0.30.0**
90
90
 
91
91
  ## Building an Agent
92
92
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  Agentbyte is an observability-first agentic AI framework for building and studying multiagent systems with a learning-first, implementation-oriented workflow.
4
4
 
5
- Current release: **0.28.0**
5
+ Current release: **0.30.0**
6
6
 
7
7
  ## Building an Agent
8
8
 
@@ -0,0 +1,2 @@
1
+ __version__ = "0.30.0"
2
+ VERSION = __version__
@@ -12,6 +12,7 @@ from typing import List, Optional, Union
12
12
 
13
13
  from agentbyte.cancellation_token import CancellationToken
14
14
  from agentbyte.context import AgentContext
15
+ from agentbyte.execution_trace import trace_stream
15
16
  from agentbyte.llm.types import ChatCompletionChunk
16
17
  from agentbyte.messages import (
17
18
  AssistantMessage,
@@ -338,6 +339,7 @@ class Agent(BaseAgent):
338
339
 
339
340
  yield message
340
341
 
342
+ @trace_stream("agent")
341
343
  async def run_stream(
342
344
  self,
343
345
  task: Optional[Union[str, UserMessage, List[Message]]] = None,
@@ -414,6 +416,15 @@ class Agent(BaseAgent):
414
416
  stacklevel=2,
415
417
  )
416
418
  effective_stream_tokens = False
419
+ elif stream_tokens and self.output_format is not None:
420
+ # Streamed chunks are not schema-constrained or parsed, so typed
421
+ # answers would lose structured_content.
422
+ warnings.warn(
423
+ "stream_tokens=True is disabled when output_format is set; "
424
+ "falling back to non-token execution to keep structured output.",
425
+ stacklevel=2,
426
+ )
427
+ effective_stream_tokens = False
417
428
 
418
429
  try:
419
430
  task_messages = self._convert_task_to_messages(task) if task else []
@@ -434,6 +445,7 @@ class Agent(BaseAgent):
434
445
  if instr:
435
446
  provider_instructions.append(instr)
436
447
  provider_tools.extend(cp_tools)
448
+ self._check_unique_tool_names(extra_tools=provider_tools)
437
449
  extra_instructions = "\n\n".join(provider_instructions) if provider_instructions else None
438
450
 
439
451
  iteration = 0
@@ -474,7 +486,8 @@ class Agent(BaseAgent):
474
486
  tools = self._get_tools_for_llm(extra_tools=provider_tools) if (self.tools or provider_tools) else None
475
487
 
476
488
  # Reserve the final iteration for a tool-free answer so an exhausted
477
- # budget yields a complete response instead of nothing.
489
+ # budget yields a complete response instead of nothing. Needs
490
+ # max_iterations > 1, otherwise the only turn would lose its tools.
478
491
  is_final_iteration = (
479
492
  self.finalize_on_exhaustion
480
493
  and self.max_iterations > 1
@@ -625,10 +638,14 @@ class Agent(BaseAgent):
625
638
  if completion_result.usage.cost_estimate is not None:
626
639
  cost_estimate += completion_result.usage.cost_estimate
627
640
 
641
+ # The finalization turn offers no tools; drop any tool calls the
642
+ # model returns anyway so they never execute or dangle in history.
628
643
  assistant_message = AssistantMessage(
629
644
  content=completion_result.message.content,
630
645
  source=self.name,
631
- tool_calls=completion_result.message.tool_calls,
646
+ tool_calls=None
647
+ if is_final_iteration
648
+ else completion_result.message.tool_calls,
632
649
  structured_content=completion_result.structured_output,
633
650
  usage=completion_result.usage,
634
651
  )
@@ -638,7 +655,7 @@ class Agent(BaseAgent):
638
655
  source=self.name,
639
656
  model=completion_result.model,
640
657
  response=completion_result.message.content,
641
- has_tool_calls=bool(completion_result.message.tool_calls),
658
+ has_tool_calls=bool(assistant_message.tool_calls),
642
659
  )
643
660
 
644
661
  working_context.add_message(assistant_message)
@@ -61,6 +61,15 @@ class BaseAgent(ABC):
61
61
  context_providers: Optional[List[ContextProvider]] = None,
62
62
  **kwargs: Any,
63
63
  ) -> None:
64
+ """Initialize the agent.
65
+
66
+ When ``finalize_on_exhaustion`` is enabled and the agent has tools, the
67
+ last of ``max_iterations`` is reserved for a tool-free final answer
68
+ (``finish_reason="max_iterations_finalized"``). This requires
69
+ ``max_iterations >= 2``: with ``max_iterations=1`` the only turn keeps
70
+ its tools, so a run that calls a tool ends with
71
+ ``finish_reason="max_iterations"`` and no final answer.
72
+ """
64
73
  self.name = name
65
74
  self.description = description
66
75
  self.instructions = instructions
@@ -124,6 +133,30 @@ class BaseAgent(ABC):
124
133
  return tool
125
134
  return None
126
135
 
136
+ def _check_unique_tool_names(
137
+ self,
138
+ extra_tools: Optional[List[BaseTool]] = None,
139
+ ) -> None:
140
+ """Raise if agent tools and context-provider tools share a name.
141
+
142
+ Duplicate names are ambiguous for the model and ``_find_tool`` always
143
+ resolves to the first match, silently shadowing later tools.
144
+ """
145
+ seen: set[str] = set()
146
+ duplicates: list[str] = []
147
+ for tool in [*self.tools, *(extra_tools or [])]:
148
+ if tool.name in seen and tool.name not in duplicates:
149
+ duplicates.append(tool.name)
150
+ seen.add(tool.name)
151
+ if duplicates:
152
+ names = ", ".join(repr(name) for name in duplicates)
153
+ raise AgentConfigurationError(
154
+ f"Duplicate tool name(s) {names} across agent tools and context "
155
+ "providers; tool names must be unique. To use several skill "
156
+ "directories, attach one SkillsProvider with all paths, e.g. "
157
+ "SkillsProvider.from_paths([...]) or AggregatingSkillsSource."
158
+ )
159
+
127
160
  def _get_tools_for_llm(
128
161
  self,
129
162
  extra_tools: Optional[List[BaseTool]] = None,
@@ -1,6 +1,9 @@
1
1
  """Evaluation framework for AgentByte."""
2
2
 
3
- from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget
3
+ from .base import BaseEvalJudge, BaseEvalRunner, BaseEvalTarget, JudgeScoringError
4
+ from .checks.process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
5
+ from .judges.trajectory import LLMTrajectoryJudge
6
+ from .targets.workflow import WorkflowEvalTarget
4
7
  from .checks import (
5
8
  CheckResult,
6
9
  EvalCheck,
@@ -16,7 +19,7 @@ from .checks import (
16
19
  tool_calls_present,
17
20
  tool_calls_present_check,
18
21
  )
19
- from .comparison import compare_configurations
22
+ from .comparison import compare_configurations, mean_scored
20
23
  from .eval_dataset import (
21
24
  EvalDatasetRecord,
22
25
  save_eval_dataset,
@@ -49,11 +52,15 @@ from .types import (
49
52
  EvalTask,
50
53
  EvalTrajectory,
51
54
  ExpectedToolCall,
55
+ ScoringStatus,
52
56
  MultiTurnEvalTask,
57
+ WorkflowEvalTask,
53
58
  PairwiseResult,
54
59
  )
55
60
 
56
61
  __all__ = [
62
+ "WorkflowEvalTask", "WorkflowEvalTarget", "LLMTrajectoryJudge",
63
+ "select_span", "terminal_status", "steps_completed", "edge_activated", "usage_limits",
57
64
  "AnswerStrategy",
58
65
  "ExpectedToolCall",
59
66
  "EvalTask",
@@ -64,6 +71,8 @@ __all__ = [
64
71
  "BaseEvalTarget",
65
72
  "BaseEvalJudge",
66
73
  "BaseEvalRunner",
74
+ "JudgeScoringError",
75
+ "ScoringStatus",
67
76
  "ExactMatchJudge",
68
77
  "ContainsJudge",
69
78
  "FuzzyMatchJudge",
@@ -82,6 +91,7 @@ __all__ = [
82
91
  "EvalSplitConfig",
83
92
  "EvalSplitStrategy",
84
93
  "compare_configurations",
94
+ "mean_scored",
85
95
  "EvalDatasetRecord",
86
96
  "save_eval_dataset",
87
97
  "tasks_from_eval_dataset",
@@ -17,6 +17,19 @@ _VALID_ANSWER_STRATEGIES: tuple[AnswerStrategy, ...] = (
17
17
  )
18
18
 
19
19
 
20
+ class JudgeScoringError(Exception):
21
+ """A judge could not produce a valid judgment.
22
+
23
+ ``reason`` is a safe, stable code (for example ``provider_error`` or
24
+ ``missing_criterion``) suitable for persistence; the message may carry more
25
+ detail and is not persisted by default.
26
+ """
27
+
28
+ def __init__(self, reason: str, message: str | None = None) -> None:
29
+ super().__init__(message or reason)
30
+ self.reason = reason
31
+
32
+
20
33
  class BaseEvalTarget(ABC):
21
34
  """Anything that can execute an evaluation task into a trajectory."""
22
35
 
@@ -96,7 +109,12 @@ class BaseEvalJudge(ABC):
96
109
  criteria: list[str] | None = None,
97
110
  cancellation_token: CancellationToken | None = None,
98
111
  ) -> EvalScore:
99
- """Score an evaluation trajectory."""
112
+ """Score an evaluation trajectory.
113
+
114
+ Returns a scored ``EvalScore``. Raises ``JudgeScoringError`` when no
115
+ valid judgment is available and ``asyncio.CancelledError`` when
116
+ cancelled; judges never substitute a placeholder score.
117
+ """
100
118
 
101
119
 
102
120
  class BaseEvalRunner(ABC):
@@ -116,4 +134,4 @@ class BaseEvalRunner(ABC):
116
134
  """Evaluate a target on multiple tasks."""
117
135
 
118
136
 
119
- __all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner"]
137
+ __all__ = ["BaseEvalTarget", "BaseEvalJudge", "BaseEvalRunner", "JudgeScoringError"]
@@ -4,6 +4,7 @@ from agentbyte.eval.report import EvalItemReport, EvalNotPassedError, EvalReport
4
4
 
5
5
  from .decorator import evaluator
6
6
  from .keyword import keyword_check
7
+ from .process import select_span, terminal_status, steps_completed, edge_activated, usage_limits
7
8
  from .local import CheckEvaluator, threshold_gate
8
9
  from .tool import (
9
10
  tool_call_args_match,
@@ -14,6 +15,7 @@ from .tool import (
14
15
  from .types import CheckResult, EvalCheck, ExpectedToolCall
15
16
 
16
17
  __all__ = [
18
+ "select_span", "terminal_status", "steps_completed", "edge_activated", "usage_limits",
17
19
  "ExpectedToolCall",
18
20
  "CheckResult",
19
21
  "EvalCheck",
@@ -93,6 +93,12 @@ def threshold_gate(
93
93
 
94
94
  async def _check(trajectory: EvalTrajectory) -> CheckResult:
95
95
  score = await judge.score(trajectory)
96
+ if not score.is_scored or score.overall is None:
97
+ return CheckResult(
98
+ passed=False,
99
+ reason=f"no valid score ({score.scoring_status}: {score.failure_reason})",
100
+ check_name=f"threshold_gate[{judge.name}]",
101
+ )
96
102
  passed = score.overall >= threshold
97
103
  return CheckResult(
98
104
  passed=passed,
@@ -0,0 +1,123 @@
1
+ """Deterministic checks over execution evidence."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from agentbyte.eval.types import EvalTrajectory
6
+ from .types import CheckResult, EvalCheck
7
+
8
+
9
+ def select_span(trajectory: EvalTrajectory, span_id: str) -> EvalTrajectory:
10
+ """Project one invocation for existing judges and checks without changing its task."""
11
+ if trajectory.trace is None:
12
+ raise ValueError("Trajectory has no execution trace")
13
+ trace = trajectory.trace.select(span_id)
14
+ root = trace.spans[0]
15
+ return trajectory.model_copy(
16
+ update={
17
+ "trace": trace,
18
+ "messages": trace.evidence_messages(),
19
+ "usage": trace.aggregate_usage(),
20
+ "success": root.status == "completed",
21
+ "error": root.error,
22
+ "metadata": {
23
+ **trajectory.metadata,
24
+ "execution_status": root.status,
25
+ "span_id": span_id,
26
+ },
27
+ }
28
+ )
29
+
30
+
31
+ def terminal_status(expected: str = "completed") -> EvalCheck:
32
+ """Require an explicit normalized terminal status."""
33
+
34
+ async def check(trajectory: EvalTrajectory) -> CheckResult:
35
+ actual = trajectory.metadata.get("execution_status")
36
+ return CheckResult(
37
+ passed=actual == expected,
38
+ reason=f"Expected {expected}; observed {actual}",
39
+ check_name="terminal_status",
40
+ )
41
+
42
+ return check
43
+
44
+
45
+ def steps_completed(*step_ids: str) -> EvalCheck:
46
+ """Require at least one completed attempt of each step; select a span to scope nesting."""
47
+
48
+ async def check(trajectory: EvalTrajectory) -> CheckResult:
49
+ actual = (
50
+ {s.step_id for s in trajectory.trace.spans if s.status == "completed"}
51
+ if trajectory.trace
52
+ else set()
53
+ )
54
+ missing = set(step_ids) - actual
55
+ return CheckResult(
56
+ passed=trajectory.trace is not None and not missing,
57
+ reason=f"Missing completed steps: {sorted(missing)}",
58
+ check_name="steps_completed",
59
+ )
60
+
61
+ return check
62
+
63
+
64
+ def edge_activated(from_step: str, to_step: str) -> EvalCheck:
65
+ """Require an observed routing edge, not merely adjacent message order."""
66
+
67
+ async def check(trajectory: EvalTrajectory) -> CheckResult:
68
+ found = trajectory.trace is not None and any(
69
+ e.kind == "EdgeActivatedEvent"
70
+ and e.data.get("from_step") == from_step
71
+ and e.data.get("to_step") == to_step
72
+ for e in trajectory.trace.events
73
+ )
74
+ return CheckResult(
75
+ passed=found,
76
+ reason=f"Edge {from_step} -> {to_step} observed: {found}",
77
+ check_name="edge_activated",
78
+ )
79
+
80
+ return check
81
+
82
+
83
+ def usage_limits(**limits: float) -> EvalCheck:
84
+ """Require usage fields to remain within bounds; unknown cost/evidence fails the gate."""
85
+ valid = {
86
+ "duration_ms",
87
+ "llm_calls",
88
+ "tokens_input",
89
+ "tokens_output",
90
+ "tokens_cached",
91
+ "total_tokens",
92
+ "tool_calls",
93
+ "memory_operations",
94
+ "cost_estimate",
95
+ }
96
+ if not limits or set(limits) - valid or any(v < 0 for v in limits.values()):
97
+ raise ValueError("Supply non-negative limits for recognized Usage fields")
98
+
99
+ async def check(trajectory: EvalTrajectory) -> CheckResult:
100
+ known = trajectory.usage is not None and (
101
+ trajectory.trace is None or trajectory.trace.usage_complete
102
+ )
103
+ passed = known and all(
104
+ getattr(trajectory.usage, k) is not None
105
+ and getattr(trajectory.usage, k) <= v
106
+ for k, v in limits.items()
107
+ )
108
+ return CheckResult(
109
+ passed=passed,
110
+ reason=f"Usage limits {limits}; complete evidence: {known}",
111
+ check_name="usage_limits",
112
+ )
113
+
114
+ return check
115
+
116
+
117
+ __all__ = [
118
+ "select_span",
119
+ "terminal_status",
120
+ "steps_completed",
121
+ "edge_activated",
122
+ "usage_limits",
123
+ ]
@@ -21,7 +21,7 @@ async def tool_calls_present(trajectory: EvalTrajectory) -> CheckResult:
21
21
 
22
22
  actual_names = {
23
23
  tool_call.tool_name
24
- for message in trajectory.messages
24
+ for message in (trajectory.trace.evidence_messages() if trajectory.trace else trajectory.messages)
25
25
  if isinstance(message, AssistantMessage) and message.tool_calls
26
26
  for tool_call in message.tool_calls
27
27
  }
@@ -69,7 +69,7 @@ async def tool_call_args_match(trajectory: EvalTrajectory) -> CheckResult:
69
69
  )
70
70
 
71
71
  actual_calls: list[tuple[str, dict[str, object]]] = []
72
- for message in trajectory.messages:
72
+ for message in (trajectory.trace.evidence_messages() if trajectory.trace else trajectory.messages):
73
73
  if not isinstance(message, AssistantMessage) or not message.tool_calls:
74
74
  continue
75
75
  for tool_call in message.tool_calls:
@@ -0,0 +1,58 @@
1
+ """Helpers for comparing evaluated configurations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from agentbyte.eval.types import EvalScore
6
+
7
+
8
+ def compare_configurations(
9
+ scores_a: list[EvalScore],
10
+ scores_b: list[EvalScore],
11
+ label_a: str = "baseline",
12
+ label_b: str = "candidate",
13
+ ) -> dict[str, float | str | int | None]:
14
+ """Compare average score between two evaluated configurations.
15
+
16
+ Averages cover scored outcomes only; failed and cancelled outcomes are
17
+ counted per side and never enter a mean. ``avg_*`` is None when a side has
18
+ no scored outcome, and ``improvement_pct`` is None unless both sides have a
19
+ nonzero baseline to compare against.
20
+ """
21
+ avg_a = mean_scored(scores_a)
22
+ avg_b = mean_scored(scores_b)
23
+
24
+ improvement_pct: float | None = None
25
+ if avg_a is not None and avg_b is not None:
26
+ if avg_a == 0:
27
+ improvement_pct = 0.0 if avg_b == 0 else 100.0
28
+ else:
29
+ improvement_pct = ((avg_b - avg_a) / avg_a) * 100.0
30
+
31
+ return {
32
+ "label_a": label_a,
33
+ "label_b": label_b,
34
+ "avg_a": avg_a,
35
+ "avg_b": avg_b,
36
+ "scored_a": _count(scores_a, "scored"),
37
+ "scored_b": _count(scores_b, "scored"),
38
+ "failed_a": _count(scores_a, "failed"),
39
+ "failed_b": _count(scores_b, "failed"),
40
+ "cancelled_a": _count(scores_a, "cancelled"),
41
+ "cancelled_b": _count(scores_b, "cancelled"),
42
+ "improvement_pct": improvement_pct,
43
+ }
44
+
45
+
46
+ def mean_scored(scores: list[EvalScore]) -> float | None:
47
+ """Mean ``overall`` across scored outcomes, or None when none were scored."""
48
+ values = [score.overall for score in scores if score.is_scored and score.overall is not None]
49
+ if not values:
50
+ return None
51
+ return sum(values) / len(values)
52
+
53
+
54
+ def _count(scores: list[EvalScore], status: str) -> int:
55
+ return sum(1 for score in scores if score.scoring_status == status)
56
+
57
+
58
+ __all__ = ["compare_configurations", "mean_scored"]
@@ -4,10 +4,12 @@ from agentbyte.eval.base import BaseEvalJudge
4
4
 
5
5
  from .composite import CompositeJudge
6
6
  from .llm import LLMEvalJudge
7
+ from .trajectory import LLMTrajectoryJudge, ProcessJudgeResponse, ProcessCriterionScore
7
8
  from .pairwise import PairwiseJudge
8
9
  from .reference import ContainsJudge, ExactMatchJudge, FuzzyMatchJudge
9
10
 
10
11
  __all__ = [
12
+ "LLMTrajectoryJudge", "ProcessJudgeResponse", "ProcessCriterionScore",
11
13
  "BaseEvalJudge",
12
14
  "ExactMatchJudge",
13
15
  "ContainsJudge",
@@ -7,7 +7,7 @@ from collections.abc import Sequence
7
7
 
8
8
  from agentbyte.cancellation_token import CancellationToken
9
9
 
10
- from agentbyte.eval.base import BaseEvalJudge
10
+ from agentbyte.eval.base import BaseEvalJudge, JudgeScoringError
11
11
  from agentbyte.eval.types import EvalScore, EvalTrajectory
12
12
 
13
13
 
@@ -45,12 +45,30 @@ class CompositeJudge(BaseEvalJudge):
45
45
  criteria: list[str] | None = None,
46
46
  cancellation_token: CancellationToken | None = None,
47
47
  ) -> EvalScore:
48
- results = await asyncio.gather(
49
- *[
50
- judge.score(trajectory, criteria, cancellation_token)
51
- for judge, _ in self.judges
52
- ]
53
- )
48
+ """Combine sub-judge scores; any sub-judge failure fails the composite.
49
+
50
+ Every sub-judge is required. A failed or cancelled component is never
51
+ dropped and the remaining weights are never renormalized into success.
52
+ """
53
+ try:
54
+ async with asyncio.TaskGroup() as group:
55
+ tasks = [
56
+ group.create_task(judge.score(trajectory, criteria, cancellation_token))
57
+ for judge, _ in self.judges
58
+ ]
59
+ except* JudgeScoringError as failures:
60
+ first = failures.exceptions[0]
61
+ raise JudgeScoringError(
62
+ "component_failed", f"Sub-judge failed: {first.reason}"
63
+ ) from first
64
+ results = [task.result() for task in tasks]
65
+ overalls: list[float] = []
66
+ for (judge, _), result in zip(self.judges, results):
67
+ if not result.is_scored or result.overall is None:
68
+ raise JudgeScoringError(
69
+ "component_failed", f"Sub-judge {judge.name!r} returned no valid score"
70
+ )
71
+ overalls.append(result.overall)
54
72
 
55
73
  dimensions: dict[str, float] = {}
56
74
  dimension_weight_totals: dict[str, float] = {}
@@ -58,11 +76,9 @@ class CompositeJudge(BaseEvalJudge):
58
76
  metadata_sub_judges: list[dict[str, float | str]] = []
59
77
  weighted_overall = 0.0
60
78
 
61
- for (judge, weight), result in zip(self.judges, results):
62
- metadata_sub_judges.append(
63
- {"name": judge.name, "weight": weight, "score": result.overall}
64
- )
65
- weighted_overall += result.overall * weight
79
+ for (judge, weight), result, overall in zip(self.judges, results, overalls):
80
+ metadata_sub_judges.append({"name": judge.name, "weight": weight, "score": overall})
81
+ weighted_overall += overall * weight
66
82
  for dimension, score in result.dimensions.items():
67
83
  dimensions[dimension] = dimensions.get(dimension, 0.0) + (score * weight)
68
84
  dimension_weight_totals[dimension] = (