gooddata-eval 1.74.1.dev3__tar.gz → 1.74.1.dev5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/PKG-INFO +2 -2
  2. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/pyproject.toml +2 -2
  3. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/alert_skill.py +10 -0
  4. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/conversation.py +20 -0
  5. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/kda_skill.py +11 -0
  6. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/metric_skill.py +8 -2
  7. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/visualization.py +6 -6
  8. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_alert_skill.py +64 -0
  9. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_conversation.py +104 -0
  10. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_kda_skill.py +85 -0
  11. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_metric_skill.py +88 -1
  12. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_visualization.py +4 -4
  13. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/.gitignore +0 -0
  14. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/AGENTS.md +0 -0
  15. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/CLAUDE.md +0 -0
  16. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/LICENSE.txt +0 -0
  17. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/Makefile +0 -0
  18. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/README.md +0 -0
  19. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/__init__.py +0 -0
  20. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/_version.py +0 -0
  21. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/__init__.py +0 -0
  22. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/agentic_runner.py +0 -0
  23. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/cli/main.py +0 -0
  24. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/__init__.py +0 -0
  25. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/_output.py +0 -0
  26. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/__init__.py +0 -0
  27. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_catalog.py +0 -0
  28. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_gate.py +0 -0
  29. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_langfuse.py +0 -0
  30. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/_trace_linker.py +0 -0
  31. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/general_question.py +0 -0
  32. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/guardrail.py +0 -0
  33. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/agentic/search_tool.py +0 -0
  34. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/__init__.py +0 -0
  35. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/render.py +0 -0
  36. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/chat/sse_client.py +0 -0
  37. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/config.py +0 -0
  38. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/connection.py +0 -0
  39. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/__init__.py +0 -0
  40. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/langfuse_source.py +0 -0
  41. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/dataset/local.py +0 -0
  42. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/__init__.py +0 -0
  43. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_deep_subset.py +0 -0
  44. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_llm_judge.py +0 -0
  45. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_maql.py +0 -0
  46. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/_text_utils.py +0 -0
  47. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/alert_skill.py +0 -0
  48. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/base.py +0 -0
  49. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/general_question.py +0 -0
  50. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/guardrail.py +0 -0
  51. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/metric_skill.py +0 -0
  52. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/search_tool.py +0 -0
  53. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/summary.py +0 -0
  54. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/evaluators/visualization.py +0 -0
  55. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/__init__.py +0 -0
  56. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/_env.py +0 -0
  57. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/client.py +0 -0
  58. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/experiment.py +0 -0
  59. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/observations.py +0 -0
  60. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/otlp.py +0 -0
  61. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/langfuse/sink.py +0 -0
  62. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/models.py +0 -0
  63. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/__init__.py +0 -0
  64. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/console.py +0 -0
  65. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/reporting/json_report.py +0 -0
  66. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/runner.py +0 -0
  67. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/scoring.py +0 -0
  68. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/summary/__init__.py +0 -0
  69. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/summary/http_client.py +0 -0
  70. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/timing.py +0 -0
  71. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/src/gooddata_eval/core/workspace.py +0 -0
  72. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/__init__.py +0 -0
  73. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/_fake_langfuse.py +0 -0
  74. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/conftest.py +0 -0
  75. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sample_dataset/metric_skill_create.json +0 -0
  76. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sample_dataset/visualization_revenue.json +0 -0
  77. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/fixtures/sse_visualization_stream.txt +0 -0
  78. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_gate.py +0 -0
  79. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_general_question.py +0 -0
  80. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_guardrail.py +0 -0
  81. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_langfuse_trace.py +0 -0
  82. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_observe_experiment.py +0 -0
  83. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_run_context.py +0 -0
  84. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_runner.py +0 -0
  85. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_agentic_search_tool.py +0 -0
  86. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_alert_skill_evaluator.py +0 -0
  87. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_chat_render.py +0 -0
  88. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_cli.py +0 -0
  89. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_connection.py +0 -0
  90. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_deep_subset.py +0 -0
  91. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_fake_langfuse.py +0 -0
  92. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_client.py +0 -0
  93. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_e2e_fake_server.py +0 -0
  94. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_env.py +0 -0
  95. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_experiment.py +0 -0
  96. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_observations.py +0 -0
  97. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_otlp.py +0 -0
  98. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_sink.py +0 -0
  99. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_langfuse_source.py +0 -0
  100. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_llm_judge.py +0 -0
  101. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_local_loader.py +0 -0
  102. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_maql_normalize.py +0 -0
  103. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_metric_skill_evaluator.py +0 -0
  104. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_models.py +0 -0
  105. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_reporting.py +0 -0
  106. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_runner.py +0 -0
  107. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_scoring.py +0 -0
  108. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_search_tool_evaluator.py +0 -0
  109. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_sse_client.py +0 -0
  110. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_summary_client.py +0 -0
  111. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_summary_evaluator.py +0 -0
  112. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_text_evaluators.py +0 -0
  113. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_timing.py +0 -0
  114. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_trace_linker.py +0 -0
  115. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_visualization_evaluator.py +0 -0
  116. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tests/test_workspace.py +0 -0
  117. {gooddata_eval-1.74.1.dev3 → gooddata_eval-1.74.1.dev5}/tox.ini +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: gooddata-eval
3
- Version: 1.74.1.dev3
3
+ Version: 1.74.1.dev5
4
4
  Summary: Evaluate the GoodData AI agent against your own questions and models.
5
5
  Project-URL: Source, https://github.com/gooddata/gooddata-python-sdk
6
6
  Author-email: GoodData <support@gooddata.com>
@@ -17,7 +17,7 @@ Classifier: Topic :: Scientific/Engineering
17
17
  Classifier: Topic :: Software Development
18
18
  Classifier: Typing :: Typed
19
19
  Requires-Python: >=3.10
20
- Requires-Dist: gooddata-sdk~=1.74.1.dev3
20
+ Requires-Dist: gooddata-sdk~=1.74.1.dev5
21
21
  Requires-Dist: httpx<1.0,>=0.27
22
22
  Requires-Dist: orjson<4.0.0,>=3.9.15
23
23
  Requires-Dist: pydantic<3.0,>=2.6
@@ -1,7 +1,7 @@
1
1
  # (C) 2026 GoodData Corporation
2
2
  [project]
3
3
  name = "gooddata-eval"
4
- version = "1.74.1.dev3"
4
+ version = "1.74.1.dev5"
5
5
  description = "Evaluate the GoodData AI agent against your own questions and models."
6
6
  readme = "README.md"
7
7
  license = "MIT"
@@ -11,7 +11,7 @@ authors = [
11
11
  keywords = ["gooddata", "ai", "evaluation", "llm", "analytics", "cli"]
12
12
  requires-python = ">=3.10"
13
13
  dependencies = [
14
- "gooddata-sdk~=1.74.1.dev3",
14
+ "gooddata-sdk~=1.74.1.dev5",
15
15
  "httpx>=0.27,<1.0",
16
16
  "orjson>=3.9.15,<4.0.0",
17
17
  "pydantic>=2.6,<3.0",
@@ -477,6 +477,8 @@ class AlertRunResult:
477
477
  alert_id: str | None
478
478
  eval: AlertEvaluation
479
479
  actual_alert_arguments: dict
480
+ total_turns: int = 0
481
+ total_steps: int = 0
480
482
  reasoning_steps: list[str] = field(default_factory=list)
481
483
  response_id: str | None = None
482
484
  tool_call_events: list[ToolCallEvent] = field(default_factory=list)
@@ -676,9 +678,13 @@ def run_agentic_alert_skill(
676
678
  # Roles follow GPT-4o's perspective: "assistant"=agent text, "user"=sim-user reply.
677
679
  conversation_history: list = []
678
680
  current_question = question
681
+ turns = 0
682
+ steps = 0
679
683
 
680
684
  for _iteration in range(max_iterations):
681
685
  chat_result = client.send_message(conv_id, current_question)
686
+ turns += 1
687
+ steps += chat_result.reasoning_step_count
682
688
  reasoning_steps.extend(chat_result.reasoning_steps or [])
683
689
  response_id = chat_result.response_id or response_id
684
690
  turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
@@ -728,6 +734,8 @@ def run_agentic_alert_skill(
728
734
  alert_id=alert_id,
729
735
  eval=ev,
730
736
  actual_alert_arguments=actual_args,
737
+ total_turns=turns,
738
+ total_steps=steps,
731
739
  reasoning_steps=reasoning_steps,
732
740
  response_id=response_id,
733
741
  tool_call_events=all_tool_call_events,
@@ -851,6 +859,8 @@ def evaluate_agentic_alert_skill(
851
859
  with ctx.observe(pt, run_idx, conversation_id=run.conversation_id, output=strict_checks) as tid:
852
860
  for score_name, value in strict_checks.items():
853
861
  ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
862
+ ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
863
+ ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
854
864
  log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k)
855
865
  ctx.quality(
856
866
  tid,
@@ -336,6 +336,7 @@ class ConversationResult:
336
336
  full_skill_coverage: bool
337
337
  conversation_success: bool
338
338
  total_clarification_turns: int
339
+ total_steps: int = 0
339
340
  reasoning_steps: list[str] = field(default_factory=list)
340
341
  response_id: str | None = None
341
342
  tool_call_events: list[ToolCallEvent] = field(default_factory=list)
@@ -365,6 +366,7 @@ def run_agentic_conversation(
365
366
  turn_results: list[TurnResult] = []
366
367
  turn_outputs: dict[str, dict] = {}
367
368
  total_clarification_turns = 0
369
+ total_steps = 0
368
370
  conversation_id: str = ""
369
371
  owns_conversation = False
370
372
  # Metrics created during this conversation, deleted after it completes so they do
@@ -434,6 +436,7 @@ def run_agentic_conversation(
434
436
  for _iter in range(max_clarification_turns + 1):
435
437
  chat_result = client.send_message(conversation_id, current_message)
436
438
  final_result = chat_result
439
+ total_steps += chat_result.reasoning_step_count
437
440
  turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
438
441
  chat_result,
439
442
  turn_offset=turn_offset,
@@ -522,6 +525,7 @@ def run_agentic_conversation(
522
525
  full_skill_coverage=full_skill_coverage,
523
526
  conversation_success=conversation_success,
524
527
  total_clarification_turns=total_clarification_turns,
528
+ total_steps=total_steps,
525
529
  reasoning_steps=reasoning_steps,
526
530
  response_id=response_id,
527
531
  tool_call_events=conversation_tool_call_events,
@@ -611,6 +615,22 @@ def evaluate_agentic_conversation(
611
615
  value=float(result.full_skill_coverage),
612
616
  data_type="BOOLEAN",
613
617
  )
618
+ # One turn per fixture turn, plus every simulated-user round the agent triggered.
619
+ # The clarification count alone hides how much of the conversation the fixture
620
+ # asked for, so the comparison needs the total.
621
+ ctx.score(
622
+ tid,
623
+ name="turns",
624
+ value=len(result.turn_results) + result.total_clarification_turns,
625
+ data_type="NUMERIC",
626
+ )
627
+ ctx.score(tid, name="steps", value=result.total_steps, data_type="NUMERIC")
628
+ ctx.score(
629
+ tid,
630
+ name="clarification_turns",
631
+ value=result.total_clarification_turns,
632
+ data_type="NUMERIC",
633
+ )
614
634
  for tr in result.turn_results:
615
635
  ctx.score(
616
636
  tid,
@@ -185,6 +185,8 @@ class KdaRunResult:
185
185
  # Wall-clock time of the turn that called create (None if create never happened) --
186
186
  # not any earlier disambiguation turn. See run_agentic_kda_skill's _run_once.
187
187
  turn_wall_clock_sec: float | None = None
188
+ total_turns: int = 0
189
+ total_steps: int = 0
188
190
  reasoning_steps: list[str] = field(default_factory=list)
189
191
  response_id: str | None = None
190
192
  tool_call_events: list[ToolCallEvent] = field(default_factory=list)
@@ -276,6 +278,9 @@ def run_agentic_kda_skill(
276
278
  all_tool_call_events.extend(result.tool_call_events or [])
277
279
  all_reasoning_step_events.extend(result.reasoning_step_events or [])
278
280
 
281
+ turns = 0
282
+ steps = 0
283
+
279
284
  for iteration in range(max_iterations):
280
285
  try:
281
286
  chat_result = client.send_message(conv_id, current_question)
@@ -291,6 +296,8 @@ def run_agentic_kda_skill(
291
296
  turn_wall_clock_sec = partial.turn_wall_clock_sec
292
297
  turn_completed = False
293
298
  break
299
+ turns += 1
300
+ steps += chat_result.reasoning_step_count
294
301
  reasoning_steps.extend(chat_result.reasoning_steps or [])
295
302
  response_id = chat_result.response_id or response_id
296
303
  _accumulate(chat_result)
@@ -330,6 +337,8 @@ def run_agentic_kda_skill(
330
337
  actual_create_args=create_args,
331
338
  actual_execute_result=execute_result,
332
339
  turn_wall_clock_sec=turn_wall_clock_sec,
340
+ total_turns=turns,
341
+ total_steps=steps,
333
342
  reasoning_steps=reasoning_steps,
334
343
  response_id=response_id,
335
344
  tool_call_events=all_tool_call_events,
@@ -442,6 +451,8 @@ def evaluate_agentic_kda_skill(
442
451
  for score_name, value in strict_checks.items():
443
452
  ctx.score(tid, name=score_name, value=float(value), data_type="BOOLEAN")
444
453
  ctx.score(tid, name="kda_disambiguated", value=float(ev.disambiguated), data_type="BOOLEAN")
454
+ ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
455
+ ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
445
456
  if turn_wall_clock_sec is not None:
446
457
  # combo_report.py reads this score directly -- no trace re-resolution needed.
447
458
  ctx.score(
@@ -162,7 +162,8 @@ class MetricRunResult:
162
162
  metric_created: bool
163
163
  actual_maql: str
164
164
  maql_correct: bool
165
- total_turns: float
165
+ total_turns: int
166
+ total_steps: int = 0
166
167
  reasoning_steps: list[str] = field(default_factory=list)
167
168
  response_id: str | None = None
168
169
  tool_call_events: list[ToolCallEvent] = field(default_factory=list)
@@ -253,6 +254,7 @@ def _execute_single_metric_run(
253
254
  metric_result: dict | None = None
254
255
  created_metric_ids: list[str] = []
255
256
  turns = 0
257
+ steps = 0
256
258
  current_question = question
257
259
  reasoning_steps: list[str] = []
258
260
  response_id: str | None = None
@@ -280,6 +282,7 @@ def _execute_single_metric_run(
280
282
  )
281
283
  all_tool_call_events.extend(chat_result.tool_call_events or [])
282
284
  all_reasoning_step_events.extend(chat_result.reasoning_step_events or [])
285
+ steps += chat_result.reasoning_step_count
283
286
  for metric_id in _extract_created_metric_ids(chat_result.tool_call_events or []):
284
287
  if metric_id not in created_metric_ids:
285
288
  created_metric_ids.append(metric_id)
@@ -330,7 +333,8 @@ def _execute_single_metric_run(
330
333
  metric_created=metric_created,
331
334
  actual_maql=actual_maql,
332
335
  maql_correct=maql_correct,
333
- total_turns=float(turns),
336
+ total_turns=turns,
337
+ total_steps=steps,
334
338
  reasoning_steps=reasoning_steps,
335
339
  response_id=response_id,
336
340
  tool_call_events=all_tool_call_events,
@@ -466,6 +470,8 @@ def evaluate_agentic_metric_skill(
466
470
  ) as tid:
467
471
  ctx.score(tid, name="metric_created", value=float(run.metric_created), data_type="BOOLEAN")
468
472
  ctx.score(tid, name="maql_correct", value=float(run.maql_correct), data_type="BOOLEAN")
473
+ ctx.score(tid, name="turns", value=run.total_turns, data_type="NUMERIC")
474
+ ctx.score(tid, name="steps", value=run.total_steps, data_type="NUMERIC")
469
475
  log_gate_scores(ctx, tid, gate=gate, pass_at_k=summary.pass_at_k, pass_power_k=summary.pass_power_k)
470
476
  ctx.quality(
471
477
  tid,
@@ -59,8 +59,8 @@ class RunResult:
59
59
  actual_output: CreatedVisualization | None
60
60
  eval_result: EvaluationResult
61
61
  best_expected: CreatedVisualization
62
- total_turns: float
63
- total_steps: float
62
+ total_turns: int
63
+ total_steps: int
64
64
  reasoning_steps: list[str] = field(default_factory=list)
65
65
  response_id: str | None = None
66
66
  tool_call_events: list[ToolCallEvent] = field(default_factory=list)
@@ -186,8 +186,8 @@ def _execute_single_run(
186
186
  max_iterations: int = _DEFAULT_MAX_ITERATIONS,
187
187
  ) -> RunResult:
188
188
  """Drive one full multi-turn conversation and evaluate the result."""
189
- total_turns = 0.0
190
- total_steps = 0.0
189
+ total_turns = 0
190
+ total_steps = 0
191
191
  all_tool_call_events: list[ToolCallEvent] = []
192
192
  all_reasoning_step_events: list[ReasoningStepEvent] = []
193
193
  reasoning_steps: list[str] = []
@@ -200,8 +200,8 @@ def _execute_single_run(
200
200
  current_result = client.send_message(conversation_id, question)
201
201
 
202
202
  for iteration in range(max_iterations):
203
- total_turns += 1.0
204
- total_steps += float(current_result.reasoning_step_count)
203
+ total_turns += 1
204
+ total_steps += current_result.reasoning_step_count
205
205
  turn_offset, tool_index_offset, reasoning_index_offset = shift_and_index_events(
206
206
  current_result,
207
207
  turn_offset=turn_offset,
@@ -996,3 +996,67 @@ def test_every_gen_ai_interval_is_accepted():
996
996
  assert AnomalyDetectionGranularity.parse(value.lower()) is AnomalyDetectionGranularity(value)
997
997
  assert AnomalyDetectionGranularity.parse(None) is None
998
998
  assert AnomalyDetectionGranularity.parse(" ") is None
999
+
1000
+
1001
+ def test_run_agentic_alert_skill_counts_the_turns_and_reasoning_steps_it_used():
1002
+ """QA-29110: the effort comparison reads these. A refusal still took a turn, and the turn
1003
+ count is what separates a wrong answer from a run max_iterations cut short."""
1004
+ mock_client = MagicMock()
1005
+ mock_client.create_conversation.return_value = "conv-1"
1006
+ mock_client.send_message.return_value = _no_alert_chat_result()
1007
+ mock_client._base = "http://host/api/v1/actions/workspaces/ws1/ai"
1008
+ mock_client._auth = {"Authorization": "Bearer tok"}
1009
+
1010
+ with _patched(mock_client, simulated_reply="Yes please"):
1011
+ summary = run_agentic_alert_skill(
1012
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1013
+ token="tok",
1014
+ workspace_id="ws1",
1015
+ question="Create alert",
1016
+ expected_output={"operator": "GREATER_THAN", "threshold": 100},
1017
+ k=1,
1018
+ max_iterations=2,
1019
+ )
1020
+
1021
+ # _no_alert_chat_result has no tool calls and non-empty text, so the run replies once and
1022
+ # stops at max_iterations: 2 turns, 1 reasoning step each.
1023
+ assert summary.best.total_turns == 2
1024
+ assert summary.best.total_steps == 2
1025
+
1026
+
1027
+ def test_alert_skill_writes_the_turn_and_step_counts_to_langfuse():
1028
+ """The counters exist to reach Langfuse; asserting only the dataclass would pass even if
1029
+ the scores were never written."""
1030
+ mock_client = MagicMock()
1031
+ mock_client.create_conversation.return_value = "conv-1"
1032
+ mock_client.send_message.return_value = _no_alert_chat_result()
1033
+ mock_client._base = "http://host/api/v1/actions/workspaces/ws1/ai"
1034
+ mock_client._auth = {"Authorization": "Bearer tok"}
1035
+ captured = {}
1036
+
1037
+ def _capture(_submit, _identity, **kwargs):
1038
+ captured["write_scores"] = kwargs["write_scores"]
1039
+
1040
+ with (
1041
+ _patched(mock_client),
1042
+ patch("gooddata_eval.core.agentic.alert_skill.submit_trace_scoring", _capture),
1043
+ pytest.raises(AlertSkillAssertionError),
1044
+ ):
1045
+ evaluate_agentic_alert_skill(
1046
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1047
+ token="tok",
1048
+ workspace_id="ws1",
1049
+ question="Create alert",
1050
+ expected_output={"operator": "GREATER_THAN", "threshold": 100},
1051
+ k=1,
1052
+ max_iterations=1,
1053
+ langfuse=MagicMock(),
1054
+ dataset_item_id="item-1",
1055
+ )
1056
+
1057
+ ctx = MagicMock()
1058
+ captured["write_scores"](ctx)
1059
+ scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
1060
+
1061
+ assert scores["turns"] == 1
1062
+ assert scores["steps"] == 1
@@ -1025,3 +1025,107 @@ def test_evaluate_agentic_conversation_attaches_reasoning_steps_to_exception_on_
1025
1025
  ],
1026
1026
  "latency_breakdown": [],
1027
1027
  }
1028
+
1029
+
1030
+ def test_run_agentic_conversation_sums_the_reasoning_steps_of_every_turn():
1031
+ """QA-29110: the effort comparison reads `steps`. A clarification round is part of the
1032
+ work the effort setting changes, so its steps count with the rest."""
1033
+ proposal_turn = ChatResult.model_validate(
1034
+ {
1035
+ "text_response": None,
1036
+ "alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
1037
+ "reasoningStepCount": 2,
1038
+ "toolCallEvents": [
1039
+ {"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
1040
+ {"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
1041
+ ],
1042
+ }
1043
+ )
1044
+ created_turn = ChatResult.model_validate(
1045
+ {
1046
+ "text_response": "Alert created.",
1047
+ "reasoningStepCount": 3,
1048
+ "toolCallEvents": [
1049
+ {"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
1050
+ ],
1051
+ }
1052
+ )
1053
+ mock_client = MagicMock()
1054
+ mock_client.create_conversation.return_value = "conv-1"
1055
+ mock_client.send_message.side_effect = [proposal_turn, created_turn]
1056
+
1057
+ with (
1058
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
1059
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
1060
+ patch(
1061
+ "gooddata_eval.core.agentic.conversation._get_sim_user_response",
1062
+ return_value="Yes, please create it.",
1063
+ ),
1064
+ ):
1065
+ result = run_agentic_conversation(
1066
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1067
+ token="tok",
1068
+ workspace_id="ws1",
1069
+ fixture=_alert_turn_fixture(),
1070
+ )
1071
+
1072
+ assert result.total_steps == 5
1073
+ assert result.total_clarification_turns == 1
1074
+
1075
+
1076
+ def test_conversation_writes_the_turn_step_and_clarification_counts_to_langfuse():
1077
+ """`turns` is not the clarification count: it is one per fixture turn plus every
1078
+ simulated-user round, so a test has to pin the sum rather than either half."""
1079
+ proposal_turn = ChatResult.model_validate(
1080
+ {
1081
+ "text_response": None,
1082
+ "alertProposals": [{"cta": "Should I create this alert?", "recipients": [{"email": "a@b.com"}]}],
1083
+ "reasoningStepCount": 2,
1084
+ "toolCallEvents": [
1085
+ {"functionName": "set_skills", "functionArguments": '{"skills": ["alert"]}', "result": None},
1086
+ {"functionName": "prepare_metric_alert_proposal", "functionArguments": "{}", "result": None},
1087
+ ],
1088
+ }
1089
+ )
1090
+ created_turn = ChatResult.model_validate(
1091
+ {
1092
+ "text_response": "Alert created.",
1093
+ "reasoningStepCount": 3,
1094
+ "toolCallEvents": [
1095
+ {"functionName": "create_metric_alert", "functionArguments": "{}", "result": '{"id": "alert-1"}'}
1096
+ ],
1097
+ }
1098
+ )
1099
+ mock_client = MagicMock()
1100
+ mock_client.create_conversation.return_value = "conv-1"
1101
+ mock_client.send_message.side_effect = [proposal_turn, created_turn]
1102
+ captured = {}
1103
+
1104
+ def _capture(_submit, _identity, **kwargs):
1105
+ captured["write_scores"] = kwargs["write_scores"]
1106
+
1107
+ with (
1108
+ patch("gooddata_eval.core.agentic.conversation.ChatClient", return_value=mock_client),
1109
+ patch("gooddata_eval.core.agentic.conversation.GoodDataSdk"),
1110
+ patch("gooddata_eval.core.agentic.conversation.submit_trace_scoring", _capture),
1111
+ patch(
1112
+ "gooddata_eval.core.agentic.conversation._get_sim_user_response",
1113
+ return_value="Yes, please create it.",
1114
+ ),
1115
+ ):
1116
+ evaluate_agentic_conversation(
1117
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1118
+ token="tok",
1119
+ workspace_id="ws1",
1120
+ fixture=_alert_turn_fixture(),
1121
+ langfuse=MagicMock(),
1122
+ dataset_item_id="item-1",
1123
+ )
1124
+
1125
+ ctx = MagicMock()
1126
+ captured["write_scores"](ctx)
1127
+ scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
1128
+
1129
+ assert scores["clarification_turns"] == 1
1130
+ assert scores["turns"] == 2 # 1 fixture turn + 1 clarification round
1131
+ assert scores["steps"] == 5
@@ -1195,3 +1195,88 @@ def test_evaluate_agentic_kda_skill_preserves_reasoning_from_a_chat_error_partia
1195
1195
 
1196
1196
  assert exc_info.value.reasoning_steps == ["analyzing before cutoff"]
1197
1197
  assert exc_info.value.response_id == "resp-3"
1198
+
1199
+
1200
+ def test_run_agentic_kda_skill_counts_the_turns_and_reasoning_steps_it_used():
1201
+ """QA-29110: the effort comparison reads these. A binary pass/fail cannot separate two
1202
+ efforts on a nightly's sample, while the reasoning-step count moves with the effort."""
1203
+ mock_client = MagicMock()
1204
+ mock_client.create_conversation.return_value = "conv-1"
1205
+ # Turn 1 asks for clarification, turn 2 runs the analysis: 2 turns, 1 step each.
1206
+ mock_client.send_message.side_effect = [
1207
+ _no_kda_chat_result("Which metric did you mean?"),
1208
+ _kda_chat_result(success=True),
1209
+ ]
1210
+
1211
+ with (
1212
+ patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
1213
+ patch("gooddata_eval.core.agentic.kda_skill.generate_simulated_kda_response", return_value="Revenue"),
1214
+ ):
1215
+ summary = run_agentic_kda_skill(
1216
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1217
+ token="tok",
1218
+ workspace_id="ws1",
1219
+ question="What drove the change?",
1220
+ expected_output=_EXPECTED,
1221
+ k=1,
1222
+ max_iterations=2,
1223
+ )
1224
+
1225
+ assert summary.best.total_turns == 2
1226
+ assert summary.best.total_steps == 2
1227
+
1228
+
1229
+ def test_run_agentic_kda_skill_reports_no_turns_when_the_first_send_fails():
1230
+ """A run that never got a reply must not report a turn it did not take."""
1231
+ mock_client = MagicMock()
1232
+ mock_client.create_conversation.return_value = "conv-1"
1233
+ mock_client.send_message.side_effect = RuntimeError("stream died")
1234
+
1235
+ with patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client):
1236
+ summary = run_agentic_kda_skill(
1237
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1238
+ token="tok",
1239
+ workspace_id="ws1",
1240
+ question="What drove the change?",
1241
+ expected_output=_EXPECTED,
1242
+ k=1,
1243
+ max_iterations=1,
1244
+ )
1245
+
1246
+ assert summary.best.total_turns == 0
1247
+ assert summary.best.total_steps == 0
1248
+
1249
+
1250
+ def test_kda_skill_writes_the_turn_and_step_counts_to_langfuse():
1251
+ """The counters exist to reach Langfuse; asserting only the dataclass would pass even if
1252
+ the scores were never written."""
1253
+ mock_client = MagicMock()
1254
+ mock_client.create_conversation.return_value = "conv-1"
1255
+ mock_client.send_message.return_value = _kda_chat_result(success=True)
1256
+ captured = {}
1257
+
1258
+ def _capture(_submit, _identity, **kwargs):
1259
+ captured["write_scores"] = kwargs["write_scores"]
1260
+
1261
+ with (
1262
+ patch("gooddata_eval.core.agentic.kda_skill.ChatClient", return_value=mock_client),
1263
+ patch("gooddata_eval.core.agentic.kda_skill.submit_trace_scoring", _capture),
1264
+ ):
1265
+ evaluate_agentic_kda_skill(
1266
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
1267
+ token="tok",
1268
+ workspace_id="ws1",
1269
+ question="What drove the change?",
1270
+ expected_output=_EXPECTED,
1271
+ k=1,
1272
+ max_iterations=1,
1273
+ langfuse=MagicMock(),
1274
+ dataset_item_id="item-1",
1275
+ )
1276
+
1277
+ ctx = MagicMock()
1278
+ captured["write_scores"](ctx)
1279
+ scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
1280
+
1281
+ assert scores["turns"] == 1
1282
+ assert scores["steps"] == 1
@@ -582,7 +582,7 @@ def test_run_agentic_metric_skill_fails_the_run_when_the_simulated_reply_cannot_
582
582
 
583
583
  assert summary.pass_at_k is False
584
584
  assert summary.best.metric_created is False
585
- assert summary.best.total_turns == 1.0
585
+ assert summary.best.total_turns == 1
586
586
  mock_client.close.assert_called_once()
587
587
  mock_sim.assert_called_once_with(
588
588
  "Which brand field should I count?", [{"maql": "SELECT {metric/foo}"}], "Create metric foo"
@@ -764,3 +764,90 @@ def test_no_timer_output_by_default(monkeypatch, capsys):
764
764
  assert "[timer]" not in capsys.readouterr().out
765
765
  # Silenced, not un-measured.
766
766
  assert summary.run_results[0].timings.agent_s == 3.0
767
+
768
+
769
+ def test_run_agentic_metric_skill_counts_the_turns_and_reasoning_steps_it_used():
770
+ """QA-29110: the effort comparison reads these. A clarification round is part of the work
771
+ the effort setting changes, so its steps count with the rest."""
772
+ clarify_turn = ChatResult.model_validate(
773
+ {"textResponse": "Which foo?", "toolCallEvents": [], "reasoningStepCount": 2}
774
+ )
775
+ created_turn = ChatResult.model_validate(
776
+ {
777
+ "textResponse": "done",
778
+ "reasoningStepCount": 3,
779
+ "toolCallEvents": [
780
+ {
781
+ "functionName": "create_metric",
782
+ "functionArguments": "{}",
783
+ "result": '{"data": {"maql": "SELECT {metric/foo}"}}',
784
+ }
785
+ ],
786
+ }
787
+ )
788
+ mock_client = _client()
789
+ mock_client.send_message.side_effect = [clarify_turn, created_turn]
790
+
791
+ with _patched(mock_client, simulated_reply="It's foo"):
792
+ summary = run_agentic_metric_skill(
793
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
794
+ token="tok",
795
+ workspace_id="ws1",
796
+ question="Create metric foo",
797
+ expected_output={"maql": "SELECT {metric/foo}"},
798
+ k=1,
799
+ max_iterations=2,
800
+ )
801
+
802
+ assert summary.best.total_turns == 2
803
+ assert summary.best.total_steps == 5
804
+
805
+
806
+ def test_metric_skill_writes_the_turn_and_step_counts_to_langfuse():
807
+ """The counters exist to reach Langfuse; asserting only the dataclass would pass even if
808
+ the scores were never written."""
809
+ mock_client = _client()
810
+ mock_client.send_message.return_value = ChatResult.model_validate(
811
+ {
812
+ "textResponse": "done",
813
+ "reasoningStepCount": 4,
814
+ "toolCallEvents": [
815
+ {
816
+ "functionName": "create_metric",
817
+ "functionArguments": "{}",
818
+ "result": '{"data": {"maql": "SELECT {metric/foo}"}}',
819
+ }
820
+ ],
821
+ }
822
+ )
823
+ captured = {}
824
+
825
+ def _capture(_submit, _identity, **kwargs):
826
+ captured["write_scores"] = kwargs["write_scores"]
827
+
828
+ with (
829
+ _patched(mock_client),
830
+ patch("gooddata_eval.core.agentic.metric_skill.submit_trace_scoring", _capture),
831
+ ):
832
+ evaluate_agentic_metric_skill(
833
+ host="http://host/api/v1/actions/workspaces/ws1/ai",
834
+ token="tok",
835
+ workspace_id="ws1",
836
+ question="Create metric foo",
837
+ expected_output={"maql": "SELECT {metric/foo}"},
838
+ k=1,
839
+ max_iterations=1,
840
+ langfuse=MagicMock(),
841
+ dataset_item_id="item-1",
842
+ )
843
+
844
+ ctx = MagicMock()
845
+ captured["write_scores"](ctx)
846
+ scores = {c.kwargs["name"]: c.kwargs["value"] for c in ctx.score.call_args_list}
847
+
848
+ assert scores["turns"] == 1
849
+ assert scores["steps"] == 4
850
+ # `==` does not separate 1 from 1.0, so the counts need their type pinned separately:
851
+ # they are counts, and a float reads as though a fraction of a turn were possible.
852
+ assert isinstance(scores["turns"], int)
853
+ assert isinstance(scores["steps"], int)
@@ -71,8 +71,8 @@ def test_execute_single_run_viz_on_first_turn():
71
71
 
72
72
  assert result.eval_result.visualization_created is True
73
73
  assert result.eval_result.strict_pass is True
74
- assert result.total_turns == 1.0
75
- assert result.total_steps == 2.0
74
+ assert result.total_turns == 1
75
+ assert result.total_steps == 2
76
76
  assert result.conversation_id == "conv-1"
77
77
  client.send_message.assert_called_once_with("conv-1", "Show revenue")
78
78
 
@@ -93,7 +93,7 @@ def test_execute_single_run_clarification_then_viz(monkeypatch):
93
93
  result = _execute_single_run(client, "conv-1", "Show me a chart", [_expected()])
94
94
 
95
95
  assert result.eval_result.visualization_created is True
96
- assert result.total_turns == 2.0
96
+ assert result.total_turns == 2
97
97
  assert client.send_message.call_count == 2
98
98
  assert client.send_message.call_args_list[1] == call("conv-1", "Revenue please")
99
99
 
@@ -111,7 +111,7 @@ def test_execute_single_run_no_viz_no_text():
111
111
  result = _execute_single_run(client, "conv-1", "Show revenue", [_expected()])
112
112
 
113
113
  assert result.eval_result.visualization_created is False
114
- assert result.total_turns == 1.0
114
+ assert result.total_turns == 1
115
115
 
116
116
 
117
117
  def test_execute_single_run_max_iterations_stops_loop(monkeypatch):