agentx-python 0.6.36__tar.gz → 0.6.38__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of agentx-python might be problematic. Click here for more details.

Files changed (81) hide show
  1. {agentx_python-0.6.36 → agentx_python-0.6.38}/PKG-INFO +16 -1
  2. {agentx_python-0.6.36 → agentx_python-0.6.38}/README.md +15 -0
  3. agentx_python-0.6.38/agentx/version.py +1 -0
  4. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/PKG-INFO +16 -1
  5. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/SOURCES.txt +1 -0
  6. agentx_python-0.6.38/tests/test_docs_match_sdk.py +176 -0
  7. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_integrations.py +3 -0
  8. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_span_tree.py +11 -1
  9. agentx_python-0.6.36/agentx/version.py +0 -1
  10. {agentx_python-0.6.36 → agentx_python-0.6.38}/LICENSE +0 -0
  11. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/__init__.py +0 -0
  12. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/agentx.py +0 -0
  13. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/cli.py +0 -0
  14. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/__init__.py +0 -0
  15. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/_term.py +0 -0
  16. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/adapters/__init__.py +0 -0
  17. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/adapters/http_endpoint.py +0 -0
  18. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/adapters/precomputed.py +0 -0
  19. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/adapters/raw.py +0 -0
  20. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/client.py +0 -0
  21. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/datasets.py +0 -0
  22. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/evaluation_settings.py +0 -0
  23. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/models.py +0 -0
  24. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/prompts.py +0 -0
  25. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/reporting.py +0 -0
  26. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/results.py +0 -0
  27. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/runner.py +0 -0
  28. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/tool_schemas.py +0 -0
  29. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/evaluations/tracing.py +0 -0
  30. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/exceptions.py +0 -0
  31. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/export.py +0 -0
  32. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/feedback.py +0 -0
  33. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/__init__.py +0 -0
  34. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/_traced_call.py +0 -0
  35. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/anthropic.py +0 -0
  36. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/autogen.py +0 -0
  37. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/crewai.py +0 -0
  38. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/databricks.py +0 -0
  39. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/google_adk.py +0 -0
  40. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/google_genai.py +0 -0
  41. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/langchain.py +0 -0
  42. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/litellm.py +0 -0
  43. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/llamaindex.py +0 -0
  44. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/moveworks.py +0 -0
  45. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/openai.py +0 -0
  46. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/integrations/openai_agents.py +0 -0
  47. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/__init__.py +0 -0
  48. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/agents.py +0 -0
  49. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/client.py +0 -0
  50. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/judge_scorers.py +0 -0
  51. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/models.py +0 -0
  52. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/online_evaluators.py +0 -0
  53. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/patterns.py +0 -0
  54. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/profile.py +0 -0
  55. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/scorers.py +0 -0
  56. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/sessions.py +0 -0
  57. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/monitor/signals.py +0 -0
  58. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/outcomes.py +0 -0
  59. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/projects.py +0 -0
  60. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/py.typed +0 -0
  61. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/resources/__init__.py +0 -0
  62. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/resources/agent.py +0 -0
  63. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/resources/conversation.py +0 -0
  64. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/resources/workforce.py +0 -0
  65. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/traces.py +0 -0
  66. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/tracing/__init__.py +0 -0
  67. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/tracing/ci_types.py +0 -0
  68. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/tracing/ingest_client.py +0 -0
  69. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/tracing/tracer.py +0 -0
  70. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx/util.py +0 -0
  71. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/dependency_links.txt +0 -0
  72. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/entry_points.txt +0 -0
  73. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/not-zip-safe +0 -0
  74. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/requires.txt +0 -0
  75. {agentx_python-0.6.36 → agentx_python-0.6.38}/agentx_python.egg-info/top_level.txt +0 -0
  76. {agentx_python-0.6.36 → agentx_python-0.6.38}/setup.cfg +0 -0
  77. {agentx_python-0.6.36 → agentx_python-0.6.38}/setup.py +0 -0
  78. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_deep_dive_fixes.py +0 -0
  79. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_integration.py +0 -0
  80. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_judge_scorers.py +0 -0
  81. {agentx_python-0.6.36 → agentx_python-0.6.38}/tests/test_selfhost_analysis_fallback.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.6.36
3
+ Version: 0.6.38
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -315,6 +315,21 @@ evaluator = client.monitor.online_evaluators.builder(
315
315
  client.monitor.online_evaluators.ratings(evaluator.id, window="7d")
316
316
  ```
317
317
 
318
+ An online evaluator and a dataset run's grading config are two profiles of one **LLM Judge Scorer**, and `client.monitor.judge_scorers` manages both in a single call - one rubric, one id, used offline and online:
319
+
320
+ ```python
321
+ scorer = client.monitor.judge_scorers.builder(
322
+ name="Helpfulness",
323
+ acceptance_criteria="Concrete, correct, cites the policy.",
324
+ live=True, sample_rate=0.1, alert_threshold=5, # the online profile
325
+ ).publish()
326
+
327
+ client.evaluations.run(dataset_id=dataset.id, subject={...}, scorer_id=scorer.id)
328
+ client.monitor.judge_scorers.ratings(scorer.id, window="7d")
329
+ ```
330
+
331
+ The two clients above are that entity's per-profile legacy views; they keep working and share the same ids. See [LLM Judge Scorers](EVALUATIONS.md#llm-judge-scorers---reusable-grading-configs) for the full surface, including which self-host engine builds serve it.
332
+
318
333
  An online evaluator can also judge **whole conversations** instead of single traces: pass `scope="session"` and the engine scores each multi-turn session once it's been idle for `idle_seconds`, re-scoring if the conversation resumes. And to close the loop with reality, two ground-truth streams feed the dashboard's Judge Calibration view (which measures how often AgentX's automated verdicts agree with what actually happened): `client.outcomes.report(...)` for after-the-fact system results (a reopened ticket, a human confirmation), and `client.feedback.report(...)` for end-user votes forwarded from your own app's UI - a "down" raises a "Negative user feedback" signal directly, no sampling or judge call involved. All self-host features.
319
334
 
320
335
  ```python
@@ -251,6 +251,21 @@ evaluator = client.monitor.online_evaluators.builder(
251
251
  client.monitor.online_evaluators.ratings(evaluator.id, window="7d")
252
252
  ```
253
253
 
254
+ An online evaluator and a dataset run's grading config are two profiles of one **LLM Judge Scorer**, and `client.monitor.judge_scorers` manages both in a single call - one rubric, one id, used offline and online:
255
+
256
+ ```python
257
+ scorer = client.monitor.judge_scorers.builder(
258
+ name="Helpfulness",
259
+ acceptance_criteria="Concrete, correct, cites the policy.",
260
+ live=True, sample_rate=0.1, alert_threshold=5, # the online profile
261
+ ).publish()
262
+
263
+ client.evaluations.run(dataset_id=dataset.id, subject={...}, scorer_id=scorer.id)
264
+ client.monitor.judge_scorers.ratings(scorer.id, window="7d")
265
+ ```
266
+
267
+ The two clients above are that entity's per-profile legacy views; they keep working and share the same ids. See [LLM Judge Scorers](EVALUATIONS.md#llm-judge-scorers---reusable-grading-configs) for the full surface, including which self-host engine builds serve it.
268
+
254
269
  An online evaluator can also judge **whole conversations** instead of single traces: pass `scope="session"` and the engine scores each multi-turn session once it's been idle for `idle_seconds`, re-scoring if the conversation resumes. And to close the loop with reality, two ground-truth streams feed the dashboard's Judge Calibration view (which measures how often AgentX's automated verdicts agree with what actually happened): `client.outcomes.report(...)` for after-the-fact system results (a reopened ticket, a human confirmation), and `client.feedback.report(...)` for end-user votes forwarded from your own app's UI - a "down" raises a "Negative user feedback" signal directly, no sampling or judge call involved. All self-host features.
255
270
 
256
271
  ```python
@@ -0,0 +1 @@
1
+ VERSION = "0.6.38"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.6.36
3
+ Version: 0.6.38
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -315,6 +315,21 @@ evaluator = client.monitor.online_evaluators.builder(
315
315
  client.monitor.online_evaluators.ratings(evaluator.id, window="7d")
316
316
  ```
317
317
 
318
+ An online evaluator and a dataset run's grading config are two profiles of one **LLM Judge Scorer**, and `client.monitor.judge_scorers` manages both in a single call - one rubric, one id, used offline and online:
319
+
320
+ ```python
321
+ scorer = client.monitor.judge_scorers.builder(
322
+ name="Helpfulness",
323
+ acceptance_criteria="Concrete, correct, cites the policy.",
324
+ live=True, sample_rate=0.1, alert_threshold=5, # the online profile
325
+ ).publish()
326
+
327
+ client.evaluations.run(dataset_id=dataset.id, subject={...}, scorer_id=scorer.id)
328
+ client.monitor.judge_scorers.ratings(scorer.id, window="7d")
329
+ ```
330
+
331
+ The two clients above are that entity's per-profile legacy views; they keep working and share the same ids. See [LLM Judge Scorers](EVALUATIONS.md#llm-judge-scorers---reusable-grading-configs) for the full surface, including which self-host engine builds serve it.
332
+
318
333
  An online evaluator can also judge **whole conversations** instead of single traces: pass `scope="session"` and the engine scores each multi-turn session once it's been idle for `idle_seconds`, re-scoring if the conversation resumes. And to close the loop with reality, two ground-truth streams feed the dashboard's Judge Calibration view (which measures how often AgentX's automated verdicts agree with what actually happened): `client.outcomes.report(...)` for after-the-fact system results (a reopened ticket, a human confirmation), and `client.feedback.report(...)` for end-user votes forwarded from your own app's UI - a "down" raises a "Negative user feedback" signal directly, no sampling or judge call involved. All self-host features.
319
334
 
320
335
  ```python
@@ -70,6 +70,7 @@ agentx_python.egg-info/not-zip-safe
70
70
  agentx_python.egg-info/requires.txt
71
71
  agentx_python.egg-info/top_level.txt
72
72
  tests/test_deep_dive_fixes.py
73
+ tests/test_docs_match_sdk.py
73
74
  tests/test_integration.py
74
75
  tests/test_integrations.py
75
76
  tests/test_judge_scorers.py
@@ -0,0 +1,176 @@
1
+ """Hold the documentation against the SDK it documents.
2
+
3
+ 0.6.36 consolidated grading configs into an LLM Judge Scorer and shipped with no doc
4
+ coverage at all: EVALUATIONS.md, TRACING.md, README.md and CICD_EVAL.md had no mention of
5
+ a judge scorer between them. Nothing caught that, because nothing checks the docs.
6
+
7
+ This does. It extracts the fenced python from the docs and resolves what they show
8
+ against the installed package - the methods, the keyword arguments, and the
9
+ cross-document links. A renamed kwarg or a dropped method is a failing test here rather
10
+ than a copy-pasted snippet that raises TypeError in someone's project.
11
+
12
+ Prose is deliberately out of scope; only fenced ```python blocks are read, which is what
13
+ a reader actually copies.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import inspect
19
+ import re
20
+ from pathlib import Path
21
+
22
+ import pytest
23
+
24
+ from agentx.evaluations.client import EvaluationsClient, _resolve_scorer_id
25
+ from agentx.evaluations.runner import EvaluationsRunner
26
+ from agentx.monitor.judge_scorers import JudgeScorerBuilder, JudgeScorersClient
27
+
28
+ ROOT = Path(__file__).resolve().parent.parent
29
+ DOCS = ("EVALUATIONS.md", "TRACING.md", "README.md", "CICD_EVAL.md")
30
+
31
+ BUILDER_PARAMS = set(inspect.signature(JudgeScorersClient.builder).parameters) - {"self"}
32
+ RUN_PARAMS = set(inspect.signature(EvaluationsRunner.run).parameters) - {"self"}
33
+
34
+ # A regex that quietly stops matching turns this file into a green light wired to nothing,
35
+ # so each sweep asserts it still found roughly what it found when written.
36
+ MIN_JUDGE_SCORER_CALLS = 10
37
+ MIN_BUILDER_KEYWORDS = 10
38
+ MIN_RUN_KEYWORDS = 3
39
+
40
+
41
+ def _fenced_python(text: str) -> str:
42
+ return "\n".join(re.findall(r"```python\n(.*?)```", text, re.S))
43
+
44
+
45
+ def _calls(src: str, prefix: str) -> list[tuple[str, str]]:
46
+ """(method, argument-text) for each ``<prefix>.<method>(...)``, parens balanced.
47
+
48
+ The docs are not importable - they carry placeholders like ``subject={...}`` - so this
49
+ reads them the way a reader does, by eye, rather than by parsing them as Python.
50
+ """
51
+ found = []
52
+ for match in re.finditer(re.escape(prefix) + r"\s*\.\s*(\w+)\(", src):
53
+ open_paren = match.end() - 1
54
+ depth, index = 0, open_paren
55
+ while index < len(src):
56
+ if src[index] == "(":
57
+ depth += 1
58
+ elif src[index] == ")":
59
+ depth -= 1
60
+ if depth == 0:
61
+ break
62
+ index += 1
63
+ found.append((match.group(1), src[open_paren + 1 : index]))
64
+ return found
65
+
66
+
67
+ def _keywords(argument_text: str) -> list[str]:
68
+ """Keyword names in a call's argument text, ignoring keys inside a value's own dict."""
69
+ return re.findall(r"(?:^|[(,]\s*|\n\s*)(\w+)\s*=", argument_text)
70
+
71
+
72
+ def _documented() -> dict[str, str]:
73
+ return {name: _fenced_python((ROOT / name).read_text()) for name in DOCS}
74
+
75
+
76
+ def _judge_scorer_calls() -> list[tuple[str, str, str]]:
77
+ return [
78
+ (doc, method, args)
79
+ for doc, src in _documented().items()
80
+ for method, args in _calls(src, "client.monitor.judge_scorers")
81
+ ]
82
+
83
+
84
+ def test_documented_judge_scorer_methods_exist():
85
+ calls = _judge_scorer_calls()
86
+ assert len(calls) >= MIN_JUDGE_SCORER_CALLS, (
87
+ f"only found {len(calls)} judge_scorers calls in the docs - the sweep is no longer "
88
+ "finding what it should, or the surface stopped being documented"
89
+ )
90
+ missing = [
91
+ f"{doc}: client.monitor.judge_scorers.{method}()"
92
+ for doc, method, _ in calls
93
+ if not hasattr(JudgeScorersClient, method)
94
+ ]
95
+ assert not missing, "documented but not on JudgeScorersClient: " + ", ".join(missing)
96
+
97
+
98
+ def test_documented_builder_keywords_are_real_parameters():
99
+ keywords = [
100
+ (doc, keyword)
101
+ for doc, method, args in _judge_scorer_calls()
102
+ if method == "builder"
103
+ for keyword in _keywords(args)
104
+ ]
105
+ assert len(keywords) >= MIN_BUILDER_KEYWORDS, (
106
+ f"only found {len(keywords)} builder keywords in the docs - the sweep is no longer "
107
+ "finding what it should"
108
+ )
109
+ unknown = [f"{doc}: builder({kw}=...)" for doc, kw in keywords if kw not in BUILDER_PARAMS]
110
+ assert not unknown, "documented but not a builder parameter: " + ", ".join(unknown)
111
+
112
+
113
+ def test_documented_run_keywords_are_real_parameters():
114
+ keywords = [
115
+ (doc, keyword)
116
+ for doc, src in _documented().items()
117
+ for method, args in _calls(src, "client.evaluations")
118
+ if method == "run"
119
+ for keyword in _keywords(args)
120
+ ]
121
+ assert len(keywords) >= MIN_RUN_KEYWORDS, (
122
+ f"only found {len(keywords)} evaluations.run keywords in the docs - the sweep is no "
123
+ "longer finding what it should"
124
+ )
125
+ unknown = [f"{doc}: run({kw}=...)" for doc, kw in keywords if kw not in RUN_PARAMS]
126
+ assert not unknown, "documented but not a run() parameter: " + ", ".join(unknown)
127
+
128
+
129
+ def test_builder_publish_is_documented_and_real():
130
+ """Every builder example ends in .publish(); it has to be there."""
131
+ assert any(".publish()" in src for src in _documented().values())
132
+ assert hasattr(JudgeScorerBuilder, "publish")
133
+
134
+
135
+ def test_legacy_views_still_exist():
136
+ """The docs tell readers the pre-consolidation clients keep working. They must."""
137
+ assert hasattr(EvaluationsClient, "settings")
138
+ from agentx.monitor.online_evaluators import MonitorOnlineEvaluatorClient
139
+
140
+ assert hasattr(MonitorOnlineEvaluatorClient, "builder")
141
+
142
+
143
+ def test_both_grader_spellings_resolve_to_one_id():
144
+ """EVALUATIONS.md states scorer_id and evaluation_settings_id are the same id, and
145
+ that passing two different ids raises. Both halves are load-bearing for readers
146
+ choosing which to write."""
147
+ assert "scorer_id" in RUN_PARAMS and "evaluation_settings_id" in RUN_PARAMS
148
+ assert _resolve_scorer_id("abc", None) == "abc"
149
+ assert _resolve_scorer_id(None, "abc") == "abc"
150
+ assert _resolve_scorer_id("abc", "abc") == "abc"
151
+ with pytest.raises(ValueError):
152
+ _resolve_scorer_id("abc", "def")
153
+
154
+
155
+ @pytest.mark.parametrize(
156
+ "link, target, heading",
157
+ [
158
+ (
159
+ "EVALUATIONS.md#llm-judge-scorers---reusable-grading-configs",
160
+ "EVALUATIONS.md",
161
+ "### LLM Judge Scorers - reusable grading configs",
162
+ ),
163
+ (
164
+ "TRACING.md#clientmonitoronline_evaluators-self-host-only",
165
+ "TRACING.md",
166
+ "### `client.monitor.online_evaluators` (self-host only)",
167
+ ),
168
+ ],
169
+ )
170
+ def test_cross_document_links_resolve(link, target, heading):
171
+ """A renamed heading silently breaks every link pointing at it."""
172
+ linking = [name for name in DOCS if link in (ROOT / name).read_text()]
173
+ assert linking, f"nothing links to {link} any more - drop this case or fix the link"
174
+ assert heading in (ROOT / target).read_text(), (
175
+ f"{linking} link to {link}, but {target} has no heading rendering to that anchor"
176
+ )
@@ -46,6 +46,9 @@ def make_tracer() -> Tracer:
46
46
  # ---------------------------------------------------------------------------
47
47
 
48
48
  def test_adk_model_error_is_captured():
49
+ # google-adk is an optional extra; skip like the crewai/litellm/llamaindex/autogen tests
50
+ # below rather than failing on ImportError (importing the module raises when it is absent).
51
+ pytest.importorskip("google.adk")
49
52
  from agentx.integrations.google_adk import AgentXADKPlugin
50
53
 
51
54
  tracer = make_tracer()
@@ -486,6 +486,9 @@ def test_google_adk_emits_real_child_spans():
486
486
  import asyncio
487
487
  import types
488
488
 
489
+ # google-adk is an optional extra; skip like every other integration test here rather
490
+ # than failing on ImportError (importing the module raises when it is absent).
491
+ pytest.importorskip("google.adk")
489
492
  from agentx.integrations.google_adk import AgentXADKPlugin
490
493
 
491
494
  tracer = make_tracer()
@@ -529,6 +532,7 @@ def test_google_adk_model_error_is_captured():
529
532
  import asyncio
530
533
  import types
531
534
 
535
+ pytest.importorskip("google.adk")
532
536
  from agentx.integrations.google_adk import AgentXADKPlugin
533
537
 
534
538
  tracer = make_tracer()
@@ -581,7 +585,13 @@ def test_trace_tool_call_emits_real_child_span():
581
585
  assert child["name"] == "policy_lookup"
582
586
  assert child["parent_span_id"] == root["span_id"]
583
587
  assert child["output"] == "digital purchases are final"
584
- assert "tool_calls" not in root or root.get("tool_calls") in (None, [])
588
+ # ...and a summary of it also lands on the ROOT's flat tool_calls list. That dual-write is
589
+ # deliberate (40c6f6e): the child span feeds the trace detail's span tree, while the
590
+ # engine's built-in "Tool failure" check and the dashboard's Tool quality column read the
591
+ # root's flat list - before it, a failed trace_tool_call() was invisible to both surfaces.
592
+ # Asserted positively, and as exactly one entry, so a third write shows up here too.
593
+ assert [tc["name"] for tc in root["tool_calls"]] == ["policy_lookup"]
594
+ assert root["tool_calls"][0]["output"] == "digital purchases are final"
585
595
 
586
596
 
587
597
  def test_trace_retrieval_emits_real_child_span():
@@ -1 +0,0 @@
1
- VERSION = "0.6.36"
File without changes
File without changes
File without changes