agentx-python 0.8.16__py3-none-any.whl → 0.8.17__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -48,7 +48,7 @@ class DatasetBuilder:
48
48
  "questions": [],
49
49
  }
50
50
  # LLM-as-judge overrides for this dataset's own grading config. Omit either to keep the
51
- # server default (raw prompt template / OpenAI gpt-5.5, see EVALUATIONS.md). judge_model
51
+ # server default (raw prompt template / gpt-5.6-luna, see EVALUATIONS.md). judge_model
52
52
  # must be one of client.evaluations.list_models() (OpenAI or Anthropic).
53
53
  if judge_prompt is not None:
54
54
  self._payload["judgePrompt"] = judge_prompt
@@ -45,7 +45,7 @@ class EvaluationSettingsBuilder:
45
45
  "evaluationCriteria": evaluation_criteria,
46
46
  }
47
47
  # LLM-as-judge overrides. Omit either to keep the server default (raw prompt template /
48
- # OpenAI gpt-5.5, see EVALUATIONS.md). judge_model must be one of
48
+ # gpt-5.6-luna, see EVALUATIONS.md). judge_model must be one of
49
49
  # client.evaluations.list_models() (OpenAI or Anthropic).
50
50
  if judge_prompt is not None:
51
51
  self._payload["judgePrompt"] = judge_prompt
@@ -113,8 +113,8 @@ class EvaluationSettings(BaseModel):
113
113
  evaluation_criteria: Optional[str] = Field(default=None, alias="evaluationCriteria")
114
114
  # Custom code scorers on this scorer's offline profile - [{ id, name, code, enabled }].
115
115
  code_scorers: Optional[List[Dict[str, Any]]] = Field(default=None, alias="codeScorers")
116
- # LLM-as-judge overrides. None means "use the server default" (raw prompt template / OpenAI
117
- # gpt-5.5). See client.evaluations.settings.builder(judge_prompt=..., judge_model=...).
116
+ # LLM-as-judge overrides. None means "use the server default" (raw prompt template /
117
+ # gpt-5.6-luna). See client.evaluations.settings.builder(judge_prompt=..., judge_model=...).
118
118
  judge_prompt: Optional[str] = Field(default=None, alias="judgePrompt")
119
119
  judge_model: Optional[str] = Field(default=None, alias="judgeModel")
120
120
  status: str = "published"
@@ -63,9 +63,6 @@ _ANALYSIS_LEVEL_LABELS = {
63
63
  "l4_final_reduce": "writing final report",
64
64
  }
65
65
 
66
- _DEFAULT_JUDGE_MODEL = "gpt-5.5"
67
-
68
-
69
66
  class GateResult:
70
67
  """Wire result of the CI gate (GET /runs/:id/gate) with attribute access for the fields a
71
68
  CI script actually branches on."""
@@ -456,16 +453,16 @@ class EvaluationRunContext:
456
453
  Args:
457
454
  mode: "auto" (default), "sync", or "batch" - how item scoring executes server-side.
458
455
  quality_mode: "quality_first" or "balanced" - how many items get a second/third judge.
459
- judges: 1-3 model ids, e.g. ``["gpt-5.5", "claude-opus-4-8"]``. Defaults to a single
460
- judge, ``["gpt-5.5"]``, rather than the dashboard's 3-judge default - SDK runs are
461
- typically lighter-weight, quick-start evaluations.
456
+ judges: 1-3 model ids from ``client.evaluations.list_models()``, e.g.
457
+ ``["gpt-5.6-luna", "claude-opus-4-8"]``. Omit to let the engine score with its
458
+ platform default model (a single judge, rather than the dashboard's 3-judge
459
+ default - SDK runs are typically lighter-weight, quick-start evaluations).
462
460
  poll_interval: seconds between status checks while waiting.
463
461
  timeout: give up waiting after this many seconds (the job keeps running server-side;
464
462
  call ``get_report()`` later to check on it).
465
463
  """
466
464
  if judges is not None and not (1 <= len(judges) <= 3):
467
465
  raise ValueError("judges must contain 1-3 model ids")
468
- resolved_judges = judges if judges is not None else [_DEFAULT_JUDGE_MODEL]
469
466
 
470
467
  _say()
471
468
  with Spinner("Analyzing - AI is reviewing your results") as spinner:
@@ -474,7 +471,7 @@ class EvaluationRunContext:
474
471
  self._run.run_id,
475
472
  mode=mode,
476
473
  quality_mode=quality_mode,
477
- judges=resolved_judges,
474
+ judges=judges,
478
475
  )
479
476
  deadline = time.monotonic() + timeout
480
477
  status = self._client.get_analysis_status(self._run.run_id)
agentx/monitor/client.py CHANGED
@@ -429,6 +429,24 @@ class MonitorClient:
429
429
  data = self._request("GET", f"/ingest/sessions/{session_id}/spans", base=self._api_root())
430
430
  return data.get("spans", []) if isinstance(data, dict) else data
431
431
 
432
+ def list_session_scores(self, session_id: str) -> List[dict]:
433
+ """Every session-level verdict on the session, newest first: session-scoped online
434
+ evaluators (kind ``online-eval:<id>``), session-scoped scorer groups
435
+ (``scorer-group:<id>``), and legacy coherence rows."""
436
+ data = self._request(
437
+ "GET", f"/agent-monitoring/sessions/{session_id}/scores", base=self._api_root()
438
+ )
439
+ return data.get("scores", []) if isinstance(data, dict) else data
440
+
441
+ def run_session_sweep(self) -> dict:
442
+ """Run the idle-session sweep once, now - the tick that scores quiet multi-turn
443
+ sessions with every enabled session-scoped evaluator and scorer group. Production
444
+ engines run this automatically every minute; the manual trigger exists for demos,
445
+ tests, and backfills. Returns ``{"judged": n}``."""
446
+ return self._request(
447
+ "POST", "/agent-monitoring/session-sweep/run", base=self._api_root(), timeout=300
448
+ )
449
+
432
450
  # ------------------------------------------------------------------
433
451
  # Model portability (self-host): replay a trace's input against other models
434
452
  # ------------------------------------------------------------------
@@ -65,7 +65,9 @@ class ScorerGroupsClient:
65
65
  online: Optional[Dict[str, Any]] = None,
66
66
  ) -> ScorerGroup:
67
67
  """``members``: [{"kind": "judge"|"pattern"|"custom", "refId": ..., "weight": 1, "gate": False}].
68
- ``online``: {"enabled": True, "sampleRate": 0.1, "alertThreshold": 5, "severity": "medium"}
68
+ ``online``: {"enabled": True, "sampleRate": 0.1, "alertThreshold": 5, "severity": "medium"}.
69
+ Add ``"scope": "session", "idleSeconds": 120`` to score whole multi-turn sessions once
70
+ idle, instead of each sampled trace.
69
71
  or None for offline-only."""
70
72
  payload: Dict[str, Any] = {"name": name, "members": members}
71
73
  if description is not None:
@@ -19,3 +19,15 @@ class MonitorSessionClient:
19
19
  def spans(self, session_id: str) -> List[dict]:
20
20
  """Every span in the session (roots and children), oldest first."""
21
21
  return self._client.list_session_spans(session_id)
22
+
23
+ def scores(self, session_id: str) -> List[dict]:
24
+ """Session-level verdicts, newest first. ``kind`` says who scored: a session-scoped
25
+ online evaluator (``online-eval:<id>``) or a session-scoped scorer group
26
+ (``scorer-group:<id>``)."""
27
+ return self._client.list_session_scores(session_id)
28
+
29
+ def run_sweep(self) -> dict:
30
+ """Trigger the idle-session sweep once (normally automatic, every minute) - scores
31
+ idle multi-turn sessions with every enabled session-scoped evaluator and scorer
32
+ group. Returns ``{"judged": n}``."""
33
+ return self._client.run_session_sweep()
agentx/version.py CHANGED
@@ -1,7 +1,7 @@
1
- VERSION = "0.8.16"
1
+ VERSION = "0.8.17"
2
2
 
3
3
  # The AgentX-trace-eval release this SDK version is tested against - what `agentx-trace-eval`
4
4
  # installs and converges to (see agentx/cli.py). Bump together with VERSION when releasing, so
5
5
  # every published SDK names a known-good engine+dashboard pair. Users can override with
6
6
  # AGENTX_TRACE_EVAL_VERSION=<tag|latest>.
7
- ENGINE_VERSION = "v0.3.13"
7
+ ENGINE_VERSION = "v0.3.16"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.8.16
3
+ Version: 0.8.17
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -10,17 +10,17 @@ agentx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
10
  agentx/testing.py,sha256=0shZEid_vhJgBegpO_loUq6APYDxQrUgD3jvcRdIDV8,7372
11
11
  agentx/traces.py,sha256=sz9gxutlDKNgf0fdCkyzKsGwK5eMAJuIEl3NIwGrsuo,2287
12
12
  agentx/util.py,sha256=lt2Kpg4Fj8whrSuAhLEahcPEIYRVpK9RCC5y9ShdTJs,703
13
- agentx/version.py,sha256=0obTIpPTOoMyuftj72non9y03CO044fCuNZQlNM6U3k,366
13
+ agentx/version.py,sha256=hFFp4l3xfBrH_LqiA3KPfmmzw-XVqa7R3936Z3truX4,366
14
14
  agentx/evaluations/__init__.py,sha256=Erv7RGFlRxGTG4rVb2uHhCqLLEX_iAWKqkZKzK6CumE,262
15
15
  agentx/evaluations/_term.py,sha256=WFpiNzdgDBeJJ-Gg-6X7TwwxllobuE4OqFUTuDQvS3Y,2529
16
16
  agentx/evaluations/client.py,sha256=f5GqsiuqknE-ENUYocBcXgV6OGMKdzmlsm76takJbwU,33112
17
- agentx/evaluations/datasets.py,sha256=2--V40JrSJoXxR1s0jI3i5rYT8ys25ouijqQQlj72D8,16053
18
- agentx/evaluations/evaluation_settings.py,sha256=geRydHqKpxwkxBVfMIOqkB6NlStyOZrYwZnMHluDwvA,6276
19
- agentx/evaluations/models.py,sha256=YUjo3glVMu9iaL9gwUt8bT2OewJGm_455ZiTwM6kaPo,32938
17
+ agentx/evaluations/datasets.py,sha256=DuX-wu5Kz953s6-605lt8N8-kn3F4WuwaCMq5ISZwbA,16051
18
+ agentx/evaluations/evaluation_settings.py,sha256=PhZtflmEAw33EHuzu4C_Vpm8BBjnIPw60y7VEJ7mU8s,6274
19
+ agentx/evaluations/models.py,sha256=v-t6_HkEQlGNc7ZGfoYXZ7FrFjepuyRODkly-ce6Qjw,32936
20
20
  agentx/evaluations/prompts.py,sha256=8xyvMpAl3mD5xuSZozuaohdnCBWe5Q02RldwiPLUO20,3493
21
21
  agentx/evaluations/reporting.py,sha256=GtNnL-1eNEQrSqd0yrHi5mGqC9_u0kUAN8Mbjdhw6yY,6389
22
22
  agentx/evaluations/results.py,sha256=w7TRUik4eXIqrznXvlF_L9hnObnHvgdRKvihfN1XeAU,4692
23
- agentx/evaluations/runner.py,sha256=jPGWOHOEhTW6PK-qMK9X8i3EqldwEOh5jDk24lLVFEY,34744
23
+ agentx/evaluations/runner.py,sha256=2lxbYKTE2OVH_DFq5yVdAH6cvW5rf3rL91VlEepWLiQ,34714
24
24
  agentx/evaluations/tool_schemas.py,sha256=ZyrnSOnx9xlSAj4swfDey9ZLS2ly-g21DByTwRAGwNA,2323
25
25
  agentx/evaluations/tracing.py,sha256=MSJD9bzfInMi7E70mc8mWUc0fx5DLkOgUzRru2mNQoc,1825
26
26
  agentx/evaluations/adapters/__init__.py,sha256=fK8Hx75usbiY03XUSmnLnSrwBXip2CQs24YZcdEgj6w,287
@@ -43,7 +43,7 @@ agentx/integrations/openai.py,sha256=1KLs-hJaeW2emHPwNkmC3zcGksL3zd87uWbcLGRW704
43
43
  agentx/integrations/openai_agents.py,sha256=tCa3kPxqux5LnNNDQ6zA05OybQrjkc_LeE08YuNPJEs,12742
44
44
  agentx/monitor/__init__.py,sha256=MLiASSvQnCy2jqVNRTgPWFhUSf8lGYRQeEgEVVhE5Tg,574
45
45
  agentx/monitor/agents.py,sha256=8Xk4jWmNTvdtjFixHiLmw4jj96uqIXWIT6KW8zXoWrc,1006
46
- agentx/monitor/client.py,sha256=qY3Dulnhsp_Izj7up4EIxYHRxOfbTVgysflTh2YRnPc,22543
46
+ agentx/monitor/client.py,sha256=V-9rPuzTFR4W-iJS73Blippx0tGdSrOQTcFt1oz29jE,23551
47
47
  agentx/monitor/improvement_groups.py,sha256=-6-FcJnyiTLqeh9z8D9wyFD1QFqEsdDdDZdDJhC6QXw,3712
48
48
  agentx/monitor/judge_scorers.py,sha256=HKCPW_9QJeyGEGwWxr_clep9vn27bwJTKnjN5lYbuSk,15972
49
49
  agentx/monitor/models.py,sha256=yTC3WTdTziHMkdeAXlTro-4YiSbBUHMgf9kNZHGfPDA,8470
@@ -52,9 +52,9 @@ agentx/monitor/patterns.py,sha256=ruBqLoA2T2KQuqRbbn1kWkIEkiOal3SvyPl9opGyhzI,41
52
52
  agentx/monitor/profile.py,sha256=uGk-4bXT2vmnEQiY5u6rWhAfzZchHBXL74VD3piVHvU,3102
53
53
  agentx/monitor/review_queue.py,sha256=Tr3GDMJLJzyFKEEeB8HPmZBsNzzYVJYElDlfBC4qd68,3766
54
54
  agentx/monitor/rules.py,sha256=oLQ5RPgMPUKJFTSZzcBEMLzDlICElUNJTyIqxZMAGmQ,3040
55
- agentx/monitor/scorer_groups.py,sha256=qiQxo6NgNJ50ABKvL1S9OQXBQ-B_YFIiLR59FuXi0hg,3533
55
+ agentx/monitor/scorer_groups.py,sha256=gRFq88B-YkdxftrWwuDLZixrpP0cXzL3ekoZ-lEYFPs,3674
56
56
  agentx/monitor/scorers.py,sha256=huIZBszrfLKI8I7VV6NFpW6piRsj9HI03rXUwWs2tgs,7336
57
- agentx/monitor/sessions.py,sha256=gePcL284m3UITxQAtHxwrKFl6bRjmRib1QK6JbZeS9s,815
57
+ agentx/monitor/sessions.py,sha256=2wTCsSE-Me0ZN_H5DJ58EJ05MQ-LL2m5XKSGIjVmgkM,1444
58
58
  agentx/monitor/signals.py,sha256=Ld11lW2lbg8NHBUHdgoY6iwCvJDK-eg4ge4qQ9y7nGY,1432
59
59
  agentx/resources/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
60
60
  agentx/resources/agent.py,sha256=ZKpxDYzJbKgOX4-W86U__b4gko7XlbwB4q6L8Az9uo0,2011
@@ -66,9 +66,9 @@ agentx/tracing/eval_scope.py,sha256=ElMbPxpqpQBVuaUnHNlw9RIQu71oyOpd0yR55bDH8eI,
66
66
  agentx/tracing/framework_detect.py,sha256=uV4O7Th-4_2UkdooAyaWCOIA0jwLJbblFQ_ucFDjeJM,2638
67
67
  agentx/tracing/ingest_client.py,sha256=spnm6bV3NXGuj2w7HwdKiJdKDJWRUFmSP3jl3rcGqkQ,18202
68
68
  agentx/tracing/tracer.py,sha256=TxY6zG6hbM81jpfrqP4imaEvoyy_hnJVUSTnoAVHMG0,56240
69
- agentx_python-0.8.16.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
70
- agentx_python-0.8.16.dist-info/METADATA,sha256=LRnDRYperDF64jcVdRcdtBH0Iwr5TtlTv0kLKbnC36o,22117
71
- agentx_python-0.8.16.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
72
- agentx_python-0.8.16.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
73
- agentx_python-0.8.16.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
74
- agentx_python-0.8.16.dist-info/RECORD,,
69
+ agentx_python-0.8.17.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
70
+ agentx_python-0.8.17.dist-info/METADATA,sha256=aqprIQoSBpSb7hcJPKt5C6pAz9R4aaHOx4dK3bWjXj8,22117
71
+ agentx_python-0.8.17.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
72
+ agentx_python-0.8.17.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
73
+ agentx_python-0.8.17.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
74
+ agentx_python-0.8.17.dist-info/RECORD,,