agentx-python 0.8.14__py3-none-any.whl → 0.8.16__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -314,10 +314,15 @@ class EvaluationsClient:
314
314
  scorer_id: Optional[str] = None,
315
315
  evaluation_settings_id: Optional[str] = None,
316
316
  split: Optional[str] = None,
317
+ additional_scorer_ids: Optional[List[str]] = None,
318
+ scorer_group_id: Optional[str] = None,
317
319
  ) -> EvaluationRun:
318
320
  """``scorer_id`` names the LLM Judge Scorer grading this run (its id doubles as the
319
321
  wire's ``evaluationSettingsId``). ``evaluation_settings_id`` is the pre-consolidation
320
- alias and keeps working. ``split`` records the named case subset this run covers."""
322
+ alias and keeps working. ``split`` records the named case subset this run covers.
323
+ ``additional_scorer_ids`` (self-host): extra judge scorers that each pass their own
324
+ verdict on every result from the same single agent execution - verdicts land in each
325
+ result row's ``judgeScorerResults`` and the run's ``scorerBreakdown``."""
321
326
  from agentx.version import VERSION
322
327
 
323
328
  grader_id = _resolve_scorer_id(scorer_id, evaluation_settings_id)
@@ -335,6 +340,12 @@ class EvaluationsClient:
335
340
  }
336
341
  if grader_id:
337
342
  payload["evaluationSettingsId"] = grader_id
343
+ if additional_scorer_ids:
344
+ payload["additionalScorerIds"] = additional_scorer_ids
345
+ # Scorer group grading (self-host): the group's weighted 0-10 aggregate fills the rating
346
+ # column and member verdicts land per row. Mutually exclusive with scorer_id (group wins).
347
+ if scorer_group_id:
348
+ payload["scorerGroupId"] = scorer_group_id
338
349
  if split:
339
350
  payload["split"] = split
340
351
  data = self._request("POST", "/runs", json=self._with_workspace(payload))
@@ -372,6 +383,7 @@ class EvaluationsClient:
372
383
  tolerance: Optional[float] = None,
373
384
  record: bool = True,
374
385
  caller: Optional[str] = "sdk",
386
+ scorer: Optional[str] = None,
375
387
  ) -> Dict[str, Any]:
376
388
  # CI gate (self-host): pass/fail a finalized run against an absolute rating floor and/or
377
389
  # the dataset's previous completed run. Recorded into gate history by default (the
@@ -389,6 +401,11 @@ class EvaluationsClient:
389
401
  params["record"] = "true"
390
402
  if caller:
391
403
  params["caller"] = caller
404
+ # Multi-judge runs (self-host): gate a named additional scorer (id or name, e.g.
405
+ # scorer="Safety") instead of the primary - failUnder/noRegression then use that
406
+ # scorer's own per-result verdicts. Unknown names are a hard 400 from the engine.
407
+ if scorer:
408
+ params["scorer"] = scorer
392
409
  return self._request("GET", f"/runs/{run_id}/gate", params=params)
393
410
 
394
411
  def analyze_run(
@@ -75,6 +75,9 @@ class Dataset(BaseModel):
75
75
  rejection_criteria: Optional[str] = Field(default=None, alias="rejectionCriteria")
76
76
  evaluation_criteria: Optional[str] = Field(default=None, alias="evaluationCriteria")
77
77
  questions: List[DatasetQuestion] = Field(default_factory=list)
78
+ # Custom code scorers attached to this dataset - [{ id, name, code, enabled }]. Retrievable,
79
+ # so a fetched dataset round-trips them (import_dataset copies them to the new dataset).
80
+ code_scorers: Optional[List[Dict[str, Any]]] = Field(default=None, alias="codeScorers")
78
81
  status: str = "published"
79
82
  version_id: Optional[str] = Field(default=None, alias="versionId")
80
83
  # Sovereignty & Portability - models selected to compare on this dataset.
@@ -108,6 +111,8 @@ class EvaluationSettings(BaseModel):
108
111
  acceptance_criteria: Optional[str] = Field(default=None, alias="acceptanceCriteria")
109
112
  rejection_criteria: Optional[str] = Field(default=None, alias="rejectionCriteria")
110
113
  evaluation_criteria: Optional[str] = Field(default=None, alias="evaluationCriteria")
114
+ # Custom code scorers on this scorer's offline profile - [{ id, name, code, enabled }].
115
+ code_scorers: Optional[List[Dict[str, Any]]] = Field(default=None, alias="codeScorers")
111
116
  # LLM-as-judge overrides. None means "use the server default" (raw prompt template / OpenAI
112
117
  # gpt-5.5). See client.evaluations.settings.builder(judge_prompt=..., judge_model=...).
113
118
  judge_prompt: Optional[str] = Field(default=None, alias="judgePrompt")
@@ -435,6 +440,8 @@ class RunResultRow(BaseModel):
435
440
  bleu_score: Optional[float] = Field(default=None, alias="bleuScore")
436
441
  rouge_score: Optional[float] = Field(default=None, alias="rougeScore")
437
442
  code_scorer_results: Optional[List[Dict[str, Any]]] = Field(default=None, alias="codeScorerResults")
443
+ # Verdicts from the run's ADDITIONAL judge scorers: [{scorerId, name, rating, justification}].
444
+ judge_scorer_results: Optional[List[Dict[str, Any]]] = Field(default=None, alias="judgeScorerResults")
438
445
  raw: Dict[str, Any] = Field(default_factory=dict)
439
446
 
440
447
  class Config:
@@ -77,6 +77,9 @@ class GateResult:
77
77
  self.baseline_average: Optional[float] = data.get("baselineAverage")
78
78
  self.baseline_run_id: Optional[str] = data.get("baselineRunId")
79
79
  self.checks: List[Dict[str, Any]] = data.get("checks", [])
80
+ # Multi-judge runs: {"id", "name"} of the additional scorer being gated when the gate
81
+ # ran with scorer=..., None when gating the primary.
82
+ self.gated_scorer: Optional[Dict[str, Any]] = data.get("gatedScorer")
80
83
 
81
84
  @property
82
85
  def exit_code(self) -> int:
@@ -354,13 +357,16 @@ class EvaluationRunContext:
354
357
  no_regression: bool = False,
355
358
  tolerance: Optional[float] = None,
356
359
  caller: str = "sdk",
360
+ scorer: Optional[str] = None,
357
361
  ) -> "GateResult":
358
362
  """CI gate (self-host): pass/fail this finalized run so a CI job can block a merge.
359
363
 
360
364
  ``fail_under`` fails the gate when the run's average rating is below the floor;
361
365
  ``no_regression=True`` fails it when the average dropped more than ``tolerance``
362
366
  (default 0.5, judge scores are noisy) below the dataset's previous completed run.
363
- At least one check is required. Prints a CI-log-friendly verdict and returns a
367
+ At least one check is required. On a multi-judge run, ``scorer`` (an additional
368
+ scorer's id or name, e.g. ``scorer="Safety"``) gates that scorer's own average
369
+ instead of the primary's - "fail if Safety is low even when the average looks fine". Prints a CI-log-friendly verdict and returns a
364
370
  :class:`GateResult` - the caller decides the exit code::
365
371
 
366
372
  report = client.evaluations.run(...).execute(my_agent).finalize()
@@ -374,6 +380,7 @@ class EvaluationRunContext:
374
380
  no_regression=no_regression,
375
381
  tolerance=tolerance,
376
382
  caller=caller,
383
+ scorer=scorer,
377
384
  )
378
385
  result = GateResult(data)
379
386
  _say()
@@ -589,6 +596,7 @@ class EvaluationsRunner:
589
596
  tolerance: Optional[float] = None,
590
597
  record: bool = True,
591
598
  caller: Optional[str] = "sdk",
599
+ scorer: Optional[str] = None,
592
600
  ) -> GateResult:
593
601
  """CI-gate any finalized run by id - the standalone form of
594
602
  ``EvaluationRunContext.gate()``, for gating a run created elsewhere or
@@ -603,6 +611,7 @@ class EvaluationsRunner:
603
611
  tolerance=tolerance,
604
612
  record=record,
605
613
  caller=caller,
614
+ scorer=scorer,
606
615
  )
607
616
  )
608
617
 
@@ -613,6 +622,8 @@ class EvaluationsRunner:
613
622
  scorer_id: Optional[str] = None,
614
623
  evaluation_settings_id: Optional[str] = None,
615
624
  split: Optional[str] = None,
625
+ additional_scorer_ids: Optional[List[str]] = None,
626
+ scorer_group_id: Optional[str] = None,
616
627
  ) -> EvaluationRunContext:
617
628
  """Start a run of ``dataset_id`` against ``subject``. Pass ``scorer_id`` (an LLM Judge
618
629
  Scorer's id, e.g. from ``client.monitor.judge_scorers``) to grade with a specific
@@ -632,7 +643,14 @@ class EvaluationsRunner:
632
643
  evaluation_settings = (
633
644
  self._client.get_evaluation_settings(grader_id) if grader_id else None
634
645
  )
635
- run = self._client.init_run(dataset_id, subject, scorer_id=grader_id, split=split)
646
+ run = self._client.init_run(
647
+ dataset_id,
648
+ subject,
649
+ scorer_id=grader_id,
650
+ split=split,
651
+ additional_scorer_ids=additional_scorer_ids,
652
+ scorer_group_id=scorer_group_id,
653
+ )
636
654
  case_count = (
637
655
  sum(1 for q in dataset.questions if split in (q.main_question.splits or []))
638
656
  if split
agentx/monitor/client.py CHANGED
@@ -67,6 +67,22 @@ class CalibrationSummary(dict):
67
67
  def review_label_count(self) -> int:
68
68
  return int(self.get("reviewLabelCount") or 0)
69
69
 
70
+ @property
71
+ def alpha(self):
72
+ """Chance-corrected agreement (Krippendorff's alpha over the binary verdict pair) -
73
+ the raw ``agreement_rate`` corrected for what a weighted coin would score on this
74
+ label mix. ``None`` below the server's sample floor (``alphaMinItems`` labeled pairs)
75
+ or when every label is identical: withheld, never fabricated. 1 = perfect, 0 = no
76
+ better than chance, negative = systematically opposed."""
77
+ return self.get("alpha")
78
+
79
+ @property
80
+ def alpha_band(self):
81
+ """Human-readable band for ``alpha`` (poor/slight/fair/moderate/substantial/
82
+ near-perfect), computed server-side so every surface reads the same alpha the
83
+ same way."""
84
+ return self.get("alphaBand")
85
+
70
86
 
71
87
  class MonitorClient:
72
88
  """Low-level HTTP client for the Monitor API (``/monitor``). Accessed via
@@ -131,6 +147,16 @@ class MonitorClient:
131
147
  # surface that matches the product; evaluations.settings and online_evaluators below
132
148
  # remain as its profile-level views.
133
149
  self.judge_scorers = JudgeScorersClient(api_key=api_key, base_url=self._api_root())
150
+ from agentx.monitor.scorer_groups import ScorerGroupsClient
151
+
152
+ # Scorer groups: mixed-kind scorers composed into one 0-10 score (weights + must-pass
153
+ # gates) - a group grades dataset runs (scorer_group_id) and, when online, live traffic.
154
+ self.scorer_groups = ScorerGroupsClient(api_key=api_key, base_url=self._api_root())
155
+ from agentx.monitor.improvement_groups import ImprovementGroupsClient
156
+
157
+ # Auto-improve: confirmed production failures -> improvement report -> code fix (via
158
+ # the AgentX-Eval-Skill auto-improve skill). Self-host only.
159
+ self.improvement_groups = ImprovementGroupsClient(api_key=api_key, base_url=self._api_root())
134
160
  self.profile = MonitorProfileClient(self)
135
161
  # Legacy view of an LLM Judge Scorer's online profile - constructed lazily so its
136
162
  # DeprecationWarning fires on first USE, not for every client that never touches it.
@@ -317,8 +343,12 @@ class MonitorClient:
317
343
  AgentX's own verdicts agreed with real-world ground truth reported later (ops outcomes
318
344
  via ``client.outcomes``, end-user downvotes, and human review labels). Returns the
319
345
  dashboard's Judge Calibration numbers with these exact keys: ``comparedCount``,
320
- ``agreementRate``, ``falsePositiveRate``, ``falseNegativeRate`` (plus
321
- ``reportedCount``/``reviewLabelCount``/``noVerdictCount``). Per-scorer calibration
346
+ ``agreementRate``, ``falsePositiveRate``, ``falseNegativeRate``, ``alpha``,
347
+ ``alphaBand``, ``alphaMinItems`` (plus
348
+ ``reportedCount``/``reviewLabelCount``/``noVerdictCount``). ``agreementRate`` is raw
349
+ agreement and inflates under class imbalance; ``alpha`` is the chance-corrected
350
+ version (Krippendorff's alpha - null until ``alphaMinItems`` labeled pairs exist).
351
+ Per-scorer calibration
322
352
  lives on ``client.monitor.judge_scorers.calibration(scorer_id)``."""
323
353
  return CalibrationSummary(
324
354
  self._request(
@@ -0,0 +1,76 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Any, Dict, List, Optional
4
+
5
+ import requests
6
+
7
+ from agentx.util import api_base, get_headers
8
+
9
+
10
+ class AgentXImprovementGroupsError(Exception):
11
+ pass
12
+
13
+
14
+ class ImprovementGroupsClient:
15
+ """Surfaced as ``client.monitor.improvement_groups``: the auto-improve loop's accumulator.
16
+
17
+ Batch lifecycle: one COLLECTING group at a time. Every Confirm verdict in signal review
18
+ automatically lands the confirmed failure there - accumulation is free, declining is
19
+ choosing Ignore. ``generate_report`` SPENDS the batch: one LLM pass clusters the confirmed
20
+ failures into issues with recommendations, the group is sealed onto that report (keeping
21
+ exactly its source cases), and the pending accumulator is thereby cleared - the next
22
+ Confirm starts a fresh batch, and the next generate makes a new report from it. The report's id is the
23
+ hand-off: paste it into the AgentX-Eval-Skill ``auto-improve`` skill, which fetches the
24
+ report (``get_report``) and triages the fixes against your agent's actual source code.
25
+
26
+ Evidence here is exclusively ONLINE - production verdicts a human confirmed - never
27
+ offline dataset runs. Self-host only.
28
+ """
29
+
30
+ def __init__(self, api_key: Optional[str] = None, base_url: Optional[str] = None):
31
+ self._api_key = api_key
32
+ self._base_url = (base_url or api_base()).rstrip("/")
33
+
34
+ def _request(self, method: str, path: str, json: Any = None, timeout: int = 120) -> Any:
35
+ resp = requests.request(
36
+ method,
37
+ f"{self._base_url}/agent-monitoring{path}",
38
+ headers={**get_headers(self._api_key), "Content-Type": "application/json"},
39
+ json=json,
40
+ timeout=timeout,
41
+ )
42
+ if resp.status_code >= 400:
43
+ try:
44
+ detail = resp.json().get("error", resp.reason)
45
+ except ValueError:
46
+ detail = resp.reason
47
+ raise AgentXImprovementGroupsError(f"Improvement group request failed ({resp.status_code}): {detail}")
48
+ return resp.json() if resp.text else {}
49
+
50
+ def list(self) -> List[Dict[str, Any]]:
51
+ return self._request("GET", "/improvement-groups").get("improvementGroups", [])
52
+
53
+ def get(self, group_id: str) -> Dict[str, Any]:
54
+ """The group with its members - each a confirmed failure's evidence snapshot."""
55
+ return self._request("GET", f"/improvement-groups/{group_id}")["improvementGroup"]
56
+
57
+ def remove_member(self, group_id: str, member_id: str) -> None:
58
+ """Prune a member before spending the group (a confirm that turned out uninteresting)."""
59
+ self._request("DELETE", f"/improvement-groups/{group_id}/members/{member_id}")
60
+
61
+ def generate_report(self, group_id: str, model: Optional[str] = None) -> Dict[str, Any]:
62
+ """Spend the group: one real LLM call clustering the confirmed failures into issues
63
+ with recommendations. Returns the report; its ``_id`` is what the auto-improve skill
64
+ takes. Explicit and billed - never called implicitly."""
65
+ payload: Dict[str, Any] = {}
66
+ if model is not None:
67
+ payload["model"] = model
68
+ return self._request("POST", f"/improvement-groups/{group_id}/report", json=payload, timeout=300)["report"]
69
+
70
+ def list_reports(self) -> List[Dict[str, Any]]:
71
+ return self._request("GET", "/improvement-reports").get("improvementReports", [])
72
+
73
+ def get_report(self, report_id: str) -> Dict[str, Any]:
74
+ """Fetch a report by the id the dashboard (or generate_report) handed out - the exact
75
+ call the auto-improve skill makes."""
76
+ return self._request("GET", f"/improvement-reports/{report_id}")["report"]
@@ -36,6 +36,11 @@ class JudgeScorer(dict):
36
36
  def offline(self) -> Dict[str, Any]:
37
37
  return self.get("offline", {})
38
38
 
39
+ @property
40
+ def code_scorers(self) -> List[Dict[str, Any]]:
41
+ """Custom code scorers on the offline profile - [{ id, name, code, enabled }]."""
42
+ return list(self.offline.get("codeScorers") or [])
43
+
39
44
  @property
40
45
  def online(self) -> Optional[Dict[str, Any]]:
41
46
  return self.get("online")
@@ -258,11 +263,19 @@ class JudgeScorersClient:
258
263
 
259
264
  def calibration(self, scorer_id: str, window: str = "7d") -> dict:
260
265
  """How this scorer's verdicts compare against recorded ground truth (triage
261
- corrections, outcomes, end-user votes) over the window."""
266
+ corrections, outcomes, end-user votes) over the window. Beyond the raw
267
+ ``agreementRate``, the response carries ``alpha``/``alphaBand`` (chance-corrected
268
+ agreement - Krippendorff's alpha, null until ``alphaMinItems`` labeled pairs exist)
269
+ and ``ratingMae`` (mean absolute error against human re-scores, over the
270
+ ``withCorrectedScore`` pairs that carry a number). ``window`` accepts "24h", "7d",
271
+ "30d", or "rubric" - only verdicts produced by the CURRENT rubric (since its criteria
272
+ were last edited, clamped to 30 days), which is what the dashboard's Tune Judge flow
273
+ uses by default; the response's ``window``/``since`` echo the boundary applied."""
262
274
  return self._request("GET", f"/online-evaluators/{self._profile_id(scorer_id)}/calibration?window={window}")
263
275
 
264
276
  def tune(self, scorer_id: str, window: str = "7d") -> dict:
265
- """Propose a rewrite of the rubric from calibration disagreements (LLM call, slow)."""
277
+ """Propose a rewrite of the rubric from calibration disagreements (LLM call, slow).
278
+ ``window`` accepts the same values as :meth:`calibration`, including "rubric"."""
266
279
  data = self._request(
267
280
  "POST", f"/online-evaluators/{self._profile_id(scorer_id)}/tune", json={"window": window}, timeout=300
268
281
  )
@@ -0,0 +1,88 @@
1
+ """Scorer groups (self-host): scorers of any kind - LLM judges, patterns, custom code/external
2
+ scorers - composed into ONE 0-10 score via per-member weights and optional must-pass gates.
3
+ Members are references: ``{"kind": "judge" | "pattern" | "custom", "refId": ..., "weight": ...,
4
+ "gate": ...}``. Grade a dataset run with a group by passing its id as ``scorer_group_id`` to
5
+ ``client.evaluations.run(...)``; give it an ``online`` profile to score sampled live traffic and
6
+ raise Signals below the alert threshold."""
7
+
8
+ from typing import Any, Dict, List, Optional
9
+
10
+ import requests
11
+
12
+
13
+ class AgentXScorerGroupsError(Exception):
14
+ pass
15
+
16
+
17
+ class ScorerGroup(dict):
18
+ """Wire object (dict subclass so unknown fields round-trip)."""
19
+
20
+ @property
21
+ def id(self) -> str:
22
+ return self["_id"]
23
+
24
+ @property
25
+ def name(self) -> str:
26
+ return self["name"]
27
+
28
+ @property
29
+ def members(self) -> List[Dict[str, Any]]:
30
+ return list(self.get("members") or [])
31
+
32
+ @property
33
+ def online(self) -> Optional[Dict[str, Any]]:
34
+ return self.get("online")
35
+
36
+
37
+ class ScorerGroupsClient:
38
+ def __init__(self, api_key: str, base_url: str):
39
+ self._api_key = api_key
40
+ self._base = base_url.rstrip("/") + "/agent-monitoring/scorer-groups"
41
+
42
+ def _request(self, method: str, url: str, json: Optional[Dict[str, Any]] = None) -> Any:
43
+ response = requests.request(
44
+ method,
45
+ url,
46
+ headers={"x-api-key": self._api_key, "content-type": "application/json"},
47
+ json=json,
48
+ timeout=30,
49
+ )
50
+ if response.status_code >= 400:
51
+ raise AgentXScorerGroupsError(f"HTTP {response.status_code}: {response.text}")
52
+ return response.json()
53
+
54
+ def list(self) -> List[ScorerGroup]:
55
+ return [ScorerGroup(g) for g in self._request("GET", self._base).get("scorerGroups", [])]
56
+
57
+ def get(self, group_id: str) -> ScorerGroup:
58
+ return ScorerGroup(self._request("GET", f"{self._base}/{group_id}")["scorerGroup"])
59
+
60
+ def create(
61
+ self,
62
+ name: str,
63
+ members: List[Dict[str, Any]],
64
+ description: Optional[str] = None,
65
+ online: Optional[Dict[str, Any]] = None,
66
+ ) -> ScorerGroup:
67
+ """``members``: [{"kind": "judge"|"pattern"|"custom", "refId": ..., "weight": 1, "gate": False}].
68
+ ``online``: {"enabled": True, "sampleRate": 0.1, "alertThreshold": 5, "severity": "medium"}
69
+ or None for offline-only."""
70
+ payload: Dict[str, Any] = {"name": name, "members": members}
71
+ if description is not None:
72
+ payload["description"] = description
73
+ if online is not None:
74
+ payload["online"] = online
75
+ return ScorerGroup(self._request("POST", self._base, json=payload)["scorerGroup"])
76
+
77
+ def update(self, group_id: str, **fields: Any) -> ScorerGroup:
78
+ """Sparse update - pass any of name/description/members/online (online=None detaches
79
+ live scoring)."""
80
+ return ScorerGroup(self._request("PUT", f"{self._base}/{group_id}", json=fields)["scorerGroup"])
81
+
82
+ def delete(self, group_id: str) -> None:
83
+ self._request("DELETE", f"{self._base}/{group_id}")
84
+
85
+ def ratings(self, group_id: str, window: str = "7d") -> Dict[str, Any]:
86
+ """Live score history for a group - ``{"window", "points": [{ts, averageRating, count}]}``,
87
+ the same shape online-evaluator ratings use. ``window``: "24h" | "7d" | "30d"."""
88
+ return self._request("GET", f"{self._base}/{group_id}/ratings?window={window}")
agentx/version.py CHANGED
@@ -1,7 +1,7 @@
1
- VERSION = "0.8.14"
1
+ VERSION = "0.8.16"
2
2
 
3
3
  # The AgentX-trace-eval release this SDK version is tested against - what `agentx-trace-eval`
4
4
  # installs and converges to (see agentx/cli.py). Bump together with VERSION when releasing, so
5
5
  # every published SDK names a known-good engine+dashboard pair. Users can override with
6
6
  # AGENTX_TRACE_EVAL_VERSION=<tag|latest>.
7
- ENGINE_VERSION = "v0.3.10"
7
+ ENGINE_VERSION = "v0.3.13"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentx-python
3
- Version: 0.8.14
3
+ Version: 0.8.16
4
4
  Summary: Official Python SDK for AgentX (https://www.agentx.so/)
5
5
  Home-page: https://github.com/AgentX-ai/AgentX-python
6
6
  Author: Robin Wang and AgentX Team
@@ -10,17 +10,17 @@ agentx/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
10
10
  agentx/testing.py,sha256=0shZEid_vhJgBegpO_loUq6APYDxQrUgD3jvcRdIDV8,7372
11
11
  agentx/traces.py,sha256=sz9gxutlDKNgf0fdCkyzKsGwK5eMAJuIEl3NIwGrsuo,2287
12
12
  agentx/util.py,sha256=lt2Kpg4Fj8whrSuAhLEahcPEIYRVpK9RCC5y9ShdTJs,703
13
- agentx/version.py,sha256=ZCO46jAagFLJA2FF86tnlc0_wrjWmgCjw4xXMTEaj3w,366
13
+ agentx/version.py,sha256=0obTIpPTOoMyuftj72non9y03CO044fCuNZQlNM6U3k,366
14
14
  agentx/evaluations/__init__.py,sha256=Erv7RGFlRxGTG4rVb2uHhCqLLEX_iAWKqkZKzK6CumE,262
15
15
  agentx/evaluations/_term.py,sha256=WFpiNzdgDBeJJ-Gg-6X7TwwxllobuE4OqFUTuDQvS3Y,2529
16
- agentx/evaluations/client.py,sha256=UlYl8OFmIjk8oqmL_LEpjnTn6CmUPBM1kgVgzkUKfT8,31999
16
+ agentx/evaluations/client.py,sha256=f5GqsiuqknE-ENUYocBcXgV6OGMKdzmlsm76takJbwU,33112
17
17
  agentx/evaluations/datasets.py,sha256=2--V40JrSJoXxR1s0jI3i5rYT8ys25ouijqQQlj72D8,16053
18
18
  agentx/evaluations/evaluation_settings.py,sha256=geRydHqKpxwkxBVfMIOqkB6NlStyOZrYwZnMHluDwvA,6276
19
- agentx/evaluations/models.py,sha256=fKaDwyXhruXRHpBDw89Vv_FeceLvSILVrA9HNtzKLF8,32266
19
+ agentx/evaluations/models.py,sha256=YUjo3glVMu9iaL9gwUt8bT2OewJGm_455ZiTwM6kaPo,32938
20
20
  agentx/evaluations/prompts.py,sha256=8xyvMpAl3mD5xuSZozuaohdnCBWe5Q02RldwiPLUO20,3493
21
21
  agentx/evaluations/reporting.py,sha256=GtNnL-1eNEQrSqd0yrHi5mGqC9_u0kUAN8Mbjdhw6yY,6389
22
22
  agentx/evaluations/results.py,sha256=w7TRUik4eXIqrznXvlF_L9hnObnHvgdRKvihfN1XeAU,4692
23
- agentx/evaluations/runner.py,sha256=qhaUiUaP_D5etjRnsuR71-uQGnwF_8J4LSPQ2lMWTU4,33882
23
+ agentx/evaluations/runner.py,sha256=jPGWOHOEhTW6PK-qMK9X8i3EqldwEOh5jDk24lLVFEY,34744
24
24
  agentx/evaluations/tool_schemas.py,sha256=ZyrnSOnx9xlSAj4swfDey9ZLS2ly-g21DByTwRAGwNA,2323
25
25
  agentx/evaluations/tracing.py,sha256=MSJD9bzfInMi7E70mc8mWUc0fx5DLkOgUzRru2mNQoc,1825
26
26
  agentx/evaluations/adapters/__init__.py,sha256=fK8Hx75usbiY03XUSmnLnSrwBXip2CQs24YZcdEgj6w,287
@@ -43,14 +43,16 @@ agentx/integrations/openai.py,sha256=1KLs-hJaeW2emHPwNkmC3zcGksL3zd87uWbcLGRW704
43
43
  agentx/integrations/openai_agents.py,sha256=tCa3kPxqux5LnNNDQ6zA05OybQrjkc_LeE08YuNPJEs,12742
44
44
  agentx/monitor/__init__.py,sha256=MLiASSvQnCy2jqVNRTgPWFhUSf8lGYRQeEgEVVhE5Tg,574
45
45
  agentx/monitor/agents.py,sha256=8Xk4jWmNTvdtjFixHiLmw4jj96uqIXWIT6KW8zXoWrc,1006
46
- agentx/monitor/client.py,sha256=nY1cvjbCqL6BZMY6ubM-Q6JUKGgUM5CdBektV99EArQ,20812
47
- agentx/monitor/judge_scorers.py,sha256=Mu6OOnfjp1SRkZMcOyriO7zPawv-O2H2l2X-0io5WLM,15025
46
+ agentx/monitor/client.py,sha256=qY3Dulnhsp_Izj7up4EIxYHRxOfbTVgysflTh2YRnPc,22543
47
+ agentx/monitor/improvement_groups.py,sha256=-6-FcJnyiTLqeh9z8D9wyFD1QFqEsdDdDZdDJhC6QXw,3712
48
+ agentx/monitor/judge_scorers.py,sha256=HKCPW_9QJeyGEGwWxr_clep9vn27bwJTKnjN5lYbuSk,15972
48
49
  agentx/monitor/models.py,sha256=yTC3WTdTziHMkdeAXlTro-4YiSbBUHMgf9kNZHGfPDA,8470
49
50
  agentx/monitor/online_evaluators.py,sha256=YWnsBmlIbYvj4LCaX4guwx4HS5y3oEzCQfo6ae9VH_w,8693
50
51
  agentx/monitor/patterns.py,sha256=ruBqLoA2T2KQuqRbbn1kWkIEkiOal3SvyPl9opGyhzI,4185
51
52
  agentx/monitor/profile.py,sha256=uGk-4bXT2vmnEQiY5u6rWhAfzZchHBXL74VD3piVHvU,3102
52
53
  agentx/monitor/review_queue.py,sha256=Tr3GDMJLJzyFKEEeB8HPmZBsNzzYVJYElDlfBC4qd68,3766
53
54
  agentx/monitor/rules.py,sha256=oLQ5RPgMPUKJFTSZzcBEMLzDlICElUNJTyIqxZMAGmQ,3040
55
+ agentx/monitor/scorer_groups.py,sha256=qiQxo6NgNJ50ABKvL1S9OQXBQ-B_YFIiLR59FuXi0hg,3533
54
56
  agentx/monitor/scorers.py,sha256=huIZBszrfLKI8I7VV6NFpW6piRsj9HI03rXUwWs2tgs,7336
55
57
  agentx/monitor/sessions.py,sha256=gePcL284m3UITxQAtHxwrKFl6bRjmRib1QK6JbZeS9s,815
56
58
  agentx/monitor/signals.py,sha256=Ld11lW2lbg8NHBUHdgoY6iwCvJDK-eg4ge4qQ9y7nGY,1432
@@ -64,9 +66,9 @@ agentx/tracing/eval_scope.py,sha256=ElMbPxpqpQBVuaUnHNlw9RIQu71oyOpd0yR55bDH8eI,
64
66
  agentx/tracing/framework_detect.py,sha256=uV4O7Th-4_2UkdooAyaWCOIA0jwLJbblFQ_ucFDjeJM,2638
65
67
  agentx/tracing/ingest_client.py,sha256=spnm6bV3NXGuj2w7HwdKiJdKDJWRUFmSP3jl3rcGqkQ,18202
66
68
  agentx/tracing/tracer.py,sha256=TxY6zG6hbM81jpfrqP4imaEvoyy_hnJVUSTnoAVHMG0,56240
67
- agentx_python-0.8.14.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
68
- agentx_python-0.8.14.dist-info/METADATA,sha256=OT_i-6ZVjbcRjixhnL6vfo8wLdtmmkXhrjuxiEY1iRc,22117
69
- agentx_python-0.8.14.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
70
- agentx_python-0.8.14.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
71
- agentx_python-0.8.14.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
72
- agentx_python-0.8.14.dist-info/RECORD,,
69
+ agentx_python-0.8.16.dist-info/licenses/LICENSE,sha256=gZVsM-nLsE8vlaY6NXXsVoo6IlCClxkToAWmhgT3y_s,10762
70
+ agentx_python-0.8.16.dist-info/METADATA,sha256=LRnDRYperDF64jcVdRcdtBH0Iwr5TtlTv0kLKbnC36o,22117
71
+ agentx_python-0.8.16.dist-info/WHEEL,sha256=aeYiig01lYGDzBgS8HxWXOg3uV61G9ijOsup-k9o1sk,91
72
+ agentx_python-0.8.16.dist-info/entry_points.txt,sha256=rQqF1JTY3T1yfviBU1r8PU-5mBdfOk6P4bZ1L4BskIM,172
73
+ agentx_python-0.8.16.dist-info/top_level.txt,sha256=s-q-HB9Gb_QdrZNacSeQyF_c25gQooMy7DlxzgLOHPk,7
74
+ agentx_python-0.8.16.dist-info/RECORD,,