agentdebugx 0.2.10__tar.gz → 0.2.11__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/PKG-INFO +1 -1
  2. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/23_status_v0_2.md +3 -2
  3. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/pyproject.toml +1 -1
  4. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/__init__.py +3 -1
  5. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/detectors.py +95 -1
  6. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/llm.py +57 -0
  7. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/LICENSE +0 -0
  8. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/README.md +0 -0
  9. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/00_overview.md +0 -0
  10. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/01_literature_survey.md +0 -0
  11. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/02_architecture.md +0 -0
  12. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/03_taxonomy.md +0 -0
  13. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/04_trace_schema.md +0 -0
  14. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/05_adapters.md +0 -0
  15. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/06_detectors.md +0 -0
  16. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/07_attribution.md +0 -0
  17. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/08_recovery.md +0 -0
  18. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/09_error_database.md +0 -0
  19. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/10_taxonomy_induction.md +0 -0
  20. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/11_multimodal.md +0 -0
  21. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/12_ui_dashboard.md +0 -0
  22. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/13_class_design.md +0 -0
  23. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/14_api_reference.md +0 -0
  24. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/15_roadmap.md +0 -0
  25. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/16_governance.md +0 -0
  26. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/17_claude_code_design_patterns.md +0 -0
  27. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/18_comparison_codex_vs_design.md +0 -0
  28. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/19_error_hub.md +0 -0
  29. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/20_deep_debug.md +0 -0
  30. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/21_integrations.md +0 -0
  31. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/22_industry_track_paper_eval_plan.md +0 -0
  32. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/ERROR_TAXONOMY.md +0 -0
  33. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/OPEN_SOURCE_DEVELOPMENT_PLAN.md +0 -0
  34. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/README.md +0 -0
  35. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/RESEARCH_SURVEY.md +0 -0
  36. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/benchmarks/e2e_v0_2_3.md +0 -0
  37. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/benchmarks/e2e_v0_2_4.md +0 -0
  38. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/benchmarks/v0_1_smoke.json +0 -0
  39. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/benchmarks/v0_1_smoke.md +0 -0
  40. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/docs/benchmarks/who_when_v0_2_6_leaderboard.md +0 -0
  41. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/__init__.py +0 -0
  42. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/base.py +0 -0
  43. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/crewai.py +0 -0
  44. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/langgraph.py +0 -0
  45. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/otel.py +0 -0
  46. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/adapters/raw.py +0 -0
  47. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/analyzers.py +0 -0
  48. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/attribution.py +0 -0
  49. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/cli.py +0 -0
  50. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/deep.py +0 -0
  51. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/events.py +0 -0
  52. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/hub/__init__.py +0 -0
  53. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/hub/backend_base.py +0 -0
  54. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/hub/backends.py +0 -0
  55. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/hub/bundle.py +0 -0
  56. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/hub/scrub.py +0 -0
  57. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/instrumentation.py +0 -0
  58. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/integrations/__init__.py +0 -0
  59. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/integrations/claude_skill.py +0 -0
  60. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/integrations/openhands.py +0 -0
  61. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/judges.py +0 -0
  62. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/models.py +0 -0
  63. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/recorder.py +0 -0
  64. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/recovery.py +0 -0
  65. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/storage.py +0 -0
  66. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/taxonomy.py +0 -0
  67. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/traceback.py +0 -0
  68. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/ui/__init__.py +0 -0
  69. {agentdebugx-0.2.10 → agentdebugx-0.2.11}/src/agentdebug/ui/server.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentdebugx
3
- Version: 0.2.10
3
+ Version: 0.2.11
4
4
  Summary: Portable error analysis, tracing, and recovery framework for agentic AI systems. Import as `agentdebug`.
5
5
  License: MIT
6
6
  License-File: LICENSE
@@ -31,6 +31,8 @@ the forward-looking plan; this doc is the rear-view mirror.
31
31
  | DeepDebug | `agentdebug.deep.DeepDebugAnalyzer` | ✅ stable | full loop + silent LLM |
32
32
  | Cascade view | `agentdebug.traceback.format_traceback` | ✅ stable | cascade + step-order + ANSI + empty |
33
33
  | Detectors | `agentdebug.detectors.RepeatedToolCall / RepeatedState / StepCountLimit` | ✅ **new 0.2.2** | threshold + window + budget |
34
+ | Detectors | `agentdebug.detectors.TopicDriftDetector` (embedding cosine vs goal) | ✅ **new 0.2.11** | stub-embedder ranking + no-goal short-circuit + embedder-raises safe + threshold boundary |
35
+ | LLM client | `agentdebug.llm.OpenAICompatClient.embed()` (+ `EmbeddingClient` Protocol) | ✅ **new 0.2.11** | mocked-httpx POST /v1/embeddings + empty-input short-circuit |
34
36
  | Hub bundle | `agentdebug.hub.Bundle / pack_bundle / unpack_bundle` | ✅ stable | round-trip |
35
37
  | Hub scrubber | `agentdebug.hub.Scrubber` | ✅ stable | 12 redactions + idempotent |
36
38
  | Hub backends | `LocalHubBackend`, `GitHubBackend`, `HuggingFaceBackend` | ✅ stable | local-bare-git + local |
@@ -49,8 +51,7 @@ across 32 source files.
49
51
 
50
52
  | Doc | Component | Why deferred | Realistic ship |
51
53
  |---|---|---|---|
52
- | [06_detectors.md](./06_detectors.md) | `trajectory_perplexity` (TrajAD) | needs token-level LM perplexity API or embedding model + baseline calibration | v0.3 |
53
- | [06_detectors.md](./06_detectors.md) | `topic_drift` (embedding cosine) | needs embedding client; consider reusing `OpenAICompatClient` `/embeddings` | v0.3 |
54
+ | [06_detectors.md](./06_detectors.md) | `trajectory_perplexity` (TrajAD) | needs token-level LM perplexity API; v0.3 |
54
55
  | [06_detectors.md](./06_detectors.md) | LTL spec monitors | requires user-supplied spec or LLM-synthesized monitors; gated on RV research | v1.2 |
55
56
  | [07_attribution.md](./07_attribution.md) | `CounterfactualAttributor` — *real* replay variant | true re-rollout requires framework-specific replay surface; the v0.2.7 LLM-simulated variant ships now, the real-replay variant is gated on adapter support (LangGraph checkpointer / OpenHands rewind) | v0.4 |
56
57
  | [07_attribution.md](./07_attribution.md) | `SBFLAttributor` — *corpus* | shipped in 0.2.8 (`tarantula`/`ochiai`/`dstar`); awaiting paired-trace adoption to gather a useful corpus in production | corpus tooling deferred to v0.4 |
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "agentdebugx"
3
- version = "0.2.10"
3
+ version = "0.2.11"
4
4
  description = "Portable error analysis, tracing, and recovery framework for agentic AI systems. Import as `agentdebug`."
5
5
  authors = ["ULab @ UIUC <ulab@illinois.edu>"]
6
6
  license = "MIT"
@@ -28,6 +28,7 @@ from agentdebug.detectors import (
28
28
  RepeatedStateDetector,
29
29
  RepeatedToolCallDetector,
30
30
  StepCountLimitDetector,
31
+ TopicDriftDetector,
31
32
  default_detectors,
32
33
  run_detectors,
33
34
  )
@@ -82,6 +83,7 @@ __all__ = [
82
83
  'SBFLAttributor',
83
84
  'SelfRefineLoop',
84
85
  'StepByStepAttributor',
86
+ 'TopicDriftDetector',
85
87
  'StepCountLimitDetector',
86
88
  'VerifierSpec',
87
89
  'build_cascade',
@@ -108,4 +110,4 @@ __all__ = [
108
110
  'get_failure_mode',
109
111
  ]
110
112
 
111
- __version__ = '0.2.10'
113
+ __version__ = '0.2.11'
@@ -15,7 +15,7 @@ from __future__ import annotations
15
15
 
16
16
  import logging
17
17
  from dataclasses import dataclass
18
- from typing import List, Optional, Protocol
18
+ from typing import Any, List, Optional, Protocol
19
19
 
20
20
  from agentdebug.models import (
21
21
  AgentEvent,
@@ -273,12 +273,106 @@ def _suggestion(mode: FailureMode) -> Optional[str]:
273
273
  return None
274
274
 
275
275
 
276
+ class TopicDriftDetector:
277
+ """Embedding-based anomaly detector for goal drift.
278
+
279
+ Embed the trajectory's goal once; embed each user-facing event payload
280
+ (LLM_RESPONSE / PLAN / OBSERVATION outputs); flag any step whose cosine
281
+ similarity with the goal drops below ``threshold``.
282
+
283
+ Maps to ``FM-2.3 task_derailment`` (MAST) / ``planning.inefficient_plan``.
284
+ Closes the anomaly family from doc 06 alongside the existing
285
+ RepeatedToolCallDetector / RepeatedStateDetector.
286
+
287
+ Skipped silently if the embedding client raises or returns no vectors —
288
+ the rest of the detector pipeline is unaffected.
289
+ """
290
+
291
+ id = 'topic_drift'
292
+
293
+ def __init__(
294
+ self,
295
+ embedding_client: Any,
296
+ *,
297
+ threshold: float = 0.35,
298
+ max_events: int = 60,
299
+ ) -> None:
300
+ # embedding_client is duck-typed to EmbeddingClient to avoid an
301
+ # import cycle (detectors.py is imported from agentdebug/__init__.py).
302
+ self.embedding_client = embedding_client
303
+ self.threshold = threshold
304
+ self.max_events = max_events
305
+
306
+ def detect(self, trajectory: AgentTrajectory) -> List[FailureFinding]:
307
+ if not trajectory.goal:
308
+ return []
309
+ contextful = [
310
+ e for e in trajectory.events
311
+ if e.event_type in {
312
+ EventType.LLM_RESPONSE, EventType.PLAN, EventType.OBSERVATION,
313
+ EventType.LLM_RESPONSE.value, EventType.PLAN.value,
314
+ EventType.OBSERVATION.value,
315
+ }
316
+ and e.output is not None and str(e.output).strip()
317
+ ]
318
+ if not contextful:
319
+ return []
320
+ contextful = contextful[-self.max_events:]
321
+ texts = [trajectory.goal] + [str(e.output)[:1000] for e in contextful]
322
+ try:
323
+ vectors = self.embedding_client.embed(texts)
324
+ except Exception as exc: # pragma: no cover - defensive
325
+ LOG.warning('topic_drift detector embed() failed: %s', exc)
326
+ return []
327
+ if not vectors or len(vectors) != len(texts):
328
+ return []
329
+ goal_vec = vectors[0]
330
+ findings: List[FailureFinding] = []
331
+ mode = SEED_FAILURE_MODES['planning.inefficient_plan']
332
+ for evt, evt_vec in zip(contextful, vectors[1:]):
333
+ sim = _cosine(goal_vec, evt_vec)
334
+ if sim >= self.threshold:
335
+ continue
336
+ findings.append(FailureFinding(
337
+ finding_id=new_id('finding'),
338
+ failure_mode=mode,
339
+ event_id=evt.event_id,
340
+ agent_name=evt.agent_name,
341
+ step_index=evt.step_index,
342
+ # Confidence proportional to how far below threshold we drifted.
343
+ confidence=min(0.95, 0.4 + (self.threshold - sim)),
344
+ evidence=[
345
+ f'goal/output cosine={sim:.3f} < threshold={self.threshold:.2f}',
346
+ ],
347
+ suggestion=_suggestion(mode),
348
+ metadata={
349
+ 'source': self.id,
350
+ 'cosine_to_goal': round(sim, 4),
351
+ 'threshold': self.threshold,
352
+ },
353
+ ))
354
+ return findings
355
+
356
+
357
+ def _cosine(a: List[float], b: List[float]) -> float:
358
+ import math
359
+ if not a or not b or len(a) != len(b):
360
+ return 0.0
361
+ dot = sum(x * y for x, y in zip(a, b))
362
+ na = math.sqrt(sum(x * x for x in a))
363
+ nb = math.sqrt(sum(y * y for y in b))
364
+ if na == 0 or nb == 0:
365
+ return 0.0
366
+ return dot / (na * nb)
367
+
368
+
276
369
  __all__ = [
277
370
  'Detector',
278
371
  'DetectorConfig',
279
372
  'RepeatedStateDetector',
280
373
  'RepeatedToolCallDetector',
281
374
  'StepCountLimitDetector',
375
+ 'TopicDriftDetector',
282
376
  'default_detectors',
283
377
  'run_detectors',
284
378
  ]
@@ -44,6 +44,25 @@ class LLMClient(Protocol):
44
44
  ...
45
45
 
46
46
 
47
+ class EmbeddingClient(Protocol):
48
+ """Subprotocol for clients that also expose ``/v1/embeddings``.
49
+
50
+ Kept separate from :class:`LLMClient` so detectors can declare a
51
+ narrower dependency and tests can stub embeddings without faking
52
+ a chat client.
53
+ """
54
+
55
+ embedding_model: str
56
+
57
+ def embed(
58
+ self,
59
+ texts: List[str],
60
+ *,
61
+ timeout: float = 60.0,
62
+ ) -> List[List[float]]:
63
+ ...
64
+
65
+
47
66
  class OpenAICompatClient:
48
67
  """OpenAI-compatible chat completions client.
49
68
 
@@ -63,12 +82,16 @@ class OpenAICompatClient:
63
82
  base_url: str,
64
83
  api_key: str,
65
84
  model: str,
85
+ embedding_model: str = 'text-embedding-3-small',
66
86
  default_max_tokens: int = 2048,
67
87
  timeout: float = 60.0,
68
88
  ) -> None:
69
89
  self.base_url = base_url.rstrip('/')
70
90
  self.api_key = api_key
71
91
  self.model = model
92
+ # Embeddings hit a separate endpoint with a separate model id; default
93
+ # to OpenAI's small embedding model since the gateway is OpenAI-compat.
94
+ self.embedding_model = embedding_model
72
95
  self.default_max_tokens = default_max_tokens
73
96
  self.timeout = timeout
74
97
 
@@ -132,6 +155,40 @@ class OpenAICompatClient:
132
155
  return CompletionResult(text=text, raw=data)
133
156
 
134
157
 
158
+ def embed(
159
+ self,
160
+ texts: List[str],
161
+ *,
162
+ timeout: Optional[float] = None,
163
+ ) -> List[List[float]]:
164
+ """OpenAI-compatible ``/v1/embeddings`` POST.
165
+
166
+ Returns a list of vectors (one per input text) in the same order.
167
+ Empty ``texts`` short-circuits to ``[]`` to save a network round-trip.
168
+ """
169
+ if not texts:
170
+ return []
171
+ url = f'{self.base_url}/embeddings'
172
+ headers = {
173
+ 'Authorization': f'Bearer {self.api_key}',
174
+ 'Content-Type': 'application/json',
175
+ }
176
+ body = {'model': self.embedding_model, 'input': list(texts)}
177
+ resp = httpx.post(
178
+ url, headers=headers, json=body, timeout=timeout or self.timeout
179
+ )
180
+ resp.raise_for_status()
181
+ data = resp.json()
182
+ rows = data.get('data') or []
183
+ out: List[List[float]] = []
184
+ for row in rows:
185
+ vec = row.get('embedding')
186
+ if not isinstance(vec, list):
187
+ continue
188
+ out.append([float(v) for v in vec])
189
+ return out
190
+
191
+
135
192
  def extract_json_block(text: str) -> Optional[Dict[str, Any]]:
136
193
  """Extract the first top-level JSON object from a possibly-fenced response."""
137
194
  if not text:
File without changes
File without changes