contextos-auditor 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextos_auditor/__init__.py +26 -0
- contextos_auditor/_internal/__init__.py +0 -0
- contextos_auditor/_internal/audit_emit.py +148 -0
- contextos_auditor/_internal/base.py +317 -0
- contextos_auditor/_internal/compat.py +101 -0
- contextos_auditor/_internal/dedupe.py +53 -0
- contextos_auditor/_internal/otel_export.py +124 -0
- contextos_auditor/_internal/pricing.py +92 -0
- contextos_auditor/_internal/redact.py +76 -0
- contextos_auditor/_internal/shadow_kit.py +309 -0
- contextos_auditor/_internal/tokens.py +57 -0
- contextos_auditor/autogen.py +180 -0
- contextos_auditor/cli.py +371 -0
- contextos_auditor/crewai.py +108 -0
- contextos_auditor/demo.py +193 -0
- contextos_auditor/langgraph.py +153 -0
- contextos_auditor/openai_agents.py +139 -0
- contextos_auditor/report.py +782 -0
- contextos_auditor/server.py +120 -0
- contextos_auditor-0.1.0.dist-info/METADATA +489 -0
- contextos_auditor-0.1.0.dist-info/RECORD +24 -0
- contextos_auditor-0.1.0.dist-info/WHEEL +4 -0
- contextos_auditor-0.1.0.dist-info/entry_points.txt +2 -0
- contextos_auditor-0.1.0.dist-info/licenses/LICENSE +155 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""contextos-auditor -- the free Agent Auditor.
|
|
2
|
+
|
|
3
|
+
Point it at a running CrewAI / LangGraph / AutoGen / OpenAI-Agents-SDK
|
|
4
|
+
session with one line, watch real token cost live, get a local report.
|
|
5
|
+
No code changes to your agent's tools/prompts, no signup, no data leaves
|
|
6
|
+
your machine.
|
|
7
|
+
|
|
8
|
+
Framework hooks live in their own submodules (imported only when you use
|
|
9
|
+
them, so an unused framework's SDK is never required):
|
|
10
|
+
|
|
11
|
+
from contextos_auditor.crewai import attach # pip install contextos-auditor[crewai]
|
|
12
|
+
from contextos_auditor.langgraph import AuditorCallback # [langgraph]
|
|
13
|
+
from contextos_auditor.openai_agents import attach # [openai-agents]
|
|
14
|
+
from contextos_auditor.autogen import new_session, wrap_client, audit_tool # [autogen]
|
|
15
|
+
|
|
16
|
+
Then from a terminal: `contextos-auditor watch` (see `contextos-auditor --help`).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from contextos_auditor._internal.audit_emit import AuditSession, load_events
|
|
22
|
+
from contextos_auditor._internal.shadow_kit import shadow_session
|
|
23
|
+
|
|
24
|
+
__version__ = "0.1.0"
|
|
25
|
+
|
|
26
|
+
__all__ = ["AuditSession", "load_events", "shadow_session", "__version__"]
|
|
File without changes
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Vendored from `dashboard/audit_emit.py` in the toku monorepo. Append-only
|
|
2
|
+
audit session writer -- unchanged logic, copied so this package installs
|
|
3
|
+
standalone outside the monorepo (see package README for the vendoring note).
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
import uuid
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class AuditSession:
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
root: Path,
|
|
21
|
+
*,
|
|
22
|
+
session_id: str,
|
|
23
|
+
model: str,
|
|
24
|
+
task: str,
|
|
25
|
+
arm: str = "baseline",
|
|
26
|
+
framework: str = "direct-api",
|
|
27
|
+
) -> None:
|
|
28
|
+
self.dir = Path(root) / session_id
|
|
29
|
+
self.dir.mkdir(parents=True, exist_ok=True)
|
|
30
|
+
self.events_path = self.dir / "events.jsonl"
|
|
31
|
+
self.session_path = self.dir / "session.json"
|
|
32
|
+
self.session_id = session_id
|
|
33
|
+
self.model = model
|
|
34
|
+
self.task = task
|
|
35
|
+
self.arm = arm
|
|
36
|
+
self.framework = framework
|
|
37
|
+
self.status = "running"
|
|
38
|
+
self.started_at = time.time()
|
|
39
|
+
# Framework adapters (crewai_adapter, etc.) hook event buses that
|
|
40
|
+
# dispatch sync handlers via a ThreadPoolExecutor, so two
|
|
41
|
+
# record_llm/emit_turn calls can genuinely run concurrently on the
|
|
42
|
+
# same AuditSession. Serialize file writes so the shared
|
|
43
|
+
# session.<pid>.tmp rename can't race across threads.
|
|
44
|
+
self._lock = threading.Lock()
|
|
45
|
+
self._write_session()
|
|
46
|
+
|
|
47
|
+
def emit_turn(
|
|
48
|
+
self,
|
|
49
|
+
turn: int,
|
|
50
|
+
*,
|
|
51
|
+
usage: dict[str, Any],
|
|
52
|
+
cost: dict[str, Any],
|
|
53
|
+
tool_calls: list[dict[str, Any]],
|
|
54
|
+
) -> None:
|
|
55
|
+
# Strip huge result bodies from non-read tools; keep read/write content
|
|
56
|
+
# needed for shadow_kit (bounded).
|
|
57
|
+
slim: list[dict[str, Any]] = []
|
|
58
|
+
for t in tool_calls:
|
|
59
|
+
row = {
|
|
60
|
+
"name": t.get("name"),
|
|
61
|
+
"args": t.get("args") or {},
|
|
62
|
+
"chars": t.get("chars"),
|
|
63
|
+
"tokens": t.get("tokens"),
|
|
64
|
+
}
|
|
65
|
+
name = t.get("name")
|
|
66
|
+
# Bounded result_text for every tool, not just read_file: KIT-504
|
|
67
|
+
# (dashboard.shadow_kit's levers_fired) needs to see the real
|
|
68
|
+
# Kit confirmation strings (edit_file/smart_patch/grep/list_dir/
|
|
69
|
+
# find_def/find_refs/run_shell all report what they did in a
|
|
70
|
+
# short marker string) to tell "lever fired" from "tool was
|
|
71
|
+
# merely called". These are cheap: unlike read_file's full file
|
|
72
|
+
# body, every other Kit tool's result is already a short,
|
|
73
|
+
# human-scale confirmation, not raw file content.
|
|
74
|
+
text = t.get("result_text") or t.get("output") or ""
|
|
75
|
+
if isinstance(text, str) and len(text) > 120_000:
|
|
76
|
+
text = text[:120_000]
|
|
77
|
+
row["result_text"] = text
|
|
78
|
+
if name == "write_file":
|
|
79
|
+
args = dict(row["args"])
|
|
80
|
+
content = str(args.get("content") or "")
|
|
81
|
+
if len(content) > 120_000:
|
|
82
|
+
args["content"] = content[:120_000]
|
|
83
|
+
row["args"] = args
|
|
84
|
+
slim.append(row)
|
|
85
|
+
event = {
|
|
86
|
+
"kind": "turn",
|
|
87
|
+
"turn": turn,
|
|
88
|
+
"ts": time.time(),
|
|
89
|
+
"usage": {
|
|
90
|
+
"prompt_tokens": int(usage.get("prompt_tokens") or 0),
|
|
91
|
+
"completion_tokens": int(usage.get("completion_tokens") or 0),
|
|
92
|
+
"total_tokens": int(usage.get("total_tokens") or 0),
|
|
93
|
+
},
|
|
94
|
+
"cost": {
|
|
95
|
+
"total_nano_aiu": int(cost.get("total_nano_aiu") or 0),
|
|
96
|
+
"input_tokens": int(cost.get("input_tokens") or 0),
|
|
97
|
+
"output_tokens": int(cost.get("output_tokens") or 0),
|
|
98
|
+
"cache_read_tokens": int(cost.get("cache_read_tokens") or 0),
|
|
99
|
+
"cache_write_tokens": int(cost.get("cache_write_tokens") or 0),
|
|
100
|
+
# AUD-011: dated, sourced $ estimate (see _internal/pricing.py) --
|
|
101
|
+
# None (not 0) when this session's model has no citable
|
|
102
|
+
# pricing snapshot, so shadow_kit never silently treats
|
|
103
|
+
# "unpriced" as "free".
|
|
104
|
+
"estimated_usd": cost.get("estimated_usd"),
|
|
105
|
+
},
|
|
106
|
+
"tool_calls": slim,
|
|
107
|
+
}
|
|
108
|
+
with self._lock:
|
|
109
|
+
with self.events_path.open("a") as f:
|
|
110
|
+
f.write(json.dumps(event, default=str) + "\n")
|
|
111
|
+
self._write_session()
|
|
112
|
+
|
|
113
|
+
def finish(self, *, success: bool | None = None, error: str = "") -> None:
|
|
114
|
+
self.status = "finished" if not error else "error"
|
|
115
|
+
with self._lock:
|
|
116
|
+
self._write_session(success=success, error=error)
|
|
117
|
+
|
|
118
|
+
def _write_session(
|
|
119
|
+
self, *, success: bool | None = None, error: str = ""
|
|
120
|
+
) -> None:
|
|
121
|
+
payload = {
|
|
122
|
+
"id": self.session_id,
|
|
123
|
+
"model": self.model,
|
|
124
|
+
"task": self.task,
|
|
125
|
+
"arm": self.arm,
|
|
126
|
+
"status": self.status,
|
|
127
|
+
"started_at": self.started_at,
|
|
128
|
+
"updated_at": time.time(),
|
|
129
|
+
"success": success,
|
|
130
|
+
"error": error,
|
|
131
|
+
"product": "agent-auditor",
|
|
132
|
+
"pricing": "free",
|
|
133
|
+
"framework": self.framework,
|
|
134
|
+
}
|
|
135
|
+
tmp = self.session_path.with_suffix(f".{os.getpid()}.{uuid.uuid4().hex[:8]}.tmp")
|
|
136
|
+
tmp.write_text(json.dumps(payload, indent=2) + "\n")
|
|
137
|
+
tmp.replace(self.session_path)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def load_events(session_dir: Path) -> list[dict]:
|
|
141
|
+
path = session_dir / "events.jsonl"
|
|
142
|
+
if not path.is_file():
|
|
143
|
+
return []
|
|
144
|
+
out: list[dict] = []
|
|
145
|
+
for line in path.read_text().splitlines():
|
|
146
|
+
if line.strip():
|
|
147
|
+
out.append(json.loads(line))
|
|
148
|
+
return out
|
|
@@ -0,0 +1,317 @@
|
|
|
1
|
+
"""Vendored from `dashboard/adapters/base.py` in the toku monorepo --
|
|
2
|
+
unchanged logic (default session directory adjusted for standalone install:
|
|
3
|
+
`./.contextos/audit` instead of `./out/audit`). See this package's README
|
|
4
|
+
for the vendoring note.
|
|
5
|
+
|
|
6
|
+
Each framework adapter (crewai.py, langgraph.py, openai_agents.py,
|
|
7
|
+
autogen.py) hooks that framework's native callback/event/tracing system and
|
|
8
|
+
forwards two kinds of observations here:
|
|
9
|
+
|
|
10
|
+
- an LLM call finishing, with whatever usage dict the framework exposes
|
|
11
|
+
- a tool call finishing, with its name/args/result
|
|
12
|
+
|
|
13
|
+
`FrameworkAuditSession` normalizes both into the same `AuditSession.emit_turn`
|
|
14
|
+
shape `contextos_auditor._internal.audit_emit` understands, so
|
|
15
|
+
`contextos_auditor._internal.shadow_kit` needs no framework-specific code.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import os
|
|
21
|
+
import sys
|
|
22
|
+
import threading
|
|
23
|
+
import time
|
|
24
|
+
import warnings
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
from typing import Any
|
|
27
|
+
|
|
28
|
+
from contextos_auditor._internal.audit_emit import AuditSession
|
|
29
|
+
from contextos_auditor._internal.dedupe import DedupeGuard
|
|
30
|
+
from contextos_auditor._internal.otel_export import build_exporter
|
|
31
|
+
from contextos_auditor._internal.pricing import estimate_usd
|
|
32
|
+
from contextos_auditor._internal.redact import redact_args, redact_text
|
|
33
|
+
|
|
34
|
+
# Tool-name aliases the file-write/read detector in shadow_kit.py understands.
|
|
35
|
+
# Framework tool names vary (write_file, WriteFileTool, apply_patch, ...); we
|
|
36
|
+
# normalize the common file-I/O ones so shadow_kit can still reconstruct
|
|
37
|
+
# file_state and estimate write_file -> edit_file waste. Anything else passes
|
|
38
|
+
# through verbatim as an opaque tool call (still visible in the UI, just not
|
|
39
|
+
# counted toward the write-waste estimate).
|
|
40
|
+
# NOTE when reading traces: this normalization is lossy in the persisted
|
|
41
|
+
# events.jsonl, so a kit-arm run that called `create_file` shows up as
|
|
42
|
+
# `write_file` -- a name the kit does not even expose. That looks alarming
|
|
43
|
+
# ("the kit arm called a baseline tool!") and is not: it is this alias map.
|
|
44
|
+
# Verified 2026-08-16 by invoking the kit's create_file tool directly and
|
|
45
|
+
# watching it record as write_file.
|
|
46
|
+
_WRITE_ALIASES = {"write_file", "writefile", "write_to_file", "create_file", "update_file"}
|
|
47
|
+
_READ_ALIASES = {"read_file", "readfile", "read_text_file"}
|
|
48
|
+
|
|
49
|
+
# AUD-009: this adapter observes a live, real agent run from inside its own
|
|
50
|
+
# process (registered directly on the framework's event bus / callback
|
|
51
|
+
# manager / model-client wrapper). If any of *our* code raises -- a
|
|
52
|
+
# malformed usage dict, an unexpected event shape from a newer/older SDK
|
|
53
|
+
# version, a full disk -- that exception must never propagate back into the
|
|
54
|
+
# agent's real execution path. An observability tool that can crash the
|
|
55
|
+
# thing it observes is disqualifying for production use, full stop.
|
|
56
|
+
#
|
|
57
|
+
# `guarded` wraps every adapter handler method and every FrameworkAuditSession
|
|
58
|
+
# entry point below: on failure it emits one `UserWarning` per distinct
|
|
59
|
+
# label per process (so failures stay discoverable, not silently swallowed
|
|
60
|
+
# forever) and returns None instead of raising.
|
|
61
|
+
_warned_labels: set[str] = set()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _warn_once(label: str, exc: Exception) -> None:
|
|
65
|
+
if label in _warned_labels:
|
|
66
|
+
return
|
|
67
|
+
_warned_labels.add(label)
|
|
68
|
+
warnings.warn(
|
|
69
|
+
f"contextos-auditor: internal error in {label}, this observation was "
|
|
70
|
+
f"dropped but your agent's real run is unaffected ({exc.__class__.__name__}: {exc}). "
|
|
71
|
+
"This warning only appears once per process; run with "
|
|
72
|
+
"`python -W always::UserWarning` for every occurrence.",
|
|
73
|
+
stacklevel=3,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def guarded(label: str):
|
|
78
|
+
"""Decorator: never let an exception out of the wrapped function. See
|
|
79
|
+
the module-level comment above for why this is not optional."""
|
|
80
|
+
|
|
81
|
+
def decorator(fn):
|
|
82
|
+
def wrapper(*args, **kwargs):
|
|
83
|
+
try:
|
|
84
|
+
return fn(*args, **kwargs)
|
|
85
|
+
except Exception as exc: # noqa: BLE001 -- intentional, see docstring
|
|
86
|
+
_warn_once(label, exc)
|
|
87
|
+
return None
|
|
88
|
+
|
|
89
|
+
wrapper.__name__ = getattr(fn, "__name__", label)
|
|
90
|
+
wrapper.__doc__ = fn.__doc__
|
|
91
|
+
return wrapper
|
|
92
|
+
|
|
93
|
+
return decorator
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _normalize_tool_name(name: str) -> str:
|
|
97
|
+
key = (name or "").strip().lower()
|
|
98
|
+
if key in _WRITE_ALIASES:
|
|
99
|
+
return "write_file"
|
|
100
|
+
if key in _READ_ALIASES:
|
|
101
|
+
return "read_file"
|
|
102
|
+
return name or "tool"
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _extract_path(args: dict[str, Any]) -> str | None:
|
|
106
|
+
for key in ("path", "file_path", "filename", "file_name"):
|
|
107
|
+
if args.get(key):
|
|
108
|
+
return str(args[key])
|
|
109
|
+
return None
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _extract_write_content(args: dict[str, Any]) -> str | None:
|
|
113
|
+
for key in ("content", "text", "new_str", "new_content"):
|
|
114
|
+
if args.get(key) is not None:
|
|
115
|
+
return str(args[key])
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
_announced = False
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _announce_capture(session_dir: Path, redacting: bool) -> None:
|
|
123
|
+
"""Tell the user, once per process, what is being written to disk.
|
|
124
|
+
|
|
125
|
+
Silence here is the wrong default: this records real tool args and
|
|
126
|
+
results (including file contents) into a plaintext file under the
|
|
127
|
+
user's cwd, and someone who never scrolled to the README's Privacy
|
|
128
|
+
section could commit that. Printing the exact path once costs two
|
|
129
|
+
lines of stderr and removes the surprise. Honour CONTEXTOS_QUIET=1 for
|
|
130
|
+
scripted/CI use.
|
|
131
|
+
"""
|
|
132
|
+
global _announced
|
|
133
|
+
if _announced or os.environ.get("CONTEXTOS_QUIET", "") not in ("", "0", "false", "False"):
|
|
134
|
+
return
|
|
135
|
+
_announced = True
|
|
136
|
+
scrub = "secret-pattern redaction ON" if redacting else "secret-pattern redaction OFF"
|
|
137
|
+
try:
|
|
138
|
+
rel: str = str(Path(session_dir).relative_to(Path.cwd()))
|
|
139
|
+
except ValueError:
|
|
140
|
+
rel = str(session_dir)
|
|
141
|
+
print(
|
|
142
|
+
f"[contextos-auditor] recording this run to ./{rel} ({scrub}).\n"
|
|
143
|
+
f"[contextos-auditor] nothing leaves your machine. add '.contextos/' to .gitignore.",
|
|
144
|
+
file=sys.stderr,
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
class FrameworkAuditSession:
|
|
149
|
+
"""Thin wrapper an adapter calls; keeps each adapter file tiny."""
|
|
150
|
+
|
|
151
|
+
def __init__(
|
|
152
|
+
self,
|
|
153
|
+
*,
|
|
154
|
+
framework: str,
|
|
155
|
+
model: str | None,
|
|
156
|
+
task: str,
|
|
157
|
+
out_dir: Path | None = None,
|
|
158
|
+
session_id: str | None = None,
|
|
159
|
+
arm: str = "baseline",
|
|
160
|
+
workspace_root: str | Path | None = None,
|
|
161
|
+
otel_endpoint: str | None = None,
|
|
162
|
+
redact_secrets: bool | None = None,
|
|
163
|
+
) -> None:
|
|
164
|
+
root = out_dir or (Path.cwd() / ".contextos" / "audit")
|
|
165
|
+
sid = session_id or f"{framework}-{int(time.time() * 1000)}"
|
|
166
|
+
self._session = AuditSession(
|
|
167
|
+
root,
|
|
168
|
+
session_id=sid,
|
|
169
|
+
model=model or "unknown",
|
|
170
|
+
task=task[:240],
|
|
171
|
+
arm=arm,
|
|
172
|
+
framework=framework,
|
|
173
|
+
)
|
|
174
|
+
self._turn = 0
|
|
175
|
+
self._pending_tools: list[dict[str, Any]] = []
|
|
176
|
+
# AUDIT-001: opt-in confinement for kit/tools.py's path-touching
|
|
177
|
+
# functions (getattr(session, "workspace_root", None), same
|
|
178
|
+
# zero-cost-when-absent pattern as dedupe_guard below). None here
|
|
179
|
+
# means "no confinement", matching every caller from before this
|
|
180
|
+
# existed.
|
|
181
|
+
self.workspace_root = workspace_root
|
|
182
|
+
# KIT-006: every FrameworkAuditSession carries a live DedupeGuard so
|
|
183
|
+
# `kit.tools`' read-only tools (read_file/grep/list_dir/find_def/
|
|
184
|
+
# find_refs) automatically short-circuit an identical back-to-back
|
|
185
|
+
# repeat instead of recomputing/resending the same result — this is
|
|
186
|
+
# what makes the fix apply to real live runs (e.g.
|
|
187
|
+
# scripts/kit_demo/run_live_enterprise.py) with no per-script
|
|
188
|
+
# wiring, not just to unit tests that build one explicitly. Purely
|
|
189
|
+
# additive: a session this is passed to only ever consults it via
|
|
190
|
+
# `getattr(session, "dedupe_guard", None)`, so anything that never
|
|
191
|
+
# looks for the attribute (e.g. this class's own record_tool below)
|
|
192
|
+
# is completely unaffected by its presence.
|
|
193
|
+
self.dedupe_guard = DedupeGuard()
|
|
194
|
+
# Framework event buses commonly dispatch callbacks from a thread
|
|
195
|
+
# pool (e.g. CrewAI's crewai_event_bus submits sync handlers via
|
|
196
|
+
# ThreadPoolExecutor) rather than calling them inline, so two
|
|
197
|
+
# record_* calls for genuinely sequential events can still race on
|
|
198
|
+
# this buffer. A lock keeps each call atomic; it does not by itself
|
|
199
|
+
# guarantee cross-call ordering — callers that need strict ordering
|
|
200
|
+
# under a threaded bus should serialize their own emits (see
|
|
201
|
+
# crewai_adapter's test for the direct-call pattern that sidesteps
|
|
202
|
+
# this entirely).
|
|
203
|
+
self._lock = threading.Lock()
|
|
204
|
+
# AUD-012: opt-in only -- explicit otel_endpoint kwarg wins, else
|
|
205
|
+
# fall back to CONTEXTOS_OTEL_ENDPOINT so a zero-code-change env var
|
|
206
|
+
# can enable it for any of the 4 adapters. build_exporter() returns
|
|
207
|
+
# None (no object at all) when neither is set, so every existing
|
|
208
|
+
# caller pays nothing for this feature's existence.
|
|
209
|
+
endpoint = otel_endpoint or os.environ.get("CONTEXTOS_OTEL_ENDPOINT")
|
|
210
|
+
self._otel = build_exporter(endpoint, framework=framework)
|
|
211
|
+
# AUD-016 / LNCH-004: pattern-based secret scrub, ON BY DEFAULT --
|
|
212
|
+
# see _internal/redact.py's module docstring for exactly what this
|
|
213
|
+
# does and does not do (never blanks read/write_file content
|
|
214
|
+
# wholesale, since that would break the waste-detection feature).
|
|
215
|
+
# Explicit False, or CONTEXTOS_REDACT_SECRETS=0, opts out.
|
|
216
|
+
if redact_secrets is None:
|
|
217
|
+
redact_secrets = os.environ.get("CONTEXTOS_REDACT_SECRETS", "1") not in ("0", "false", "False", "no", "off")
|
|
218
|
+
self._redact_secrets = redact_secrets
|
|
219
|
+
_announce_capture(self._session.dir, redact_secrets)
|
|
220
|
+
|
|
221
|
+
@property
|
|
222
|
+
def session_id(self) -> str:
|
|
223
|
+
return self._session.session_id
|
|
224
|
+
|
|
225
|
+
@guarded("FrameworkAuditSession.record_tool")
|
|
226
|
+
def record_tool(self, name: str, args: Any, result: Any) -> None:
|
|
227
|
+
"""Buffer a tool call; flushed into the next `record_llm` turn.
|
|
228
|
+
|
|
229
|
+
Tool calls between two LLM calls are attributed to the *next* LLM
|
|
230
|
+
turn's event (matching how the observation-only harness scripts
|
|
231
|
+
already interleave usage + tool_calls — see scripts/copilot_ctx_ab.py).
|
|
232
|
+
"""
|
|
233
|
+
args_dict = args if isinstance(args, dict) else {"value": args}
|
|
234
|
+
norm_name = _normalize_tool_name(name)
|
|
235
|
+
row: dict[str, Any] = {"name": norm_name, "args": args_dict}
|
|
236
|
+
# KIT-504: every tool's result_text gets forwarded (audit_emit.py
|
|
237
|
+
# bounds it further before persisting), not just read_file's — Kit's
|
|
238
|
+
# own lever tools (edit_file/smart_patch/grep/list_dir/find_def/
|
|
239
|
+
# find_refs/run_shell) report what they actually did in a short
|
|
240
|
+
# confirmation string, and dashboard.shadow_kit's levers_fired needs
|
|
241
|
+
# that text to tell "fired" from "merely called, but failed/no-op".
|
|
242
|
+
row["result_text"] = result if isinstance(result, str) else str(result)
|
|
243
|
+
if norm_name == "write_file":
|
|
244
|
+
path = _extract_path(args_dict)
|
|
245
|
+
content = _extract_write_content(args_dict)
|
|
246
|
+
if path is not None and content is not None:
|
|
247
|
+
row["args"] = {"path": path, "content": content}
|
|
248
|
+
row["chars"] = len(str(result)) if result is not None else 0
|
|
249
|
+
if self._redact_secrets:
|
|
250
|
+
row["args"] = redact_args(row["args"])
|
|
251
|
+
row["result_text"] = redact_text(row["result_text"])
|
|
252
|
+
with self._lock:
|
|
253
|
+
self._pending_tools.append(row)
|
|
254
|
+
|
|
255
|
+
@guarded("FrameworkAuditSession.record_llm")
|
|
256
|
+
def record_llm(self, usage: dict[str, Any] | None, model: str | None = None) -> None:
|
|
257
|
+
"""Flush buffered tool calls together with this LLM call's usage."""
|
|
258
|
+
# Most adapters don't know the real model name until the first LLM
|
|
259
|
+
# call actually returns one (e.g. CrewAI's LLMCallCompletedEvent.model)
|
|
260
|
+
# -- construction time only has whatever model_hint the caller passed,
|
|
261
|
+
# which defaults to "unknown". Backfill it here so `watch`/the
|
|
262
|
+
# dashboard show the real model instead of "unknown" for the entire
|
|
263
|
+
# session's lifetime.
|
|
264
|
+
if model and self._session.model in (None, "unknown"):
|
|
265
|
+
self._session.model = model
|
|
266
|
+
usage = usage or {}
|
|
267
|
+
prompt = int(usage.get("prompt_tokens") or usage.get("input_tokens") or 0)
|
|
268
|
+
completion = int(
|
|
269
|
+
usage.get("completion_tokens") or usage.get("output_tokens") or 0
|
|
270
|
+
)
|
|
271
|
+
total = int(usage.get("total_tokens") or (prompt + completion))
|
|
272
|
+
with self._lock:
|
|
273
|
+
self._turn += 1
|
|
274
|
+
turn = self._turn
|
|
275
|
+
tool_calls = self._pending_tools
|
|
276
|
+
self._pending_tools = []
|
|
277
|
+
self._session.emit_turn(
|
|
278
|
+
turn,
|
|
279
|
+
usage={
|
|
280
|
+
"prompt_tokens": prompt,
|
|
281
|
+
"completion_tokens": completion,
|
|
282
|
+
"total_tokens": total,
|
|
283
|
+
},
|
|
284
|
+
cost={
|
|
285
|
+
"total_nano_aiu": 0,
|
|
286
|
+
"estimated_usd": estimate_usd(self._session.model, prompt, completion),
|
|
287
|
+
},
|
|
288
|
+
tool_calls=tool_calls,
|
|
289
|
+
)
|
|
290
|
+
if self._otel is not None:
|
|
291
|
+
for call in tool_calls:
|
|
292
|
+
self._otel.export_tool_call(name=call["name"], turn=turn)
|
|
293
|
+
self._otel.export_llm_turn(
|
|
294
|
+
turn=turn,
|
|
295
|
+
model=self._session.model,
|
|
296
|
+
prompt_tokens=prompt,
|
|
297
|
+
completion_tokens=completion,
|
|
298
|
+
total_tokens=total,
|
|
299
|
+
estimated_usd=estimate_usd(self._session.model, prompt, completion),
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
@guarded("FrameworkAuditSession.finish")
|
|
303
|
+
def finish(self, *, success: bool | None = None, error: str = "") -> None:
|
|
304
|
+
with self._lock:
|
|
305
|
+
leftover = self._pending_tools
|
|
306
|
+
self._pending_tools = []
|
|
307
|
+
if leftover:
|
|
308
|
+
self._turn += 1
|
|
309
|
+
turn = self._turn
|
|
310
|
+
if leftover:
|
|
311
|
+
self._session.emit_turn(
|
|
312
|
+
turn,
|
|
313
|
+
usage={"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
|
|
314
|
+
cost={"total_nano_aiu": 0},
|
|
315
|
+
tool_calls=leftover,
|
|
316
|
+
)
|
|
317
|
+
self._session.finish(success=success, error=error)
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""AUD-007: SDK-version drift is a real, quiet failure mode for an
|
|
2
|
+
observation-only adapter -- if a framework changes an event's field names
|
|
3
|
+
or a client wrapper's method signature between versions, the Auditor can
|
|
4
|
+
silently under-count tokens/tool calls instead of raising a loud error,
|
|
5
|
+
because it never sits in the agent's execution path (that's the whole
|
|
6
|
+
point of it being unobtrusive).
|
|
7
|
+
|
|
8
|
+
This module is the single place that knows which SDK versions each
|
|
9
|
+
vendored adapter was actually tested against. It never blocks or raises --
|
|
10
|
+
an outdated/newer SDK should still be usable -- it only emits one
|
|
11
|
+
`UserWarning` per (framework, installed version) so a user investigating
|
|
12
|
+
"my savings numbers look wrong" has an obvious first thing to check, and
|
|
13
|
+
`contextos-auditor doctor` surfaces the same information proactively.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import importlib
|
|
19
|
+
import warnings
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True)
|
|
24
|
+
class _TestedRange:
|
|
25
|
+
module: str
|
|
26
|
+
min_version: tuple[int, ...]
|
|
27
|
+
max_version_exclusive: tuple[int, ...]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Update these when the corresponding _internal/*.py vendor copy is
|
|
31
|
+
# re-synced against a newer framework release and re-verified against its
|
|
32
|
+
# real test suite (see each adapter's module docstring + the README's
|
|
33
|
+
# "vendored, not auto-synced" disclaimer).
|
|
34
|
+
_TESTED: dict[str, _TestedRange] = {
|
|
35
|
+
"crewai": _TestedRange("crewai", (0, 100, 0), (2, 0, 0)),
|
|
36
|
+
"langgraph": _TestedRange("langchain_core", (0, 3, 0), (2, 0, 0)),
|
|
37
|
+
"openai_agents": _TestedRange("agents", (0, 1, 0), (1, 0, 0)),
|
|
38
|
+
"autogen": _TestedRange("autogen_core", (0, 4, 0), (1, 0, 0)),
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
_warned: set[str] = set()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _parse_version(raw: str) -> tuple[int, ...]:
|
|
45
|
+
parts: list[int] = []
|
|
46
|
+
for chunk in raw.split(".")[:3]:
|
|
47
|
+
digits = ""
|
|
48
|
+
for ch in chunk:
|
|
49
|
+
if ch.isdigit():
|
|
50
|
+
digits += ch
|
|
51
|
+
else:
|
|
52
|
+
break
|
|
53
|
+
parts.append(int(digits) if digits else 0)
|
|
54
|
+
while len(parts) < 3:
|
|
55
|
+
parts.append(0)
|
|
56
|
+
return tuple(parts)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def installed_version(framework: str) -> str | None:
|
|
60
|
+
"""Best-effort installed-version lookup for the SDK module backing
|
|
61
|
+
`framework` (one of "crewai"/"langgraph"/"openai_agents"/"autogen").
|
|
62
|
+
Returns None if the SDK isn't installed at all."""
|
|
63
|
+
spec = _TESTED.get(framework)
|
|
64
|
+
if spec is None:
|
|
65
|
+
return None
|
|
66
|
+
try:
|
|
67
|
+
mod = importlib.import_module(spec.module)
|
|
68
|
+
except ImportError:
|
|
69
|
+
return None
|
|
70
|
+
return getattr(mod, "__version__", None)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def check_compat(framework: str) -> None:
|
|
74
|
+
"""Warn (once per framework per process) if the installed SDK version
|
|
75
|
+
falls outside the range this adapter was last verified against. Never
|
|
76
|
+
raises -- an out-of-range SDK is a "double-check your numbers" signal,
|
|
77
|
+
not a hard failure, since many minor-version bumps won't actually
|
|
78
|
+
change the event shapes this adapter reads."""
|
|
79
|
+
spec = _TESTED.get(framework)
|
|
80
|
+
if spec is None or framework in _warned:
|
|
81
|
+
return
|
|
82
|
+
|
|
83
|
+
raw = installed_version(framework)
|
|
84
|
+
if raw is None:
|
|
85
|
+
return # not installed / no __version__ -- nothing useful to say
|
|
86
|
+
|
|
87
|
+
version = _parse_version(raw)
|
|
88
|
+
if spec.min_version <= version < spec.max_version_exclusive:
|
|
89
|
+
return
|
|
90
|
+
|
|
91
|
+
_warned.add(framework)
|
|
92
|
+
tested_lo = ".".join(str(p) for p in spec.min_version)
|
|
93
|
+
tested_hi = ".".join(str(p) for p in spec.max_version_exclusive)
|
|
94
|
+
warnings.warn(
|
|
95
|
+
f"contextos-auditor: {spec.module} {raw} is outside the range this "
|
|
96
|
+
f"{framework} adapter was last verified against ([{tested_lo}, {tested_hi})). "
|
|
97
|
+
"It will likely still work, but if the numbers you see look wrong, "
|
|
98
|
+
"this version drift is the first thing to check -- run "
|
|
99
|
+
"`contextos-auditor doctor` for the full compatibility report.",
|
|
100
|
+
stacklevel=3,
|
|
101
|
+
)
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Vendored (trimmed) from `kit/dedupe.py` in the toku monorepo.
|
|
2
|
+
|
|
3
|
+
Per-session duplicate-tool-call detector. Kept here mainly so
|
|
4
|
+
`FrameworkAuditSession` (in `_internal/base.py`) has a `dedupe_guard`
|
|
5
|
+
attribute for parity with the monorepo's adapters — it is not required for
|
|
6
|
+
the Auditor's own read-only observation and estimation, only consulted by
|
|
7
|
+
tools that explicitly opt in via `getattr(session, "dedupe_guard", None)`.
|
|
8
|
+
See the monorepo's `kit/dedupe.py` for the full incident history (KIT-006
|
|
9
|
+
through KIT-039) this design encodes.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _signature(name: str, args: dict[str, Any]) -> tuple:
|
|
21
|
+
return (name, tuple(sorted((k, repr(v)) for k, v in args.items())))
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass
|
|
25
|
+
class DedupeGuard:
|
|
26
|
+
WRITE_DEDUPE_TTL_S = 3.0
|
|
27
|
+
|
|
28
|
+
_seen: dict[tuple, int] = field(default_factory=dict)
|
|
29
|
+
_epoch: int = 0
|
|
30
|
+
_write_seen: dict[tuple, float] = field(default_factory=dict)
|
|
31
|
+
_lock: threading.Lock = field(default_factory=threading.Lock)
|
|
32
|
+
|
|
33
|
+
def check(self, name: str, args: dict[str, Any]) -> bool:
|
|
34
|
+
sig = _signature(name, args)
|
|
35
|
+
with self._lock:
|
|
36
|
+
is_dup = self._seen.get(sig) == self._epoch
|
|
37
|
+
self._seen[sig] = self._epoch
|
|
38
|
+
return is_dup
|
|
39
|
+
|
|
40
|
+
def bump(self) -> None:
|
|
41
|
+
with self._lock:
|
|
42
|
+
self._epoch += 1
|
|
43
|
+
|
|
44
|
+
def check_write(self, name: str, args: dict[str, Any], *, ttl: float | None = None) -> bool:
|
|
45
|
+
ttl = self.WRITE_DEDUPE_TTL_S if ttl is None else ttl
|
|
46
|
+
sig = _signature(name, args)
|
|
47
|
+
now = time.monotonic()
|
|
48
|
+
with self._lock:
|
|
49
|
+
last = self._write_seen.get(sig)
|
|
50
|
+
is_dup = last is not None and (now - last) < ttl
|
|
51
|
+
if not is_dup:
|
|
52
|
+
self._write_seen[sig] = now
|
|
53
|
+
return is_dup
|