aftersight 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aftersight/__init__.py +21 -0
- aftersight/assets/NAVIGATE.md +129 -0
- aftersight/assets/skill/SKILL.md +39 -0
- aftersight/autostart.py +8 -0
- aftersight/cli.py +91 -0
- aftersight/config.py +60 -0
- aftersight/constants.py +125 -0
- aftersight/events/__init__.py +0 -0
- aftersight/events/schemas.py +59 -0
- aftersight/events/sink.py +58 -0
- aftersight/integrations/__init__.py +0 -0
- aftersight/integrations/otel.py +249 -0
- aftersight/integrations/stdlog.py +49 -0
- aftersight/py.typed +0 -0
- aftersight/run/__init__.py +0 -0
- aftersight/run/api.py +211 -0
- aftersight/run/run.py +164 -0
- aftersight/run/store.py +134 -0
- aftersight/utils.py +55 -0
- aftersight/writers/__init__.py +0 -0
- aftersight/writers/analytics.py +108 -0
- aftersight/writers/blobs.py +42 -0
- aftersight/writers/outline.py +120 -0
- aftersight/writers/trace.py +40 -0
- aftersight/writers/transcript.py +194 -0
- aftersight-0.1.0.dist-info/METADATA +157 -0
- aftersight-0.1.0.dist-info/RECORD +30 -0
- aftersight-0.1.0.dist-info/WHEEL +4 -0
- aftersight-0.1.0.dist-info/entry_points.txt +2 -0
- aftersight-0.1.0.dist-info/licenses/LICENSE +21 -0
aftersight/__init__.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Observability infrastructure for self-improving agents.
|
|
2
|
+
|
|
3
|
+
import aftersight
|
|
4
|
+
aftersight.start()
|
|
5
|
+
|
|
6
|
+
Or, with no code change at all:
|
|
7
|
+
|
|
8
|
+
aftersight run python my_agent.py
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from aftersight.run.api import (
|
|
12
|
+
artifact_dir,
|
|
13
|
+
current,
|
|
14
|
+
log,
|
|
15
|
+
span,
|
|
16
|
+
start,
|
|
17
|
+
trace,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
__all__ = ["start", "span", "trace", "log", "current", "artifact_dir"]
|
|
21
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# Navigating these runs
|
|
2
|
+
|
|
3
|
+
You are a coding agent. This directory is the execution history of the agent
|
|
4
|
+
system in this repo. Read this file once, then use `rg` and `jq`.
|
|
5
|
+
|
|
6
|
+
## Layout
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
index.jsonl one line per run, start here for "which run?"
|
|
10
|
+
latest -> runs/<run_id> most recent run
|
|
11
|
+
sessions/<session_id> -> the run folder that session resumed into
|
|
12
|
+
runs/<run_id>/
|
|
13
|
+
outline.md folded map of the run, read this first
|
|
14
|
+
agent.logs full transcript, nothing truncated
|
|
15
|
+
trace.jsonl same events, machine-readable, has parent_seq
|
|
16
|
+
analytics.json status, timings, per-agent cost, error list
|
|
17
|
+
meta.json framework, git sha, argv, attempts[]
|
|
18
|
+
blobs/<sha>.txt payloads over 8 KB, plain text
|
|
19
|
+
artifacts/ whatever the app chose to keep
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Run ids are `YYYYMMDD_HHMMSS_hash`, so they sort chronologically. Glob run
|
|
23
|
+
files as `runs/*/...` and never from the root: `latest` and `sessions/` are
|
|
24
|
+
symlinks into `runs/`, so a root glob counts every run twice.
|
|
25
|
+
|
|
26
|
+
## The three-step loop
|
|
27
|
+
|
|
28
|
+
1. **Which run?** `index.jsonl`
|
|
29
|
+
2. **What happened in it?** `runs/<run_id>/outline.md`, then jump to a `#seq`
|
|
30
|
+
3. **Exactly what went in and out?** `runs/<run_id>/agent.logs` at that `#seq`
|
|
31
|
+
|
|
32
|
+
## Anchors
|
|
33
|
+
|
|
34
|
+
Every event has a zero-padded sequence number. It is the same number in
|
|
35
|
+
`outline.md`, `agent.logs` and `trace.jsonl`, so you can move between them:
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
rg -n '^#0016 ' latest/agent.logs # jump to one event
|
|
39
|
+
rg -n '#0016' runs/<run_id>/ # every mention of it
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Multi-line payloads are framed. The closing line repeats the anchor, so a hit
|
|
43
|
+
landing inside a block can find both edges:
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
#0004 14:22:35.512 +2.4s llm.response triage stop=end_turn · 386 tok · $0.0005
|
|
47
|
+
┌──────────────────────────────────
|
|
48
|
+
{"decision": "work", ...}
|
|
49
|
+
└─ #0004 end · 86 B ───────────────
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Column 3 is the delta from the previous event. Scan it to find slow steps. A
|
|
53
|
+
negative delta means an event whose span declared its kind late and was therefore
|
|
54
|
+
placed at its end rather than its start; the timestamp is still correct.
|
|
55
|
+
|
|
56
|
+
## Recipes
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
# failed runs, newest last
|
|
60
|
+
jq -r 'select(.status!="ok") | "\(.run_id) \(.errors) errs $\(.cost_usd) \(.last_error)"' index.jsonl
|
|
61
|
+
|
|
62
|
+
# the most common failure across all history
|
|
63
|
+
rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
|
|
64
|
+
|
|
65
|
+
# every error in one run, with its anchor
|
|
66
|
+
# filter on payload.error_type, not .status, agent.end also carries status "error"
|
|
67
|
+
jq -r 'select(.payload.error_type) | "#\(.seq|tostring|("0000"+.)[-4:]) \(.name) \(.payload.error_type): \(.payload.error)"' latest/trace.jsonl
|
|
68
|
+
|
|
69
|
+
# what a tool was actually called with, every time
|
|
70
|
+
rg -N '^#\d+ .* tool\.call run_python' runs/*/agent.logs
|
|
71
|
+
|
|
72
|
+
# read one event and its payload block
|
|
73
|
+
rg -n -A40 '^#0023 ' latest/agent.logs
|
|
74
|
+
|
|
75
|
+
# where the money went
|
|
76
|
+
jq -r '.cost.by_agent | to_entries[] | "\(.key) $\(.value.cost_usd)"' latest/analytics.json
|
|
77
|
+
|
|
78
|
+
# slowest steps in a run, exclude run.* or the run total wins every time
|
|
79
|
+
jq -r 'select(.dur_ms!=null and (.type|startswith("run.")|not)) | "\(.dur_ms)ms #\(.seq|tostring|("0000"+.)[-4:]) \(.type) \(.name)"' latest/trace.jsonl | sort -rn | head
|
|
80
|
+
|
|
81
|
+
# did this session get resumed, and why did the earlier attempt end?
|
|
82
|
+
jq '.attempts' latest/meta.json
|
|
83
|
+
rg -n '^==== attempt|^---- attempt' latest/agent.logs
|
|
84
|
+
|
|
85
|
+
# a prompt too big to inline
|
|
86
|
+
rg -n 'blobs/' latest/agent.logs # find the pointer
|
|
87
|
+
rg -n 'PCA' latest/blobs/*.txt # then grep the blob like any text file
|
|
88
|
+
|
|
89
|
+
# same failure across runs, or just this one?
|
|
90
|
+
rg -lN 'TimeoutError' runs/*/agent.logs
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Counting things correctly
|
|
94
|
+
|
|
95
|
+
Three traps, each of which silently produces a wrong answer rather than an error:
|
|
96
|
+
|
|
97
|
+
- `status: "error"` appears on `agent.end` too. Count failures by
|
|
98
|
+
`payload.error_type`.
|
|
99
|
+
- `run.end` carries the run total in `dur_ms`. Exclude `run.*` when ranking steps.
|
|
100
|
+
- `agent.end` carries a cost rollup. Sum leaf costs, or read
|
|
101
|
+
`analytics.json → cost.by_agent`.
|
|
102
|
+
|
|
103
|
+
## Event types
|
|
104
|
+
|
|
105
|
+
`run.start` `run.resume` `run.end` · `agent.start` `agent.end` ·
|
|
106
|
+
`llm.prompt` `llm.response` · `tool.call` `tool.result` `tool.error` ·
|
|
107
|
+
`error` · `artifact` · `log`
|
|
108
|
+
|
|
109
|
+
Namespaced on purpose: `rg 'tool\.'` gets all tool activity, `rg '\.error'`
|
|
110
|
+
gets every failure shape.
|
|
111
|
+
|
|
112
|
+
## Resume
|
|
113
|
+
|
|
114
|
+
A session that is resumed appends into the same run folder under a new
|
|
115
|
+
`==== attempt N ====` banner. Sequence numbers continue across attempts and
|
|
116
|
+
are never reused. An attempt that died without `run.end` is marked
|
|
117
|
+
`---- attempt N ended ... without run.end ----`, which is how you tell a crash
|
|
118
|
+
from a clean failure.
|
|
119
|
+
|
|
120
|
+
## When you are looking for a systemic problem
|
|
121
|
+
|
|
122
|
+
Do not read one run. Aggregate first:
|
|
123
|
+
|
|
124
|
+
```sh
|
|
125
|
+
rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
|
|
126
|
+
jq -s 'group_by(.status)[] | {status: .[0].status, n: length}' index.jsonl
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Then open the cheapest run that shows the pattern, not the biggest.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: aftersight
|
|
3
|
+
description: Use when investigating how this project's agent behaved: a failed or slow agent run, a recurring tool failure, unexpected model output, rising cost, or any question about what an agent actually sent and received. Reads the execution traces in .runs/.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Reading this project's agent traces
|
|
7
|
+
|
|
8
|
+
Agent runs in this repo record to `.runs/`. That directory is the evidence for
|
|
9
|
+
what an agent actually did. Read it before theorising about behaviour.
|
|
10
|
+
|
|
11
|
+
Start with `.runs/NAVIGATE.md`. It documents the layout, the `#seq` anchor
|
|
12
|
+
scheme and a set of verified `rg`/`jq` recipes.
|
|
13
|
+
|
|
14
|
+
## The loop
|
|
15
|
+
|
|
16
|
+
1. `.runs/index.jsonl` holds one line per run. Pick the run.
|
|
17
|
+
2. `.runs/runs/<run_id>/outline.md` is the folded map. Find the `#seq` that
|
|
18
|
+
matters.
|
|
19
|
+
3. `.runs/runs/<run_id>/agent.logs` at that anchor has the exact prompt,
|
|
20
|
+
completion, tool arguments or traceback.
|
|
21
|
+
|
|
22
|
+
## Rules that keep answers correct
|
|
23
|
+
|
|
24
|
+
- Aggregate before you read. One run shows an incident; `runs/*/trace.jsonl`
|
|
25
|
+
shows whether it is systemic.
|
|
26
|
+
- Glob as `runs/*/…`. `latest` and `sessions/` are symlinks into `runs/`, so
|
|
27
|
+
globbing the root double-counts.
|
|
28
|
+
- Count failures by `payload.error_type`, not `status == "error"`, because
|
|
29
|
+
`agent.end` carries that status too.
|
|
30
|
+
- Exclude `run.*` when ranking slow steps; `run.end` holds the run total.
|
|
31
|
+
- A prompt over 8 KB lives in `blobs/` as plain text. Grep it like any file.
|
|
32
|
+
- Sequence numbers continue across a resumed session, and an attempt that died
|
|
33
|
+
without `run.end` is marked as such. A crash and a clean failure are different
|
|
34
|
+
findings.
|
|
35
|
+
|
|
36
|
+
## When asked to improve the agent
|
|
37
|
+
|
|
38
|
+
Ground every claim in an anchor. "Tool `x` timed out in 4 of the last 9 runs
|
|
39
|
+
(`#0044`, `#0071`, …)" is actionable; "the executor seems flaky" is not.
|
aftersight/autostart.py
ADDED
aftersight/cli.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Two commands. Anything else `rg` and `jq` already do better.
|
|
2
|
+
|
|
3
|
+
aftersight run python my_agent.py # record, no code change
|
|
4
|
+
aftersight skill # teach your coding agent to read it
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import os
|
|
11
|
+
import shutil
|
|
12
|
+
import subprocess
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from aftersight import constants
|
|
18
|
+
|
|
19
|
+
_SITECUSTOMIZE = '''\
|
|
20
|
+
# Generated by `aftersight run`. Chains to any real sitecustomize.
|
|
21
|
+
import os, sys
|
|
22
|
+
|
|
23
|
+
_here = os.path.dirname(os.path.abspath(__file__))
|
|
24
|
+
sys.path[:] = [p for p in sys.path if os.path.abspath(p) != _here]
|
|
25
|
+
sys.modules.pop("sitecustomize", None)
|
|
26
|
+
try:
|
|
27
|
+
import sitecustomize # noqa: F401 the project's own, if it has one
|
|
28
|
+
except ImportError:
|
|
29
|
+
pass
|
|
30
|
+
try:
|
|
31
|
+
import aftersight.autostart # noqa: F401
|
|
32
|
+
except Exception as exc: # noqa: BLE001
|
|
33
|
+
print(f"aftersight: autostart failed: {exc}", file=sys.stderr)
|
|
34
|
+
'''
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def cmd_run(args: argparse.Namespace) -> int:
|
|
38
|
+
"""Records a program without editing it.
|
|
39
|
+
|
|
40
|
+
A generated `sitecustomize` on PYTHONPATH is how coverage-style tools attach
|
|
41
|
+
to an interpreter they do not control; it also survives `uv run`, `pytest`
|
|
42
|
+
and any wrapper that eventually execs Python.
|
|
43
|
+
"""
|
|
44
|
+
if not args.command:
|
|
45
|
+
print("usage: aftersight run <command> [args...]", file=sys.stderr)
|
|
46
|
+
return 2
|
|
47
|
+
staging = tempfile.mkdtemp(prefix="aftersight-")
|
|
48
|
+
try:
|
|
49
|
+
(Path(staging) / "sitecustomize.py").write_text(_SITECUSTOMIZE)
|
|
50
|
+
env = dict(os.environ)
|
|
51
|
+
env[constants.ENV_AUTOSTART] = "1"
|
|
52
|
+
env["PYTHONPATH"] = os.pathsep.join(
|
|
53
|
+
[staging] + ([env["PYTHONPATH"]] if env.get("PYTHONPATH") else []))
|
|
54
|
+
if args.session:
|
|
55
|
+
env[constants.ENV_SESSION] = args.session
|
|
56
|
+
if args.root:
|
|
57
|
+
env[constants.ENV_ROOT] = args.root
|
|
58
|
+
return subprocess.run(args.command, env=env).returncode
|
|
59
|
+
finally:
|
|
60
|
+
shutil.rmtree(staging, ignore_errors=True)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def cmd_skill(args: argparse.Namespace) -> int:
|
|
64
|
+
source = Path(__file__).resolve().parent / "assets" / "skill" / "SKILL.md"
|
|
65
|
+
target = Path(args.into) / "aftersight" / "SKILL.md"
|
|
66
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
67
|
+
target.write_text(source.read_text())
|
|
68
|
+
print(f"installed {target}")
|
|
69
|
+
return 0
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def main(argv: list[str] | None = None) -> int:
|
|
73
|
+
parser = argparse.ArgumentParser(prog="aftersight", description=__doc__)
|
|
74
|
+
sub = parser.add_subparsers(dest="cmd", required=True)
|
|
75
|
+
|
|
76
|
+
run = sub.add_parser("run", help="run a command with recording enabled")
|
|
77
|
+
run.add_argument("--session", help="session id; a resumed session reuses its run folder")
|
|
78
|
+
run.add_argument("--root", help="telemetry root (default .runs)")
|
|
79
|
+
run.add_argument("command", nargs=argparse.REMAINDER)
|
|
80
|
+
run.set_defaults(func=cmd_run)
|
|
81
|
+
|
|
82
|
+
skill = sub.add_parser("skill", help="install the Claude Code skill")
|
|
83
|
+
skill.add_argument("--into", default=".claude/skills", help="skills directory")
|
|
84
|
+
skill.set_defaults(func=cmd_skill)
|
|
85
|
+
|
|
86
|
+
args = parser.parse_args(argv)
|
|
87
|
+
return args.func(args)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
if __name__ == "__main__":
|
|
91
|
+
raise SystemExit(main())
|
aftersight/config.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
from dataclasses import dataclass, replace
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from aftersight import constants
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Config:
|
|
13
|
+
enabled: bool = True
|
|
14
|
+
root: Path = Path(constants.DEFAULT_ROOT)
|
|
15
|
+
keep_runs: int = constants.DEFAULT_KEEP_RUNS
|
|
16
|
+
blob_max_bytes: int = constants.DEFAULT_BLOB_MAX_BYTES
|
|
17
|
+
blob_preview_lines: int = constants.DEFAULT_BLOB_PREVIEW_LINES
|
|
18
|
+
capture_otel: bool = True
|
|
19
|
+
capture_logging: bool = True
|
|
20
|
+
log_level: int = logging.WARNING
|
|
21
|
+
redact: bool = True
|
|
22
|
+
quiet: bool = False
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _env_bool(name: str, default: bool) -> bool:
|
|
26
|
+
raw = os.environ.get(name)
|
|
27
|
+
if raw is None:
|
|
28
|
+
return default
|
|
29
|
+
lowered = raw.strip().lower()
|
|
30
|
+
if lowered in constants.TRUTHY:
|
|
31
|
+
return True
|
|
32
|
+
if lowered in constants.FALSY:
|
|
33
|
+
return False
|
|
34
|
+
return default
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _env_int(name: str, default: int) -> int:
|
|
38
|
+
try:
|
|
39
|
+
return int(os.environ[name])
|
|
40
|
+
except (KeyError, ValueError):
|
|
41
|
+
return default
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def load(**overrides) -> Config:
|
|
45
|
+
"""Defaults, then environment, then explicit `start()` arguments."""
|
|
46
|
+
cfg = Config(
|
|
47
|
+
enabled=_env_bool(constants.ENV_ENABLED, True),
|
|
48
|
+
root=Path(os.environ.get(constants.ENV_ROOT, constants.DEFAULT_ROOT)),
|
|
49
|
+
keep_runs=_env_int(constants.ENV_KEEP_RUNS, constants.DEFAULT_KEEP_RUNS),
|
|
50
|
+
blob_max_bytes=_env_int(constants.ENV_BLOB_MAX, constants.DEFAULT_BLOB_MAX_BYTES),
|
|
51
|
+
capture_otel=_env_bool(constants.ENV_OTEL, True),
|
|
52
|
+
capture_logging=_env_bool(constants.ENV_LOGGING, True),
|
|
53
|
+
redact=_env_bool(constants.ENV_REDACT, True),
|
|
54
|
+
quiet=_env_bool(constants.ENV_QUIET, False),
|
|
55
|
+
)
|
|
56
|
+
known = {f for f in Config.__dataclass_fields__}
|
|
57
|
+
clean = {k: v for k, v in overrides.items() if k in known and v is not None}
|
|
58
|
+
if "root" in clean:
|
|
59
|
+
clean["root"] = Path(clean["root"])
|
|
60
|
+
return replace(cfg, **clean) if clean else cfg
|
aftersight/constants.py
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
# Dotted on purpose: `rg 'tool\.'` finds all tool activity, `rg '\.error'`
|
|
4
|
+
# finds every failure shape.
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class EventType:
|
|
8
|
+
RUN_START = "run.start"
|
|
9
|
+
RUN_RESUME = "run.resume"
|
|
10
|
+
RUN_END = "run.end"
|
|
11
|
+
AGENT_START = "agent.start"
|
|
12
|
+
AGENT_END = "agent.end"
|
|
13
|
+
LLM_PROMPT = "llm.prompt"
|
|
14
|
+
LLM_RESPONSE = "llm.response"
|
|
15
|
+
TOOL_CALL = "tool.call"
|
|
16
|
+
TOOL_RESULT = "tool.result"
|
|
17
|
+
TOOL_ERROR = "tool.error"
|
|
18
|
+
ERROR = "error"
|
|
19
|
+
LOG = "log"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# Pointers live outside RUNS_DIR so a `runs/*/...` glob cannot count a run twice.
|
|
23
|
+
|
|
24
|
+
DEFAULT_ROOT = ".runs"
|
|
25
|
+
RUNS_DIR = "runs"
|
|
26
|
+
SESSIONS_DIR = "sessions"
|
|
27
|
+
LATEST_LINK = "latest"
|
|
28
|
+
NAVIGATE_FILE = "NAVIGATE.md"
|
|
29
|
+
INDEX_FILE = "index.jsonl"
|
|
30
|
+
GITIGNORE_MARK = "# aftersight"
|
|
31
|
+
|
|
32
|
+
TRANSCRIPT_FILE = "agent.logs"
|
|
33
|
+
TRACE_FILE = "trace.jsonl"
|
|
34
|
+
OUTLINE_FILE = "outline.md"
|
|
35
|
+
ANALYTICS_FILE = "analytics.json"
|
|
36
|
+
META_FILE = "meta.json"
|
|
37
|
+
BLOBS_DIR = "blobs"
|
|
38
|
+
ARTIFACTS_DIR = "artifacts"
|
|
39
|
+
|
|
40
|
+
RUN_ID_TIME_FORMAT = "%Y%m%d_%H%M%S"
|
|
41
|
+
RUN_ID_SUFFIX_LEN = 4
|
|
42
|
+
|
|
43
|
+
DEFAULT_KEEP_RUNS = 50
|
|
44
|
+
DEFAULT_BLOB_MAX_BYTES = 8192
|
|
45
|
+
DEFAULT_BLOB_PREVIEW_LINES = 12
|
|
46
|
+
|
|
47
|
+
ENV_ENABLED = "AFTERSIGHT"
|
|
48
|
+
ENV_ROOT = "AFTERSIGHT_ROOT"
|
|
49
|
+
ENV_KEEP_RUNS = "AFTERSIGHT_KEEP_RUNS"
|
|
50
|
+
ENV_BLOB_MAX = "AFTERSIGHT_BLOB_MAX"
|
|
51
|
+
ENV_OTEL = "AFTERSIGHT_OTEL"
|
|
52
|
+
ENV_LOGGING = "AFTERSIGHT_LOGGING"
|
|
53
|
+
ENV_REDACT = "AFTERSIGHT_REDACT"
|
|
54
|
+
ENV_QUIET = "AFTERSIGHT_QUIET"
|
|
55
|
+
ENV_AUTOSTART = "AFTERSIGHT_AUTOSTART"
|
|
56
|
+
ENV_SESSION = "AFTERSIGHT_SESSION"
|
|
57
|
+
|
|
58
|
+
TRUTHY = {"1", "true", "yes", "on"}
|
|
59
|
+
FALSY = {"0", "false", "no", "off"}
|
|
60
|
+
|
|
61
|
+
FRAME_WIDTH = 70
|
|
62
|
+
INLINE_ARG_LIMIT = 200
|
|
63
|
+
INLINE_BLOCK_LIMIT = 120
|
|
64
|
+
OUTLINE_NAME_COL = 28
|
|
65
|
+
|
|
66
|
+
#: Payload keys the transcript renders itself, so the generic key=value tail
|
|
67
|
+
#: does not repeat them.
|
|
68
|
+
RENDERED_PAYLOAD_KEYS = {"text", "text_ref", "preview", "bytes"}
|
|
69
|
+
|
|
70
|
+
TRANSCRIPT_TIME_FORMAT = "%H:%M:%S"
|
|
71
|
+
BANNER_TIME_FORMAT = "%Y-%m-%d %H:%M:%S"
|
|
72
|
+
|
|
73
|
+
REDACTION_MASK = "[REDACTED]"
|
|
74
|
+
|
|
75
|
+
#: A pattern with a capture group masks only the group, so the surrounding
|
|
76
|
+
#: `api_key=` stays readable.
|
|
77
|
+
SECRET_PATTERNS = [
|
|
78
|
+
r"sk-[A-Za-z0-9_\-]{16,}",
|
|
79
|
+
r"(?i:bearer)\s+[A-Za-z0-9._\-]{16,}",
|
|
80
|
+
r"(?i:(?:api[_-]?key|secret|token|password)\"?\s*[:=]\s*\"?)([A-Za-z0-9._\-]{12,})",
|
|
81
|
+
r"gh[pousr]_[A-Za-z0-9]{16,}",
|
|
82
|
+
r"AKIA[0-9A-Z]{16}",
|
|
83
|
+
]
|
|
84
|
+
|
|
85
|
+
# Three are read: OTel `gen_ai.*` semantic conventions, OpenInference `llm.*`,
|
|
86
|
+
# and a generic `input.value` / `output.value` fallback. Add an attribute name
|
|
87
|
+
# here rather than at a call site.
|
|
88
|
+
|
|
89
|
+
SPAN_KIND_ATTRS = ("openinference.span.kind", "gen_ai.operation.name",
|
|
90
|
+
"traceloop.span.kind")
|
|
91
|
+
|
|
92
|
+
LLM_SPAN_KINDS = {"LLM", "chat", "text_completion", "generate_content"}
|
|
93
|
+
TOOL_SPAN_KINDS = {"TOOL", "execute_tool"}
|
|
94
|
+
CONTAINER_SPAN_KINDS = {"AGENT", "CHAIN", "GRAPH", "WORKFLOW", "invoke_agent",
|
|
95
|
+
"create_agent"}
|
|
96
|
+
|
|
97
|
+
#: Indexed message attributes, joined into one text block.
|
|
98
|
+
PROMPT_TEMPLATES = ("gen_ai.prompt.{i}.content", "llm.input_messages.{i}.message.content")
|
|
99
|
+
COMPLETION_TEMPLATES = ("gen_ai.completion.{i}.content",
|
|
100
|
+
"llm.output_messages.{i}.message.content")
|
|
101
|
+
|
|
102
|
+
PROMPT_ATTRS = ("gen_ai.prompt", "input.value", "llm.prompts")
|
|
103
|
+
COMPLETION_ATTRS = ("gen_ai.completion", "output.value")
|
|
104
|
+
MODEL_ATTRS = ("gen_ai.request.model", "llm.model_name")
|
|
105
|
+
TOOL_NAME_ATTRS = ("tool.name", "gen_ai.tool.name")
|
|
106
|
+
TOOL_ARG_ATTRS = ("tool.parameters", "input.value")
|
|
107
|
+
COST_ATTRS = ("gen_ai.usage.cost", "llm.cost.total")
|
|
108
|
+
FINISH_REASON_ATTRS = ("gen_ai.response.finish_reasons",)
|
|
109
|
+
INPUT_TOKEN_ATTRS = ("gen_ai.usage.input_tokens", "gen_ai.usage.prompt_tokens",
|
|
110
|
+
"llm.token_count.prompt")
|
|
111
|
+
OUTPUT_TOKEN_ATTRS = ("gen_ai.usage.output_tokens", "gen_ai.usage.completion_tokens",
|
|
112
|
+
"llm.token_count.completion")
|
|
113
|
+
|
|
114
|
+
MAX_INDEXED_MESSAGES = 64
|
|
115
|
+
|
|
116
|
+
#: OpenInference instrumentors activated when already installed. Nothing is
|
|
117
|
+
#: installed on the user's behalf.
|
|
118
|
+
INSTRUMENTORS = [
|
|
119
|
+
("openinference.instrumentation.agno", "AgnoInstrumentor", "agno"),
|
|
120
|
+
("openinference.instrumentation.langchain", "LangChainInstrumentor", "langchain"),
|
|
121
|
+
("openinference.instrumentation.crewai", "CrewAIInstrumentor", "crewai"),
|
|
122
|
+
("openinference.instrumentation.llama_index", "LlamaIndexInstrumentor", "llama-index"),
|
|
123
|
+
("openinference.instrumentation.openai_agents", "OpenAIAgentsInstrumentor", "openai-agents"),
|
|
124
|
+
("openinference.instrumentation.smolagents", "SmolagentsInstrumentor", "smolagents"),
|
|
125
|
+
]
|
|
File without changes
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from datetime import datetime
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class Event:
|
|
10
|
+
run_id: str
|
|
11
|
+
seq: int
|
|
12
|
+
attempt: int
|
|
13
|
+
ts: float
|
|
14
|
+
type: str
|
|
15
|
+
name: str | None = None
|
|
16
|
+
parent_seq: int | None = None
|
|
17
|
+
status: str | None = None
|
|
18
|
+
dur_ms: int | None = None
|
|
19
|
+
payload: dict[str, Any] = field(default_factory=dict)
|
|
20
|
+
|
|
21
|
+
@property
|
|
22
|
+
def anchor(self) -> str:
|
|
23
|
+
"""`#0016`. The same string in outline.md, agent.logs and trace.jsonl."""
|
|
24
|
+
return f"#{self.seq:04d}"
|
|
25
|
+
|
|
26
|
+
def datetime(self) -> datetime:
|
|
27
|
+
"""Local time with an explicit offset. A user correlating a run with their
|
|
28
|
+
own terminal reads a local clock, and the offset keeps it unambiguous."""
|
|
29
|
+
return datetime.fromtimestamp(self.ts).astimezone()
|
|
30
|
+
|
|
31
|
+
def to_dict(self) -> dict[str, Any]:
|
|
32
|
+
return {
|
|
33
|
+
"run_id": self.run_id,
|
|
34
|
+
"seq": self.seq,
|
|
35
|
+
"attempt": self.attempt,
|
|
36
|
+
"ts": round(self.ts, 3),
|
|
37
|
+
"iso": self.datetime().isoformat(),
|
|
38
|
+
"type": self.type,
|
|
39
|
+
"name": self.name,
|
|
40
|
+
"parent_seq": self.parent_seq,
|
|
41
|
+
"status": self.status,
|
|
42
|
+
"dur_ms": self.dur_ms,
|
|
43
|
+
"payload": self.payload,
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def from_dict(cls, data: dict[str, Any]) -> "Event":
|
|
48
|
+
return cls(
|
|
49
|
+
run_id=data["run_id"],
|
|
50
|
+
seq=data["seq"],
|
|
51
|
+
attempt=data.get("attempt", 1),
|
|
52
|
+
ts=data["ts"],
|
|
53
|
+
type=data["type"],
|
|
54
|
+
name=data.get("name"),
|
|
55
|
+
parent_seq=data.get("parent_seq"),
|
|
56
|
+
status=data.get("status"),
|
|
57
|
+
dur_ms=data.get("dur_ms"),
|
|
58
|
+
payload=data.get("payload", {}),
|
|
59
|
+
)
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Sequence allocation and fan-out to the writers.
|
|
2
|
+
|
|
3
|
+
`emit()` is thread-safe: concurrent agents get unique, contiguous sequence
|
|
4
|
+
numbers. Sequence numbers continue across a resume, which is why `start_seq`
|
|
5
|
+
exists.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
from collections.abc import Callable
|
|
13
|
+
from typing import Any
|
|
14
|
+
|
|
15
|
+
from aftersight.events.schemas import Event
|
|
16
|
+
from aftersight.utils import log_error, redact
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class EventSink:
|
|
20
|
+
def __init__(self, run_id: str, attempt: int = 1, *, start_seq: int = 0,
|
|
21
|
+
redact_payloads: bool = True):
|
|
22
|
+
self.run_id = run_id
|
|
23
|
+
self.attempt = attempt
|
|
24
|
+
self._seq = start_seq
|
|
25
|
+
self._redact = redact_payloads
|
|
26
|
+
self._writers: list[Callable[[Event], None]] = []
|
|
27
|
+
self._lock = threading.Lock()
|
|
28
|
+
|
|
29
|
+
def add_writer(self, writer: Callable[[Event], None]) -> None:
|
|
30
|
+
self._writers.append(writer)
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
def last_seq(self) -> int:
|
|
34
|
+
return self._seq
|
|
35
|
+
|
|
36
|
+
def emit(self, type: str, name: str | None = None, *, parent_seq: int | None = None,
|
|
37
|
+
status: str | None = None, dur_ms: int | None = None,
|
|
38
|
+
ts: float | None = None, **payload: Any) -> Event:
|
|
39
|
+
with self._lock:
|
|
40
|
+
self._seq += 1
|
|
41
|
+
event = Event(
|
|
42
|
+
run_id=self.run_id,
|
|
43
|
+
seq=self._seq,
|
|
44
|
+
attempt=self.attempt,
|
|
45
|
+
ts=time.time() if ts is None else ts,
|
|
46
|
+
type=type,
|
|
47
|
+
name=name,
|
|
48
|
+
parent_seq=parent_seq,
|
|
49
|
+
status=status,
|
|
50
|
+
dur_ms=dur_ms,
|
|
51
|
+
payload=redact(payload) if self._redact else payload,
|
|
52
|
+
)
|
|
53
|
+
for write in self._writers:
|
|
54
|
+
try:
|
|
55
|
+
write(event)
|
|
56
|
+
except Exception as exc: # noqa: BLE001 (never break the host app)
|
|
57
|
+
log_error(f"writer {write!r} failed on {event.anchor}: {exc}")
|
|
58
|
+
return event
|
|
File without changes
|