aftersight 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aftersight-0.1.0/.github/workflows/docs.yml +43 -0
- aftersight-0.1.0/.github/workflows/release.yml +26 -0
- aftersight-0.1.0/.gitignore +10 -0
- aftersight-0.1.0/AGENTS.md +101 -0
- aftersight-0.1.0/LICENSE +21 -0
- aftersight-0.1.0/PKG-INFO +157 -0
- aftersight-0.1.0/README.md +131 -0
- aftersight-0.1.0/aftersight/__init__.py +21 -0
- aftersight-0.1.0/aftersight/assets/NAVIGATE.md +129 -0
- aftersight-0.1.0/aftersight/assets/skill/SKILL.md +39 -0
- aftersight-0.1.0/aftersight/autostart.py +8 -0
- aftersight-0.1.0/aftersight/cli.py +91 -0
- aftersight-0.1.0/aftersight/config.py +60 -0
- aftersight-0.1.0/aftersight/constants.py +125 -0
- aftersight-0.1.0/aftersight/events/__init__.py +0 -0
- aftersight-0.1.0/aftersight/events/schemas.py +59 -0
- aftersight-0.1.0/aftersight/events/sink.py +58 -0
- aftersight-0.1.0/aftersight/integrations/__init__.py +0 -0
- aftersight-0.1.0/aftersight/integrations/otel.py +249 -0
- aftersight-0.1.0/aftersight/integrations/stdlog.py +49 -0
- aftersight-0.1.0/aftersight/py.typed +0 -0
- aftersight-0.1.0/aftersight/run/__init__.py +0 -0
- aftersight-0.1.0/aftersight/run/api.py +211 -0
- aftersight-0.1.0/aftersight/run/run.py +164 -0
- aftersight-0.1.0/aftersight/run/store.py +134 -0
- aftersight-0.1.0/aftersight/utils.py +55 -0
- aftersight-0.1.0/aftersight/writers/__init__.py +0 -0
- aftersight-0.1.0/aftersight/writers/analytics.py +108 -0
- aftersight-0.1.0/aftersight/writers/blobs.py +42 -0
- aftersight-0.1.0/aftersight/writers/outline.py +120 -0
- aftersight-0.1.0/aftersight/writers/trace.py +40 -0
- aftersight-0.1.0/aftersight/writers/transcript.py +194 -0
- aftersight-0.1.0/docs/design/architecture.md +129 -0
- aftersight-0.1.0/docs/design/decisions.md +208 -0
- aftersight-0.1.0/docs/frameworks.md +151 -0
- aftersight-0.1.0/docs/index.md +36 -0
- aftersight-0.1.0/docs/navigating.md +122 -0
- aftersight-0.1.0/docs/quickstart.md +120 -0
- aftersight-0.1.0/docs/reference/api.md +172 -0
- aftersight-0.1.0/docs/reference/cli.md +49 -0
- aftersight-0.1.0/docs/reference/configuration.md +75 -0
- aftersight-0.1.0/docs/reference/run-folder.md +178 -0
- aftersight-0.1.0/docs/superpowers/plans/2026-08-30-public-release.md +616 -0
- aftersight-0.1.0/docs/superpowers/specs/2026-08-30-public-release-design.md +207 -0
- aftersight-0.1.0/docs/troubleshooting.md +100 -0
- aftersight-0.1.0/docs/why.md +97 -0
- aftersight-0.1.0/examples/generate.py +139 -0
- aftersight-0.1.0/examples/starters/claude_agent_sdk_starter.py +40 -0
- aftersight-0.1.0/examples/starters/langchain_starter.py +38 -0
- aftersight-0.1.0/examples/starters/openai_agents_starter.py +37 -0
- aftersight-0.1.0/examples/starters/openhands_starter.py +37 -0
- aftersight-0.1.0/examples/starters/pydantic_ai_starter.py +29 -0
- aftersight-0.1.0/examples/telemetry_root/NAVIGATE.md +129 -0
- aftersight-0.1.0/examples/telemetry_root/index.jsonl +1 -0
- aftersight-0.1.0/examples/telemetry_root/latest/agent.logs +67 -0
- aftersight-0.1.0/examples/telemetry_root/latest/analytics.json +74 -0
- aftersight-0.1.0/examples/telemetry_root/latest/blobs/db797ac6.txt +2803 -0
- aftersight-0.1.0/examples/telemetry_root/latest/meta.json +41 -0
- aftersight-0.1.0/examples/telemetry_root/latest/outline.md +22 -0
- aftersight-0.1.0/examples/telemetry_root/latest/trace.jsonl +25 -0
- aftersight-0.1.0/mkdocs.yml +77 -0
- aftersight-0.1.0/pyproject.toml +42 -0
- aftersight-0.1.0/tests/test_formats.py +216 -0
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
name: docs
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
paths:
|
|
7
|
+
- docs/**
|
|
8
|
+
- mkdocs.yml
|
|
9
|
+
- .github/workflows/docs.yml
|
|
10
|
+
workflow_dispatch:
|
|
11
|
+
|
|
12
|
+
permissions:
|
|
13
|
+
contents: read
|
|
14
|
+
pages: write
|
|
15
|
+
id-token: write
|
|
16
|
+
|
|
17
|
+
concurrency:
|
|
18
|
+
group: pages
|
|
19
|
+
cancel-in-progress: false
|
|
20
|
+
|
|
21
|
+
jobs:
|
|
22
|
+
build:
|
|
23
|
+
runs-on: ubuntu-latest
|
|
24
|
+
steps:
|
|
25
|
+
- uses: actions/checkout@v4
|
|
26
|
+
- uses: actions/setup-python@v5
|
|
27
|
+
with:
|
|
28
|
+
python-version: "3.12"
|
|
29
|
+
- run: pip install -e '.[docs]'
|
|
30
|
+
- run: mkdocs build --strict
|
|
31
|
+
- uses: actions/upload-pages-artifact@v3
|
|
32
|
+
with:
|
|
33
|
+
path: site
|
|
34
|
+
|
|
35
|
+
deploy:
|
|
36
|
+
needs: build
|
|
37
|
+
runs-on: ubuntu-latest
|
|
38
|
+
environment:
|
|
39
|
+
name: github-pages
|
|
40
|
+
url: ${{ steps.deployment.outputs.page_url }}
|
|
41
|
+
steps:
|
|
42
|
+
- id: deployment
|
|
43
|
+
uses: actions/deploy-pages@v4
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
publish:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
# Trusted publishing: PyPI verifies this workflow through OIDC, so there is
|
|
12
|
+
# no API token to store or rotate. Register the publisher on PyPI with
|
|
13
|
+
# owner nebulaanish, repository aftersight, workflow release.yml and
|
|
14
|
+
# environment pypi, all four of which must match what is below.
|
|
15
|
+
environment: pypi
|
|
16
|
+
permissions:
|
|
17
|
+
id-token: write
|
|
18
|
+
steps:
|
|
19
|
+
- uses: actions/checkout@v4
|
|
20
|
+
- uses: actions/setup-python@v5
|
|
21
|
+
with:
|
|
22
|
+
python-version: "3.12"
|
|
23
|
+
- run: pip install -e . build
|
|
24
|
+
- run: python tests/test_formats.py
|
|
25
|
+
- run: python -m build
|
|
26
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# AGENTS.md
|
|
2
|
+
|
|
3
|
+
Behavioral guidelines to reduce common LLM coding mistakes. Merge with project-specific instructions as needed.
|
|
4
|
+
|
|
5
|
+
**Tradeoff:** These guidelines bias toward caution over speed. For trivial tasks, use judgment.
|
|
6
|
+
|
|
7
|
+
## 1. Think Before Coding
|
|
8
|
+
|
|
9
|
+
**Don't assume. Don't hide confusion. Surface tradeoffs.**
|
|
10
|
+
|
|
11
|
+
Before implementing:
|
|
12
|
+
- State your assumptions explicitly. If uncertain, ask.
|
|
13
|
+
- If multiple interpretations exist, present them - don't pick silently.
|
|
14
|
+
- If a simpler approach exists, say so. Push back when warranted.
|
|
15
|
+
- If something is unclear, stop. Name what's confusing. Ask.
|
|
16
|
+
|
|
17
|
+
## 2. Simplicity First
|
|
18
|
+
|
|
19
|
+
**Minimum code that solves the problem. Nothing speculative.**
|
|
20
|
+
|
|
21
|
+
- No features beyond what was asked.
|
|
22
|
+
- No abstractions for single-use code.
|
|
23
|
+
- No "flexibility" or "configurability" that wasn't requested.
|
|
24
|
+
- No error handling for impossible scenarios.
|
|
25
|
+
- If you write 200 lines and it could be 50, rewrite it.
|
|
26
|
+
|
|
27
|
+
Ask yourself: "Would a senior engineer say this is overcomplicated?" If yes, simplify.
|
|
28
|
+
|
|
29
|
+
## 3. Surgical Changes
|
|
30
|
+
|
|
31
|
+
**Touch only what you must. Clean up only your own mess.**
|
|
32
|
+
|
|
33
|
+
When editing existing code:
|
|
34
|
+
- Don't "improve" adjacent code, comments, or formatting.
|
|
35
|
+
- Don't refactor things that aren't broken.
|
|
36
|
+
- Match existing style, even if you'd do it differently.
|
|
37
|
+
- If you notice unrelated dead code, mention it - don't delete it.
|
|
38
|
+
|
|
39
|
+
When your changes create orphans:
|
|
40
|
+
- Remove imports/variables/functions that YOUR changes made unused.
|
|
41
|
+
- Don't remove pre-existing dead code unless asked.
|
|
42
|
+
|
|
43
|
+
The test: Every changed line should trace directly to the user's request.
|
|
44
|
+
|
|
45
|
+
## 4. Goal-Driven Execution
|
|
46
|
+
|
|
47
|
+
**Define success criteria. Loop until verified.**
|
|
48
|
+
|
|
49
|
+
Transform tasks into verifiable goals:
|
|
50
|
+
- "Add validation" → "Write tests for invalid inputs, then make them pass"
|
|
51
|
+
- "Fix the bug" → "Write a test that reproduces it, then make it pass"
|
|
52
|
+
- "Refactor X" → "Ensure tests pass before and after"
|
|
53
|
+
|
|
54
|
+
For multi-step tasks, state a brief plan:
|
|
55
|
+
```
|
|
56
|
+
1. [Step] → verify: [check]
|
|
57
|
+
2. [Step] → verify: [check]
|
|
58
|
+
3. [Step] → verify: [check]
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Strong success criteria let you loop independently. Weak criteria ("make it work") require constant clarification.
|
|
62
|
+
|
|
63
|
+
---
|
|
64
|
+
|
|
65
|
+
**These guidelines are working if:** fewer unnecessary changes in diffs, fewer rewrites due to overcomplication, and clarifying questions come before implementation rather than after mistakes.
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# Guidelines: What Defines a Valid Test Case
|
|
70
|
+
|
|
71
|
+
Strictly enforce this standard when writing or evaluating tests. Reject any test case exhibiting invalid anti-patterns.
|
|
72
|
+
|
|
73
|
+
---
|
|
74
|
+
|
|
75
|
+
## 1. Valid Test Case Criteria
|
|
76
|
+
A valid test must satisfy ALL of the following:
|
|
77
|
+
|
|
78
|
+
- **Single Concept:** Tests exactly ONE behavior per test function.
|
|
79
|
+
- **Black-Box Testing:** Verifies public interfaces and inputs/outputs—never internal private implementation details.
|
|
80
|
+
- **100% Deterministic:** Zero reliance on ambient clock (`Date.now()`), random seeds, external APIs, or dirty DB state.
|
|
81
|
+
- **Explicit AAA:** Structured strictly as Arrange (setup), Act (trigger), Assert (check).
|
|
82
|
+
- **Refactor-Resistant:** Must pass if internal code is completely refactored without breaking the public contract.
|
|
83
|
+
- **High Information Density:** Focuses on realistic boundaries (`0`, `null`, empty arrays, max limits) or explicit error modes.
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## 2. Invalid Test Case Anti-Patterns (Immediate Rejection)
|
|
88
|
+
|
|
89
|
+
- ❌ **Implementation Coupling:** Asserting internal helper execution order or checking private class state.
|
|
90
|
+
- ❌ **Tautologies:** Re-implementing production logic inside the test setup or asserting simple pass-through getters.
|
|
91
|
+
- ❌ **Multi-Scenario Bloat:** Single tests containing sequential user actions, >3 independent assertions, or multiple business rules.
|
|
92
|
+
- ❌ **Over-Mocking:** Mocking simple domain models or internal helper classes instead of using real instances or factories.
|
|
93
|
+
- ❌ **Flakiness:** Relying on arbitrary `sleep()` calls, live network calls, or unordered array checks.
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Output Verification Checklist
|
|
98
|
+
Before outputting code, verify:
|
|
99
|
+
1. Is the test name `<Method>_<Scenario>_<ExpectedResult>`?
|
|
100
|
+
2. Does it test what the code DOES, not HOW it does it?
|
|
101
|
+
3. If the implementation changes internally, will this test still pass?
|
aftersight-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Balaram
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: aftersight
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Observability infrastructure for self-improving agents
|
|
5
|
+
Project-URL: Homepage, https://github.com/nebulaanish/aftersight
|
|
6
|
+
Project-URL: Documentation, https://nebulaanish.github.io/aftersight/
|
|
7
|
+
Project-URL: Source, https://github.com/nebulaanish/aftersight
|
|
8
|
+
Project-URL: Issues, https://github.com/nebulaanish/aftersight/issues
|
|
9
|
+
Author-email: Balaram <balaramneu@gmail.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agents,llm,observability,opentelemetry,tracing
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Software Development :: Debuggers
|
|
17
|
+
Classifier: Topic :: System :: Monitoring
|
|
18
|
+
Classifier: Typing :: Typed
|
|
19
|
+
Requires-Python: >=3.10
|
|
20
|
+
Requires-Dist: opentelemetry-api>=1.20
|
|
21
|
+
Requires-Dist: opentelemetry-sdk>=1.20
|
|
22
|
+
Provides-Extra: docs
|
|
23
|
+
Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
|
|
24
|
+
Requires-Dist: mkdocs<2; extra == 'docs'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# aftersight
|
|
28
|
+
|
|
29
|
+
Observability infrastructure for self-improving agents.
|
|
30
|
+
|
|
31
|
+
An agent cannot improve on a run it cannot read. Aftersight writes every
|
|
32
|
+
run into the repository as plain files, so the agent can read its own
|
|
33
|
+
history with the tools it already has.
|
|
34
|
+
|
|
35
|
+
## The problem
|
|
36
|
+
|
|
37
|
+
Coding agents already explore repositories with file search, shell commands and
|
|
38
|
+
plain text. Agent telemetry usually lives somewhere else: behind a dashboard,
|
|
39
|
+
an API or a remote MCP server.
|
|
40
|
+
|
|
41
|
+
Aftersight writes the evidence into the repository instead. A coding agent can
|
|
42
|
+
use the tools it already knows to:
|
|
43
|
+
|
|
44
|
+
1. diagnose why one run failed;
|
|
45
|
+
2. find failures that recur across runs;
|
|
46
|
+
3. inspect the exact inputs and outputs of models and tools.
|
|
47
|
+
|
|
48
|
+
## Start in one command
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
pip install aftersight
|
|
52
|
+
aftersight run python my_agent.py
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
No account, API key, service or initialization step is required.
|
|
56
|
+
|
|
57
|
+
## Ask your coding agent
|
|
58
|
+
|
|
59
|
+
Each telemetry root includes a `NAVIGATE.md` with verified `rg` and `jq`
|
|
60
|
+
recipes. Tell your coding agent:
|
|
61
|
+
|
|
62
|
+
> Read `.runs/NAVIGATE.md` and find out why the latest run failed.
|
|
63
|
+
|
|
64
|
+
For Claude Code, the optional command below installs the bundled navigation
|
|
65
|
+
skill:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
aftersight skill
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## What gets recorded
|
|
72
|
+
|
|
73
|
+
```text
|
|
74
|
+
.runs/
|
|
75
|
+
NAVIGATE.md
|
|
76
|
+
index.jsonl
|
|
77
|
+
latest -> runs/<run_id>
|
|
78
|
+
runs/<run_id>/
|
|
79
|
+
outline.md
|
|
80
|
+
agent.logs
|
|
81
|
+
trace.jsonl
|
|
82
|
+
analytics.json
|
|
83
|
+
meta.json
|
|
84
|
+
blobs/
|
|
85
|
+
artifacts/
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`outline.md` is the short map. `agent.logs` is the complete readable
|
|
89
|
+
transcript. `trace.jsonl`, `analytics.json` and `index.jsonl` are stable
|
|
90
|
+
machine-readable views for scripts and frontends. The same `#seq` anchor
|
|
91
|
+
identifies an event in every projection.
|
|
92
|
+
|
|
93
|
+
## Use it from Python
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import aftersight
|
|
97
|
+
|
|
98
|
+
aftersight.start()
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Add explicit detail only where it helps:
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
with aftersight.span("planner"):
|
|
105
|
+
with aftersight.span("web_search", kind="tool", args={"q": query}) as span:
|
|
106
|
+
span.output = search(query)
|
|
107
|
+
|
|
108
|
+
@aftersight.trace
|
|
109
|
+
def read_file(path): ...
|
|
110
|
+
|
|
111
|
+
aftersight.log("cache miss", key=key)
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## Works with OpenTelemetry
|
|
115
|
+
|
|
116
|
+
Aftersight attaches a span processor to the application's existing
|
|
117
|
+
OpenTelemetry setup and reads common `gen_ai.*`, OpenInference and generic
|
|
118
|
+
input/output attributes. Frameworks that already emit compatible spans need no
|
|
119
|
+
aftersight-specific adapter. An existing tracer provider and its exporters are
|
|
120
|
+
left in place.
|
|
121
|
+
|
|
122
|
+
## How it compares
|
|
123
|
+
|
|
124
|
+
[LangSmith](https://docs.langchain.com/langsmith/observability-concepts),
|
|
125
|
+
[Langfuse](https://langfuse.com/docs/api-and-data-platform/features/observations-api),
|
|
126
|
+
[Phoenix](https://arize.com/docs/phoenix/tracing/how-to-tracing/importing-and-exporting-traces/extract-data-from-spans)
|
|
127
|
+
and [Braintrust](https://www.braintrust.dev/docs/integrations/developer-tools/mcp)
|
|
128
|
+
provide mature observability with dashboards and programmatic access through
|
|
129
|
+
APIs, exports or MCP.
|
|
130
|
+
|
|
131
|
+
Aftersight has a narrower default: ordinary files in the repository, with no
|
|
132
|
+
hosted project, credentials or remote query round trips. Those platforms are a
|
|
133
|
+
better fit when you need centralized durability, team dashboards or managed
|
|
134
|
+
evaluation workflows. See [Why aftersight exists](docs/why.md) for the fuller
|
|
135
|
+
comparison.
|
|
136
|
+
|
|
137
|
+
## Local by default
|
|
138
|
+
|
|
139
|
+
Aftersight itself makes no network requests and does not upload telemetry.
|
|
140
|
+
Payload redaction is enabled by default, and the run folder is added to
|
|
141
|
+
`.gitignore` when it is first created.
|
|
142
|
+
|
|
143
|
+
If the application already has an OpenTelemetry exporter, that exporter may
|
|
144
|
+
still send its own spans. Aftersight can run in production, but local disk can
|
|
145
|
+
disappear with an ephemeral, replaced or failed host, so it should not be the
|
|
146
|
+
only durable production record.
|
|
147
|
+
|
|
148
|
+
## Documentation
|
|
149
|
+
|
|
150
|
+
- [Quickstart](docs/quickstart.md)
|
|
151
|
+
- [Framework support](docs/frameworks.md)
|
|
152
|
+
- [Navigating runs](docs/navigating.md)
|
|
153
|
+
- [Python API](docs/reference/api.md)
|
|
154
|
+
- [Run folder format](docs/reference/run-folder.md)
|
|
155
|
+
- [Architecture](docs/design/architecture.md)
|
|
156
|
+
|
|
157
|
+
Full documentation: **https://nebulaanish.github.io/aftersight/**
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# aftersight
|
|
2
|
+
|
|
3
|
+
Observability infrastructure for self-improving agents.
|
|
4
|
+
|
|
5
|
+
An agent cannot improve on a run it cannot read. Aftersight writes every
|
|
6
|
+
run into the repository as plain files, so the agent can read its own
|
|
7
|
+
history with the tools it already has.
|
|
8
|
+
|
|
9
|
+
## The problem
|
|
10
|
+
|
|
11
|
+
Coding agents already explore repositories with file search, shell commands and
|
|
12
|
+
plain text. Agent telemetry usually lives somewhere else: behind a dashboard,
|
|
13
|
+
an API or a remote MCP server.
|
|
14
|
+
|
|
15
|
+
Aftersight writes the evidence into the repository instead. A coding agent can
|
|
16
|
+
use the tools it already knows to:
|
|
17
|
+
|
|
18
|
+
1. diagnose why one run failed;
|
|
19
|
+
2. find failures that recur across runs;
|
|
20
|
+
3. inspect the exact inputs and outputs of models and tools.
|
|
21
|
+
|
|
22
|
+
## Start in one command
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install aftersight
|
|
26
|
+
aftersight run python my_agent.py
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
No account, API key, service or initialization step is required.
|
|
30
|
+
|
|
31
|
+
## Ask your coding agent
|
|
32
|
+
|
|
33
|
+
Each telemetry root includes a `NAVIGATE.md` with verified `rg` and `jq`
|
|
34
|
+
recipes. Tell your coding agent:
|
|
35
|
+
|
|
36
|
+
> Read `.runs/NAVIGATE.md` and find out why the latest run failed.
|
|
37
|
+
|
|
38
|
+
For Claude Code, the optional command below installs the bundled navigation
|
|
39
|
+
skill:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
aftersight skill
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## What gets recorded
|
|
46
|
+
|
|
47
|
+
```text
|
|
48
|
+
.runs/
|
|
49
|
+
NAVIGATE.md
|
|
50
|
+
index.jsonl
|
|
51
|
+
latest -> runs/<run_id>
|
|
52
|
+
runs/<run_id>/
|
|
53
|
+
outline.md
|
|
54
|
+
agent.logs
|
|
55
|
+
trace.jsonl
|
|
56
|
+
analytics.json
|
|
57
|
+
meta.json
|
|
58
|
+
blobs/
|
|
59
|
+
artifacts/
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
`outline.md` is the short map. `agent.logs` is the complete readable
|
|
63
|
+
transcript. `trace.jsonl`, `analytics.json` and `index.jsonl` are stable
|
|
64
|
+
machine-readable views for scripts and frontends. The same `#seq` anchor
|
|
65
|
+
identifies an event in every projection.
|
|
66
|
+
|
|
67
|
+
## Use it from Python
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import aftersight
|
|
71
|
+
|
|
72
|
+
aftersight.start()
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Add explicit detail only where it helps:
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
with aftersight.span("planner"):
|
|
79
|
+
with aftersight.span("web_search", kind="tool", args={"q": query}) as span:
|
|
80
|
+
span.output = search(query)
|
|
81
|
+
|
|
82
|
+
@aftersight.trace
|
|
83
|
+
def read_file(path): ...
|
|
84
|
+
|
|
85
|
+
aftersight.log("cache miss", key=key)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## Works with OpenTelemetry
|
|
89
|
+
|
|
90
|
+
Aftersight attaches a span processor to the application's existing
|
|
91
|
+
OpenTelemetry setup and reads common `gen_ai.*`, OpenInference and generic
|
|
92
|
+
input/output attributes. Frameworks that already emit compatible spans need no
|
|
93
|
+
aftersight-specific adapter. An existing tracer provider and its exporters are
|
|
94
|
+
left in place.
|
|
95
|
+
|
|
96
|
+
## How it compares
|
|
97
|
+
|
|
98
|
+
[LangSmith](https://docs.langchain.com/langsmith/observability-concepts),
|
|
99
|
+
[Langfuse](https://langfuse.com/docs/api-and-data-platform/features/observations-api),
|
|
100
|
+
[Phoenix](https://arize.com/docs/phoenix/tracing/how-to-tracing/importing-and-exporting-traces/extract-data-from-spans)
|
|
101
|
+
and [Braintrust](https://www.braintrust.dev/docs/integrations/developer-tools/mcp)
|
|
102
|
+
provide mature observability with dashboards and programmatic access through
|
|
103
|
+
APIs, exports or MCP.
|
|
104
|
+
|
|
105
|
+
Aftersight has a narrower default: ordinary files in the repository, with no
|
|
106
|
+
hosted project, credentials or remote query round trips. Those platforms are a
|
|
107
|
+
better fit when you need centralized durability, team dashboards or managed
|
|
108
|
+
evaluation workflows. See [Why aftersight exists](docs/why.md) for the fuller
|
|
109
|
+
comparison.
|
|
110
|
+
|
|
111
|
+
## Local by default
|
|
112
|
+
|
|
113
|
+
Aftersight itself makes no network requests and does not upload telemetry.
|
|
114
|
+
Payload redaction is enabled by default, and the run folder is added to
|
|
115
|
+
`.gitignore` when it is first created.
|
|
116
|
+
|
|
117
|
+
If the application already has an OpenTelemetry exporter, that exporter may
|
|
118
|
+
still send its own spans. Aftersight can run in production, but local disk can
|
|
119
|
+
disappear with an ephemeral, replaced or failed host, so it should not be the
|
|
120
|
+
only durable production record.
|
|
121
|
+
|
|
122
|
+
## Documentation
|
|
123
|
+
|
|
124
|
+
- [Quickstart](docs/quickstart.md)
|
|
125
|
+
- [Framework support](docs/frameworks.md)
|
|
126
|
+
- [Navigating runs](docs/navigating.md)
|
|
127
|
+
- [Python API](docs/reference/api.md)
|
|
128
|
+
- [Run folder format](docs/reference/run-folder.md)
|
|
129
|
+
- [Architecture](docs/design/architecture.md)
|
|
130
|
+
|
|
131
|
+
Full documentation: **https://nebulaanish.github.io/aftersight/**
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Observability infrastructure for self-improving agents.
|
|
2
|
+
|
|
3
|
+
import aftersight
|
|
4
|
+
aftersight.start()
|
|
5
|
+
|
|
6
|
+
Or, with no code change at all:
|
|
7
|
+
|
|
8
|
+
aftersight run python my_agent.py
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from aftersight.run.api import (
|
|
12
|
+
artifact_dir,
|
|
13
|
+
current,
|
|
14
|
+
log,
|
|
15
|
+
span,
|
|
16
|
+
start,
|
|
17
|
+
trace,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
__all__ = ["start", "span", "trace", "log", "current", "artifact_dir"]
|
|
21
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# Navigating these runs
|
|
2
|
+
|
|
3
|
+
You are a coding agent. This directory is the execution history of the agent
|
|
4
|
+
system in this repo. Read this file once, then use `rg` and `jq`.
|
|
5
|
+
|
|
6
|
+
## Layout
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
index.jsonl one line per run, start here for "which run?"
|
|
10
|
+
latest -> runs/<run_id> most recent run
|
|
11
|
+
sessions/<session_id> -> the run folder that session resumed into
|
|
12
|
+
runs/<run_id>/
|
|
13
|
+
outline.md folded map of the run, read this first
|
|
14
|
+
agent.logs full transcript, nothing truncated
|
|
15
|
+
trace.jsonl same events, machine-readable, has parent_seq
|
|
16
|
+
analytics.json status, timings, per-agent cost, error list
|
|
17
|
+
meta.json framework, git sha, argv, attempts[]
|
|
18
|
+
blobs/<sha>.txt payloads over 8 KB, plain text
|
|
19
|
+
artifacts/ whatever the app chose to keep
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Run ids are `YYYYMMDD_HHMMSS_hash`, so they sort chronologically. Glob run
|
|
23
|
+
files as `runs/*/...` and never from the root: `latest` and `sessions/` are
|
|
24
|
+
symlinks into `runs/`, so a root glob counts every run twice.
|
|
25
|
+
|
|
26
|
+
## The three-step loop
|
|
27
|
+
|
|
28
|
+
1. **Which run?** `index.jsonl`
|
|
29
|
+
2. **What happened in it?** `runs/<run_id>/outline.md`, then jump to a `#seq`
|
|
30
|
+
3. **Exactly what went in and out?** `runs/<run_id>/agent.logs` at that `#seq`
|
|
31
|
+
|
|
32
|
+
## Anchors
|
|
33
|
+
|
|
34
|
+
Every event has a zero-padded sequence number. It is the same number in
|
|
35
|
+
`outline.md`, `agent.logs` and `trace.jsonl`, so you can move between them:
|
|
36
|
+
|
|
37
|
+
```sh
|
|
38
|
+
rg -n '^#0016 ' latest/agent.logs # jump to one event
|
|
39
|
+
rg -n '#0016' runs/<run_id>/ # every mention of it
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
Multi-line payloads are framed. The closing line repeats the anchor, so a hit
|
|
43
|
+
landing inside a block can find both edges:
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
#0004 14:22:35.512 +2.4s llm.response triage stop=end_turn · 386 tok · $0.0005
|
|
47
|
+
┌──────────────────────────────────
|
|
48
|
+
{"decision": "work", ...}
|
|
49
|
+
└─ #0004 end · 86 B ───────────────
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Column 3 is the delta from the previous event. Scan it to find slow steps. A
|
|
53
|
+
negative delta means an event whose span declared its kind late and was therefore
|
|
54
|
+
placed at its end rather than its start; the timestamp is still correct.
|
|
55
|
+
|
|
56
|
+
## Recipes
|
|
57
|
+
|
|
58
|
+
```sh
|
|
59
|
+
# failed runs, newest last
|
|
60
|
+
jq -r 'select(.status!="ok") | "\(.run_id) \(.errors) errs $\(.cost_usd) \(.last_error)"' index.jsonl
|
|
61
|
+
|
|
62
|
+
# the most common failure across all history
|
|
63
|
+
rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
|
|
64
|
+
|
|
65
|
+
# every error in one run, with its anchor
|
|
66
|
+
# filter on payload.error_type, not .status, agent.end also carries status "error"
|
|
67
|
+
jq -r 'select(.payload.error_type) | "#\(.seq|tostring|("0000"+.)[-4:]) \(.name) \(.payload.error_type): \(.payload.error)"' latest/trace.jsonl
|
|
68
|
+
|
|
69
|
+
# what a tool was actually called with, every time
|
|
70
|
+
rg -N '^#\d+ .* tool\.call run_python' runs/*/agent.logs
|
|
71
|
+
|
|
72
|
+
# read one event and its payload block
|
|
73
|
+
rg -n -A40 '^#0023 ' latest/agent.logs
|
|
74
|
+
|
|
75
|
+
# where the money went
|
|
76
|
+
jq -r '.cost.by_agent | to_entries[] | "\(.key) $\(.value.cost_usd)"' latest/analytics.json
|
|
77
|
+
|
|
78
|
+
# slowest steps in a run, exclude run.* or the run total wins every time
|
|
79
|
+
jq -r 'select(.dur_ms!=null and (.type|startswith("run.")|not)) | "\(.dur_ms)ms #\(.seq|tostring|("0000"+.)[-4:]) \(.type) \(.name)"' latest/trace.jsonl | sort -rn | head
|
|
80
|
+
|
|
81
|
+
# did this session get resumed, and why did the earlier attempt end?
|
|
82
|
+
jq '.attempts' latest/meta.json
|
|
83
|
+
rg -n '^==== attempt|^---- attempt' latest/agent.logs
|
|
84
|
+
|
|
85
|
+
# a prompt too big to inline
|
|
86
|
+
rg -n 'blobs/' latest/agent.logs # find the pointer
|
|
87
|
+
rg -n 'PCA' latest/blobs/*.txt # then grep the blob like any text file
|
|
88
|
+
|
|
89
|
+
# same failure across runs, or just this one?
|
|
90
|
+
rg -lN 'TimeoutError' runs/*/agent.logs
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Counting things correctly
|
|
94
|
+
|
|
95
|
+
Three traps, each of which silently produces a wrong answer rather than an error:
|
|
96
|
+
|
|
97
|
+
- `status: "error"` appears on `agent.end` too. Count failures by
|
|
98
|
+
`payload.error_type`.
|
|
99
|
+
- `run.end` carries the run total in `dur_ms`. Exclude `run.*` when ranking steps.
|
|
100
|
+
- `agent.end` carries a cost rollup. Sum leaf costs, or read
|
|
101
|
+
`analytics.json → cost.by_agent`.
|
|
102
|
+
|
|
103
|
+
## Event types
|
|
104
|
+
|
|
105
|
+
`run.start` `run.resume` `run.end` · `agent.start` `agent.end` ·
|
|
106
|
+
`llm.prompt` `llm.response` · `tool.call` `tool.result` `tool.error` ·
|
|
107
|
+
`error` · `artifact` · `log`
|
|
108
|
+
|
|
109
|
+
Namespaced on purpose: `rg 'tool\.'` gets all tool activity, `rg '\.error'`
|
|
110
|
+
gets every failure shape.
|
|
111
|
+
|
|
112
|
+
## Resume
|
|
113
|
+
|
|
114
|
+
A session that is resumed appends into the same run folder under a new
|
|
115
|
+
`==== attempt N ====` banner. Sequence numbers continue across attempts and
|
|
116
|
+
are never reused. An attempt that died without `run.end` is marked
|
|
117
|
+
`---- attempt N ended ... without run.end ----`, which is how you tell a crash
|
|
118
|
+
from a clean failure.
|
|
119
|
+
|
|
120
|
+
## When you are looking for a systemic problem
|
|
121
|
+
|
|
122
|
+
Do not read one run. Aggregate first:
|
|
123
|
+
|
|
124
|
+
```sh
|
|
125
|
+
rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
|
|
126
|
+
jq -s 'group_by(.status)[] | {status: .[0].status, n: length}' index.jsonl
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Then open the cheapest run that shows the pattern, not the biggest.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: aftersight
|
|
3
|
+
description: Use when investigating how this project's agent behaved: a failed or slow agent run, a recurring tool failure, unexpected model output, rising cost, or any question about what an agent actually sent and received. Reads the execution traces in .runs/.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Reading this project's agent traces
|
|
7
|
+
|
|
8
|
+
Agent runs in this repo record to `.runs/`. That directory is the evidence for
|
|
9
|
+
what an agent actually did. Read it before theorising about behaviour.
|
|
10
|
+
|
|
11
|
+
Start with `.runs/NAVIGATE.md`. It documents the layout, the `#seq` anchor
|
|
12
|
+
scheme and a set of verified `rg`/`jq` recipes.
|
|
13
|
+
|
|
14
|
+
## The loop
|
|
15
|
+
|
|
16
|
+
1. `.runs/index.jsonl` holds one line per run. Pick the run.
|
|
17
|
+
2. `.runs/runs/<run_id>/outline.md` is the folded map. Find the `#seq` that
|
|
18
|
+
matters.
|
|
19
|
+
3. `.runs/runs/<run_id>/agent.logs` at that anchor has the exact prompt,
|
|
20
|
+
completion, tool arguments or traceback.
|
|
21
|
+
|
|
22
|
+
## Rules that keep answers correct
|
|
23
|
+
|
|
24
|
+
- Aggregate before you read. One run shows an incident; `runs/*/trace.jsonl`
|
|
25
|
+
shows whether it is systemic.
|
|
26
|
+
- Glob as `runs/*/…`. `latest` and `sessions/` are symlinks into `runs/`, so
|
|
27
|
+
globbing the root double-counts.
|
|
28
|
+
- Count failures by `payload.error_type`, not `status == "error"`, because
|
|
29
|
+
`agent.end` carries that status too.
|
|
30
|
+
- Exclude `run.*` when ranking slow steps; `run.end` holds the run total.
|
|
31
|
+
- A prompt over 8 KB lives in `blobs/` as plain text. Grep it like any file.
|
|
32
|
+
- Sequence numbers continue across a resumed session, and an attempt that died
|
|
33
|
+
without `run.end` is marked as such. A crash and a clean failure are different
|
|
34
|
+
findings.
|
|
35
|
+
|
|
36
|
+
## When asked to improve the agent
|
|
37
|
+
|
|
38
|
+
Ground every claim in an anchor. "Tool `x` timed out in 4 of the last 9 runs
|
|
39
|
+
(`#0044`, `#0071`, …)" is actionable; "the executor seems flaky" is not.
|