aftersight 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. aftersight-0.1.0/.github/workflows/docs.yml +43 -0
  2. aftersight-0.1.0/.github/workflows/release.yml +26 -0
  3. aftersight-0.1.0/.gitignore +10 -0
  4. aftersight-0.1.0/AGENTS.md +101 -0
  5. aftersight-0.1.0/LICENSE +21 -0
  6. aftersight-0.1.0/PKG-INFO +157 -0
  7. aftersight-0.1.0/README.md +131 -0
  8. aftersight-0.1.0/aftersight/__init__.py +21 -0
  9. aftersight-0.1.0/aftersight/assets/NAVIGATE.md +129 -0
  10. aftersight-0.1.0/aftersight/assets/skill/SKILL.md +39 -0
  11. aftersight-0.1.0/aftersight/autostart.py +8 -0
  12. aftersight-0.1.0/aftersight/cli.py +91 -0
  13. aftersight-0.1.0/aftersight/config.py +60 -0
  14. aftersight-0.1.0/aftersight/constants.py +125 -0
  15. aftersight-0.1.0/aftersight/events/__init__.py +0 -0
  16. aftersight-0.1.0/aftersight/events/schemas.py +59 -0
  17. aftersight-0.1.0/aftersight/events/sink.py +58 -0
  18. aftersight-0.1.0/aftersight/integrations/__init__.py +0 -0
  19. aftersight-0.1.0/aftersight/integrations/otel.py +249 -0
  20. aftersight-0.1.0/aftersight/integrations/stdlog.py +49 -0
  21. aftersight-0.1.0/aftersight/py.typed +0 -0
  22. aftersight-0.1.0/aftersight/run/__init__.py +0 -0
  23. aftersight-0.1.0/aftersight/run/api.py +211 -0
  24. aftersight-0.1.0/aftersight/run/run.py +164 -0
  25. aftersight-0.1.0/aftersight/run/store.py +134 -0
  26. aftersight-0.1.0/aftersight/utils.py +55 -0
  27. aftersight-0.1.0/aftersight/writers/__init__.py +0 -0
  28. aftersight-0.1.0/aftersight/writers/analytics.py +108 -0
  29. aftersight-0.1.0/aftersight/writers/blobs.py +42 -0
  30. aftersight-0.1.0/aftersight/writers/outline.py +120 -0
  31. aftersight-0.1.0/aftersight/writers/trace.py +40 -0
  32. aftersight-0.1.0/aftersight/writers/transcript.py +194 -0
  33. aftersight-0.1.0/docs/design/architecture.md +129 -0
  34. aftersight-0.1.0/docs/design/decisions.md +208 -0
  35. aftersight-0.1.0/docs/frameworks.md +151 -0
  36. aftersight-0.1.0/docs/index.md +36 -0
  37. aftersight-0.1.0/docs/navigating.md +122 -0
  38. aftersight-0.1.0/docs/quickstart.md +120 -0
  39. aftersight-0.1.0/docs/reference/api.md +172 -0
  40. aftersight-0.1.0/docs/reference/cli.md +49 -0
  41. aftersight-0.1.0/docs/reference/configuration.md +75 -0
  42. aftersight-0.1.0/docs/reference/run-folder.md +178 -0
  43. aftersight-0.1.0/docs/superpowers/plans/2026-08-30-public-release.md +616 -0
  44. aftersight-0.1.0/docs/superpowers/specs/2026-08-30-public-release-design.md +207 -0
  45. aftersight-0.1.0/docs/troubleshooting.md +100 -0
  46. aftersight-0.1.0/docs/why.md +97 -0
  47. aftersight-0.1.0/examples/generate.py +139 -0
  48. aftersight-0.1.0/examples/starters/claude_agent_sdk_starter.py +40 -0
  49. aftersight-0.1.0/examples/starters/langchain_starter.py +38 -0
  50. aftersight-0.1.0/examples/starters/openai_agents_starter.py +37 -0
  51. aftersight-0.1.0/examples/starters/openhands_starter.py +37 -0
  52. aftersight-0.1.0/examples/starters/pydantic_ai_starter.py +29 -0
  53. aftersight-0.1.0/examples/telemetry_root/NAVIGATE.md +129 -0
  54. aftersight-0.1.0/examples/telemetry_root/index.jsonl +1 -0
  55. aftersight-0.1.0/examples/telemetry_root/latest/agent.logs +67 -0
  56. aftersight-0.1.0/examples/telemetry_root/latest/analytics.json +74 -0
  57. aftersight-0.1.0/examples/telemetry_root/latest/blobs/db797ac6.txt +2803 -0
  58. aftersight-0.1.0/examples/telemetry_root/latest/meta.json +41 -0
  59. aftersight-0.1.0/examples/telemetry_root/latest/outline.md +22 -0
  60. aftersight-0.1.0/examples/telemetry_root/latest/trace.jsonl +25 -0
  61. aftersight-0.1.0/mkdocs.yml +77 -0
  62. aftersight-0.1.0/pyproject.toml +42 -0
  63. aftersight-0.1.0/tests/test_formats.py +216 -0
@@ -0,0 +1,43 @@
1
+ name: docs
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ paths:
7
+ - docs/**
8
+ - mkdocs.yml
9
+ - .github/workflows/docs.yml
10
+ workflow_dispatch:
11
+
12
+ permissions:
13
+ contents: read
14
+ pages: write
15
+ id-token: write
16
+
17
+ concurrency:
18
+ group: pages
19
+ cancel-in-progress: false
20
+
21
+ jobs:
22
+ build:
23
+ runs-on: ubuntu-latest
24
+ steps:
25
+ - uses: actions/checkout@v4
26
+ - uses: actions/setup-python@v5
27
+ with:
28
+ python-version: "3.12"
29
+ - run: pip install -e '.[docs]'
30
+ - run: mkdocs build --strict
31
+ - uses: actions/upload-pages-artifact@v3
32
+ with:
33
+ path: site
34
+
35
+ deploy:
36
+ needs: build
37
+ runs-on: ubuntu-latest
38
+ environment:
39
+ name: github-pages
40
+ url: ${{ steps.deployment.outputs.page_url }}
41
+ steps:
42
+ - id: deployment
43
+ uses: actions/deploy-pages@v4
@@ -0,0 +1,26 @@
1
+ name: release
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+ workflow_dispatch:
7
+
8
+ jobs:
9
+ publish:
10
+ runs-on: ubuntu-latest
11
+ # Trusted publishing: PyPI verifies this workflow through OIDC, so there is
12
+ # no API token to store or rotate. Register the publisher on PyPI with
13
+ # owner nebulaanish, repository aftersight, workflow release.yml and
14
+ # environment pypi, all four of which must match what is below.
15
+ environment: pypi
16
+ permissions:
17
+ id-token: write
18
+ steps:
19
+ - uses: actions/checkout@v4
20
+ - uses: actions/setup-python@v5
21
+ with:
22
+ python-version: "3.12"
23
+ - run: pip install -e . build
24
+ - run: python tests/test_formats.py
25
+ - run: python -m build
26
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,10 @@
1
+ .runs/
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ .venv/
6
+ dist/
7
+ build/
8
+ .pytest_cache/
9
+ .ruff_cache/
10
+ site/
@@ -0,0 +1,101 @@
1
+ # AGENTS.md
2
+
3
+ Behavioral guidelines to reduce common LLM coding mistakes. Merge with project-specific instructions as needed.
4
+
5
+ **Tradeoff:** These guidelines bias toward caution over speed. For trivial tasks, use judgment.
6
+
7
+ ## 1. Think Before Coding
8
+
9
+ **Don't assume. Don't hide confusion. Surface tradeoffs.**
10
+
11
+ Before implementing:
12
+ - State your assumptions explicitly. If uncertain, ask.
13
+ - If multiple interpretations exist, present them - don't pick silently.
14
+ - If a simpler approach exists, say so. Push back when warranted.
15
+ - If something is unclear, stop. Name what's confusing. Ask.
16
+
17
+ ## 2. Simplicity First
18
+
19
+ **Minimum code that solves the problem. Nothing speculative.**
20
+
21
+ - No features beyond what was asked.
22
+ - No abstractions for single-use code.
23
+ - No "flexibility" or "configurability" that wasn't requested.
24
+ - No error handling for impossible scenarios.
25
+ - If you write 200 lines and it could be 50, rewrite it.
26
+
27
+ Ask yourself: "Would a senior engineer say this is overcomplicated?" If yes, simplify.
28
+
29
+ ## 3. Surgical Changes
30
+
31
+ **Touch only what you must. Clean up only your own mess.**
32
+
33
+ When editing existing code:
34
+ - Don't "improve" adjacent code, comments, or formatting.
35
+ - Don't refactor things that aren't broken.
36
+ - Match existing style, even if you'd do it differently.
37
+ - If you notice unrelated dead code, mention it - don't delete it.
38
+
39
+ When your changes create orphans:
40
+ - Remove imports/variables/functions that YOUR changes made unused.
41
+ - Don't remove pre-existing dead code unless asked.
42
+
43
+ The test: Every changed line should trace directly to the user's request.
44
+
45
+ ## 4. Goal-Driven Execution
46
+
47
+ **Define success criteria. Loop until verified.**
48
+
49
+ Transform tasks into verifiable goals:
50
+ - "Add validation" → "Write tests for invalid inputs, then make them pass"
51
+ - "Fix the bug" → "Write a test that reproduces it, then make it pass"
52
+ - "Refactor X" → "Ensure tests pass before and after"
53
+
54
+ For multi-step tasks, state a brief plan:
55
+ ```
56
+ 1. [Step] → verify: [check]
57
+ 2. [Step] → verify: [check]
58
+ 3. [Step] → verify: [check]
59
+ ```
60
+
61
+ Strong success criteria let you loop independently. Weak criteria ("make it work") require constant clarification.
62
+
63
+ ---
64
+
65
+ **These guidelines are working if:** fewer unnecessary changes in diffs, fewer rewrites due to overcomplication, and clarifying questions come before implementation rather than after mistakes.
66
+
67
+
68
+
69
+ # Guidelines: What Defines a Valid Test Case
70
+
71
+ Strictly enforce this standard when writing or evaluating tests. Reject any test case exhibiting invalid anti-patterns.
72
+
73
+ ---
74
+
75
+ ## 1. Valid Test Case Criteria
76
+ A valid test must satisfy ALL of the following:
77
+
78
+ - **Single Concept:** Tests exactly ONE behavior per test function.
79
+ - **Black-Box Testing:** Verifies public interfaces and inputs/outputs—never internal private implementation details.
80
+ - **100% Deterministic:** Zero reliance on ambient clock (`Date.now()`), random seeds, external APIs, or dirty DB state.
81
+ - **Explicit AAA:** Structured strictly as Arrange (setup), Act (trigger), Assert (check).
82
+ - **Refactor-Resistant:** Must pass if internal code is completely refactored without breaking the public contract.
83
+ - **High Information Density:** Focuses on realistic boundaries (`0`, `null`, empty arrays, max limits) or explicit error modes.
84
+
85
+ ---
86
+
87
+ ## 2. Invalid Test Case Anti-Patterns (Immediate Rejection)
88
+
89
+ - ❌ **Implementation Coupling:** Asserting internal helper execution order or checking private class state.
90
+ - ❌ **Tautologies:** Re-implementing production logic inside the test setup or asserting simple pass-through getters.
91
+ - ❌ **Multi-Scenario Bloat:** Single tests containing sequential user actions, >3 independent assertions, or multiple business rules.
92
+ - ❌ **Over-Mocking:** Mocking simple domain models or internal helper classes instead of using real instances or factories.
93
+ - ❌ **Flakiness:** Relying on arbitrary `sleep()` calls, live network calls, or unordered array checks.
94
+
95
+ ---
96
+
97
+ ## Output Verification Checklist
98
+ Before outputting code, verify:
99
+ 1. Is the test name `<Method>_<Scenario>_<ExpectedResult>`?
100
+ 2. Does it test what the code DOES, not HOW it does it?
101
+ 3. If the implementation changes internally, will this test still pass?
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Balaram
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,157 @@
1
+ Metadata-Version: 2.5
2
+ Name: aftersight
3
+ Version: 0.1.0
4
+ Summary: Observability infrastructure for self-improving agents
5
+ Project-URL: Homepage, https://github.com/nebulaanish/aftersight
6
+ Project-URL: Documentation, https://nebulaanish.github.io/aftersight/
7
+ Project-URL: Source, https://github.com/nebulaanish/aftersight
8
+ Project-URL: Issues, https://github.com/nebulaanish/aftersight/issues
9
+ Author-email: Balaram <balaramneu@gmail.com>
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: agents,llm,observability,opentelemetry,tracing
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Topic :: Software Development :: Debuggers
17
+ Classifier: Topic :: System :: Monitoring
18
+ Classifier: Typing :: Typed
19
+ Requires-Python: >=3.10
20
+ Requires-Dist: opentelemetry-api>=1.20
21
+ Requires-Dist: opentelemetry-sdk>=1.20
22
+ Provides-Extra: docs
23
+ Requires-Dist: mkdocs-material>=9.5; extra == 'docs'
24
+ Requires-Dist: mkdocs<2; extra == 'docs'
25
+ Description-Content-Type: text/markdown
26
+
27
+ # aftersight
28
+
29
+ Observability infrastructure for self-improving agents.
30
+
31
+ An agent cannot improve on a run it cannot read. Aftersight writes every
32
+ run into the repository as plain files, so the agent can read its own
33
+ history with the tools it already has.
34
+
35
+ ## The problem
36
+
37
+ Coding agents already explore repositories with file search, shell commands and
38
+ plain text. Agent telemetry usually lives somewhere else: behind a dashboard,
39
+ an API or a remote MCP server.
40
+
41
+ Aftersight writes the evidence into the repository instead. A coding agent can
42
+ use the tools it already knows to:
43
+
44
+ 1. diagnose why one run failed;
45
+ 2. find failures that recur across runs;
46
+ 3. inspect the exact inputs and outputs of models and tools.
47
+
48
+ ## Start in one command
49
+
50
+ ```bash
51
+ pip install aftersight
52
+ aftersight run python my_agent.py
53
+ ```
54
+
55
+ No account, API key, service or initialization step is required.
56
+
57
+ ## Ask your coding agent
58
+
59
+ Each telemetry root includes a `NAVIGATE.md` with verified `rg` and `jq`
60
+ recipes. Tell your coding agent:
61
+
62
+ > Read `.runs/NAVIGATE.md` and find out why the latest run failed.
63
+
64
+ For Claude Code, the optional command below installs the bundled navigation
65
+ skill:
66
+
67
+ ```bash
68
+ aftersight skill
69
+ ```
70
+
71
+ ## What gets recorded
72
+
73
+ ```text
74
+ .runs/
75
+ NAVIGATE.md
76
+ index.jsonl
77
+ latest -> runs/<run_id>
78
+ runs/<run_id>/
79
+ outline.md
80
+ agent.logs
81
+ trace.jsonl
82
+ analytics.json
83
+ meta.json
84
+ blobs/
85
+ artifacts/
86
+ ```
87
+
88
+ `outline.md` is the short map. `agent.logs` is the complete readable
89
+ transcript. `trace.jsonl`, `analytics.json` and `index.jsonl` are stable
90
+ machine-readable views for scripts and frontends. The same `#seq` anchor
91
+ identifies an event in every projection.
92
+
93
+ ## Use it from Python
94
+
95
+ ```python
96
+ import aftersight
97
+
98
+ aftersight.start()
99
+ ```
100
+
101
+ Add explicit detail only where it helps:
102
+
103
+ ```python
104
+ with aftersight.span("planner"):
105
+ with aftersight.span("web_search", kind="tool", args={"q": query}) as span:
106
+ span.output = search(query)
107
+
108
+ @aftersight.trace
109
+ def read_file(path): ...
110
+
111
+ aftersight.log("cache miss", key=key)
112
+ ```
113
+
114
+ ## Works with OpenTelemetry
115
+
116
+ Aftersight attaches a span processor to the application's existing
117
+ OpenTelemetry setup and reads common `gen_ai.*`, OpenInference and generic
118
+ input/output attributes. Frameworks that already emit compatible spans need no
119
+ aftersight-specific adapter. An existing tracer provider and its exporters are
120
+ left in place.
121
+
122
+ ## How it compares
123
+
124
+ [LangSmith](https://docs.langchain.com/langsmith/observability-concepts),
125
+ [Langfuse](https://langfuse.com/docs/api-and-data-platform/features/observations-api),
126
+ [Phoenix](https://arize.com/docs/phoenix/tracing/how-to-tracing/importing-and-exporting-traces/extract-data-from-spans)
127
+ and [Braintrust](https://www.braintrust.dev/docs/integrations/developer-tools/mcp)
128
+ provide mature observability with dashboards and programmatic access through
129
+ APIs, exports or MCP.
130
+
131
+ Aftersight has a narrower default: ordinary files in the repository, with no
132
+ hosted project, credentials or remote query round trips. Those platforms are a
133
+ better fit when you need centralized durability, team dashboards or managed
134
+ evaluation workflows. See [Why aftersight exists](docs/why.md) for the fuller
135
+ comparison.
136
+
137
+ ## Local by default
138
+
139
+ Aftersight itself makes no network requests and does not upload telemetry.
140
+ Payload redaction is enabled by default, and the run folder is added to
141
+ `.gitignore` when it is first created.
142
+
143
+ If the application already has an OpenTelemetry exporter, that exporter may
144
+ still send its own spans. Aftersight can run in production, but local disk can
145
+ disappear with an ephemeral, replaced or failed host, so it should not be the
146
+ only durable production record.
147
+
148
+ ## Documentation
149
+
150
+ - [Quickstart](docs/quickstart.md)
151
+ - [Framework support](docs/frameworks.md)
152
+ - [Navigating runs](docs/navigating.md)
153
+ - [Python API](docs/reference/api.md)
154
+ - [Run folder format](docs/reference/run-folder.md)
155
+ - [Architecture](docs/design/architecture.md)
156
+
157
+ Full documentation: **https://nebulaanish.github.io/aftersight/**
@@ -0,0 +1,131 @@
1
+ # aftersight
2
+
3
+ Observability infrastructure for self-improving agents.
4
+
5
+ An agent cannot improve on a run it cannot read. Aftersight writes every
6
+ run into the repository as plain files, so the agent can read its own
7
+ history with the tools it already has.
8
+
9
+ ## The problem
10
+
11
+ Coding agents already explore repositories with file search, shell commands and
12
+ plain text. Agent telemetry usually lives somewhere else: behind a dashboard,
13
+ an API or a remote MCP server.
14
+
15
+ Aftersight writes the evidence into the repository instead. A coding agent can
16
+ use the tools it already knows to:
17
+
18
+ 1. diagnose why one run failed;
19
+ 2. find failures that recur across runs;
20
+ 3. inspect the exact inputs and outputs of models and tools.
21
+
22
+ ## Start in one command
23
+
24
+ ```bash
25
+ pip install aftersight
26
+ aftersight run python my_agent.py
27
+ ```
28
+
29
+ No account, API key, service or initialization step is required.
30
+
31
+ ## Ask your coding agent
32
+
33
+ Each telemetry root includes a `NAVIGATE.md` with verified `rg` and `jq`
34
+ recipes. Tell your coding agent:
35
+
36
+ > Read `.runs/NAVIGATE.md` and find out why the latest run failed.
37
+
38
+ For Claude Code, the optional command below installs the bundled navigation
39
+ skill:
40
+
41
+ ```bash
42
+ aftersight skill
43
+ ```
44
+
45
+ ## What gets recorded
46
+
47
+ ```text
48
+ .runs/
49
+ NAVIGATE.md
50
+ index.jsonl
51
+ latest -> runs/<run_id>
52
+ runs/<run_id>/
53
+ outline.md
54
+ agent.logs
55
+ trace.jsonl
56
+ analytics.json
57
+ meta.json
58
+ blobs/
59
+ artifacts/
60
+ ```
61
+
62
+ `outline.md` is the short map. `agent.logs` is the complete readable
63
+ transcript. `trace.jsonl`, `analytics.json` and `index.jsonl` are stable
64
+ machine-readable views for scripts and frontends. The same `#seq` anchor
65
+ identifies an event in every projection.
66
+
67
+ ## Use it from Python
68
+
69
+ ```python
70
+ import aftersight
71
+
72
+ aftersight.start()
73
+ ```
74
+
75
+ Add explicit detail only where it helps:
76
+
77
+ ```python
78
+ with aftersight.span("planner"):
79
+ with aftersight.span("web_search", kind="tool", args={"q": query}) as span:
80
+ span.output = search(query)
81
+
82
+ @aftersight.trace
83
+ def read_file(path): ...
84
+
85
+ aftersight.log("cache miss", key=key)
86
+ ```
87
+
88
+ ## Works with OpenTelemetry
89
+
90
+ Aftersight attaches a span processor to the application's existing
91
+ OpenTelemetry setup and reads common `gen_ai.*`, OpenInference and generic
92
+ input/output attributes. Frameworks that already emit compatible spans need no
93
+ aftersight-specific adapter. An existing tracer provider and its exporters are
94
+ left in place.
95
+
96
+ ## How it compares
97
+
98
+ [LangSmith](https://docs.langchain.com/langsmith/observability-concepts),
99
+ [Langfuse](https://langfuse.com/docs/api-and-data-platform/features/observations-api),
100
+ [Phoenix](https://arize.com/docs/phoenix/tracing/how-to-tracing/importing-and-exporting-traces/extract-data-from-spans)
101
+ and [Braintrust](https://www.braintrust.dev/docs/integrations/developer-tools/mcp)
102
+ provide mature observability with dashboards and programmatic access through
103
+ APIs, exports or MCP.
104
+
105
+ Aftersight has a narrower default: ordinary files in the repository, with no
106
+ hosted project, credentials or remote query round trips. Those platforms are a
107
+ better fit when you need centralized durability, team dashboards or managed
108
+ evaluation workflows. See [Why aftersight exists](docs/why.md) for the fuller
109
+ comparison.
110
+
111
+ ## Local by default
112
+
113
+ Aftersight itself makes no network requests and does not upload telemetry.
114
+ Payload redaction is enabled by default, and the run folder is added to
115
+ `.gitignore` when it is first created.
116
+
117
+ If the application already has an OpenTelemetry exporter, that exporter may
118
+ still send its own spans. Aftersight can run in production, but local disk can
119
+ disappear with an ephemeral, replaced or failed host, so it should not be the
120
+ only durable production record.
121
+
122
+ ## Documentation
123
+
124
+ - [Quickstart](docs/quickstart.md)
125
+ - [Framework support](docs/frameworks.md)
126
+ - [Navigating runs](docs/navigating.md)
127
+ - [Python API](docs/reference/api.md)
128
+ - [Run folder format](docs/reference/run-folder.md)
129
+ - [Architecture](docs/design/architecture.md)
130
+
131
+ Full documentation: **https://nebulaanish.github.io/aftersight/**
@@ -0,0 +1,21 @@
1
+ """Observability infrastructure for self-improving agents.
2
+
3
+ import aftersight
4
+ aftersight.start()
5
+
6
+ Or, with no code change at all:
7
+
8
+ aftersight run python my_agent.py
9
+ """
10
+
11
+ from aftersight.run.api import (
12
+ artifact_dir,
13
+ current,
14
+ log,
15
+ span,
16
+ start,
17
+ trace,
18
+ )
19
+
20
+ __all__ = ["start", "span", "trace", "log", "current", "artifact_dir"]
21
+ __version__ = "0.1.0"
@@ -0,0 +1,129 @@
1
+ # Navigating these runs
2
+
3
+ You are a coding agent. This directory is the execution history of the agent
4
+ system in this repo. Read this file once, then use `rg` and `jq`.
5
+
6
+ ## Layout
7
+
8
+ ```
9
+ index.jsonl one line per run, start here for "which run?"
10
+ latest -> runs/<run_id> most recent run
11
+ sessions/<session_id> -> the run folder that session resumed into
12
+ runs/<run_id>/
13
+ outline.md folded map of the run, read this first
14
+ agent.logs full transcript, nothing truncated
15
+ trace.jsonl same events, machine-readable, has parent_seq
16
+ analytics.json status, timings, per-agent cost, error list
17
+ meta.json framework, git sha, argv, attempts[]
18
+ blobs/<sha>.txt payloads over 8 KB, plain text
19
+ artifacts/ whatever the app chose to keep
20
+ ```
21
+
22
+ Run ids are `YYYYMMDD_HHMMSS_hash`, so they sort chronologically. Glob run
23
+ files as `runs/*/...` and never from the root: `latest` and `sessions/` are
24
+ symlinks into `runs/`, so a root glob counts every run twice.
25
+
26
+ ## The three-step loop
27
+
28
+ 1. **Which run?** `index.jsonl`
29
+ 2. **What happened in it?** `runs/<run_id>/outline.md`, then jump to a `#seq`
30
+ 3. **Exactly what went in and out?** `runs/<run_id>/agent.logs` at that `#seq`
31
+
32
+ ## Anchors
33
+
34
+ Every event has a zero-padded sequence number. It is the same number in
35
+ `outline.md`, `agent.logs` and `trace.jsonl`, so you can move between them:
36
+
37
+ ```sh
38
+ rg -n '^#0016 ' latest/agent.logs # jump to one event
39
+ rg -n '#0016' runs/<run_id>/ # every mention of it
40
+ ```
41
+
42
+ Multi-line payloads are framed. The closing line repeats the anchor, so a hit
43
+ landing inside a block can find both edges:
44
+
45
+ ```
46
+ #0004 14:22:35.512 +2.4s llm.response triage stop=end_turn · 386 tok · $0.0005
47
+ ┌──────────────────────────────────
48
+ {"decision": "work", ...}
49
+ └─ #0004 end · 86 B ───────────────
50
+ ```
51
+
52
+ Column 3 is the delta from the previous event. Scan it to find slow steps. A
53
+ negative delta means an event whose span declared its kind late and was therefore
54
+ placed at its end rather than its start; the timestamp is still correct.
55
+
56
+ ## Recipes
57
+
58
+ ```sh
59
+ # failed runs, newest last
60
+ jq -r 'select(.status!="ok") | "\(.run_id) \(.errors) errs $\(.cost_usd) \(.last_error)"' index.jsonl
61
+
62
+ # the most common failure across all history
63
+ rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
64
+
65
+ # every error in one run, with its anchor
66
+ # filter on payload.error_type, not .status, agent.end also carries status "error"
67
+ jq -r 'select(.payload.error_type) | "#\(.seq|tostring|("0000"+.)[-4:]) \(.name) \(.payload.error_type): \(.payload.error)"' latest/trace.jsonl
68
+
69
+ # what a tool was actually called with, every time
70
+ rg -N '^#\d+ .* tool\.call run_python' runs/*/agent.logs
71
+
72
+ # read one event and its payload block
73
+ rg -n -A40 '^#0023 ' latest/agent.logs
74
+
75
+ # where the money went
76
+ jq -r '.cost.by_agent | to_entries[] | "\(.key) $\(.value.cost_usd)"' latest/analytics.json
77
+
78
+ # slowest steps in a run, exclude run.* or the run total wins every time
79
+ jq -r 'select(.dur_ms!=null and (.type|startswith("run.")|not)) | "\(.dur_ms)ms #\(.seq|tostring|("0000"+.)[-4:]) \(.type) \(.name)"' latest/trace.jsonl | sort -rn | head
80
+
81
+ # did this session get resumed, and why did the earlier attempt end?
82
+ jq '.attempts' latest/meta.json
83
+ rg -n '^==== attempt|^---- attempt' latest/agent.logs
84
+
85
+ # a prompt too big to inline
86
+ rg -n 'blobs/' latest/agent.logs # find the pointer
87
+ rg -n 'PCA' latest/blobs/*.txt # then grep the blob like any text file
88
+
89
+ # same failure across runs, or just this one?
90
+ rg -lN 'TimeoutError' runs/*/agent.logs
91
+ ```
92
+
93
+ ## Counting things correctly
94
+
95
+ Three traps, each of which silently produces a wrong answer rather than an error:
96
+
97
+ - `status: "error"` appears on `agent.end` too. Count failures by
98
+ `payload.error_type`.
99
+ - `run.end` carries the run total in `dur_ms`. Exclude `run.*` when ranking steps.
100
+ - `agent.end` carries a cost rollup. Sum leaf costs, or read
101
+ `analytics.json → cost.by_agent`.
102
+
103
+ ## Event types
104
+
105
+ `run.start` `run.resume` `run.end` · `agent.start` `agent.end` ·
106
+ `llm.prompt` `llm.response` · `tool.call` `tool.result` `tool.error` ·
107
+ `error` · `artifact` · `log`
108
+
109
+ Namespaced on purpose: `rg 'tool\.'` gets all tool activity, `rg '\.error'`
110
+ gets every failure shape.
111
+
112
+ ## Resume
113
+
114
+ A session that is resumed appends into the same run folder under a new
115
+ `==== attempt N ====` banner. Sequence numbers continue across attempts and
116
+ are never reused. An attempt that died without `run.end` is marked
117
+ `---- attempt N ended ... without run.end ----`, which is how you tell a crash
118
+ from a clean failure.
119
+
120
+ ## When you are looking for a systemic problem
121
+
122
+ Do not read one run. Aggregate first:
123
+
124
+ ```sh
125
+ rg -oN '"error_type": "[A-Za-z]+"' runs/*/trace.jsonl | cut -d'"' -f4 | sort | uniq -c | sort -rn
126
+ jq -s 'group_by(.status)[] | {status: .[0].status, n: length}' index.jsonl
127
+ ```
128
+
129
+ Then open the cheapest run that shows the pattern, not the biggest.
@@ -0,0 +1,39 @@
1
+ ---
2
+ name: aftersight
3
+ description: Use when investigating how this project's agent behaved: a failed or slow agent run, a recurring tool failure, unexpected model output, rising cost, or any question about what an agent actually sent and received. Reads the execution traces in .runs/.
4
+ ---
5
+
6
+ # Reading this project's agent traces
7
+
8
+ Agent runs in this repo record to `.runs/`. That directory is the evidence for
9
+ what an agent actually did. Read it before theorising about behaviour.
10
+
11
+ Start with `.runs/NAVIGATE.md`. It documents the layout, the `#seq` anchor
12
+ scheme and a set of verified `rg`/`jq` recipes.
13
+
14
+ ## The loop
15
+
16
+ 1. `.runs/index.jsonl` holds one line per run. Pick the run.
17
+ 2. `.runs/runs/<run_id>/outline.md` is the folded map. Find the `#seq` that
18
+ matters.
19
+ 3. `.runs/runs/<run_id>/agent.logs` at that anchor has the exact prompt,
20
+ completion, tool arguments or traceback.
21
+
22
+ ## Rules that keep answers correct
23
+
24
+ - Aggregate before you read. One run shows an incident; `runs/*/trace.jsonl`
25
+ shows whether it is systemic.
26
+ - Glob as `runs/*/…`. `latest` and `sessions/` are symlinks into `runs/`, so
27
+ globbing the root double-counts.
28
+ - Count failures by `payload.error_type`, not `status == "error"`, because
29
+ `agent.end` carries that status too.
30
+ - Exclude `run.*` when ranking slow steps; `run.end` holds the run total.
31
+ - A prompt over 8 KB lives in `blobs/` as plain text. Grep it like any file.
32
+ - Sequence numbers continue across a resumed session, and an attempt that died
33
+ without `run.end` is marked as such. A crash and a clean failure are different
34
+ findings.
35
+
36
+ ## When asked to improve the agent
37
+
38
+ Ground every claim in an anchor. "Tool `x` timed out in 4 of the last 9 runs
39
+ (`#0044`, `#0071`, …)" is actionable; "the executor seems flaky" is not.