agent-readable 0.2.1__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {agent_readable-0.2.1 → agent_readable-0.3.0}/.github/workflows/ci.yml +3 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/.gitignore +3 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/CHANGELOG.md +50 -1
- {agent_readable-0.2.1 → agent_readable-0.3.0}/PKG-INFO +4 -3
- {agent_readable-0.2.1 → agent_readable-0.3.0}/README.md +2 -1
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/authoring.md +6 -6
- agent_readable-0.3.0/docs/benchmark.md +79 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/examples.md +2 -2
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/getting-started.md +16 -11
- {agent_readable-0.2.1 → agent_readable-0.3.0}/pyproject.toml +12 -1
- agent_readable-0.3.0/scripts/benchmark.py +243 -0
- agent_readable-0.3.0/src/agent_readable/__main__.py +77 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_model.py +130 -17
- {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_protocol.py +37 -47
- {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_render.py +12 -2
- {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/test_cli.py +31 -23
- {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/test_protocol.py +251 -27
- agent_readable-0.2.1/src/agent_readable/__main__.py +0 -59
- {agent_readable-0.2.1 → agent_readable-0.3.0}/.github/workflows/publish.yml +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/LICENSE +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/agent_help_vs_help.gif +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/faq.md +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/why.md +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/any_class.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/duck_type.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/modules_and_functions.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/sqlite_connection.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/temperature.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/__init__.py +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/py.typed +0 -0
- {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/__init__.py +0 -0
|
@@ -7,6 +7,54 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.3.0] - 2026-08-23
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- Console script `agent-readable`, so one-off use is
|
|
15
|
+
`uvx agent-readable sqlite3:Connection` instead of
|
|
16
|
+
`uvx --from agent-readable python -m agent_readable sqlite3:Connection`.
|
|
17
|
+
`python -m agent_readable` keeps working.
|
|
18
|
+
- `--version` flag for the CLI.
|
|
19
|
+
- Public class attributes and module constants are now listed in the Public API
|
|
20
|
+
section with their current value: `repr` for exact primitive types (never
|
|
21
|
+
executing a custom `__repr__`), the type name otherwise.
|
|
22
|
+
- Enum members are listed (e.g. ``- `RED` member: 1``), and the misleading
|
|
23
|
+
`EnumMeta.__call__` constructor signature is no longer shown for enum
|
|
24
|
+
classes.
|
|
25
|
+
- Module docs now include C builtins (`inspect.isroutine` instead of
|
|
26
|
+
`inspect.isfunction`), so `agent_help(math)` lists `sin` alongside
|
|
27
|
+
pure-Python functions, and module-level constants such as `math.pi` are no
|
|
28
|
+
longer dropped by the origin heuristic.
|
|
29
|
+
- A constructor that introspection renders as the placeholder `(*args,
|
|
30
|
+
**kwargs)` — e.g. masked by a metaclass `__call__` — now falls back to the
|
|
31
|
+
class's `__init__`/`__new__` signature.
|
|
32
|
+
- Mypy type-checking job in CI.
|
|
33
|
+
- Benchmark harness (`scripts/benchmark.py`) and methodology
|
|
34
|
+
(`docs/benchmark.md`) for measuring whether `agent_help()` context reduces
|
|
35
|
+
hallucinated-API failures versus a baseline prompt. Status: not yet run;
|
|
36
|
+
no numbers are claimed until a real run is recorded.
|
|
37
|
+
|
|
38
|
+
### Changed
|
|
39
|
+
|
|
40
|
+
- **Breaking:** `__agent_notes__()` sections are now always appended, including
|
|
41
|
+
after a custom `__agent_help__()`'s output. Previously the notes were
|
|
42
|
+
silently dropped when a custom `__agent_help__` was defined and a
|
|
43
|
+
`UserWarning` was emitted; defining both is now a supported combination and
|
|
44
|
+
the warning is gone. For full verbatim control of the entire output, do not
|
|
45
|
+
define `__agent_notes__()` anywhere in the MRO.
|
|
46
|
+
- The CLI prints a one-line `error: ...` message to stderr and exits with
|
|
47
|
+
status 2 for targets that cannot be imported or resolved, instead of raising
|
|
48
|
+
a traceback.
|
|
49
|
+
|
|
50
|
+
### Fixed
|
|
51
|
+
|
|
52
|
+
- A raising `__agent_notes__()` no longer breaks `agent_help()` for the class
|
|
53
|
+
and all of its subclasses; the broken notes are skipped, mirroring the
|
|
54
|
+
existing `__agent_help__()` fallback.
|
|
55
|
+
- `functools.cached_property` members are no longer silently dropped from the
|
|
56
|
+
Public API; they render as properties with the wrapped function's docstring.
|
|
57
|
+
|
|
10
58
|
## [0.2.1] - 2026-07-04
|
|
11
59
|
|
|
12
60
|
### Documentation
|
|
@@ -86,7 +134,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
86
134
|
`AgentReadableMixin`, `__agent_notes__` accumulation across the MRO, module
|
|
87
135
|
support, and the `python -m agent_readable` CLI.
|
|
88
136
|
|
|
89
|
-
[Unreleased]: https://github.com/zydo/agent-readable/compare/v0.
|
|
137
|
+
[Unreleased]: https://github.com/zydo/agent-readable/compare/v0.3.0...HEAD
|
|
138
|
+
[0.3.0]: https://github.com/zydo/agent-readable/compare/v0.2.1...v0.3.0
|
|
90
139
|
[0.2.1]: https://github.com/zydo/agent-readable/compare/v0.2.0...v0.2.1
|
|
91
140
|
[0.1.2]: https://github.com/zydo/agent-readable/compare/v0.1.1...v0.1.2
|
|
92
141
|
[0.1.1]: https://github.com/zydo/agent-readable/compare/v0.1.0...v0.1.1
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: agent-readable
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: A lightweight Python protocol for agent-oriented documentation
|
|
5
5
|
Project-URL: Repository, https://github.com/zydo/agent-readable
|
|
6
6
|
Project-URL: Issues, https://github.com/zydo/agent-readable/issues
|
|
@@ -40,7 +40,7 @@ npx skills add zydo/skills --skill agent-readable
|
|
|
40
40
|
<!-- markdownlint-disable MD033 -->
|
|
41
41
|
<p align="center">
|
|
42
42
|
<strong><code>logging.Logger</code> compared with <code>agent_help()</code> and <code>help()</code></strong><br>
|
|
43
|
-
<img src="docs/agent_help_vs_help.gif" alt="agent_help vs help">
|
|
43
|
+
<img src="https://raw.githubusercontent.com/zydo/agent-readable/main/docs/agent_help_vs_help.gif" alt="agent_help vs help">
|
|
44
44
|
</p>
|
|
45
45
|
<!-- markdownlint-enable MD033 -->
|
|
46
46
|
|
|
@@ -86,6 +86,7 @@ class Sensor:
|
|
|
86
86
|
- [Examples](docs/examples.md): wrapping existing classes, inherited notes, duck typing, plain classes, modules, functions, and methods.
|
|
87
87
|
- [Authoring guide](docs/authoring.md): `__agent_help__`, `__agent_notes__`, class docstring hints, freshness guidance, and API reference.
|
|
88
88
|
- [FAQ](docs/faq.md): common questions about agent skills, docstrings, `AGENTS.md`, third-party libraries, and constrained decoding.
|
|
89
|
+
- [Benchmark](docs/benchmark.md): methodology and harness measuring whether `agent_help()` context reduces hallucinated-API failures.
|
|
89
90
|
|
|
90
91
|
## Other Languages
|
|
91
92
|
|
|
@@ -13,7 +13,7 @@ npx skills add zydo/skills --skill agent-readable
|
|
|
13
13
|
<!-- markdownlint-disable MD033 -->
|
|
14
14
|
<p align="center">
|
|
15
15
|
<strong><code>logging.Logger</code> compared with <code>agent_help()</code> and <code>help()</code></strong><br>
|
|
16
|
-
<img src="docs/agent_help_vs_help.gif" alt="agent_help vs help">
|
|
16
|
+
<img src="https://raw.githubusercontent.com/zydo/agent-readable/main/docs/agent_help_vs_help.gif" alt="agent_help vs help">
|
|
17
17
|
</p>
|
|
18
18
|
<!-- markdownlint-enable MD033 -->
|
|
19
19
|
|
|
@@ -59,6 +59,7 @@ class Sensor:
|
|
|
59
59
|
- [Examples](docs/examples.md): wrapping existing classes, inherited notes, duck typing, plain classes, modules, functions, and methods.
|
|
60
60
|
- [Authoring guide](docs/authoring.md): `__agent_help__`, `__agent_notes__`, class docstring hints, freshness guidance, and API reference.
|
|
61
61
|
- [FAQ](docs/faq.md): common questions about agent skills, docstrings, `AGENTS.md`, third-party libraries, and constrained decoding.
|
|
62
|
+
- [Benchmark](docs/benchmark.md): methodology and harness measuring whether `agent_help()` context reduces hallucinated-API failures.
|
|
62
63
|
|
|
63
64
|
## Other Languages
|
|
64
65
|
|
|
@@ -28,13 +28,13 @@ The two dunders intentionally encode different composition rules:
|
|
|
28
28
|
|
|
29
29
|
| Aspect | `__agent_help__()` | `__agent_notes__()` |
|
|
30
30
|
| --------------- | ------------------------------------------ | ----------------------------------------------------------------------------- |
|
|
31
|
-
| Semantics |
|
|
31
|
+
| Semantics | Replaces the auto-generated base document | Additive: appended after the auto-doc or custom `__agent_help__()` output |
|
|
32
32
|
| Composition | Single class wins, closest in MRO | Accumulated across the MRO; leaf class wins on conflict |
|
|
33
|
-
| When to use |
|
|
34
|
-
| Skipped when | Always called if defined |
|
|
33
|
+
| When to use | Custom control over the base document | Extra do/don't rules on top of any base |
|
|
34
|
+
| Skipped when | Always called if defined | Never skipped when defined; a notes method that raises is skipped |
|
|
35
35
|
| Mixin required? | No | No |
|
|
36
36
|
|
|
37
|
-
When a class defines both `__agent_help__()` and `__agent_notes__()`, the
|
|
37
|
+
When a class defines both `__agent_help__()` and `__agent_notes__()`, the custom help replaces the auto-generated base document and the notes are appended after it — the same additive behavior as the auto-doc path, so no combination of the two dunders silently drops notes. If you need full verbatim control of the entire output, do not define `__agent_notes__()` anywhere in the class hierarchy.
|
|
38
38
|
|
|
39
39
|
## Class docstring hints
|
|
40
40
|
|
|
@@ -58,9 +58,9 @@ This way, even agents that only see the source or call `help()` are reminded to
|
|
|
58
58
|
|
|
59
59
|
Render agent-oriented Markdown for a class, module, function, or method.
|
|
60
60
|
|
|
61
|
-
For classes, the output includes purpose, constructor, public API, default usage rules, and accumulated `__agent_notes__()` sections when present. For modules, it includes purpose, public functions,
|
|
61
|
+
For classes, the output includes purpose, constructor, public API (methods, properties, cached properties, constants, and enum members), default usage rules, and accumulated `__agent_notes__()` sections when present. For modules, it includes purpose, public functions (including C builtins), public classes, and constants. For functions and methods, it includes signature, docstring, and default usage rules.
|
|
62
62
|
|
|
63
|
-
For classes and instances, if `__agent_help__()` is defined through the mixin or duck typing, it is called and its return value is used
|
|
63
|
+
For classes and instances, if `__agent_help__()` is defined through the mixin or duck typing, it is called and its return value is used as the base document. Duck-typed implementations replace the auto-generated base; `__agent_notes__()` sections from the MRO are still appended after it (the mixin default already embeds notes, so its output is used as-is). If `__agent_help__()` raises, `agent_help()` falls back to the auto-generated path, which does include notes. A `__agent_notes__()` that raises is skipped rather than fatal, so one broken notes method cannot take down help for a class and its subclasses.
|
|
64
64
|
|
|
65
65
|
For modules, if the module defines a `__agent_help__` attribute, either callable or string, it is used. Otherwise, auto-generated docs are produced from the module docstring and its public functions and classes. When the module defines `__all__`, that list is the authoritative public API, so re-exported symbols are included. Otherwise public members are discovered by introspection, skipping private names and anything defined outside the module.
|
|
66
66
|
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Benchmark: does `agent_help()` context reduce API hallucination?
|
|
2
|
+
|
|
3
|
+
**Status: not yet run.** This page documents the methodology and the harness
|
|
4
|
+
(`scripts/benchmark.py`). No numbers are published until a real run has been
|
|
5
|
+
executed and recorded here — claims in the README about hallucination and
|
|
6
|
+
token efficiency remain unevidenced until then.
|
|
7
|
+
|
|
8
|
+
## The question
|
|
9
|
+
|
|
10
|
+
When a coding agent writes Python against an API it has not seen (a new
|
|
11
|
+
library, a recent release, a project-private module), does injecting
|
|
12
|
+
`agent_help(target)` output into the prompt reduce failures compared with the
|
|
13
|
+
same prompt without it?
|
|
14
|
+
|
|
15
|
+
The failure modes scored are the two documented in [Why it matters](why.md):
|
|
16
|
+
|
|
17
|
+
- **What exists** — hallucinated attributes and methods, measured as
|
|
18
|
+
`AttributeError` at runtime.
|
|
19
|
+
- **How to use it** — wrong signatures or call shapes, measured as `TypeError`
|
|
20
|
+
at runtime.
|
|
21
|
+
|
|
22
|
+
## Method
|
|
23
|
+
|
|
24
|
+
For each task (one target library plus a natural-language coding prompt) and
|
|
25
|
+
each condition, the harness asks the model for a single runnable script and
|
|
26
|
+
executes it against the installed library:
|
|
27
|
+
|
|
28
|
+
| Condition | Prompt |
|
|
29
|
+
| --- | --- |
|
|
30
|
+
| `baseline` | The task prompt only. |
|
|
31
|
+
| `agent_help` | The task prompt plus `agent_help(target)` output labeled as reference documentation for the installed version. |
|
|
32
|
+
|
|
33
|
+
Every generated script runs in a fresh subprocess against the same
|
|
34
|
+
environment. Outcomes:
|
|
35
|
+
|
|
36
|
+
| Outcome | Meaning |
|
|
37
|
+
| --- | --- |
|
|
38
|
+
| `ok` | Script exits 0. |
|
|
39
|
+
| `attributeerror` | Runtime `AttributeError` — most directly indicates a hallucinated API. |
|
|
40
|
+
| `typeerror` | Runtime `TypeError` — wrong signature or call shape. |
|
|
41
|
+
| `nameerror`, `importerror` | Reference or import failures. |
|
|
42
|
+
| `other_error`, `timeout`, `no_code`, `api_error` | Everything else; reported but not counted as API hallucination. |
|
|
43
|
+
|
|
44
|
+
Headline metric: the `ok` rate per condition, plus the split between
|
|
45
|
+
`attributeerror` and `typeerror`. Results are recorded below with the date,
|
|
46
|
+
model, sample count, and library versions.
|
|
47
|
+
|
|
48
|
+
## Running it
|
|
49
|
+
|
|
50
|
+
The harness calls the Anthropic API (needs `ANTHROPIC_API_KEY` or an
|
|
51
|
+
`ant auth login` profile) and requires the target libraries to be installed.
|
|
52
|
+
It is deliberately not part of CI.
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
uv run --with anthropic --with icalendar --with feedparser \
|
|
56
|
+
python scripts/benchmark.py --samples 5 --out .localonly/benchmark-results.json
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Custom task sets are JSON files of `{module, target, task}` entries passed via
|
|
60
|
+
`--tasks`. Choose targets the model plausibly has weak training coverage of:
|
|
61
|
+
small libraries, recent releases, or project-private code — for well-known
|
|
62
|
+
stable APIs the model may succeed in both conditions and the benchmark
|
|
63
|
+
measures nothing.
|
|
64
|
+
|
|
65
|
+
## Threats to validity
|
|
66
|
+
|
|
67
|
+
- **Target familiarity.** A model that already knows a library reduces the
|
|
68
|
+
gap; a model that misremembers a *changed* API may fail both conditions.
|
|
69
|
+
Unfamiliar targets are the population of interest.
|
|
70
|
+
- **Task wording.** The injected documentation is the only intended
|
|
71
|
+
difference; prompts are otherwise identical.
|
|
72
|
+
- **Execution-only scoring.** A script can exit 0 and still be semantically
|
|
73
|
+
wrong; `ok` is an upper bound on correctness, not a proof of it.
|
|
74
|
+
- **Sample size.** Per-task differences at small sample counts are noise;
|
|
75
|
+
read the aggregate.
|
|
76
|
+
|
|
77
|
+
## Results
|
|
78
|
+
|
|
79
|
+
_None yet._
|
|
@@ -154,11 +154,11 @@ class RateLimiter:
|
|
|
154
154
|
print(agent_help(RateLimiter))
|
|
155
155
|
```
|
|
156
156
|
|
|
157
|
-
Use `__agent_help__()` when you
|
|
157
|
+
Use `__agent_help__()` when the auto-generated base document is not what you want and you prefer to write the base by hand. Use `__agent_notes__()` when the auto-generated API docs are fine and you just want extra rules on top. `__agent_notes__()` sections defined anywhere in the MRO are appended after custom `__agent_help__()` output too, so mixed setups do not lose notes.
|
|
158
158
|
|
|
159
159
|
## Example 4: Any class, no setup required
|
|
160
160
|
|
|
161
|
-
Even without the mixin or duck typing, `agent_help()` generates structured Markdown from introspection: a curated public API list with current signatures, free of inherited dunders and MRO clutter. Full example: [`examples/any_class.py`](../examples/any_class.py).
|
|
161
|
+
Even without the mixin or duck typing, `agent_help()` generates structured Markdown from introspection: a curated public API list with current signatures, free of inherited dunders and MRO clutter. Methods, properties, cached properties, public constants (with their values), and enum members are all listed. Full example: [`examples/any_class.py`](../examples/any_class.py).
|
|
162
162
|
|
|
163
163
|
```python
|
|
164
164
|
import logging
|
|
@@ -23,22 +23,22 @@ The examples below use `sqlite3:Connection` as the target; replace it with any i
|
|
|
23
23
|
With `uvx`:
|
|
24
24
|
|
|
25
25
|
```bash
|
|
26
|
-
uvx --from agent-readable
|
|
26
|
+
uvx --from agent-readable agent-readable sqlite3:Connection
|
|
27
27
|
```
|
|
28
28
|
|
|
29
29
|
With `uv tool run`:
|
|
30
30
|
|
|
31
31
|
```bash
|
|
32
|
-
uv tool run --from agent-readable
|
|
32
|
+
uv tool run --from agent-readable agent-readable sqlite3:Connection
|
|
33
33
|
```
|
|
34
34
|
|
|
35
35
|
With `pipx`:
|
|
36
36
|
|
|
37
37
|
```bash
|
|
38
|
-
pipx run --spec agent-readable
|
|
38
|
+
pipx run --spec agent-readable agent-readable sqlite3:Connection
|
|
39
39
|
```
|
|
40
40
|
|
|
41
|
-
`uvx` is shorthand for `uv tool run
|
|
41
|
+
`uvx` is shorthand for `uv tool run`, and `python -m agent_readable` works identically wherever the package is installed. These commands are useful for standard-library targets and packages available inside the temporary environment. For your own project classes, run from an environment where that project is importable.
|
|
42
42
|
|
|
43
43
|
To let your coding agent automatically call `agent_help()` before using an unfamiliar API, install the companion skill:
|
|
44
44
|
|
|
@@ -123,25 +123,30 @@ print(agent_help(Sensor))
|
|
|
123
123
|
|
|
124
124
|
## CLI
|
|
125
125
|
|
|
126
|
+
Wherever the package is installed, both `agent-readable` and `python -m agent_readable` work:
|
|
127
|
+
|
|
126
128
|
```bash
|
|
127
129
|
# Any stdlib class
|
|
128
|
-
|
|
130
|
+
agent-readable sqlite3:Connection
|
|
129
131
|
|
|
130
132
|
# A class in your own package
|
|
131
|
-
|
|
133
|
+
agent-readable my_package.temperature:CalibratedSensor
|
|
132
134
|
|
|
133
135
|
# The library itself
|
|
134
|
-
|
|
136
|
+
agent-readable agent_readable:AgentReadableMixin
|
|
135
137
|
|
|
136
138
|
# Any module
|
|
137
|
-
|
|
139
|
+
agent-readable pathlib
|
|
138
140
|
|
|
139
141
|
# A function or method
|
|
140
|
-
|
|
141
|
-
|
|
142
|
+
agent-readable json:dumps
|
|
143
|
+
agent-readable pathlib:Path.read_text
|
|
144
|
+
|
|
145
|
+
# Installed version
|
|
146
|
+
agent-readable --version
|
|
142
147
|
```
|
|
143
148
|
|
|
144
|
-
The CLI writes agent-oriented documentation for the target to stdout.
|
|
149
|
+
The CLI writes agent-oriented documentation for the target to stdout. A target that cannot be imported or resolved prints a one-line error to stderr and exits with status 2.
|
|
145
150
|
|
|
146
151
|
## Other Languages
|
|
147
152
|
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "agent-readable"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "A lightweight Python protocol for agent-oriented documentation"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -47,6 +47,9 @@ Repository = "https://github.com/zydo/agent-readable"
|
|
|
47
47
|
Issues = "https://github.com/zydo/agent-readable/issues"
|
|
48
48
|
Changelog = "https://github.com/zydo/agent-readable/blob/main/CHANGELOG.md"
|
|
49
49
|
|
|
50
|
+
[project.scripts]
|
|
51
|
+
agent-readable = "agent_readable.__main__:main"
|
|
52
|
+
|
|
50
53
|
[tool.hatch.build.targets.wheel]
|
|
51
54
|
packages = ["src/agent_readable"]
|
|
52
55
|
|
|
@@ -61,9 +64,17 @@ target-version = "py310"
|
|
|
61
64
|
select = ["E", "F", "I", "UP", "B", "RUF"]
|
|
62
65
|
external = ["S"]
|
|
63
66
|
|
|
67
|
+
[tool.mypy]
|
|
68
|
+
files = ["src", "tests"]
|
|
69
|
+
python_version = "3.10"
|
|
70
|
+
check_untyped_defs = true
|
|
71
|
+
warn_redundant_casts = true
|
|
72
|
+
no_implicit_optional = true
|
|
73
|
+
|
|
64
74
|
[dependency-groups]
|
|
65
75
|
dev = [
|
|
66
76
|
"pytest>=8.0",
|
|
67
77
|
"pytest-cov>=7.1.0",
|
|
68
78
|
"ruff>=0.15.15",
|
|
79
|
+
"mypy>=1.19.0",
|
|
69
80
|
]
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Benchmark: does `agent_help()` context reduce API-hallucination failures?
|
|
2
|
+
|
|
3
|
+
Compares two conditions on identical code-generation tasks against unfamiliar
|
|
4
|
+
libraries:
|
|
5
|
+
|
|
6
|
+
- ``baseline``: a natural-language task prompt only.
|
|
7
|
+
- ``agent_help``: the same prompt with `agent_help(target)` output injected as
|
|
8
|
+
reference documentation.
|
|
9
|
+
|
|
10
|
+
Each generated script is executed against the installed library and scored by
|
|
11
|
+
outcome: clean exit, AttributeError (hallucinated API), TypeError (wrong
|
|
12
|
+
signature/usage), other error, timeout, or no code returned.
|
|
13
|
+
|
|
14
|
+
Not wired into CI: it calls a paid API and needs the target libraries
|
|
15
|
+
installed. See docs/benchmark.md for methodology and current status.
|
|
16
|
+
|
|
17
|
+
Usage:
|
|
18
|
+
|
|
19
|
+
uv run --with anthropic --with <target-libraries> \
|
|
20
|
+
python scripts/benchmark.py [--tasks tasks.json] \
|
|
21
|
+
[--samples 3] [--out results.json]
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import argparse
|
|
27
|
+
import json
|
|
28
|
+
import re
|
|
29
|
+
import subprocess
|
|
30
|
+
import sys
|
|
31
|
+
import tempfile
|
|
32
|
+
from dataclasses import asdict, dataclass
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any
|
|
35
|
+
|
|
36
|
+
import anthropic
|
|
37
|
+
|
|
38
|
+
from agent_readable import agent_help
|
|
39
|
+
|
|
40
|
+
MODEL = "claude-opus-5"
|
|
41
|
+
RUN_TIMEOUT_S = 30
|
|
42
|
+
|
|
43
|
+
BASELINE_PROMPT = """You are writing Python code using the `{module}` library.
|
|
44
|
+
|
|
45
|
+
Task: {task}
|
|
46
|
+
|
|
47
|
+
Reply with exactly one Python code block containing a complete, runnable
|
|
48
|
+
script. No explanations, no example usage outside the block.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
TREATMENT_PROMPT = """You are writing Python code using the `{module}` library.
|
|
52
|
+
|
|
53
|
+
Reference documentation for `{target}`, generated from the installed version:
|
|
54
|
+
|
|
55
|
+
{doc}
|
|
56
|
+
|
|
57
|
+
Task: {task}
|
|
58
|
+
|
|
59
|
+
Reply with exactly one Python code block containing a complete, runnable
|
|
60
|
+
script. No explanations, no example usage outside the block.
|
|
61
|
+
"""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class Task:
|
|
66
|
+
"""One code-generation task against one importable target."""
|
|
67
|
+
|
|
68
|
+
module: str
|
|
69
|
+
target: str
|
|
70
|
+
task: str
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class SampleResult:
|
|
75
|
+
condition: str
|
|
76
|
+
module: str
|
|
77
|
+
sample: int
|
|
78
|
+
outcome: str
|
|
79
|
+
detail: str = ""
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# Tasks deliberately favor small, unfamiliar libraries so the model cannot
|
|
83
|
+
# answer from training data. Add entries via --tasks <file> with the same
|
|
84
|
+
# shape; the listed libraries must be installed in the run environment.
|
|
85
|
+
DEFAULT_TASKS = (
|
|
86
|
+
Task(
|
|
87
|
+
module="icalendar",
|
|
88
|
+
target="icalendar:Calendar",
|
|
89
|
+
task="Parse the ICS string 'BEGIN:VCALENDAR\\r\\nEND:VCALENDAR\\r\\n' "
|
|
90
|
+
"and print the calendar's PRODID property (or 'none' if absent).",
|
|
91
|
+
),
|
|
92
|
+
Task(
|
|
93
|
+
module="feedparser",
|
|
94
|
+
target="feedparser",
|
|
95
|
+
task='Parse the Atom feed string \'<feed xmlns="http://www.w3.org/2005/Atom">'
|
|
96
|
+
"<title>t</title></feed>' and print the feed title.",
|
|
97
|
+
),
|
|
98
|
+
)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def generate(client: anthropic.Anthropic, prompt: str) -> str:
|
|
102
|
+
"""Call the model and return its text, or raise RuntimeError on refusal."""
|
|
103
|
+
response = client.messages.create(
|
|
104
|
+
model=MODEL,
|
|
105
|
+
max_tokens=4096,
|
|
106
|
+
messages=[{"role": "user", "content": prompt}],
|
|
107
|
+
)
|
|
108
|
+
if response.stop_reason == "refusal":
|
|
109
|
+
raise RuntimeError(f"model refused ({response.stop_details})")
|
|
110
|
+
return "".join(block.text for block in response.content if block.type == "text")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def extract_code(response_text: str) -> str | None:
|
|
114
|
+
"""Return the first fenced code block, or None when the reply has none."""
|
|
115
|
+
for match in re.finditer(r"```(?:python)?\s*\n(.*?)```", response_text, re.DOTALL):
|
|
116
|
+
return match.group(1)
|
|
117
|
+
return None
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def classify(code: str) -> SampleResult:
|
|
121
|
+
"""Execute one generated script and classify the outcome."""
|
|
122
|
+
with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f: # noqa: SIM115 - executed before deletion below
|
|
123
|
+
f.write(code)
|
|
124
|
+
path = f.name
|
|
125
|
+
try:
|
|
126
|
+
proc = subprocess.run(
|
|
127
|
+
[sys.executable, path],
|
|
128
|
+
capture_output=True,
|
|
129
|
+
text=True,
|
|
130
|
+
timeout=RUN_TIMEOUT_S,
|
|
131
|
+
)
|
|
132
|
+
except subprocess.TimeoutExpired:
|
|
133
|
+
return SampleResult("", "", 0, "timeout")
|
|
134
|
+
finally:
|
|
135
|
+
Path(path).unlink(missing_ok=True)
|
|
136
|
+
|
|
137
|
+
if proc.returncode == 0:
|
|
138
|
+
return SampleResult("", "", 0, "ok")
|
|
139
|
+
stderr_tail = proc.stderr.strip().splitlines()[-1] if proc.stderr else ""
|
|
140
|
+
for exc in ("AttributeError", "TypeError", "ImportError", "NameError"):
|
|
141
|
+
if re.search(rf"\b{exc}\b", stderr_tail):
|
|
142
|
+
return SampleResult("", "", 0, exc.lower(), stderr_tail[:200])
|
|
143
|
+
return SampleResult("", "", 0, "other_error", stderr_tail[:200])
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def run_task(
|
|
147
|
+
client: anthropic.Anthropic,
|
|
148
|
+
task: Task,
|
|
149
|
+
samples: int,
|
|
150
|
+
doc: str,
|
|
151
|
+
) -> list[SampleResult]:
|
|
152
|
+
"""Run one task under both conditions for `samples` repetitions each."""
|
|
153
|
+
results: list[SampleResult] = []
|
|
154
|
+
prompts = {
|
|
155
|
+
"baseline": BASELINE_PROMPT.format(module=task.module, task=task.task),
|
|
156
|
+
"agent_help": TREATMENT_PROMPT.format(
|
|
157
|
+
module=task.module, doc=doc, target=task.target, task=task.task
|
|
158
|
+
),
|
|
159
|
+
}
|
|
160
|
+
for condition, prompt in prompts.items():
|
|
161
|
+
for i in range(samples):
|
|
162
|
+
result = SampleResult(condition, task.module, i, "no_code")
|
|
163
|
+
try:
|
|
164
|
+
code = extract_code(generate(client, prompt))
|
|
165
|
+
if code:
|
|
166
|
+
result = classify(code)
|
|
167
|
+
except RuntimeError as exc:
|
|
168
|
+
result = SampleResult(condition, task.module, i, "api_error", str(exc))
|
|
169
|
+
result.condition = condition
|
|
170
|
+
result.module = task.module
|
|
171
|
+
result.sample = i
|
|
172
|
+
results.append(result)
|
|
173
|
+
print(
|
|
174
|
+
f" {condition:>10} #{i} {task.module}: {result.outcome}"
|
|
175
|
+
f"{' — ' + result.detail if result.detail else ''}",
|
|
176
|
+
file=sys.stderr,
|
|
177
|
+
)
|
|
178
|
+
return results
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def summarize(results: list[SampleResult]) -> str:
|
|
182
|
+
"""Render the per-condition outcome table."""
|
|
183
|
+
by_condition: dict[str, dict[str, int]] = {}
|
|
184
|
+
for r in results:
|
|
185
|
+
counts = by_condition.setdefault(r.condition, {})
|
|
186
|
+
counts[r.outcome] = counts.get(r.outcome, 0) + 1
|
|
187
|
+
|
|
188
|
+
outcomes = sorted({r.outcome for r in results})
|
|
189
|
+
conditions = list(by_condition)
|
|
190
|
+
header = f"{'outcome':<15}" + "".join(f"{c:>12}" for c in conditions)
|
|
191
|
+
rows = [
|
|
192
|
+
f"{outcome:<15}"
|
|
193
|
+
+ "".join(f"{by_condition[c].get(outcome, 0):>12}" for c in conditions)
|
|
194
|
+
for outcome in outcomes
|
|
195
|
+
]
|
|
196
|
+
total = len(results) // max(len(conditions), 1)
|
|
197
|
+
footer = f"{'ok rate':<15}" + "".join(
|
|
198
|
+
f"{by_condition[c].get('ok', 0) / total:>12.0%}" if total else " 0%"
|
|
199
|
+
for c in conditions
|
|
200
|
+
)
|
|
201
|
+
return "\n".join([header, *rows, footer])
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def load_tasks(path: str | None) -> tuple[Task, ...]:
|
|
205
|
+
if path is None:
|
|
206
|
+
return DEFAULT_TASKS
|
|
207
|
+
raw: Any = json.loads(Path(path).read_text())
|
|
208
|
+
return tuple(Task(**entry) for entry in raw)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def main() -> None:
|
|
212
|
+
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
213
|
+
parser.add_argument("--tasks", help="JSON file of {module, target, task} entries")
|
|
214
|
+
parser.add_argument(
|
|
215
|
+
"--samples", type=int, default=3, help="repetitions per condition"
|
|
216
|
+
)
|
|
217
|
+
parser.add_argument("--out", help="write full results JSON here")
|
|
218
|
+
args = parser.parse_args()
|
|
219
|
+
|
|
220
|
+
client = anthropic.Anthropic()
|
|
221
|
+
tasks = load_tasks(args.tasks)
|
|
222
|
+
all_results: list[SampleResult] = []
|
|
223
|
+
for task in tasks:
|
|
224
|
+
print(f"== {task.module} ({task.target}) ==", file=sys.stderr)
|
|
225
|
+
doc = agent_help(_import_target(task.target))
|
|
226
|
+
all_results.extend(run_task(client, task, args.samples, doc))
|
|
227
|
+
|
|
228
|
+
print(summarize(all_results))
|
|
229
|
+
if args.out:
|
|
230
|
+
payload = json.dumps([asdict(r) for r in all_results], indent=2)
|
|
231
|
+
Path(args.out).write_text(payload)
|
|
232
|
+
print(f"\nFull results written to {args.out}", file=sys.stderr)
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _import_target(dotted: str) -> Any:
|
|
236
|
+
"""Resolve a CLI-style target path to a live object."""
|
|
237
|
+
from agent_readable.__main__ import _resolve
|
|
238
|
+
|
|
239
|
+
return _resolve(dotted)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
if __name__ == "__main__":
|
|
243
|
+
main()
|