agent-readable 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. {agent_readable-0.2.1 → agent_readable-0.3.0}/.github/workflows/ci.yml +3 -0
  2. {agent_readable-0.2.1 → agent_readable-0.3.0}/.gitignore +3 -0
  3. {agent_readable-0.2.1 → agent_readable-0.3.0}/CHANGELOG.md +50 -1
  4. {agent_readable-0.2.1 → agent_readable-0.3.0}/PKG-INFO +4 -3
  5. {agent_readable-0.2.1 → agent_readable-0.3.0}/README.md +2 -1
  6. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/authoring.md +6 -6
  7. agent_readable-0.3.0/docs/benchmark.md +79 -0
  8. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/examples.md +2 -2
  9. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/getting-started.md +16 -11
  10. {agent_readable-0.2.1 → agent_readable-0.3.0}/pyproject.toml +12 -1
  11. agent_readable-0.3.0/scripts/benchmark.py +243 -0
  12. agent_readable-0.3.0/src/agent_readable/__main__.py +77 -0
  13. {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_model.py +130 -17
  14. {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_protocol.py +37 -47
  15. {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/_render.py +12 -2
  16. {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/test_cli.py +31 -23
  17. {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/test_protocol.py +251 -27
  18. agent_readable-0.2.1/src/agent_readable/__main__.py +0 -59
  19. {agent_readable-0.2.1 → agent_readable-0.3.0}/.github/workflows/publish.yml +0 -0
  20. {agent_readable-0.2.1 → agent_readable-0.3.0}/LICENSE +0 -0
  21. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/agent_help_vs_help.gif +0 -0
  22. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/faq.md +0 -0
  23. {agent_readable-0.2.1 → agent_readable-0.3.0}/docs/why.md +0 -0
  24. {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/any_class.py +0 -0
  25. {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/duck_type.py +0 -0
  26. {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/modules_and_functions.py +0 -0
  27. {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/sqlite_connection.py +0 -0
  28. {agent_readable-0.2.1 → agent_readable-0.3.0}/examples/temperature.py +0 -0
  29. {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/__init__.py +0 -0
  30. {agent_readable-0.2.1 → agent_readable-0.3.0}/src/agent_readable/py.typed +0 -0
  31. {agent_readable-0.2.1 → agent_readable-0.3.0}/tests/__init__.py +0 -0
@@ -24,6 +24,9 @@ jobs:
24
24
  - name: Ruff format check
25
25
  run: uv run ruff format --check
26
26
 
27
+ - name: Mypy
28
+ run: uv run mypy
29
+
27
30
  test:
28
31
  runs-on: ubuntu-latest
29
32
  strategy:
@@ -43,3 +43,6 @@ hatch.toml
43
43
 
44
44
  # No external dependencies for this project, and uv.lock is only needed for local dev.
45
45
  uv.lock
46
+
47
+ # Local scratch files (audits, drafts, planning docs)
48
+ .localonly/
@@ -7,6 +7,54 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.3.0] - 2026-08-23
11
+
12
+ ### Added
13
+
14
+ - Console script `agent-readable`, so one-off use is
15
+ `uvx agent-readable sqlite3:Connection` instead of
16
+ `uvx --from agent-readable python -m agent_readable sqlite3:Connection`.
17
+ `python -m agent_readable` keeps working.
18
+ - `--version` flag for the CLI.
19
+ - Public class attributes and module constants are now listed in the Public API
20
+ section with their current value: `repr` for exact primitive types (never
21
+ executing a custom `__repr__`), the type name otherwise.
22
+ - Enum members are listed (e.g. ``- `RED` member: 1``), and the misleading
23
+ `EnumMeta.__call__` constructor signature is no longer shown for enum
24
+ classes.
25
+ - Module docs now include C builtins (`inspect.isroutine` instead of
26
+ `inspect.isfunction`), so `agent_help(math)` lists `sin` alongside
27
+ pure-Python functions, and module-level constants such as `math.pi` are no
28
+ longer dropped by the origin heuristic.
29
+ - A constructor that introspection renders as the placeholder `(*args,
30
+ **kwargs)` — e.g. masked by a metaclass `__call__` — now falls back to the
31
+ class's `__init__`/`__new__` signature.
32
+ - Mypy type-checking job in CI.
33
+ - Benchmark harness (`scripts/benchmark.py`) and methodology
34
+ (`docs/benchmark.md`) for measuring whether `agent_help()` context reduces
35
+ hallucinated-API failures versus a baseline prompt. Status: not yet run;
36
+ no numbers are claimed until a real run is recorded.
37
+
38
+ ### Changed
39
+
40
+ - **Breaking:** `__agent_notes__()` sections are now always appended, including
41
+ after a custom `__agent_help__()`'s output. Previously the notes were
42
+ silently dropped when a custom `__agent_help__` was defined and a
43
+ `UserWarning` was emitted; defining both is now a supported combination and
44
+ the warning is gone. For full verbatim control of the entire output, do not
45
+ define `__agent_notes__()` anywhere in the MRO.
46
+ - The CLI prints a one-line `error: ...` message to stderr and exits with
47
+ status 2 for targets that cannot be imported or resolved, instead of raising
48
+ a traceback.
49
+
50
+ ### Fixed
51
+
52
+ - A raising `__agent_notes__()` no longer breaks `agent_help()` for the class
53
+ and all of its subclasses; the broken notes are skipped, mirroring the
54
+ existing `__agent_help__()` fallback.
55
+ - `functools.cached_property` members are no longer silently dropped from the
56
+ Public API; they render as properties with the wrapped function's docstring.
57
+
10
58
  ## [0.2.1] - 2026-07-04
11
59
 
12
60
  ### Documentation
@@ -86,7 +134,8 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
86
134
  `AgentReadableMixin`, `__agent_notes__` accumulation across the MRO, module
87
135
  support, and the `python -m agent_readable` CLI.
88
136
 
89
- [Unreleased]: https://github.com/zydo/agent-readable/compare/v0.2.1...HEAD
137
+ [Unreleased]: https://github.com/zydo/agent-readable/compare/v0.3.0...HEAD
138
+ [0.3.0]: https://github.com/zydo/agent-readable/compare/v0.2.1...v0.3.0
90
139
  [0.2.1]: https://github.com/zydo/agent-readable/compare/v0.2.0...v0.2.1
91
140
  [0.1.2]: https://github.com/zydo/agent-readable/compare/v0.1.1...v0.1.2
92
141
  [0.1.1]: https://github.com/zydo/agent-readable/compare/v0.1.0...v0.1.1
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: agent-readable
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: A lightweight Python protocol for agent-oriented documentation
5
5
  Project-URL: Repository, https://github.com/zydo/agent-readable
6
6
  Project-URL: Issues, https://github.com/zydo/agent-readable/issues
@@ -40,7 +40,7 @@ npx skills add zydo/skills --skill agent-readable
40
40
  <!-- markdownlint-disable MD033 -->
41
41
  <p align="center">
42
42
  <strong><code>logging.Logger</code> compared with <code>agent_help()</code> and <code>help()</code></strong><br>
43
- <img src="docs/agent_help_vs_help.gif" alt="agent_help vs help">
43
+ <img src="https://raw.githubusercontent.com/zydo/agent-readable/main/docs/agent_help_vs_help.gif" alt="agent_help vs help">
44
44
  </p>
45
45
  <!-- markdownlint-enable MD033 -->
46
46
 
@@ -86,6 +86,7 @@ class Sensor:
86
86
  - [Examples](docs/examples.md): wrapping existing classes, inherited notes, duck typing, plain classes, modules, functions, and methods.
87
87
  - [Authoring guide](docs/authoring.md): `__agent_help__`, `__agent_notes__`, class docstring hints, freshness guidance, and API reference.
88
88
  - [FAQ](docs/faq.md): common questions about agent skills, docstrings, `AGENTS.md`, third-party libraries, and constrained decoding.
89
+ - [Benchmark](docs/benchmark.md): methodology and harness measuring whether `agent_help()` context reduces hallucinated-API failures.
89
90
 
90
91
  ## Other Languages
91
92
 
@@ -13,7 +13,7 @@ npx skills add zydo/skills --skill agent-readable
13
13
  <!-- markdownlint-disable MD033 -->
14
14
  <p align="center">
15
15
  <strong><code>logging.Logger</code> compared with <code>agent_help()</code> and <code>help()</code></strong><br>
16
- <img src="docs/agent_help_vs_help.gif" alt="agent_help vs help">
16
+ <img src="https://raw.githubusercontent.com/zydo/agent-readable/main/docs/agent_help_vs_help.gif" alt="agent_help vs help">
17
17
  </p>
18
18
  <!-- markdownlint-enable MD033 -->
19
19
 
@@ -59,6 +59,7 @@ class Sensor:
59
59
  - [Examples](docs/examples.md): wrapping existing classes, inherited notes, duck typing, plain classes, modules, functions, and methods.
60
60
  - [Authoring guide](docs/authoring.md): `__agent_help__`, `__agent_notes__`, class docstring hints, freshness guidance, and API reference.
61
61
  - [FAQ](docs/faq.md): common questions about agent skills, docstrings, `AGENTS.md`, third-party libraries, and constrained decoding.
62
+ - [Benchmark](docs/benchmark.md): methodology and harness measuring whether `agent_help()` context reduces hallucinated-API failures.
62
63
 
63
64
  ## Other Languages
64
65
 
@@ -28,13 +28,13 @@ The two dunders intentionally encode different composition rules:
28
28
 
29
29
  | Aspect | `__agent_help__()` | `__agent_notes__()` |
30
30
  | --------------- | ------------------------------------------ | ----------------------------------------------------------------------------- |
31
- | Semantics | Replacement: returned string is the output | Additive: appended to auto-generated docs |
31
+ | Semantics | Replaces the auto-generated base document | Additive: appended after the auto-doc or custom `__agent_help__()` output |
32
32
  | Composition | Single class wins, closest in MRO | Accumulated across the MRO; leaf class wins on conflict |
33
- | When to use | Total control over rendered text | Auto-doc plus extra do/don't rules |
34
- | Skipped when | Always called if defined | Silently dropped, with `UserWarning`, when custom `__agent_help__` is present |
33
+ | When to use | Custom control over the base document | Extra do/don't rules on top of any base |
34
+ | Skipped when | Always called if defined | Never skipped when defined; a notes method that raises is skipped |
35
35
  | Mixin required? | No | No |
36
36
 
37
- When a class defines both `__agent_help__()` and `__agent_notes__()`, the notes are silently dropped because `__agent_help__()` owns the output and the auto-doc plus notes path never runs. A `UserWarning` is emitted, but warnings are easy to miss in agent shells, CI logs, and notebooks. Treat "both defined" as a review error. Fix it by folding the notes into `__agent_help__()`, or by dropping custom `__agent_help__()` and letting the auto-doc plus notes path run.
37
+ When a class defines both `__agent_help__()` and `__agent_notes__()`, the custom help replaces the auto-generated base document and the notes are appended after it — the same additive behavior as the auto-doc path, so no combination of the two dunders silently drops notes. If you need full verbatim control of the entire output, do not define `__agent_notes__()` anywhere in the class hierarchy.
38
38
 
39
39
  ## Class docstring hints
40
40
 
@@ -58,9 +58,9 @@ This way, even agents that only see the source or call `help()` are reminded to
58
58
 
59
59
  Render agent-oriented Markdown for a class, module, function, or method.
60
60
 
61
- For classes, the output includes purpose, constructor, public API, default usage rules, and accumulated `__agent_notes__()` sections when present. For modules, it includes purpose, public functions, and public classes. For functions and methods, it includes signature, docstring, and default usage rules.
61
+ For classes, the output includes purpose, constructor, public API (methods, properties, cached properties, constants, and enum members), default usage rules, and accumulated `__agent_notes__()` sections when present. For modules, it includes purpose, public functions (including C builtins), public classes, and constants. For functions and methods, it includes signature, docstring, and default usage rules.
62
62
 
63
- For classes and instances, if `__agent_help__()` is defined through the mixin or duck typing, it is called and its return value is used verbatim. Duck-typed implementations are responsible for their own formatting, and notes are not auto-appended. If such a class also defines `__agent_notes__()`, a `UserWarning` is emitted because those notes are silently dropped. Fold them into `__agent_help__()`, or drop the custom `__agent_help__()` to use the auto-doc path. If `__agent_help__()` raises, `agent_help()` falls back to the auto-generated path, which does include notes.
63
+ For classes and instances, if `__agent_help__()` is defined through the mixin or duck typing, it is called and its return value is used as the base document. Duck-typed implementations replace the auto-generated base; `__agent_notes__()` sections from the MRO are still appended after it (the mixin default already embeds notes, so its output is used as-is). If `__agent_help__()` raises, `agent_help()` falls back to the auto-generated path, which does include notes. A `__agent_notes__()` that raises is skipped rather than fatal, so one broken notes method cannot take down help for a class and its subclasses.
64
64
 
65
65
  For modules, if the module defines a `__agent_help__` attribute, either callable or string, it is used. Otherwise, auto-generated docs are produced from the module docstring and its public functions and classes. When the module defines `__all__`, that list is the authoritative public API, so re-exported symbols are included. Otherwise public members are discovered by introspection, skipping private names and anything defined outside the module.
66
66
 
@@ -0,0 +1,79 @@
1
+ # Benchmark: does `agent_help()` context reduce API hallucination?
2
+
3
+ **Status: not yet run.** This page documents the methodology and the harness
4
+ (`scripts/benchmark.py`). No numbers are published until a real run has been
5
+ executed and recorded here — claims in the README about hallucination and
6
+ token efficiency remain unevidenced until then.
7
+
8
+ ## The question
9
+
10
+ When a coding agent writes Python against an API it has not seen (a new
11
+ library, a recent release, a project-private module), does injecting
12
+ `agent_help(target)` output into the prompt reduce failures compared with the
13
+ same prompt without it?
14
+
15
+ The failure modes scored are the two documented in [Why it matters](why.md):
16
+
17
+ - **What exists** — hallucinated attributes and methods, measured as
18
+ `AttributeError` at runtime.
19
+ - **How to use it** — wrong signatures or call shapes, measured as `TypeError`
20
+ at runtime.
21
+
22
+ ## Method
23
+
24
+ For each task (one target library plus a natural-language coding prompt) and
25
+ each condition, the harness asks the model for a single runnable script and
26
+ executes it against the installed library:
27
+
28
+ | Condition | Prompt |
29
+ | --- | --- |
30
+ | `baseline` | The task prompt only. |
31
+ | `agent_help` | The task prompt plus `agent_help(target)` output labeled as reference documentation for the installed version. |
32
+
33
+ Every generated script runs in a fresh subprocess against the same
34
+ environment. Outcomes:
35
+
36
+ | Outcome | Meaning |
37
+ | --- | --- |
38
+ | `ok` | Script exits 0. |
39
+ | `attributeerror` | Runtime `AttributeError` — most directly indicates a hallucinated API. |
40
+ | `typeerror` | Runtime `TypeError` — wrong signature or call shape. |
41
+ | `nameerror`, `importerror` | Reference or import failures. |
42
+ | `other_error`, `timeout`, `no_code`, `api_error` | Everything else; reported but not counted as API hallucination. |
43
+
44
+ Headline metric: the `ok` rate per condition, plus the split between
45
+ `attributeerror` and `typeerror`. Results are recorded below with the date,
46
+ model, sample count, and library versions.
47
+
48
+ ## Running it
49
+
50
+ The harness calls the Anthropic API (needs `ANTHROPIC_API_KEY` or an
51
+ `ant auth login` profile) and requires the target libraries to be installed.
52
+ It is deliberately not part of CI.
53
+
54
+ ```bash
55
+ uv run --with anthropic --with icalendar --with feedparser \
56
+ python scripts/benchmark.py --samples 5 --out .localonly/benchmark-results.json
57
+ ```
58
+
59
+ Custom task sets are JSON files of `{module, target, task}` entries passed via
60
+ `--tasks`. Choose targets the model plausibly has weak training coverage of:
61
+ small libraries, recent releases, or project-private code — for well-known
62
+ stable APIs the model may succeed in both conditions and the benchmark
63
+ measures nothing.
64
+
65
+ ## Threats to validity
66
+
67
+ - **Target familiarity.** A model that already knows a library reduces the
68
+ gap; a model that misremembers a *changed* API may fail both conditions.
69
+ Unfamiliar targets are the population of interest.
70
+ - **Task wording.** The injected documentation is the only intended
71
+ difference; prompts are otherwise identical.
72
+ - **Execution-only scoring.** A script can exit 0 and still be semantically
73
+ wrong; `ok` is an upper bound on correctness, not a proof of it.
74
+ - **Sample size.** Per-task differences at small sample counts are noise;
75
+ read the aggregate.
76
+
77
+ ## Results
78
+
79
+ _None yet._
@@ -154,11 +154,11 @@ class RateLimiter:
154
154
  print(agent_help(RateLimiter))
155
155
  ```
156
156
 
157
- Use `__agent_help__()` when you need total control over the rendered output. Use `__agent_notes__()` when you want auto-generated API docs plus your extra rules.
157
+ Use `__agent_help__()` when the auto-generated base document is not what you want and you prefer to write the base by hand. Use `__agent_notes__()` when the auto-generated API docs are fine and you just want extra rules on top. `__agent_notes__()` sections defined anywhere in the MRO are appended after custom `__agent_help__()` output too, so mixed setups do not lose notes.
158
158
 
159
159
  ## Example 4: Any class, no setup required
160
160
 
161
- Even without the mixin or duck typing, `agent_help()` generates structured Markdown from introspection: a curated public API list with current signatures, free of inherited dunders and MRO clutter. Full example: [`examples/any_class.py`](../examples/any_class.py).
161
+ Even without the mixin or duck typing, `agent_help()` generates structured Markdown from introspection: a curated public API list with current signatures, free of inherited dunders and MRO clutter. Methods, properties, cached properties, public constants (with their values), and enum members are all listed. Full example: [`examples/any_class.py`](../examples/any_class.py).
162
162
 
163
163
  ```python
164
164
  import logging
@@ -23,22 +23,22 @@ The examples below use `sqlite3:Connection` as the target; replace it with any i
23
23
  With `uvx`:
24
24
 
25
25
  ```bash
26
- uvx --from agent-readable python -m agent_readable sqlite3:Connection
26
+ uvx --from agent-readable agent-readable sqlite3:Connection
27
27
  ```
28
28
 
29
29
  With `uv tool run`:
30
30
 
31
31
  ```bash
32
- uv tool run --from agent-readable python -m agent_readable sqlite3:Connection
32
+ uv tool run --from agent-readable agent-readable sqlite3:Connection
33
33
  ```
34
34
 
35
35
  With `pipx`:
36
36
 
37
37
  ```bash
38
- pipx run --spec agent-readable python -m agent_readable sqlite3:Connection
38
+ pipx run --spec agent-readable agent-readable sqlite3:Connection
39
39
  ```
40
40
 
41
- `uvx` is shorthand for `uv tool run`. These commands are useful for standard-library targets and packages available inside the temporary environment. For your own project classes, run from an environment where that project is importable.
41
+ `uvx` is shorthand for `uv tool run`, and `python -m agent_readable` works identically wherever the package is installed. These commands are useful for standard-library targets and packages available inside the temporary environment. For your own project classes, run from an environment where that project is importable.
42
42
 
43
43
  To let your coding agent automatically call `agent_help()` before using an unfamiliar API, install the companion skill:
44
44
 
@@ -123,25 +123,30 @@ print(agent_help(Sensor))
123
123
 
124
124
  ## CLI
125
125
 
126
+ Wherever the package is installed, both `agent-readable` and `python -m agent_readable` work:
127
+
126
128
  ```bash
127
129
  # Any stdlib class
128
- python -m agent_readable sqlite3:Connection
130
+ agent-readable sqlite3:Connection
129
131
 
130
132
  # A class in your own package
131
- python -m agent_readable my_package.temperature:CalibratedSensor
133
+ agent-readable my_package.temperature:CalibratedSensor
132
134
 
133
135
  # The library itself
134
- python -m agent_readable agent_readable:AgentReadableMixin
136
+ agent-readable agent_readable:AgentReadableMixin
135
137
 
136
138
  # Any module
137
- python -m agent_readable pathlib
139
+ agent-readable pathlib
138
140
 
139
141
  # A function or method
140
- python -m agent_readable json:dumps
141
- python -m agent_readable pathlib:Path.read_text
142
+ agent-readable json:dumps
143
+ agent-readable pathlib:Path.read_text
144
+
145
+ # Installed version
146
+ agent-readable --version
142
147
  ```
143
148
 
144
- The CLI writes agent-oriented documentation for the target to stdout.
149
+ The CLI writes agent-oriented documentation for the target to stdout. A target that cannot be imported or resolved prints a one-line error to stderr and exits with status 2.
145
150
 
146
151
  ## Other Languages
147
152
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "agent-readable"
7
- version = "0.2.1"
7
+ version = "0.3.0"
8
8
  description = "A lightweight Python protocol for agent-oriented documentation"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -47,6 +47,9 @@ Repository = "https://github.com/zydo/agent-readable"
47
47
  Issues = "https://github.com/zydo/agent-readable/issues"
48
48
  Changelog = "https://github.com/zydo/agent-readable/blob/main/CHANGELOG.md"
49
49
 
50
+ [project.scripts]
51
+ agent-readable = "agent_readable.__main__:main"
52
+
50
53
  [tool.hatch.build.targets.wheel]
51
54
  packages = ["src/agent_readable"]
52
55
 
@@ -61,9 +64,17 @@ target-version = "py310"
61
64
  select = ["E", "F", "I", "UP", "B", "RUF"]
62
65
  external = ["S"]
63
66
 
67
+ [tool.mypy]
68
+ files = ["src", "tests"]
69
+ python_version = "3.10"
70
+ check_untyped_defs = true
71
+ warn_redundant_casts = true
72
+ no_implicit_optional = true
73
+
64
74
  [dependency-groups]
65
75
  dev = [
66
76
  "pytest>=8.0",
67
77
  "pytest-cov>=7.1.0",
68
78
  "ruff>=0.15.15",
79
+ "mypy>=1.19.0",
69
80
  ]
@@ -0,0 +1,243 @@
1
+ """Benchmark: does `agent_help()` context reduce API-hallucination failures?
2
+
3
+ Compares two conditions on identical code-generation tasks against unfamiliar
4
+ libraries:
5
+
6
+ - ``baseline``: a natural-language task prompt only.
7
+ - ``agent_help``: the same prompt with `agent_help(target)` output injected as
8
+ reference documentation.
9
+
10
+ Each generated script is executed against the installed library and scored by
11
+ outcome: clean exit, AttributeError (hallucinated API), TypeError (wrong
12
+ signature/usage), other error, timeout, or no code returned.
13
+
14
+ Not wired into CI: it calls a paid API and needs the target libraries
15
+ installed. See docs/benchmark.md for methodology and current status.
16
+
17
+ Usage:
18
+
19
+ uv run --with anthropic --with <target-libraries> \
20
+ python scripts/benchmark.py [--tasks tasks.json] \
21
+ [--samples 3] [--out results.json]
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import argparse
27
+ import json
28
+ import re
29
+ import subprocess
30
+ import sys
31
+ import tempfile
32
+ from dataclasses import asdict, dataclass
33
+ from pathlib import Path
34
+ from typing import Any
35
+
36
+ import anthropic
37
+
38
+ from agent_readable import agent_help
39
+
40
+ MODEL = "claude-opus-5"
41
+ RUN_TIMEOUT_S = 30
42
+
43
+ BASELINE_PROMPT = """You are writing Python code using the `{module}` library.
44
+
45
+ Task: {task}
46
+
47
+ Reply with exactly one Python code block containing a complete, runnable
48
+ script. No explanations, no example usage outside the block.
49
+ """
50
+
51
+ TREATMENT_PROMPT = """You are writing Python code using the `{module}` library.
52
+
53
+ Reference documentation for `{target}`, generated from the installed version:
54
+
55
+ {doc}
56
+
57
+ Task: {task}
58
+
59
+ Reply with exactly one Python code block containing a complete, runnable
60
+ script. No explanations, no example usage outside the block.
61
+ """
62
+
63
+
64
+ @dataclass(frozen=True)
65
+ class Task:
66
+ """One code-generation task against one importable target."""
67
+
68
+ module: str
69
+ target: str
70
+ task: str
71
+
72
+
73
+ @dataclass
74
+ class SampleResult:
75
+ condition: str
76
+ module: str
77
+ sample: int
78
+ outcome: str
79
+ detail: str = ""
80
+
81
+
82
+ # Tasks deliberately favor small, unfamiliar libraries so the model cannot
83
+ # answer from training data. Add entries via --tasks <file> with the same
84
+ # shape; the listed libraries must be installed in the run environment.
85
+ DEFAULT_TASKS = (
86
+ Task(
87
+ module="icalendar",
88
+ target="icalendar:Calendar",
89
+ task="Parse the ICS string 'BEGIN:VCALENDAR\\r\\nEND:VCALENDAR\\r\\n' "
90
+ "and print the calendar's PRODID property (or 'none' if absent).",
91
+ ),
92
+ Task(
93
+ module="feedparser",
94
+ target="feedparser",
95
+ task='Parse the Atom feed string \'<feed xmlns="http://www.w3.org/2005/Atom">'
96
+ "<title>t</title></feed>' and print the feed title.",
97
+ ),
98
+ )
99
+
100
+
101
+ def generate(client: anthropic.Anthropic, prompt: str) -> str:
102
+ """Call the model and return its text, or raise RuntimeError on refusal."""
103
+ response = client.messages.create(
104
+ model=MODEL,
105
+ max_tokens=4096,
106
+ messages=[{"role": "user", "content": prompt}],
107
+ )
108
+ if response.stop_reason == "refusal":
109
+ raise RuntimeError(f"model refused ({response.stop_details})")
110
+ return "".join(block.text for block in response.content if block.type == "text")
111
+
112
+
113
+ def extract_code(response_text: str) -> str | None:
114
+ """Return the first fenced code block, or None when the reply has none."""
115
+ for match in re.finditer(r"```(?:python)?\s*\n(.*?)```", response_text, re.DOTALL):
116
+ return match.group(1)
117
+ return None
118
+
119
+
120
+ def classify(code: str) -> SampleResult:
121
+ """Execute one generated script and classify the outcome."""
122
+ with tempfile.NamedTemporaryFile("w", suffix=".py", delete=False) as f: # noqa: SIM115 - executed before deletion below
123
+ f.write(code)
124
+ path = f.name
125
+ try:
126
+ proc = subprocess.run(
127
+ [sys.executable, path],
128
+ capture_output=True,
129
+ text=True,
130
+ timeout=RUN_TIMEOUT_S,
131
+ )
132
+ except subprocess.TimeoutExpired:
133
+ return SampleResult("", "", 0, "timeout")
134
+ finally:
135
+ Path(path).unlink(missing_ok=True)
136
+
137
+ if proc.returncode == 0:
138
+ return SampleResult("", "", 0, "ok")
139
+ stderr_tail = proc.stderr.strip().splitlines()[-1] if proc.stderr else ""
140
+ for exc in ("AttributeError", "TypeError", "ImportError", "NameError"):
141
+ if re.search(rf"\b{exc}\b", stderr_tail):
142
+ return SampleResult("", "", 0, exc.lower(), stderr_tail[:200])
143
+ return SampleResult("", "", 0, "other_error", stderr_tail[:200])
144
+
145
+
146
+ def run_task(
147
+ client: anthropic.Anthropic,
148
+ task: Task,
149
+ samples: int,
150
+ doc: str,
151
+ ) -> list[SampleResult]:
152
+ """Run one task under both conditions for `samples` repetitions each."""
153
+ results: list[SampleResult] = []
154
+ prompts = {
155
+ "baseline": BASELINE_PROMPT.format(module=task.module, task=task.task),
156
+ "agent_help": TREATMENT_PROMPT.format(
157
+ module=task.module, doc=doc, target=task.target, task=task.task
158
+ ),
159
+ }
160
+ for condition, prompt in prompts.items():
161
+ for i in range(samples):
162
+ result = SampleResult(condition, task.module, i, "no_code")
163
+ try:
164
+ code = extract_code(generate(client, prompt))
165
+ if code:
166
+ result = classify(code)
167
+ except RuntimeError as exc:
168
+ result = SampleResult(condition, task.module, i, "api_error", str(exc))
169
+ result.condition = condition
170
+ result.module = task.module
171
+ result.sample = i
172
+ results.append(result)
173
+ print(
174
+ f" {condition:>10} #{i} {task.module}: {result.outcome}"
175
+ f"{' — ' + result.detail if result.detail else ''}",
176
+ file=sys.stderr,
177
+ )
178
+ return results
179
+
180
+
181
+ def summarize(results: list[SampleResult]) -> str:
182
+ """Render the per-condition outcome table."""
183
+ by_condition: dict[str, dict[str, int]] = {}
184
+ for r in results:
185
+ counts = by_condition.setdefault(r.condition, {})
186
+ counts[r.outcome] = counts.get(r.outcome, 0) + 1
187
+
188
+ outcomes = sorted({r.outcome for r in results})
189
+ conditions = list(by_condition)
190
+ header = f"{'outcome':<15}" + "".join(f"{c:>12}" for c in conditions)
191
+ rows = [
192
+ f"{outcome:<15}"
193
+ + "".join(f"{by_condition[c].get(outcome, 0):>12}" for c in conditions)
194
+ for outcome in outcomes
195
+ ]
196
+ total = len(results) // max(len(conditions), 1)
197
+ footer = f"{'ok rate':<15}" + "".join(
198
+ f"{by_condition[c].get('ok', 0) / total:>12.0%}" if total else " 0%"
199
+ for c in conditions
200
+ )
201
+ return "\n".join([header, *rows, footer])
202
+
203
+
204
+ def load_tasks(path: str | None) -> tuple[Task, ...]:
205
+ if path is None:
206
+ return DEFAULT_TASKS
207
+ raw: Any = json.loads(Path(path).read_text())
208
+ return tuple(Task(**entry) for entry in raw)
209
+
210
+
211
+ def main() -> None:
212
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
213
+ parser.add_argument("--tasks", help="JSON file of {module, target, task} entries")
214
+ parser.add_argument(
215
+ "--samples", type=int, default=3, help="repetitions per condition"
216
+ )
217
+ parser.add_argument("--out", help="write full results JSON here")
218
+ args = parser.parse_args()
219
+
220
+ client = anthropic.Anthropic()
221
+ tasks = load_tasks(args.tasks)
222
+ all_results: list[SampleResult] = []
223
+ for task in tasks:
224
+ print(f"== {task.module} ({task.target}) ==", file=sys.stderr)
225
+ doc = agent_help(_import_target(task.target))
226
+ all_results.extend(run_task(client, task, args.samples, doc))
227
+
228
+ print(summarize(all_results))
229
+ if args.out:
230
+ payload = json.dumps([asdict(r) for r in all_results], indent=2)
231
+ Path(args.out).write_text(payload)
232
+ print(f"\nFull results written to {args.out}", file=sys.stderr)
233
+
234
+
235
+ def _import_target(dotted: str) -> Any:
236
+ """Resolve a CLI-style target path to a live object."""
237
+ from agent_readable.__main__ import _resolve
238
+
239
+ return _resolve(dotted)
240
+
241
+
242
+ if __name__ == "__main__":
243
+ main()