thunc 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. thunc-0.2.2/.github/demo-agent.gif +0 -0
  2. thunc-0.2.2/.github/demo-function.gif +0 -0
  3. {thunc-0.2.1 → thunc-0.2.2}/CHANGELOG.md +69 -0
  4. {thunc-0.2.1 → thunc-0.2.2}/CONTRIBUTING.md +76 -13
  5. {thunc-0.2.1 → thunc-0.2.2}/PKG-INFO +119 -53
  6. {thunc-0.2.1 → thunc-0.2.2}/README.md +117 -52
  7. thunc-0.2.2/benchmarks/README.md +32 -0
  8. thunc-0.2.2/benchmarks/__init__.py +1 -0
  9. thunc-0.2.2/benchmarks/__main__.py +60 -0
  10. thunc-0.2.2/benchmarks/bench_agent.py +62 -0
  11. thunc-0.2.2/benchmarks/bench_backends.py +84 -0
  12. thunc-0.2.2/benchmarks/bench_cache.py +34 -0
  13. thunc-0.2.2/benchmarks/bench_calls.py +127 -0
  14. thunc-0.2.2/benchmarks/bench_schema.py +85 -0
  15. thunc-0.2.2/benchmarks/bench_startup.py +30 -0
  16. thunc-0.2.2/benchmarks/bench_tools.py +102 -0
  17. thunc-0.2.2/benchmarks/fakes.py +101 -0
  18. thunc-0.2.2/benchmarks/harness.py +190 -0
  19. thunc-0.2.2/design/og-card.html +61 -0
  20. thunc-0.2.2/docs/docs/agents.html +189 -0
  21. thunc-0.2.2/docs/docs/api.html +150 -0
  22. thunc-0.2.2/docs/docs/backends.html +122 -0
  23. thunc-0.2.2/docs/docs/caching.html +109 -0
  24. thunc-0.2.2/docs/docs/functions.html +172 -0
  25. thunc-0.2.2/docs/docs/index.html +139 -0
  26. thunc-0.2.2/docs/docs/jev.html +182 -0
  27. thunc-0.2.2/docs/docs/temporal.html +147 -0
  28. thunc-0.2.2/docs/index.html +215 -0
  29. thunc-0.2.2/docs/jev.html +14 -0
  30. thunc-0.2.2/docs/og.png +0 -0
  31. thunc-0.2.2/docs/site.css +218 -0
  32. thunc-0.2.2/docs/site.js +94 -0
  33. thunc-0.2.2/docs/sitemap.xml +12 -0
  34. thunc-0.2.2/docs/temporal.html +14 -0
  35. {thunc-0.2.1 → thunc-0.2.2}/examples/repo_guide.py +4 -2
  36. thunc-0.2.2/live_tests/bench_tooluse.py +903 -0
  37. thunc-0.2.2/live_tests/bench_tooluse_report.md +385 -0
  38. {thunc-0.2.1 → thunc-0.2.2}/pyproject.toml +2 -1
  39. thunc-0.2.2/tests/fake_claude.py +135 -0
  40. {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_runtime.py +37 -0
  41. {thunc-0.2.1 → thunc-0.2.2}/tests/test_agent.py +252 -6
  42. {thunc-0.2.1 → thunc-0.2.2}/tests/test_backends.py +206 -69
  43. thunc-0.2.2/tests/test_benchmarks.py +13 -0
  44. thunc-0.2.2/tests/test_claude_code_agent.py +267 -0
  45. {thunc-0.2.1 → thunc-0.2.2}/tests/test_native.py +6 -0
  46. {thunc-0.2.1 → thunc-0.2.2}/tests/test_permissions.py +19 -0
  47. thunc-0.2.2/tests/test_profiling.py +119 -0
  48. {thunc-0.2.1 → thunc-0.2.2}/thunc/__init__.py +1 -1
  49. {thunc-0.2.1 → thunc-0.2.2}/thunc/__main__.py +50 -2
  50. {thunc-0.2.1 → thunc-0.2.2}/thunc/agent.py +115 -26
  51. {thunc-0.2.1 → thunc-0.2.2}/thunc/backends.py +150 -30
  52. thunc-0.2.2/thunc/claude_code.py +351 -0
  53. {thunc-0.2.1 → thunc-0.2.2}/thunc/core.py +25 -4
  54. thunc-0.2.2/thunc/errors.py +15 -0
  55. {thunc-0.2.1 → thunc-0.2.2}/thunc/execution.py +4 -0
  56. thunc-0.2.2/thunc/mcp_relay.py +77 -0
  57. {thunc-0.2.1 → thunc-0.2.2}/thunc/native.py +54 -18
  58. {thunc-0.2.1 → thunc-0.2.2}/thunc/permissions.py +23 -7
  59. thunc-0.2.2/thunc/profiling.py +261 -0
  60. {thunc-0.2.1 → thunc-0.2.2}/thunc/tools.py +140 -41
  61. {thunc-0.2.1 → thunc-0.2.2}/uv.lock +1 -1
  62. thunc-0.2.1/docs/index.html +0 -279
  63. thunc-0.2.1/docs/jev.html +0 -325
  64. thunc-0.2.1/docs/sitemap.xml +0 -15
  65. thunc-0.2.1/docs/temporal.html +0 -40
  66. thunc-0.2.1/thunc/errors.py +0 -2
  67. {thunc-0.2.1 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
  68. {thunc-0.2.1 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
  69. {thunc-0.2.1 → thunc-0.2.2}/.github/social-preview.png +0 -0
  70. {thunc-0.2.1 → thunc-0.2.2}/.github/workflows/ci.yml +0 -0
  71. {thunc-0.2.1 → thunc-0.2.2}/.github/workflows/publish.yml +0 -0
  72. {thunc-0.2.1 → thunc-0.2.2}/.gitignore +0 -0
  73. {thunc-0.2.1 → thunc-0.2.2}/LICENSE +0 -0
  74. {thunc-0.2.1 → thunc-0.2.2}/docs/.nojekyll +0 -0
  75. {thunc-0.2.1 → thunc-0.2.2}/docs/apple-touch-icon.png +0 -0
  76. {thunc-0.2.1 → thunc-0.2.2}/docs/favicon-96.png +0 -0
  77. {thunc-0.2.1 → thunc-0.2.2}/docs/favicon.ico +0 -0
  78. {thunc-0.2.1 → thunc-0.2.2}/docs/favicon.svg +0 -0
  79. {thunc-0.2.1 → thunc-0.2.2}/docs/google2cd177e5e85b3c3e.html +0 -0
  80. {thunc-0.2.1 → thunc-0.2.2}/examples/dynamic_prompts.py +0 -0
  81. {thunc-0.2.1 → thunc-0.2.2}/examples/hello.py +0 -0
  82. {thunc-0.2.1 → thunc-0.2.2}/examples/jev_hello.py +0 -0
  83. {thunc-0.2.1 → thunc-0.2.2}/examples/jev_inbox.py +0 -0
  84. {thunc-0.2.1 → thunc-0.2.2}/examples/jev_with_claude.py +0 -0
  85. {thunc-0.2.1 → thunc-0.2.2}/examples/log_triage.py +0 -0
  86. {thunc-0.2.1 → thunc-0.2.2}/examples/support_inbox.py +0 -0
  87. {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/README.md +0 -0
  88. {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/application.py +0 -0
  89. {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/client.py +0 -0
  90. {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/pipeline.py +0 -0
  91. {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/worker.py +0 -0
  92. {thunc-0.2.1 → thunc-0.2.2}/live_tests/conftest.py +0 -0
  93. {thunc-0.2.1 → thunc-0.2.2}/live_tests/eval_prompts.py +0 -0
  94. {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_agent.py +0 -0
  95. {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_bool_decision.py +0 -0
  96. {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_dict_output.py +0 -0
  97. {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_hello.py +0 -0
  98. {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_literal_choice.py +0 -0
  99. {thunc-0.2.1 → thunc-0.2.2}/pytest-temporal.ini +0 -0
  100. {thunc-0.2.1 → thunc-0.2.2}/tests/conftest.py +0 -0
  101. {thunc-0.2.1 → thunc-0.2.2}/tests/future_types.py +0 -0
  102. {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/conftest.py +0 -0
  103. {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/process_worker.py +0 -0
  104. {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_contract.py +0 -0
  105. {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_storage.py +0 -0
  106. {thunc-0.2.1 → thunc-0.2.2}/tests/test_cache.py +0 -0
  107. {thunc-0.2.1 → thunc-0.2.2}/tests/test_calls.py +0 -0
  108. {thunc-0.2.1 → thunc-0.2.2}/tests/test_cli.py +0 -0
  109. {thunc-0.2.1 → thunc-0.2.2}/tests/test_execution.py +0 -0
  110. {thunc-0.2.1 → thunc-0.2.2}/tests/test_jev.py +0 -0
  111. {thunc-0.2.1 → thunc-0.2.2}/tests/test_schema.py +0 -0
  112. {thunc-0.2.1 → thunc-0.2.2}/thunc/cache.py +0 -0
  113. {thunc-0.2.1 → thunc-0.2.2}/thunc/config.py +0 -0
  114. {thunc-0.2.1 → thunc-0.2.2}/thunc/decorator.py +0 -0
  115. {thunc-0.2.1 → thunc-0.2.2}/thunc/prompts.py +0 -0
  116. {thunc-0.2.1 → thunc-0.2.2}/thunc/py.typed +0 -0
  117. {thunc-0.2.1 → thunc-0.2.2}/thunc/runs.py +0 -0
  118. {thunc-0.2.1 → thunc-0.2.2}/thunc/schema.py +0 -0
  119. {thunc-0.2.1 → thunc-0.2.2}/thunc/store.py +0 -0
  120. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/__init__.py +0 -0
  121. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/activities.py +0 -0
  122. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/adapters.py +0 -0
  123. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/client.py +0 -0
  124. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/effects.py +0 -0
  125. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/models.py +0 -0
  126. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/registry.py +0 -0
  127. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/storage.py +0 -0
  128. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/worker.py +0 -0
  129. {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/workflows.py +0 -0
Binary file
Binary file
@@ -3,6 +3,75 @@
3
3
  All notable changes to thunc. The full notes for each release are on the
4
4
  [releases page](https://github.com/Eltarras/thunc/releases).
5
5
 
6
+ ## 0.2.2 (beta)
7
+
8
+ **Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
9
+ calls instead of writing each action as JSON text, a failed step is retried instead of ending the
10
+ run, and the agent's tools fill gaps a benchmark found. In that tool-use benchmark
11
+ (`live_tests/bench_tooluse.py`, 8 tasks, Claude Sonnet 5.5, 3 runs each), agents on Claude Code went
12
+ from 12 of 24 runs passing to 24 of 24, from 99 to 10 seconds a task, and from $0.084 to $0.022 a
13
+ task; Claude Code itself took 11 seconds and $0.069. The API backends also reuse their connections,
14
+ Codex answers return sooner, and `thunc run --profile` shows where a program's time goes.
15
+
16
+ ### Behavior changes
17
+
18
+ Nothing is removed, but these defaults change:
19
+
20
+ - **Agents on `claude-code` make native tool calls** (see Added). With a `claude` CLI too old for
21
+ them, or with MCP servers turned off by a policy, a run falls back to the text protocol with a
22
+ warning. `protocol="text"` keeps the old way.
23
+ - **`list` and `search` leave out what git ignores** in a git repository (build output, caches,
24
+ vendored code). A folder named explicitly is still listed and searched.
25
+ - **Long command output keeps its start and its end** (the first error and the summary), not only
26
+ the end.
27
+ - **The `claude-code` backend loads none of your Claude Code settings** (`--setting-sources ""`):
28
+ no `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so an
29
+ agent's `workdir` can't give it instructions unless `follow=` asks for them.
30
+
31
+ ### Added
32
+
33
+ - **Native calls on `claude-code`**: the agent's tools are an MCP server that one `claude -p`
34
+ process per run calls; thunc carries out each call with its own tools, permissions and run
35
+ record. The CLI runs in the agent's `workdir`. `protocol="text"` keeps the old way, and durable
36
+ runs on Claude Code still use it. When Claude Code can't start native calls (an older CLI, or MCP
37
+ servers turned off by a policy), a run falls back to the text protocol with a warning and a
38
+ `fallback` entry in its record; `protocol="native"` raises instead.
39
+ - **`run` takes `cwd`**, a folder inside `workdir` to run the command in.
40
+ - **The `shell` permission** runs command lines through the system shell, so pipes, `&&`, `cd`
41
+ and redirects work. Off by default; it can't be combined with `!run:` rules.
42
+ - **`search` takes `glob`** (`*.py` by file name, `src/**/*.ts` by path) to limit the files
43
+ searched.
44
+ - **`thunc run --profile`**: runs a script (or `-m module`) and prints a performance report to
45
+ stderr when it ends: per function, calls, cache hits, retries, failures, total/mean/p95/max time
46
+ and the split between model time and thunc's own; for agents, steps and time in each tool; and
47
+ the share of wall time spent in thunc, with the overlap from `thunc.map`.
48
+
49
+ ### Changed
50
+
51
+ - **A failed step is retried.** A timeout, lost connection, rate limit, server error or CLI call
52
+ that ended in an error (`thunc.errors.TransientError`) is retried twice in an agent run, with a
53
+ note in the run record, before the run fails. A text-protocol step on Claude Code or Codex may
54
+ take 120 seconds before it's retried, instead of the whole `timeout`.
55
+ - **The `anthropic` and `openai` backends reuse their connections.** One SDK client is shared by
56
+ every call in the process (`thunc.map`'s threads and agent runs included), instead of a new
57
+ client, and so a new TCP and TLS handshake, for each call. A new client is made when the API key,
58
+ the SDK's environment variables (`ANTHROPIC_*`, `OPENAI_*`) or the process change. In a local
59
+ benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
60
+ - **Agents on the text protocol can act several times per reply.** On Codex and `protocol="text"`
61
+ (and on Claude Code when it falls back to the text protocol), a reply can be a JSON array of
62
+ independent actions (reading three files) instead of one. They run in order, at most 16 per reply,
63
+ and every result comes back together, as with native tool calls. Each turn resends the whole
64
+ transcript and, on the CLI backends, starts the CLI, so fewer turns save both. A single JSON
65
+ action works as before. On Codex, with `live_tests/eval_prompts.py` (default prompt, 5 runs of
66
+ each task), every run batched its first reads: replies went from 6.0 / 4.8 / 5.0 to 5.0 / 3.0 /
67
+ 3.6 (fix / review / analysis) and the mean time from 37 / 27 / 27 s to 29 / 19 / 21 s, with the
68
+ same work done and 30/30 passing.
69
+ - **The `codex` backend returns as soon as the answer arrives.** It reads Codex's JSON events as
70
+ they come (`codex exec --json`) instead of waiting for the process to exit and reading the answer
71
+ from a file. Codex takes about 0.4 s to shut down after answering; that now happens in the
72
+ background. Over 8 alternating pairs of real calls the new way was faster every time, by a median
73
+ of 0.67 s on a call of about 4 s. Every agent turn on Codex is one call, so the saving repeats.
74
+
6
75
  ## 0.2.1 (beta)
7
76
 
8
77
  **Durable agents with Temporal.** An optional `thunc[temporal]` runtime records each model turn
@@ -16,34 +16,59 @@ Thanks for helping. thunc is in beta, so feedback on the API is as useful as cod
16
16
 
17
17
  ```bash
18
18
  git clone https://github.com/Eltarras/thunc && cd thunc
19
- python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
19
+ python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
20
20
  ```
21
21
 
22
- The README's [Development](README.md#development) section lists the commands for tests and checks.
22
+ `temporal-test` installs the pinned Temporal SDK the durable-run tests use. Without it the tests in
23
+ `tests/temporal/` are skipped and everything else still runs. The README's
24
+ [Development](README.md#development) section lists the commands for live tests.
23
25
 
24
26
  ## Before you open a PR
25
27
 
26
- CI runs these on Python 3.10 to 3.14, and all must pass:
28
+ CI runs these on Python 3.10 to 3.14 on Linux, and all must pass. Run each on its own and check
29
+ its exit code:
27
30
 
28
31
  ```bash
29
32
  .venv/bin/pytest
30
- .venv/bin/ruff check . && .venv/bin/ruff format --check .
33
+ .venv/bin/ruff check .
34
+ .venv/bin/ruff format --check .
31
35
  .venv/bin/mypy --strict thunc
36
+ .venv/bin/mypy --strict --platform win32 thunc
37
+ ```
38
+
39
+ CI also runs the offline tests on Windows (Python 3.13), since the agent lock, command handling and
40
+ paths have Windows-only code. If you touch `thunc/temporal/`, run the integration tests against a
41
+ real local Temporal service too (they download it once and make no model calls):
42
+
43
+ ```bash
44
+ THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
32
45
  ```
33
46
 
34
47
  ## Ground rules
35
48
 
36
49
  - **Standard library only in the core.** A provider SDK is an optional extra in `pyproject.toml`,
37
- imported inside its backend function, never at module level.
50
+ imported inside its backend function, never at module level. The same goes for Temporal:
51
+ only `thunc/temporal/` imports `temporalio`, and `thunc.call`, `@thunc.function` and
52
+ `agent.run()` must keep working without it.
38
53
  - **Tests in `tests/` never call a real model.** Use the `fake` fixture in `tests/conftest.py`,
39
- which returns scripted replies and records the prompts. Real calls belong in `live_tests/`,
40
- which CI doesn't run.
54
+ which returns scripted replies and records the prompts. Tests of native tool calls replace only
55
+ the SDK client and use the SDK's real types. Real calls belong in `live_tests/`, which CI
56
+ doesn't run.
57
+ - **Agent permissions are a safety boundary.** A change to `permissions.py`, `tools.py` or the
58
+ effect journal in `thunc/temporal/effects.py` needs a test that fails when the rule is broken.
59
+ Check it by breaking the rule on purpose and watching the test fail.
60
+ - **Durable runs replay.** Workflow code in `thunc/temporal/workflows.py` stays deterministic;
61
+ model calls, tools and file access happen in activities. A file or memory change goes through
62
+ the intent and receipt journal, and a command whose outcome is uncertain is never rerun
63
+ automatically. Before changing orchestration or the saved state format, version it and replay
64
+ saved histories (see the [Temporal guide](examples/temporal/README.md)).
41
65
  - **User data stays out of the instructions.** Inputs are sent separately from the prompt
42
66
  (see `_build_prompt` in `thunc/core.py`). Don't add code paths that paste inputs into
43
67
  instructions.
44
68
  - **Failures are loud.** When no valid answer arrives, raise `ThuncError`; never return a
45
69
  default value.
46
- - **Update the README** when you change the public API or the supported return types.
70
+ - **Update the README** when you change the public API or the supported return types, and the
71
+ [Temporal guide](examples/temporal/README.md) when you change durable runs.
47
72
  - **One change per PR**, with a description of what it does and how you tested it.
48
73
 
49
74
  ## Adding a backend
@@ -58,7 +83,14 @@ CI runs these on Python 3.10 to 3.14, and all must pass:
58
83
  `thunc/config.py` if needed.
59
84
  3. Add an optional extra for its SDK in `pyproject.toml`, and add the extra to the CI install.
60
85
  4. Add offline tests with a stubbed SDK module, like the existing ones in `tests/test_backends.py`.
61
- 5. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
86
+ 5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
87
+ `Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
88
+ the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
89
+ A CLI that can call MCP tools can get native calls the way Claude Code does
90
+ (`thunc/claude_code.py`, tested with a fake CLI in `tests/test_claude_code_agent.py`).
91
+ Raise `TransientError` for failures worth asking again, so agent runs retry them.
92
+ Agents and durable runs refuse typed backends.
93
+ 6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
62
94
  the PR that you did.
63
95
 
64
96
  ## Known issues
@@ -73,8 +105,7 @@ model is asked again instead of thunc guessing:
73
105
  - Quoted numbers (`"4"` for an `int`), labels in the wrong case (`Bug` for `bug`), prose around
74
106
  JSON, Python-style values (`['a']`, `None`), trailing commas, curly quotes, non-ASCII digits.
75
107
  - Unquoted text for `str | None`, and exclamations like `Yes!` for a `bool`.
76
- - Several `<think>` blocks in a row, several lines of prose before a fence, `~~~`, indented or
77
- four-backtick fences.
108
+ - Several `<think>` blocks in a row, `~~~`, indented or four-backtick fences.
78
109
  - A one-key object whose key is a field of the expected dataclass: `{"customer": {...}}` for an
79
110
  `Order` with a `customer` field is an `Order` with a bad customer, not a wrapper.
80
111
  - A reply after the closing fence (for example a reasoning paragraph): it could be a second answer.
@@ -86,7 +117,11 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
86
117
  **Open: behavior that might change.**
87
118
 
88
119
  - `thunc.map` loses every result when one item fails, and returns coroutines for `async` functions.
89
- - Backend errors (timeouts, connection failures) are never retried; only bad replies are.
120
+ - Backend errors (timeouts, connection failures) are never retried by `thunc.call`; only bad
121
+ replies are. Local agent runs retry a step that failed with a `TransientError` twice, and durable
122
+ runs retry transient provider failures.
123
+ - Several lines of prose before a code fence are read as a preamble, though `_unfence` in
124
+ `thunc/schema.py` documents one line, and no test pins either behavior.
90
125
  - A union takes the first option that accepts the reply as JSON, in the order written: `3` for
91
126
  `float | int` is `3.0`. An unquoted label is only tried after that, so `2` for
92
127
  `Literal["2"] | int` is the int `2`, not the label.
@@ -109,10 +144,38 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
109
144
  - A list of dataclasses passed as an input is shown to the model as Python reprs, not JSON.
110
145
  - A string return annotation naming a class defined after the function raises `NameError` when
111
146
  the decorator runs.
112
- - `make_dataclass` classes, and `Annotated` fields on Python 3.10, aren't supported.
147
+ - On Python 3.10 and 3.11, a `make_dataclass` class whose field types are strings naming anything
148
+ but builtins (`"Item"`, `"Annotated[int, 'x']"`) raises `NameError`: before 3.12 its module is
149
+ `types`, not the caller's.
150
+ - `Annotated[...]` works on dataclass fields but not as a return type.
113
151
  - The Python 3.10 fallback for string annotations (`_hints` and `resolve_strings` in
114
152
  `thunc/schema.py`) can be removed when 3.10 support is dropped.
115
153
 
154
+ **Open: agents and durable runs.**
155
+
156
+ - Durable runs on `claude-code` use the JSON text protocol, not native calls: the MCP path keeps
157
+ one CLI process for the whole run, which a durable step can't snapshot. The text protocol is the
158
+ weaker one: in `live_tests/bench_tooluse.py` on Sonnet 5.5 it passed 12 of 24 runs, and 20 of 24
159
+ with step retries (measured before action arrays), against 24 of 24 for native calls.
160
+ - The `shell` permission isn't tested on Windows: its test is skipped there, so `cmd /c` has never
161
+ run in CI.
162
+ - With `shell`, a file read with `cat` doesn't count as read for `edit`, which refuses until the
163
+ agent reads it with `read` (the no-blind-overwrite rule). Agents work around it with a short
164
+ `read`, at the cost of a step.
165
+ - The agent loops on the Claude and OpenAI APIs have no live tool-use benchmark yet (only
166
+ `live_tests/eval_prompts.py`); their `max_tokens`, effort and stop-reason handling were reviewed
167
+ from the code only (`live_tests/bench_tooluse_report.md`, finding 7).
168
+ - Durable agents can't use `tools=` yet: the effects of the program's own functions can't be
169
+ journaled.
170
+ - Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
171
+ tombstones are kept forever. There's no context summarization either, so a long run fails once
172
+ its saved state passes 16 MiB.
173
+ - A durable workspace can only restart on the same volume and absolute paths; nothing moves it
174
+ between hosts.
175
+ - A hard worker kill can leave a command's subprocesses running.
176
+ - Durable runs aren't tested on Windows (the code is only type-checked for it), and there are no
177
+ live provider tests for durable runs yet.
178
+
116
179
  ## Reporting bugs
117
180
 
118
181
  Open an [issue](https://github.com/Eltarras/thunc/issues) with:
@@ -1,8 +1,9 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: thunc
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: think + function: call an LLM like a typed Python function.
5
5
  Project-URL: Homepage, https://eltarras.github.io/thunc/
6
+ Project-URL: Documentation, https://eltarras.github.io/thunc/docs/
6
7
  Project-URL: Repository, https://github.com/Eltarras/thunc
7
8
  Project-URL: Issues, https://github.com/Eltarras/thunc/issues
8
9
  Author: Hussein Eltarras
@@ -32,20 +33,24 @@ Requires-Dist: pytest-asyncio>=0.24; extra == 'temporal-test'
32
33
  Requires-Dist: temporalio==1.34.0; extra == 'temporal-test'
33
34
  Description-Content-Type: text/markdown
34
35
 
35
- ![thunc: call an LLM like a typed Python function](https://raw.githubusercontent.com/Eltarras/thunc/main/.github/social-preview.png)
36
-
37
- [![CI](https://github.com/Eltarras/thunc/actions/workflows/ci.yml/badge.svg)](https://github.com/Eltarras/thunc/actions/workflows/ci.yml)
38
- [![PyPI](https://img.shields.io/pypi/v/thunc)](https://pypi.org/project/thunc/)
39
- [![Website](https://img.shields.io/badge/website-eltarras.github.io%2Fthunc-f5b14c)](https://eltarras.github.io/thunc/)
36
+ ![A thunc function returning list[Item] is called with "2 oat lattes and a croissant pls. oh, one more latte!" and returns two Item dataclasses: oat latte ×3 and croissant ×1.](https://raw.githubusercontent.com/Eltarras/thunc/main/.github/demo-function.gif)
40
37
 
41
38
  **think + function.** Call an LLM like a typed Python function.
42
39
 
43
- > **Status: beta (v0.2).** Expect bugs; the API may change. Feedback and issues are welcome.
40
+ [![PyPI](https://img.shields.io/pypi/v/thunc)](https://pypi.org/project/thunc/)
41
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue)](https://pypi.org/project/thunc/)
42
+ [![CI](https://github.com/Eltarras/thunc/actions/workflows/ci.yml/badge.svg)](https://github.com/Eltarras/thunc/actions/workflows/ci.yml)
43
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](https://github.com/Eltarras/thunc/blob/main/LICENSE)
44
+ [![Docs](https://img.shields.io/badge/docs-eltarras.github.io%2Fthunc-9b481b)](https://eltarras.github.io/thunc/docs/)
45
+
46
+ ```bash
47
+ pip install thunc
48
+ ```
44
49
 
45
50
  ```python
46
51
  import thunc
47
52
 
48
- thunc.configure(backend="claude-code")
53
+ thunc.configure(backend="claude-code") # or "codex", "anthropic", "openai"
49
54
 
50
55
 
51
56
  @thunc.function
@@ -57,26 +62,43 @@ def urgency(ticket: str) -> int:
57
62
  urgency("I was charged twice!") # -> 4, a checked int
58
63
  ```
59
64
 
60
- The answer is parsed into the declared type. If it doesn't fit, the model is asked again, and
61
- after that `thunc.ThuncError` is raised. The library uses the standard library only and needs
62
- Python 3.10+.
65
+ The docstring is the prompt and the return annotation is the type. The answer is parsed into that
66
+ type; if it doesn't fit, the model is asked again, and after that `thunc.ThuncError` is raised. No
67
+ dependencies, Python 3.10+.
68
+
69
+ It also runs [agents](#agents): typed functions that can read, edit and test your code before they
70
+ answer.
63
71
 
64
- It runs on the Claude API, the OpenAI API, a local model (LM Studio, or any server that speaks the
65
- OpenAI API), or your Claude Code or Codex login.
72
+ The [docs](https://eltarras.github.io/thunc/docs/) cover everything below, a page per topic.
66
73
 
67
- ## Install
74
+ ## Quickstart
75
+
76
+ No API key needed if you have Claude Code or Codex installed: thunc can use their login.
68
77
 
69
78
  ```bash
70
- pip install thunc # standard library only
71
- pip install "thunc[anthropic]" # adds the Claude API backend
72
- pip install "thunc[openai]" # adds the OpenAI API backend
73
- pip install "thunc[temporal]" # adds durable agents on Temporal
79
+ pip install thunc
80
+ THUNC_BACKEND=claude-code python3 -c 'import thunc; print(thunc.call("Say hello in five words or fewer."))'
74
81
  ```
75
82
 
76
- ## Try it
83
+ Swap in the backend you have:
84
+
85
+ | You have | Set | Install |
86
+ |---|---|---|
87
+ | [Claude Code](https://claude.com/claude-code), logged in | `THUNC_BACKEND=claude-code` | `pip install thunc` |
88
+ | [Codex](https://github.com/openai/codex), logged in | `THUNC_BACKEND=codex` | `pip install thunc` |
89
+ | An Anthropic API key | `ANTHROPIC_API_KEY` | `pip install "thunc[anthropic]"` |
90
+ | An OpenAI API key | `OPENAI_API_KEY` | `pip install "thunc[openai]"` |
91
+ | A local model (LM Studio) | `OPENAI_BASE_URL` (see **Local models** below) | `pip install "thunc[openai]"` |
92
+
93
+ With an API key set, thunc picks that backend on its own, so `THUNC_BACKEND` isn't needed. In code,
94
+ `thunc.configure(backend=...)` does the same. Durable agents on Temporal add
95
+ `pip install "thunc[temporal]"`.
96
+
97
+ > **Beta (v0.2).** The API may still change. Bug reports and feedback are welcome in
98
+ > [issues](https://github.com/Eltarras/thunc/issues).
77
99
 
78
- Clone the repo and run the examples from its root. No install is needed; the examples run through
79
- your local [Claude Code](https://claude.com/claude-code) login:
100
+ **More examples.** Clone the repo and run them from its root, with no install, through your
101
+ Claude Code login:
80
102
 
81
103
  ```bash
82
104
  git clone https://github.com/Eltarras/thunc && cd thunc
@@ -85,9 +107,6 @@ python3 -m examples.support_inbox
85
107
  THUNC_BACKEND=codex python3 -m examples.log_triage
86
108
  ```
87
109
 
88
- With an API key instead, install the SDK and pick the backend: `THUNC_BACKEND=openai` with
89
- `OPENAI_API_KEY`, or `THUNC_BACKEND=anthropic` with `ANTHROPIC_API_KEY`.
90
-
91
110
  ## Two ways to write a prompt
92
111
 
93
112
  | | When | |
@@ -184,7 +203,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
184
203
  `pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
185
204
  any server that speaks the OpenAI Responses API.
186
205
  - `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
187
- Both run with their own tools turned off, so the model can only answer. `codex` also ignores
206
+ Both run with their own tools turned off, so the model can only answer; an agent on
207
+ `claude-code` gets only its thunc tools, as native calls (see Agents). `codex` also ignores
188
208
  `~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
189
209
  login still works. Pick the model with `configure(model=...)` or `model=`.
190
210
  - `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
@@ -198,7 +218,7 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
198
218
  CLI always uses `jev-latest`), and an answer that fails `ensure=` isn't retried, since Jev would
199
219
  give the same one. It's only used when you choose it: `backend="jev"` or `THUNC_BACKEND=jev`.
200
220
  Setup (install the CLI, log in, check it works): the
201
- [Jev guide](https://eltarras.github.io/thunc/jev.html).
221
+ [Jev guide](https://eltarras.github.io/thunc/docs/jev.html).
202
222
 
203
223
  **Local models:** the `openai` backend works with a local server through `OPENAI_BASE_URL`. This
204
224
  has been tested with [LM Studio](https://lmstudio.ai) running `openai/gpt-oss-20b`:
@@ -214,6 +234,31 @@ giving it alone.
214
234
  The backend can also be set with `THUNC_BACKEND`. With none set, `ANTHROPIC_API_KEY` (or a
215
235
  `configure(api_key=...)` alone) selects `anthropic`, and otherwise `OPENAI_API_KEY` selects `openai`.
216
236
 
237
+ **Profiling:** run your program with `thunc run --profile` to see where the time went when it ends:
238
+
239
+ ```bash
240
+ thunc run --profile support_inbox.py --limit 20 # a script and its arguments
241
+ thunc run --profile -m myapp.triage # a module, as with python -m
242
+ ```
243
+
244
+ The report goes to stderr: per function, the calls, cache hits, retries and failures, the total,
245
+ mean, p95 and slowest time, and how much of it was the model and how much thunc's own work
246
+ (building the prompt, parsing, the cache). Agent runs get their steps, model time and time in each
247
+ tool. It also says what share of the program's wall time was spent in thunc, and how much calls
248
+ overlapped under `thunc.map`. Without `--profile`, `thunc run` just runs the program, and nothing
249
+ is recorded. The program's exit code is passed through.
250
+
251
+ ```
252
+ CALLS
253
+ FUNCTION CALLS CACHED RETRIES FAILED TOTAL MEAN P95 MAX MODEL LOCAL
254
+ urgency 11 1 1 0 1.70s 155ms 309ms 309ms 1.69s 12ms
255
+
256
+ In thunc: 774ms of 980ms wall time (79%); the rest was the program's own code
257
+ Model time: 1.69s, 99% of the time in calls (anthropic/default model 1.69s)
258
+ Concurrency: calls overlapped 2.2x on average (thunc.map or threads)
259
+ Slowest: urgency took 309ms
260
+ ```
261
+
217
262
  **Type checking:** signatures and return types are visible to mypy and Pyright. mypy reports
218
263
  empty bodies; turn that off with `disable_error_code = ["empty-body"]`.
219
264
 
@@ -221,6 +266,8 @@ empty bodies; turn that off with `disable_error_code = ["empty-body"]`.
221
266
 
222
267
  > **New in 0.2.** Agents are new; their API may change in a later release as feedback comes in.
223
268
 
269
+ ![An agent allowed to write src/** and run pytest is asked to run the tests and fix any problems. pytest shows 1 failure; it reads src/pricing.py, fixes one line, reruns pytest (3 passed) and returns True.](https://raw.githubusercontent.com/Eltarras/thunc/main/.github/demo-agent.gif)
270
+
224
271
  An agent is a typed function that can look around before it answers. Give it a name and a working
225
272
  directory, declare its tasks the way you write `@thunc.function`, and call them from Python:
226
273
 
@@ -253,9 +300,15 @@ repo.call(f"Where is {setting} set?", returns=str)
253
300
  Each call is one run. The model takes one step at a time (list a folder, search, read or edit a
254
301
  file) and ends by calling `finish` with a value of the return type, which is checked like any thunc
255
302
  result. It runs on the Claude and OpenAI APIs through their own tool calls (the
256
- model can make several at once, and the fixed part of the prompt is cached), and on Claude Code and
257
- Codex by replying with one JSON action at a time. `protocol="text"` uses the second way on an API
258
- too, for example with a server behind `OPENAI_BASE_URL` that has no function calling.
303
+ model can make several at once, and the fixed part of the prompt is cached). On Claude Code the
304
+ calls are native too: the agent's tools are an MCP server that one `claude -p` process per run
305
+ calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
306
+ can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
307
+ text protocol below instead, with a warning, and so do later runs in the process;
308
+ `protocol="native"` fails instead. On Codex the model replies with JSON actions as text: one at a
309
+ time, or several independent ones (reading three files) as a JSON array, which saves turns.
310
+ `protocol="text"` uses that way on any backend, for example with a server behind `OPENAI_BASE_URL`
311
+ that has no function calling. (Durable runs on Claude Code use it too.)
259
312
  The `jev` backend only answers typed questions and cannot run agents, even for a task returning
260
313
  `bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
261
314
  or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
@@ -272,25 +325,29 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
272
325
  | `write:docs/**`, `write` | create and edit matching files (all files with no path); also lets it read them |
273
326
  | `read:src/**` | read only these; any `read:` rule replaces the read-everything default |
274
327
  | `run:pytest`, `run:git log`, `run` | run commands that start with these words (`run:git log` allows `git log --oneline`, not `git push`); `run` alone allows any |
328
+ | `shell` | run any command line in a shell (`sh -c`, or `cmd /c` on Windows), so pipes, `&&`, `cd` and redirects work; off by default, and it can't be combined with `!run:` rules |
275
329
  | `!read:.env*`, `!write:...`, `!run:git push`, `!memory` | deny; a deny always wins, and `!read` also stops writing |
276
330
 
277
331
  `*` stays within one folder, `**` crosses folders, and paths are relative to `workdir`. The agent
278
332
  is told its permissions, and an action they don't allow is refused with the reason, after which
279
333
  the run carries on. Bad rules fail when the agent is declared.
280
- - **Tools:** `list`, `read` and `search`; `write` (create a file, or replace one) and `edit`
281
- (replace text that appears exactly once) when a write rule allows it; `run` when a run rule
282
- allows it; and `remember`. Every path
283
- must stay inside `workdir`: `..`, absolute paths and symlinks that point outside are refused, and
284
- the rules are checked on where a link really leads. Files the agent may not read are left out of
285
- `list` and `search`.
334
+ - **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
335
+ such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
336
+ exactly once) when a write rule allows it; `run` when a run or shell rule allows it; and
337
+ `remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
338
+ outside are refused, and the rules are checked on where a link really leads. Files the agent may
339
+ not read are left out of `list` and `search`, and so is what git ignores, in a git repository
340
+ (build output, caches, vendored code); a folder named explicitly is still listed and searched.
286
341
  - **No blind overwrites.** A file is only replaced or edited after the agent read it in the same
287
342
  run, and only if it hasn't changed on disk since. There is no undo, so run agents that write in a
288
343
  git repository with a clean tree, and review their changes with `git diff`.
289
- - **Commands** run in `workdir` without a shell, so `&&`, pipes, redirects and `$VARIABLES` don't
290
- work (the agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and
291
- temp-folder variables, and whatever you pass in `env=`, so your API keys don't reach them. Each
292
- has a time limit (`command_timeout=120` seconds) that also stops the processes it started, and
293
- the agent sees the exit code and the output, its end kept when it's long.
344
+ - **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
345
+ permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
346
+ agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
347
+ variables, and whatever you pass in `env=`, so your API keys don't reach them. Each has a time
348
+ limit (`command_timeout=120` seconds) that also stops the processes it started, and the agent
349
+ sees the exit code and the output: the start and the end when it's long, since the first error is
350
+ often at the start and the summary at the end.
294
351
  - **A permitted command can do anything its program can.** `run:pytest` runs the project's code,
295
352
  which can read or change any file your user account can, whatever the read and write rules say.
296
353
  Permissions limit which tools the model uses; they aren't a sandbox. For untrusted input, run the
@@ -341,24 +398,29 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
341
398
 
342
399
  - **Failures are loud.** A run that hits `max_steps`, never gives a valid value, or loses its
343
400
  backend raises `thunc.AgentError` (a `ThuncError`), whose `.run` is the record up to that point.
344
- With tracing on, each run is also one line with every model reply.
401
+ A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
402
+ a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
403
+ each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
404
+ seconds before it's retried. With tracing on, each run is also one line with every model reply.
345
405
 
346
406
  **How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
347
407
  small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
348
- system prompt: bare (no working method), the default, and the task's preset. Five runs of each on
349
- 4 October 2026:
408
+ system prompt: bare (no working method), the default, and the task's preset. Five runs of each, on
409
+ the Claude API on 4 October 2026 and on Claude Code on 5 October 2026:
350
410
 
351
- | | Claude API (Opus 5.5, native calls) | Claude Code (text protocol) |
411
+ | | Claude API (Opus 5.5, native calls) | Claude Code (Sonnet 5.5, native calls) |
352
412
  |---|---|---|
353
413
  | Passed | 45/45: every task, every version | 45/45 |
354
- | Steps (bare / default / preset) | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.4 / 2.8, analysis 3.0 / 3.0 / 3.0 | fix 5.2 / 6.0 / 6.0, review 2.2 / 2.0 / 3.8, analysis 4.2 / 3.8 / 4.2 |
414
+ | Steps (bare / default / preset) | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.4 / 2.8, analysis 3.0 / 3.0 / 3.0 | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.0 / 2.0, analysis 3.0 / 2.8 / 2.6 |
355
415
  | Cost | $0.76 for all 45 runs (cache reads were 257,553 of 312,294 input tokens) | |
356
416
 
357
417
  Every version passed every time, so these tasks are too easy to tell the versions apart: the result
358
- says the prompt does no harm, not that it helps. The one difference is that the review preset reads
359
- more of the code before answering. Each review flagged the renamed function as a minor issue
360
- (outside code importing the old name breaks), never as blocking. Harder tasks are needed to measure
361
- more.
418
+ says the prompt does no harm, not that it helps. On the API, the review preset read more of the code
419
+ before answering, and each review flagged the renamed function as a minor issue (outside code
420
+ importing the old name breaks), never as blocking. On Claude Code, only the review preset flagged
421
+ it (2 of 5 runs, as minor). On the text protocol these tasks took more steps (fix 5.2 / 6.0 / 6.0
422
+ on Claude Code before native calls). For harder tasks that do tell harnesses apart, see the tool-use
423
+ benchmark in `live_tests/bench_tooluse.py` and its report.
362
424
 
363
425
  ## Durable agents with Temporal
364
426
 
@@ -428,15 +490,17 @@ thunc/
428
490
  permissions.py the agent's permission rules
429
491
  runs.py thunc.Run and AgentError: what a run did
430
492
  native.py how a run talks to its backend: native tool calls or the text protocol
493
+ claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
431
494
  store.py the agent's folder: memory, settings, run records, the lock
432
495
  prompts.py the agent's system prompt
433
- __main__.py the thunc command: thunc cache list / clear
496
+ __main__.py the thunc command: thunc run [--profile], thunc cache list / clear
497
+ profiling.py thunc run --profile: timing records and the report
434
498
  core.py thunc.call, thunc.map, tracing
435
499
  cache.py the answer cache: saving, listing, clearing
436
500
  schema.py return types: describe, parse, validate
437
501
  config.py settings and backend selection
438
502
  backends.py anthropic, openai, claude-code, codex, jev
439
- errors.py ThuncError
503
+ errors.py ThuncError, and TransientError for failures worth asking again
440
504
  tests/ offline: a fake backend, never a real model
441
505
  live_tests/ against a real model: hello, a yes/no decision, labels and ratings, messy text to a dict
442
506
  examples/
@@ -452,12 +516,14 @@ examples/
452
516
  ## Development
453
517
 
454
518
  ```bash
455
- python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
519
+ python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
456
520
  .venv/bin/pytest # offline tests (these run in CI)
521
+ THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal # a real local Temporal service; no model calls
457
522
  .venv/bin/pytest live_tests # real model calls through your Claude Code login; costs quota
458
523
  THUNC_BACKEND=anthropic .venv/bin/pytest live_tests # the same, through the Claude API (needs ANTHROPIC_API_KEY)
459
524
  THUNC_BACKEND=openai .venv/bin/pytest live_tests # the same, through the OpenAI API (needs OPENAI_API_KEY)
460
- .venv/bin/ruff check . && .venv/bin/mypy --strict thunc
525
+ .venv/bin/ruff check . && .venv/bin/ruff format --check .
526
+ .venv/bin/mypy --strict thunc && .venv/bin/mypy --strict --platform win32 thunc
461
527
  ```
462
528
 
463
529
  ## License