thunc 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (135) hide show
  1. thunc-0.2.3/.github/release-notes/v0.2.2.md +74 -0
  2. thunc-0.2.3/.github/release-notes/v0.2.3.md +80 -0
  3. thunc-0.2.3/.github/workflows/release-notes.yml +46 -0
  4. {thunc-0.2.2 → thunc-0.2.3}/CHANGELOG.md +117 -0
  5. {thunc-0.2.2 → thunc-0.2.3}/CONTRIBUTING.md +17 -16
  6. {thunc-0.2.2 → thunc-0.2.3}/PKG-INFO +35 -15
  7. {thunc-0.2.2 → thunc-0.2.3}/README.md +34 -14
  8. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/agents.html +17 -12
  9. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/api.html +13 -9
  10. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/backends.html +6 -5
  11. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/caching.html +25 -9
  12. thunc-0.2.3/docs/docs/changelog.html +214 -0
  13. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/functions.html +5 -4
  14. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/index.html +9 -5
  15. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/jev.html +4 -3
  16. {thunc-0.2.2 → thunc-0.2.3}/docs/docs/temporal.html +9 -3
  17. {thunc-0.2.2 → thunc-0.2.3}/docs/index.html +1 -1
  18. {thunc-0.2.2 → thunc-0.2.3}/docs/site.css +6 -0
  19. {thunc-0.2.2 → thunc-0.2.3}/docs/sitemap.xml +10 -9
  20. {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/README.md +14 -0
  21. {thunc-0.2.2 → thunc-0.2.3}/live_tests/bench_tooluse.py +167 -9
  22. {thunc-0.2.2 → thunc-0.2.3}/live_tests/bench_tooluse_report.md +76 -0
  23. {thunc-0.2.2 → thunc-0.2.3}/pyproject.toml +1 -1
  24. {thunc-0.2.2 → thunc-0.2.3}/tests/conftest.py +18 -0
  25. {thunc-0.2.2 → thunc-0.2.3}/tests/fake_claude.py +28 -1
  26. thunc-0.2.3/tests/fake_codex_mcp.py +142 -0
  27. thunc-0.2.3/tests/temporal/test_contract.py +250 -0
  28. {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/test_runtime.py +110 -0
  29. thunc-0.2.3/tests/temporal/test_segments.py +160 -0
  30. {thunc-0.2.2 → thunc-0.2.3}/tests/test_agent.py +228 -11
  31. {thunc-0.2.2 → thunc-0.2.3}/tests/test_backends.py +144 -4
  32. {thunc-0.2.2 → thunc-0.2.3}/tests/test_claude_code_agent.py +14 -3
  33. thunc-0.2.3/tests/test_codex_agent.py +208 -0
  34. {thunc-0.2.2 → thunc-0.2.3}/tests/test_execution.py +34 -0
  35. {thunc-0.2.2 → thunc-0.2.3}/tests/test_native.py +190 -7
  36. thunc-0.2.3/tests/test_text_protocol.py +272 -0
  37. {thunc-0.2.2 → thunc-0.2.3}/thunc/__init__.py +1 -1
  38. {thunc-0.2.2 → thunc-0.2.3}/thunc/agent.py +59 -25
  39. {thunc-0.2.2 → thunc-0.2.3}/thunc/backends.py +129 -33
  40. {thunc-0.2.2 → thunc-0.2.3}/thunc/claude_code.py +65 -130
  41. thunc-0.2.3/thunc/codex.py +276 -0
  42. {thunc-0.2.2 → thunc-0.2.3}/thunc/core.py +15 -1
  43. {thunc-0.2.2 → thunc-0.2.3}/thunc/execution.py +23 -0
  44. {thunc-0.2.2 → thunc-0.2.3}/thunc/native.py +275 -40
  45. thunc-0.2.3/thunc/relay.py +179 -0
  46. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/activities.py +193 -10
  47. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/effects.py +11 -3
  48. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/registry.py +37 -7
  49. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/workflows.py +1 -1
  50. {thunc-0.2.2 → thunc-0.2.3}/thunc/tools.py +132 -25
  51. {thunc-0.2.2 → thunc-0.2.3}/uv.lock +1 -1
  52. thunc-0.2.2/tests/temporal/test_contract.py +0 -118
  53. {thunc-0.2.2 → thunc-0.2.3}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
  54. {thunc-0.2.2 → thunc-0.2.3}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
  55. {thunc-0.2.2 → thunc-0.2.3}/.github/demo-agent.gif +0 -0
  56. {thunc-0.2.2 → thunc-0.2.3}/.github/demo-function.gif +0 -0
  57. {thunc-0.2.2 → thunc-0.2.3}/.github/social-preview.png +0 -0
  58. {thunc-0.2.2 → thunc-0.2.3}/.github/workflows/ci.yml +0 -0
  59. {thunc-0.2.2 → thunc-0.2.3}/.github/workflows/publish.yml +0 -0
  60. {thunc-0.2.2 → thunc-0.2.3}/.gitignore +0 -0
  61. {thunc-0.2.2 → thunc-0.2.3}/LICENSE +0 -0
  62. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/README.md +0 -0
  63. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/__init__.py +0 -0
  64. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/__main__.py +0 -0
  65. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_agent.py +0 -0
  66. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_backends.py +0 -0
  67. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_cache.py +0 -0
  68. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_calls.py +0 -0
  69. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_schema.py +0 -0
  70. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_startup.py +0 -0
  71. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_tools.py +0 -0
  72. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/fakes.py +0 -0
  73. {thunc-0.2.2 → thunc-0.2.3}/benchmarks/harness.py +0 -0
  74. {thunc-0.2.2 → thunc-0.2.3}/design/og-card.html +0 -0
  75. {thunc-0.2.2 → thunc-0.2.3}/docs/.nojekyll +0 -0
  76. {thunc-0.2.2 → thunc-0.2.3}/docs/apple-touch-icon.png +0 -0
  77. {thunc-0.2.2 → thunc-0.2.3}/docs/favicon-96.png +0 -0
  78. {thunc-0.2.2 → thunc-0.2.3}/docs/favicon.ico +0 -0
  79. {thunc-0.2.2 → thunc-0.2.3}/docs/favicon.svg +0 -0
  80. {thunc-0.2.2 → thunc-0.2.3}/docs/google2cd177e5e85b3c3e.html +0 -0
  81. {thunc-0.2.2 → thunc-0.2.3}/docs/jev.html +0 -0
  82. {thunc-0.2.2 → thunc-0.2.3}/docs/og.png +0 -0
  83. {thunc-0.2.2 → thunc-0.2.3}/docs/site.js +0 -0
  84. {thunc-0.2.2 → thunc-0.2.3}/docs/temporal.html +0 -0
  85. {thunc-0.2.2 → thunc-0.2.3}/examples/dynamic_prompts.py +0 -0
  86. {thunc-0.2.2 → thunc-0.2.3}/examples/hello.py +0 -0
  87. {thunc-0.2.2 → thunc-0.2.3}/examples/jev_hello.py +0 -0
  88. {thunc-0.2.2 → thunc-0.2.3}/examples/jev_inbox.py +0 -0
  89. {thunc-0.2.2 → thunc-0.2.3}/examples/jev_with_claude.py +0 -0
  90. {thunc-0.2.2 → thunc-0.2.3}/examples/log_triage.py +0 -0
  91. {thunc-0.2.2 → thunc-0.2.3}/examples/repo_guide.py +0 -0
  92. {thunc-0.2.2 → thunc-0.2.3}/examples/support_inbox.py +0 -0
  93. {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/application.py +0 -0
  94. {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/client.py +0 -0
  95. {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/pipeline.py +0 -0
  96. {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/worker.py +0 -0
  97. {thunc-0.2.2 → thunc-0.2.3}/live_tests/conftest.py +0 -0
  98. {thunc-0.2.2 → thunc-0.2.3}/live_tests/eval_prompts.py +0 -0
  99. {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_agent.py +0 -0
  100. {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_bool_decision.py +0 -0
  101. {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_dict_output.py +0 -0
  102. {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_hello.py +0 -0
  103. {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_literal_choice.py +0 -0
  104. {thunc-0.2.2 → thunc-0.2.3}/pytest-temporal.ini +0 -0
  105. {thunc-0.2.2 → thunc-0.2.3}/tests/future_types.py +0 -0
  106. {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/conftest.py +0 -0
  107. {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/process_worker.py +0 -0
  108. {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/test_storage.py +0 -0
  109. {thunc-0.2.2 → thunc-0.2.3}/tests/test_benchmarks.py +0 -0
  110. {thunc-0.2.2 → thunc-0.2.3}/tests/test_cache.py +0 -0
  111. {thunc-0.2.2 → thunc-0.2.3}/tests/test_calls.py +0 -0
  112. {thunc-0.2.2 → thunc-0.2.3}/tests/test_cli.py +0 -0
  113. {thunc-0.2.2 → thunc-0.2.3}/tests/test_jev.py +0 -0
  114. {thunc-0.2.2 → thunc-0.2.3}/tests/test_permissions.py +0 -0
  115. {thunc-0.2.2 → thunc-0.2.3}/tests/test_profiling.py +0 -0
  116. {thunc-0.2.2 → thunc-0.2.3}/tests/test_schema.py +0 -0
  117. {thunc-0.2.2 → thunc-0.2.3}/thunc/__main__.py +0 -0
  118. {thunc-0.2.2 → thunc-0.2.3}/thunc/cache.py +0 -0
  119. {thunc-0.2.2 → thunc-0.2.3}/thunc/config.py +0 -0
  120. {thunc-0.2.2 → thunc-0.2.3}/thunc/decorator.py +0 -0
  121. {thunc-0.2.2 → thunc-0.2.3}/thunc/errors.py +0 -0
  122. {thunc-0.2.2 → thunc-0.2.3}/thunc/mcp_relay.py +0 -0
  123. {thunc-0.2.2 → thunc-0.2.3}/thunc/permissions.py +0 -0
  124. {thunc-0.2.2 → thunc-0.2.3}/thunc/profiling.py +0 -0
  125. {thunc-0.2.2 → thunc-0.2.3}/thunc/prompts.py +0 -0
  126. {thunc-0.2.2 → thunc-0.2.3}/thunc/py.typed +0 -0
  127. {thunc-0.2.2 → thunc-0.2.3}/thunc/runs.py +0 -0
  128. {thunc-0.2.2 → thunc-0.2.3}/thunc/schema.py +0 -0
  129. {thunc-0.2.2 → thunc-0.2.3}/thunc/store.py +0 -0
  130. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/__init__.py +0 -0
  131. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/adapters.py +0 -0
  132. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/client.py +0 -0
  133. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/models.py +0 -0
  134. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/storage.py +0 -0
  135. {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/worker.py +0 -0
@@ -0,0 +1,74 @@
1
+ thunc 0.2.2 makes **agents use their tools reliably**, especially on Claude Code, and makes calls faster. Agents on Claude Code now make native tool calls instead of writing each action as JSON text, a failed step is retried instead of ending the run, and the agent's tools fill gaps that a new tool-use benchmark found.
2
+
3
+ | 8 tool-heavy tasks, Claude Sonnet 5.5 | Passed | Seconds per task | $ per task |
4
+ |---|---|---|---|
5
+ | thunc 0.2.1 agents on Claude Code | 12/24 | 99 | 0.084 |
6
+ | **thunc 0.2.2 agents on Claude Code** | **24/24** | **10** | **0.022** |
7
+ | Claude Code itself, for reference | 24/24 | 11 | 0.069 |
8
+
9
+ On Claude Opus 5.5 both versions passed every task, but 0.2.2 took 15 seconds and $0.051 a task instead of 44 seconds and $0.264.
10
+
11
+ ```bash
12
+ pip install --upgrade thunc
13
+ ```
14
+
15
+ ```python
16
+ import thunc
17
+
18
+ thunc.configure(backend="claude-code") # agents now make native tool calls here
19
+ fixer = thunc.Agent("fixer", workdir=".", permissions=["write:src/**", "run:pytest"])
20
+
21
+
22
+ @fixer.task
23
+ def make_tests_pass() -> str:
24
+ """Run the tests in services/api and fix what fails. Say what you changed."""
25
+ ...
26
+
27
+
28
+ run = fixer.run(make_tests_pass) # run.commands, run.files_changed, and any retries in run.session
29
+ ```
30
+
31
+ ```bash
32
+ thunc run --profile my_script.py # where the time went: model time, thunc's own, per tool
33
+ ```
34
+
35
+ **Agents on Claude Code make native tool calls** ([#46](https://github.com/Eltarras/thunc/pull/46))
36
+ - The agent's tools are an MCP server that one `claude -p` process per run calls, and thunc carries out each call with its own tools, permissions and run record. Before, the model wrote each action as JSON text. Current models slip back into their trained tool calls: they invented tool results and ran on to the timeout, or the CLI refused a tool call it couldn't parse.
37
+ - Calls from one model reply count as one step. A turn without a tool call gets a nudge, and a turn that ends in an error gets a second chance, both in the same session.
38
+ - **Fallback:** with a `claude` CLI too old for these options, or with MCP servers turned off by a policy, a run falls back to the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
39
+ - The CLI runs in the agent's `workdir`, so the environment details it adds to the prompt name the right folder.
40
+
41
+ **Steadier runs** ([#46](https://github.com/Eltarras/thunc/pull/46))
42
+ - **A failed step is retried** twice before the run fails, and each retry is logged in the run record. This covers timeouts, lost connections, rate limits, server errors and CLI calls that end in an error (`thunc.errors.TransientError`). Before, one bad call ended the run.
43
+ - A text-protocol step on Claude Code or Codex now times out after 120 seconds instead of the whole 300-second `timeout`.
44
+
45
+ **Better tools for agents** ([#46](https://github.com/Eltarras/thunc/pull/46))
46
+ - **`run` takes `cwd`**: a folder inside `workdir` to run the command in.
47
+ - **The `shell` permission** (opt-in) runs command lines through the system shell, so pipes, `&&`, `cd` and redirects work. It can't be combined with `!run:` rules, since a shell command line can't be checked word by word.
48
+ - **`search` takes `glob`**: `*.py` matches by file name, `src/**/*.ts` by path.
49
+ - **Git-aware `list` and `search`**: in a git repository they leave out what git ignores, so a stale `build/` copy no longer crowds out the real source.
50
+ - **Long command output keeps its start and its end**, so an agent sees the first error as well as the summary.
51
+
52
+ **Faster calls** ([#45](https://github.com/Eltarras/thunc/pull/45))
53
+ - **The `anthropic` and `openai` backends reuse their connections**: one SDK client per process instead of a new TCP and TLS handshake each call. In a local benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
54
+ - **Text-protocol agents can act several times per reply** (Codex and `protocol="text"`), with a JSON array of independent actions. On Codex, the prompt eval took fewer replies (fix / review / analysis: 6.0 / 4.8 / 5.0 → 5.0 / 3.0 / 3.6) and less time (37 / 27 / 27 s → 29 / 19 / 21 s).
55
+ - **The `codex` backend returns as soon as the answer arrives**, without waiting about 0.4 s for Codex to shut down.
56
+ - **`thunc run --profile SCRIPT`** (or `-m module`) reports where a program's time went, per function: calls, cache hits, retries, failures, model time against thunc's own, and time per tool for agents. `python -m benchmarks` runs a dependency-free benchmark suite.
57
+
58
+ **Tested**
59
+ - Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows. Temporal integration tests run on Python 3.10 and 3.14.
60
+ - The MCP path has an end-to-end test with a fake `claude` that starts the real tool server, covering parallel and serial calls, permissions, errors and the fallback.
61
+ - Live: the tool-use benchmark (`live_tests/bench_tooluse.py`, with its report in `live_tests/bench_tooluse_report.md`) on Sonnet 5.5 and Opus 5.5, the prompt eval on Claude Code (45/45), and the live tests on Claude Code.
62
+ - The agent loops on the Claude and OpenAI APIs haven't had a live tool-use benchmark yet.
63
+
64
+ **Behaviour changes**
65
+ - Agents on `claude-code` make native tool calls by default, and fall back to the text protocol if they can't.
66
+ - In a git repository, `list` and `search` leave out what git ignores. A folder named explicitly is still listed and searched.
67
+ - Long command output keeps its start as well as its end.
68
+ - The `claude-code` backend loads none of your Claude Code settings (`--setting-sources ""`). No `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so a folder you point an agent at can't instruct it unless you pass `follow=`.
69
+ - Durable runs on Claude Code still use the text protocol.
70
+ - Nothing is removed.
71
+
72
+ Still beta: expect bugs, and the API may change. Known issues are listed in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
73
+
74
+ **Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.1...v0.2.2
@@ -0,0 +1,80 @@
1
+ thunc 0.2.3 is the last 0.2 release. It makes **native tool calls the way agents work wherever they run**: on Codex now, as on Claude Code, and in durable runs on Claude Code, which pick up after a crash mid-call. Durable agents can take `tools=`, the Claude API agent path no longer loses runs to `max_tokens` or a stalled reply, and `edit` can change many places at once.
2
+
3
+ | 8 tool-heavy tasks, Claude Sonnet 5.5, 3 runs each | Passed | Seconds per task | $ per task |
4
+ |---|---|---|---|
5
+ | thunc 0.2.2 agents, text protocol (Codex, durable runs, fallbacks) | 20/24 | 47 | 0.084 |
6
+ | **thunc 0.2.3 agents, text protocol** | **24/24** | **27** | 0.082 |
7
+ | thunc 0.2.3 agents on Claude Code, native calls | 24/24 | 8 | 0.021 |
8
+ | Claude Code itself, for reference | 24/24 | 10 | 0.078 |
9
+
10
+ Through the Claude API, thunc's agents passed 8 of 8 on Sonnet 5.5 ($0.023 a task) and on Opus 5.5 ($0.049 a task). On Codex, native calls took 30 seconds a task, against 45 for the text protocol 0.2.2 used there and 27 for Codex itself (16 of 16 each).
11
+
12
+ ```bash
13
+ pip install --upgrade thunc
14
+ ```
15
+
16
+ ```python
17
+ import thunc
18
+
19
+ thunc.configure(backend="codex") # agents now make native tool calls here too
20
+
21
+
22
+ def find_issue(title: str) -> int: # a tool is your own code, with a docstring for the model
23
+ """Find an open issue by title. Returns its number, or 0."""
24
+ return tracker.find(title)
25
+
26
+
27
+ triage = thunc.Agent("triage", workdir=".", tools=[find_issue], effort="high")
28
+
29
+
30
+ @triage.task
31
+ def flaky_tests() -> list[str]:
32
+ """Run the tests three times and list the ones that fail only sometimes."""
33
+ ...
34
+ ```
35
+
36
+ ```python
37
+ # On a Temporal worker: durable runs take the agent's own tools now, and name those that may run
38
+ # again after a crash.
39
+ registry.agent_task("triage", flaky_tests, version="1", workspace_id="repo", retry_safe_tools=["find_issue"])
40
+ ```
41
+
42
+ **Native tool calls on Codex** ([#55](https://github.com/Eltarras/thunc/pull/55))
43
+ - The agent's tools are an MCP server that Codex calls, through the same relay as Claude Code: thunc carries out each call with its own tools, permissions and run record. Codex's own tools and your `~/.codex` config stay out, its sandbox stays read-only, and only thunc's server is approved to run without asking.
44
+ - A turn that ends without `finish`, or fails, is continued with `codex exec resume`. The run's Codex session is deleted when the run ends.
45
+ - **Fallback:** when Codex can't start them, a run uses the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
46
+
47
+ **Durable runs** ([#56](https://github.com/Eltarras/thunc/pull/56), [#57](https://github.com/Eltarras/thunc/pull/57))
48
+ - **On Claude Code, native calls in segments.** One activity keeps one `claude -p` process for many model replies, and checkpoints the run and Claude Code's session before each reply's calls. If the worker stops or the CLI dies, the retried activity restores the session and continues it with `--resume`. The call that was in flight, asked for again, gets its old journal entry: it's replayed, waits for `resolve()`, or runs. Never twice.
49
+ - **`tools=`.** Each call of the agent's own functions is journaled like a command: recorded before it runs and after, replayed when completed, and waiting for `resolve()` if a worker stopped mid-call. `retry_safe_tools=` names the ones that may simply run again.
50
+
51
+ **The Claude API agent path** ([#52](https://github.com/Eltarras/thunc/pull/52))
52
+ - **`Agent(effort=...)`** on every backend. Claude 4.6 and later on the API get `"high"` by default; Claude Opus 5.5's own default is `"medium"`, which is low for agentic coding.
53
+ - Replies are streamed with room for 64,000 tokens. A reply cut off at `max_tokens` gets an error result for its calls instead of ending the run, `pause_turn` carries on, and a reply that sends nothing for `timeout` seconds is stopped and asked again (one Opus reply in the benchmark sent nothing for an hour).
54
+ - Old tool results are cleared on long runs (context editing), tools are strict where the model and schema allow, and connection errors, rate limits and server errors are retried as a step.
55
+
56
+ **Steadier agents everywhere** ([#51](https://github.com/Eltarras/thunc/pull/51), [#52](https://github.com/Eltarras/thunc/pull/52), [#53](https://github.com/Eltarras/thunc/pull/53))
57
+ - **The text protocol reads replies that aren't only the action**: the first complete action amid prose, a fence, `<invoke>` markup or made-up results, with a note to the model to keep to JSON. On Claude Code, a text step stops as soon as its action has arrived.
58
+ - **`finish` in the same reply as other calls is refused** (except beside `remember`): in the benchmark, a run returned a guess written before its own read came back.
59
+ - **The model is told when few steps are left**, in its last three replies before `max_steps`.
60
+
61
+ **Better tools** ([#50](https://github.com/Eltarras/thunc/pull/50), [#51](https://github.com/Eltarras/thunc/pull/51), [#49](https://github.com/Eltarras/thunc/pull/49))
62
+ - **`edit` takes `replace_all`, and `edits`** for several changes to one file in one call, all made or none.
63
+ - **`edit` no longer needs a prior `read`**: it only changes text the agent quotes exactly. `write` still replaces only a file read with `read`.
64
+ - **Large system prompts work on Claude Code**: the prompt goes in a file, not on the command line, where large `follow=` files could pass the operating system's limit.
65
+
66
+ **Tested**
67
+ - Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows; Temporal integration tests on Python 3.10 and 3.14, including a Claude Code run whose CLI dies mid-run. Tests can't reach a real `claude`, `codex` or `jev`.
68
+ - Live: the tool-use benchmark on Claude Code (Sonnet 5.5), the Claude API (Sonnet 5.5 and Opus 5.5) and Codex, with the results in `live_tests/bench_tooluse_report.md`; a durable run on real Claude Code with the CLI killed after its first tool result, which finished with each effect done once.
69
+
70
+ **Behaviour changes**
71
+ - Agents on `codex` make native tool calls by default, and fall back to the text protocol if they can't.
72
+ - Durable runs on `claude-code` make native tool calls. Runs already in progress keep the text protocol.
73
+ - Agents on the Claude API think at effort `high` on Claude 4.6 and later.
74
+ - `edit` no longer needs a prior `read`; `write` still does.
75
+ - `finish` in the same reply as other calls is refused, except beside `remember`.
76
+ - Nothing is removed.
77
+
78
+ **Known issues:** on Codex, native calls count each tool call as a step, so a long task reaches `max_steps` sooner, and their tokens aren't reported. Durable runs on Codex use the text protocol. On a fix in a long file, native calls take about twice Claude Code's steps. The full list is in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
79
+
80
+ **Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.2...v0.2.3
@@ -0,0 +1,46 @@
1
+ name: Release notes
2
+
3
+ # Run by hand (Actions > Release notes > Run workflow) with a version like 0.2.2. It writes the
4
+ # notes in .github/release-notes/v<version>.md to that release, and sets its title to
5
+ # "v<version> (beta)". A release that doesn't exist yet is created as a draft pre-release on the
6
+ # commit the workflow runs from, for you to review and publish. Editing or drafting a release
7
+ # doesn't trigger publish.yml, which runs only when you publish one.
8
+
9
+ on:
10
+ workflow_dispatch:
11
+ inputs:
12
+ version:
13
+ description: "Version, like 0.2.2"
14
+ required: true
15
+
16
+ permissions:
17
+ contents: write # create and edit releases
18
+
19
+ jobs:
20
+ notes:
21
+ runs-on: ubuntu-latest
22
+ steps:
23
+ - uses: actions/checkout@v4
24
+ - name: Write the notes to the release
25
+ env:
26
+ GH_TOKEN: ${{ github.token }}
27
+ VERSION: ${{ inputs.version }}
28
+ run: |
29
+ if [[ ! "$VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
30
+ echo "::error::version must look like 0.2.2, not '$VERSION'"
31
+ exit 1
32
+ fi
33
+ tag="v$VERSION"
34
+ notes=".github/release-notes/$tag.md"
35
+ if [[ ! -f "$notes" ]]; then
36
+ echo "::error::$notes doesn't exist; add it first"
37
+ exit 1
38
+ fi
39
+ if gh release view "$tag" > /dev/null 2>&1; then
40
+ gh release edit "$tag" --title "$tag (beta)" --notes-file "$notes"
41
+ echo "Updated the notes of $tag."
42
+ else
43
+ gh release create "$tag" --draft --prerelease --target "$GITHUB_SHA" \
44
+ --title "$tag (beta)" --notes-file "$notes"
45
+ echo "Created $tag as a draft pre-release."
46
+ fi
@@ -3,6 +3,123 @@
3
3
  All notable changes to thunc. The full notes for each release are on the
4
4
  [releases page](https://github.com/Eltarras/thunc/releases).
5
5
 
6
+ ## 0.2.3 (beta)
7
+
8
+ **Native tool calls everywhere agents run, durable runs that keep them, and the last of 0.2.**
9
+ Agents on Codex make native tool calls through the same MCP relay as Claude Code, durable runs on
10
+ Claude Code do too and pick up after a crash mid-call, and durable agents can take `tools=`. On the
11
+ Claude API, agents stream their replies, think at effort `high`, and no longer lose a run to
12
+ `max_tokens` or a stalled reply. `edit` can replace every occurrence or make several changes at
13
+ once, and no longer needs a prior `read`. In the tool-use benchmark (`live_tests/bench_tooluse.py`,
14
+ 8 tasks, Claude Sonnet 5.5, 3 runs each), every harness passed 24 of 24, the text protocol included
15
+ (20 of 24 in 0.2.2) in about half the time; through the Claude API, Sonnet 5.5 and Opus 5.5 passed
16
+ 8 of 8; on Codex, native calls took 30 seconds a task against 45 for the text protocol.
17
+
18
+ ### Behavior changes
19
+
20
+ - **Agents on `codex` make native tool calls**, as on Claude Code since 0.2.2 (see Added). When
21
+ Codex can't start them, a run falls back to the text protocol with a warning; `protocol="text"`
22
+ keeps the old way, and durable runs on Codex still use it.
23
+ - **`finish` called in the same reply as other calls is refused** (except beside `remember`): its
24
+ value can't account for results the model hasn't seen yet. The other calls run, and the model is
25
+ told to call `finish` on its own. In the tool-use benchmark, a run on the text protocol batched
26
+ `[search, read, finish 0.0]` and returned the guess. Every protocol and durable runs get it.
27
+ - **Agents on the Claude API think at effort `high` by default** on Claude 4.6 and later. Claude
28
+ Opus 5.5's own default is `medium`, which is low for agentic coding. `effort=` changes it (below).
29
+ - **`edit` no longer needs the file to have been read first.** It only changes text the agent
30
+ quotes exactly, so it can't overwrite what the agent hasn't seen. With the `shell` permission,
31
+ agents often read files with `cat`, and `edit` refused them until they read the file again with
32
+ `read`: 7 times in 24 runs of the tool-use benchmark. A file the agent did read must still not
33
+ have changed on disk since, and `write` still replaces only a file read with `read` (a file
34
+ changed by an edit alone still counts as unread). The `run` tool's description, with `shell`, now
35
+ says to read files with `read`.
36
+
37
+ ### Added
38
+
39
+ - **Durable runs on Claude Code make native tool calls.** A run goes in segments: one activity keeps
40
+ one `claude -p` process for many model replies, and before each reply's calls are carried out it
41
+ saves a checkpoint of the run and of Claude Code's session. If the worker stops or the CLI dies,
42
+ the retried activity restores the session and continues it with `--resume`; Claude Code marks the
43
+ call that was in flight as interrupted, the model asks for it again, and it gets the journal entry
44
+ it had, so it's replayed, waits for `resolve()`, or runs (one the model doesn't ask for again
45
+ keeps its entry). The session is removed when the run ends. When Claude Code can't start native
46
+ calls, the run goes on with the text protocol. Runs already in progress keep the text protocol.
47
+ - **Durable agents can have `tools=`.** Each call of one of the agent's own functions is journaled
48
+ like a command: the intent is recorded before it runs and its result after, so a retried activity
49
+ replays the result instead of calling it again, and a call interrupted by a worker stopping waits
50
+ for `resolve()`. `registry.agent_task(..., retry_safe_tools=["find_issue"])` names the tools that
51
+ may run again instead. A tool's description, arguments and retry marking are part of the task's
52
+ fingerprint, so changing one needs a new version; tasks without tools keep their fingerprint.
53
+ - **Native calls on `codex`**: the agent's tools are an MCP server (thunc's relay, given with
54
+ `-c mcp_servers.thunc.*`) that Codex calls; thunc carries out each call with its own tools,
55
+ permissions and run record. Each `codex exec` is a turn: one that ends without `finish` is
56
+ continued with `codex exec resume`, as is one that fails (twice at most). Codex's own tools and
57
+ your `~/.codex` config stay out and its sandbox stays read-only; only thunc's server is approved
58
+ to run without asking, with a tool timeout above `command_timeout`. Resuming needs the session
59
+ saved, so thunc deletes the run's Codex session (`codex delete --force`) when the run ends.
60
+ - **`Agent(effort=...)`**: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`, on every backend
61
+ (`output_config.effort` on the Claude API, `reasoning.effort` on OpenAI, `--effort` on Claude Code,
62
+ `model_reasoning_effort` on Codex; the last two go up to `"xhigh"`). Recorded in `agent.json` only
63
+ when set, so durable tasks registered without it keep their fingerprint.
64
+ - **The model is told when few steps are left.** In its last three replies before `max_steps`, the
65
+ last tool result says how many replies remain, so the model can finish with what it has instead
66
+ of being cut off. Every protocol and durable runs get it; the run record keeps each tool's output.
67
+ - **`edit` can replace every occurrence, and make several changes in one call.** With
68
+ `"replace_all": true`, every occurrence of `old` is replaced and the result gives the count. With
69
+ `"edits": [{"old": ..., "new": ..., "replace_all"?: ...}, ...]` (at most 50) instead of `old` and
70
+ `new`, the changes apply in order, each to the text the ones before it left; if one fails, none is
71
+ made, and the error names it. In the tool-use benchmark, models renamed a symbol by writing a
72
+ throwaway script instead of making 26 separate edits, and took twice Claude Code's turns on a
73
+ multi-spot fix. In a durable run, a multi-edit is one effect, recovered as a whole.
74
+
75
+ ### Changed
76
+
77
+ - **The text protocol reads replies that aren't only the action** (finding 1 of the tool-use
78
+ benchmark report). Models trained for native tool calls often wrap the action in prose, a code
79
+ fence or `<invoke>` markup, or carry on past it with results they make up: on Sonnet 5.5, 25% of
80
+ text-protocol replies were sent back as "not valid JSON" with a correct action inside. Now the
81
+ first complete action (or array of them) in the reply is used, with literal newlines in its
82
+ strings accepted, and a reply written only as `<invoke name="...">` markup is read as its calls,
83
+ each argument in its tool's type (or as one JSON `args` parameter). A reply with no action in it
84
+ is still sent back, as before. A reply read this way runs, but its results carry a note to reply
85
+ with the JSON action alone: without it, a model that slipped into markup was never corrected, and
86
+ on Sonnet 5.5 fell into repeating empty markup until the step timed out.
87
+ This is the text protocol on Codex, `protocol="text"`, durable runs on Codex, and Claude
88
+ Code's fallback from native calls.
89
+ - **A text-protocol step on Claude Code stops once its action is complete.** The reply is streamed
90
+ (`--output-format stream-json --include-partial-messages`) and the CLI is stopped as soon as a
91
+ complete action has arrived, rather than left to make up the tool's result until the step times
92
+ out (7 of 8 replayed first steps on Sonnet ran past 120 seconds that way). A batch that has begun
93
+ is waited for, and two blocks of `<invoke>` markup end the step too (a model repeating itself).
94
+ Plain `@thunc.function` calls on Claude Code aren't streamed.
95
+ - **The Claude API agent path keeps runs going** (finding 7 of the tool-use benchmark report):
96
+ - Replies are streamed with `max_tokens=64000` (was 16,000 without streaming), so a large write
97
+ fits.
98
+ - A reply cut off at `max_tokens` no longer ends the run: its tool calls aren't run and get an
99
+ error result saying so, and the model is asked again. Two in a row end the run.
100
+ - `pause_turn` is asked to carry on (up to 6 times in a row) instead of ending the run.
101
+ `model_context_window_exceeded` ends it with a message that says so.
102
+ - On Claude 4.6 and later, the API clears old tool results on long runs (context editing, beta
103
+ `context-management-2025-06-27`).
104
+ - Tools are `strict` (arguments guaranteed to match their schema) on the models that support it,
105
+ when the schema allows it: the built-in tools except `edit`, `remember`, and `finish` and custom
106
+ tools whose schema is closed.
107
+ - A reply that sends nothing for `timeout` seconds (300 by default) is stopped and asked again. The
108
+ SDK's read timeout doesn't catch it, as the API's keep-alive pings count as reading: in the
109
+ tool-use benchmark, one reply on Claude Opus 5.5 sent nothing for an hour.
110
+ - A lost connection, a rate limit or a server error on the Claude and OpenAI APIs is retried as a
111
+ step, like the CLI backends' errors, instead of ending the run.
112
+
113
+ ### Fixed
114
+
115
+ - **A large system prompt no longer stops the `claude-code` backend from starting.** It went on the
116
+ command line, so large `follow=` files and memory could pass the operating system's limit on its
117
+ length (128 KB for one argument on Linux, 32,767 characters for the whole line on Windows), and
118
+ starting `claude` failed with a raw `OSError: Argument list too long`. The prompt now goes in a
119
+ temporary file (`--system-prompt-file`), as it already did for agents' native calls and on Codex.
120
+ This covers `@thunc.function` and `thunc.call`, agents on the text protocol, and durable runs on
121
+ Claude Code. A command line that is still too long raises a `ThuncError` that gives its size.
122
+
6
123
  ## 0.2.2 (beta)
7
124
 
8
125
  **Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
@@ -86,8 +86,9 @@ THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
86
86
  5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
87
87
  `Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
88
88
  the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
89
- A CLI that can call MCP tools can get native calls the way Claude Code does
90
- (`thunc/claude_code.py`, tested with a fake CLI in `tests/test_claude_code_agent.py`).
89
+ A CLI that can call MCP tools can get native calls the way Claude Code and Codex do: through the
90
+ relay in `thunc/relay.py` (see `thunc/claude_code.py` and `thunc/codex.py`, tested with fake CLIs
91
+ in `tests/test_claude_code_agent.py` and `tests/test_codex_agent.py`).
91
92
  Raise `TransientError` for failures worth asking again, so agent runs retry them.
92
93
  Agents and durable runs refuse typed backends.
93
94
  6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
@@ -153,20 +154,20 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
153
154
 
154
155
  **Open: agents and durable runs.**
155
156
 
156
- - Durable runs on `claude-code` use the JSON text protocol, not native calls: the MCP path keeps
157
- one CLI process for the whole run, which a durable step can't snapshot. The text protocol is the
158
- weaker one: in `live_tests/bench_tooluse.py` on Sonnet 5.5 it passed 12 of 24 runs, and 20 of 24
159
- with step retries (measured before action arrays), against 24 of 24 for native calls.
160
- - The `shell` permission isn't tested on Windows: its test is skipped there, so `cmd /c` has never
161
- run in CI.
162
- - With `shell`, a file read with `cat` doesn't count as read for `edit`, which refuses until the
163
- agent reads it with `read` (the no-blind-overwrite rule). Agents work around it with a short
164
- `read`, at the cost of a step.
165
- - The agent loops on the Claude and OpenAI APIs have no live tool-use benchmark yet (only
166
- `live_tests/eval_prompts.py`); their `max_tokens`, effort and stop-reason handling were reviewed
167
- from the code only (`live_tests/bench_tooluse_report.md`, finding 7).
168
- - Durable agents can't use `tools=` yet: the effects of the program's own functions can't be
169
- journaled.
157
+ - On Codex, native calls count each tool call as a step: Codex's events don't say which calls came
158
+ from the same model reply. A long task reaches `max_steps` sooner than on Claude Code (22 steps on
159
+ `rename` in the tool-use benchmark, against `max_steps=40`).
160
+ - Tokens aren't reported for native runs on Codex: `codex exec --json` reports usage at the end of
161
+ a turn, and a run ends inside one when the model calls `finish`.
162
+ - Durable runs on `codex` use the JSON text protocol, not native calls.
163
+ - On a fix in a long file (`deep_fix` in the tool-use benchmark), native calls take about twice
164
+ Claude Code's steps (8 against 4.3), reading around the file in pages. A larger `read` limit is
165
+ the next thing to measure.
166
+ - Durable runs on Claude Code save the whole session file at each checkpoint, so a long run's saved
167
+ state grows with every reply, and counts toward the 16 MiB limit. A CLI that dies leaves Claude
168
+ Code's own `~/.claude/sessions/<pid>.json` behind.
169
+ - The `shell` permission's own test (pipes, `cd`, redirects) is skipped on Windows; only a single
170
+ command line runs through `cmd /c` in CI.
170
171
  - Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
171
172
  tombstones are kept forever. There's no context summarization either, so a long run fails once
172
173
  its saved state passes 16 MiB.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: thunc
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: think + function: call an LLM like a typed Python function.
5
5
  Project-URL: Homepage, https://eltarras.github.io/thunc/
6
6
  Project-URL: Documentation, https://eltarras.github.io/thunc/docs/
@@ -203,8 +203,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
203
203
  `pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
204
204
  any server that speaks the OpenAI Responses API.
205
205
  - `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
206
- Both run with their own tools turned off, so the model can only answer; an agent on
207
- `claude-code` gets only its thunc tools, as native calls (see Agents). `codex` also ignores
206
+ Both run with their own tools turned off, so the model can only answer; an agent on either
207
+ gets only its thunc tools, as native calls (see Agents). `codex` also ignores
208
208
  `~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
209
209
  login still works. Pick the model with `configure(model=...)` or `model=`.
210
210
  - `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
@@ -305,10 +305,15 @@ calls are native too: the agent's tools are an MCP server that one `claude -p` p
305
305
  calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
306
306
  can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
307
307
  text protocol below instead, with a warning, and so do later runs in the process;
308
- `protocol="native"` fails instead. On Codex the model replies with JSON actions as text: one at a
309
- time, or several independent ones (reading three files) as a JSON array, which saves turns.
310
- `protocol="text"` uses that way on any backend, for example with a server behind `OPENAI_BASE_URL`
311
- that has no function calling. (Durable runs on Claude Code use it too.)
308
+ `protocol="native"` fails instead. On Codex it's the same: each `codex exec` is a turn in which
309
+ the model calls the agent's tools through the MCP server, a turn that ends without `finish` is
310
+ continued with `codex exec resume`, Codex's own tools stay off and its sandbox read-only, and the
311
+ run's Codex session is deleted when the run ends. With `protocol="text"` the model replies with
312
+ JSON actions as text instead: one at a time, or several independent ones (reading three files) as
313
+ a JSON array, which saves turns. That works on any backend, for example with a server behind
314
+ `OPENAI_BASE_URL` that has no function calling. (Durable runs on Codex use it.) A reply that wraps its
315
+ action in prose or tool-call markup, or carries on past it, is read for its first complete action,
316
+ and on Claude Code the step stops as soon as that action has arrived.
312
317
  The `jev` backend only answers typed questions and cannot run agents, even for a task returning
313
318
  `bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
314
319
  or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
@@ -333,14 +338,18 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
333
338
  the run carries on. Bad rules fail when the agent is declared.
334
339
  - **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
335
340
  such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
336
- exactly once) when a write rule allows it; `run` when a run or shell rule allows it; and
341
+ exactly once, or every occurrence with `replace_all`; several changes to one file can go in one
342
+ call as `edits`, all made or none) when a write rule allows it; `run` when a run or shell rule
343
+ allows it; and
337
344
  `remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
338
345
  outside are refused, and the rules are checked on where a link really leads. Files the agent may
339
346
  not read are left out of `list` and `search`, and so is what git ignores, in a git repository
340
347
  (build output, caches, vendored code); a folder named explicitly is still listed and searched.
341
- - **No blind overwrites.** A file is only replaced or edited after the agent read it in the same
342
- run, and only if it hasn't changed on disk since. There is no undo, so run agents that write in a
343
- git repository with a clean tree, and review their changes with `git diff`.
348
+ - **No blind overwrites.** A file is only replaced (`write`) after the agent read it with `read` in
349
+ the same run, and only if it hasn't changed on disk since. An `edit` needs no read, because it
350
+ only changes text the agent quotes exactly, but a file the agent did read must not have changed
351
+ since. There is no undo, so run agents that write in a git repository with a clean tree, and
352
+ review their changes with `git diff`.
344
353
  - **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
345
354
  permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
346
355
  agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
@@ -379,9 +388,15 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
379
388
  `system=thunc.prompts.CODING + "\n\nTarget Python 3.10."`.
380
389
  - **Time:** `max_steps=40` bounds the model replies in a run, and `timeout=` (seconds) bounds the
381
390
  run's time. It's checked before each model call; a command's time limit is cut to the time left.
391
+ In its last three replies before `max_steps`, the model is told how many are left, so it can
392
+ finish with what it has.
393
+ - **`effort=`** sets how hard the model thinks: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`
394
+ (`openai` and `codex` go up to `"xhigh"`). By default it's `"high"` on the `anthropic` backend for
395
+ Claude 4.6 and later (Claude Opus 5.5's own default is `"medium"`, low for agentic coding), and
396
+ each backend's own default elsewhere.
382
397
  - **Options:** `thunc.Agent(name, *, workdir, system=None, permissions=(), env=None,
383
398
  command_timeout=120, follow=False, protocol=None, tools=(), timeout=None, max_steps=40,
384
- retries=2, backend=None, model=None)`, and `@agent.task(instructions=..., ensure=...)`.
399
+ retries=2, backend=None, model=None, effort=None)`, and `@agent.task(instructions=..., ensure=...)`.
385
400
  `@thunc.agent(name, workdir=..., instructions=..., ensure=..., **options)` takes the same options.
386
401
  `async def` tasks work.
387
402
  - **What happened in a run.** Calling a task returns its value. `agent.run(task, *args)` runs it
@@ -401,7 +416,9 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
401
416
  A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
402
417
  a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
403
418
  each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
404
- seconds before it's retried. With tracing on, each run is also one line with every model reply.
419
+ seconds before it's retried. On the Claude API, a reply cut off at `max_tokens` doesn't end the
420
+ run: its tool calls get an error result saying so (twice in a row does). With tracing on, each run
421
+ is also one line with every model reply.
405
422
 
406
423
  **How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
407
424
  small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
@@ -458,8 +475,9 @@ print(run.value)
458
475
  ```
459
476
 
460
477
  Durable mode requires a Temporal service, a worker, and persistent storage on the
461
- same volume. File changes and memory updates use recovery receipts. Commands with
462
- uncertain outcomes pause for operator resolution instead of blindly running twice.
478
+ same volume. File changes and memory updates use recovery receipts. Commands, and the agent's
479
+ own `tools=` functions, with uncertain outcomes pause for operator resolution instead of blindly
480
+ running twice (`retry_safe_tools=` names functions that may run again).
463
481
  Temporal does not back up your workspace or guarantee exactly-once external effects.
464
482
 
465
483
  The [Temporal guide and runnable example](examples/temporal/README.md) cover service
@@ -491,6 +509,8 @@ thunc/
491
509
  runs.py thunc.Run and AgentError: what a run did
492
510
  native.py how a run talks to its backend: native tool calls or the text protocol
493
511
  claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
512
+ codex.py native tool calls on Codex, the same way
513
+ relay.py the run's end of the MCP server: the connection the calls come through
494
514
  store.py the agent's folder: memory, settings, run records, the lock
495
515
  prompts.py the agent's system prompt
496
516
  __main__.py the thunc command: thunc run [--profile], thunc cache list / clear
@@ -168,8 +168,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
168
168
  `pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
169
169
  any server that speaks the OpenAI Responses API.
170
170
  - `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
171
- Both run with their own tools turned off, so the model can only answer; an agent on
172
- `claude-code` gets only its thunc tools, as native calls (see Agents). `codex` also ignores
171
+ Both run with their own tools turned off, so the model can only answer; an agent on either
172
+ gets only its thunc tools, as native calls (see Agents). `codex` also ignores
173
173
  `~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
174
174
  login still works. Pick the model with `configure(model=...)` or `model=`.
175
175
  - `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
@@ -270,10 +270,15 @@ calls are native too: the agent's tools are an MCP server that one `claude -p` p
270
270
  calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
271
271
  can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
272
272
  text protocol below instead, with a warning, and so do later runs in the process;
273
- `protocol="native"` fails instead. On Codex the model replies with JSON actions as text: one at a
274
- time, or several independent ones (reading three files) as a JSON array, which saves turns.
275
- `protocol="text"` uses that way on any backend, for example with a server behind `OPENAI_BASE_URL`
276
- that has no function calling. (Durable runs on Claude Code use it too.)
273
+ `protocol="native"` fails instead. On Codex it's the same: each `codex exec` is a turn in which
274
+ the model calls the agent's tools through the MCP server, a turn that ends without `finish` is
275
+ continued with `codex exec resume`, Codex's own tools stay off and its sandbox read-only, and the
276
+ run's Codex session is deleted when the run ends. With `protocol="text"` the model replies with
277
+ JSON actions as text instead: one at a time, or several independent ones (reading three files) as
278
+ a JSON array, which saves turns. That works on any backend, for example with a server behind
279
+ `OPENAI_BASE_URL` that has no function calling. (Durable runs on Codex use it.) A reply that wraps its
280
+ action in prose or tool-call markup, or carries on past it, is read for its first complete action,
281
+ and on Claude Code the step stops as soon as that action has arrived.
277
282
  The `jev` backend only answers typed questions and cannot run agents, even for a task returning
278
283
  `bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
279
284
  or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
@@ -298,14 +303,18 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
298
303
  the run carries on. Bad rules fail when the agent is declared.
299
304
  - **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
300
305
  such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
301
- exactly once) when a write rule allows it; `run` when a run or shell rule allows it; and
306
+ exactly once, or every occurrence with `replace_all`; several changes to one file can go in one
307
+ call as `edits`, all made or none) when a write rule allows it; `run` when a run or shell rule
308
+ allows it; and
302
309
  `remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
303
310
  outside are refused, and the rules are checked on where a link really leads. Files the agent may
304
311
  not read are left out of `list` and `search`, and so is what git ignores, in a git repository
305
312
  (build output, caches, vendored code); a folder named explicitly is still listed and searched.
306
- - **No blind overwrites.** A file is only replaced or edited after the agent read it in the same
307
- run, and only if it hasn't changed on disk since. There is no undo, so run agents that write in a
308
- git repository with a clean tree, and review their changes with `git diff`.
313
+ - **No blind overwrites.** A file is only replaced (`write`) after the agent read it with `read` in
314
+ the same run, and only if it hasn't changed on disk since. An `edit` needs no read, because it
315
+ only changes text the agent quotes exactly, but a file the agent did read must not have changed
316
+ since. There is no undo, so run agents that write in a git repository with a clean tree, and
317
+ review their changes with `git diff`.
309
318
  - **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
310
319
  permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
311
320
  agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
@@ -344,9 +353,15 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
344
353
  `system=thunc.prompts.CODING + "\n\nTarget Python 3.10."`.
345
354
  - **Time:** `max_steps=40` bounds the model replies in a run, and `timeout=` (seconds) bounds the
346
355
  run's time. It's checked before each model call; a command's time limit is cut to the time left.
356
+ In its last three replies before `max_steps`, the model is told how many are left, so it can
357
+ finish with what it has.
358
+ - **`effort=`** sets how hard the model thinks: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`
359
+ (`openai` and `codex` go up to `"xhigh"`). By default it's `"high"` on the `anthropic` backend for
360
+ Claude 4.6 and later (Claude Opus 5.5's own default is `"medium"`, low for agentic coding), and
361
+ each backend's own default elsewhere.
347
362
  - **Options:** `thunc.Agent(name, *, workdir, system=None, permissions=(), env=None,
348
363
  command_timeout=120, follow=False, protocol=None, tools=(), timeout=None, max_steps=40,
349
- retries=2, backend=None, model=None)`, and `@agent.task(instructions=..., ensure=...)`.
364
+ retries=2, backend=None, model=None, effort=None)`, and `@agent.task(instructions=..., ensure=...)`.
350
365
  `@thunc.agent(name, workdir=..., instructions=..., ensure=..., **options)` takes the same options.
351
366
  `async def` tasks work.
352
367
  - **What happened in a run.** Calling a task returns its value. `agent.run(task, *args)` runs it
@@ -366,7 +381,9 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
366
381
  A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
367
382
  a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
368
383
  each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
369
- seconds before it's retried. With tracing on, each run is also one line with every model reply.
384
+ seconds before it's retried. On the Claude API, a reply cut off at `max_tokens` doesn't end the
385
+ run: its tool calls get an error result saying so (twice in a row does). With tracing on, each run
386
+ is also one line with every model reply.
370
387
 
371
388
  **How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
372
389
  small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
@@ -423,8 +440,9 @@ print(run.value)
423
440
  ```
424
441
 
425
442
  Durable mode requires a Temporal service, a worker, and persistent storage on the
426
- same volume. File changes and memory updates use recovery receipts. Commands with
427
- uncertain outcomes pause for operator resolution instead of blindly running twice.
443
+ same volume. File changes and memory updates use recovery receipts. Commands, and the agent's
444
+ own `tools=` functions, with uncertain outcomes pause for operator resolution instead of blindly
445
+ running twice (`retry_safe_tools=` names functions that may run again).
428
446
  Temporal does not back up your workspace or guarantee exactly-once external effects.
429
447
 
430
448
  The [Temporal guide and runnable example](examples/temporal/README.md) cover service
@@ -456,6 +474,8 @@ thunc/
456
474
  runs.py thunc.Run and AgentError: what a run did
457
475
  native.py how a run talks to its backend: native tool calls or the text protocol
458
476
  claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
477
+ codex.py native tool calls on Codex, the same way
478
+ relay.py the run's end of the MCP server: the connection the calls come through
459
479
  store.py the agent's folder: memory, settings, run records, the lock
460
480
  prompts.py the agent's system prompt
461
481
  __main__.py the thunc command: thunc run [--profile], thunc cache list / clear