thunc 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thunc-0.2.3/.github/release-notes/v0.2.2.md +74 -0
- thunc-0.2.3/.github/release-notes/v0.2.3.md +80 -0
- thunc-0.2.3/.github/workflows/release-notes.yml +46 -0
- {thunc-0.2.2 → thunc-0.2.3}/CHANGELOG.md +117 -0
- {thunc-0.2.2 → thunc-0.2.3}/CONTRIBUTING.md +17 -16
- {thunc-0.2.2 → thunc-0.2.3}/PKG-INFO +35 -15
- {thunc-0.2.2 → thunc-0.2.3}/README.md +34 -14
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/agents.html +17 -12
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/api.html +13 -9
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/backends.html +6 -5
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/caching.html +25 -9
- thunc-0.2.3/docs/docs/changelog.html +214 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/functions.html +5 -4
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/index.html +9 -5
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/jev.html +4 -3
- {thunc-0.2.2 → thunc-0.2.3}/docs/docs/temporal.html +9 -3
- {thunc-0.2.2 → thunc-0.2.3}/docs/index.html +1 -1
- {thunc-0.2.2 → thunc-0.2.3}/docs/site.css +6 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/sitemap.xml +10 -9
- {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/README.md +14 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/bench_tooluse.py +167 -9
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/bench_tooluse_report.md +76 -0
- {thunc-0.2.2 → thunc-0.2.3}/pyproject.toml +1 -1
- {thunc-0.2.2 → thunc-0.2.3}/tests/conftest.py +18 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/fake_claude.py +28 -1
- thunc-0.2.3/tests/fake_codex_mcp.py +142 -0
- thunc-0.2.3/tests/temporal/test_contract.py +250 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/test_runtime.py +110 -0
- thunc-0.2.3/tests/temporal/test_segments.py +160 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_agent.py +228 -11
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_backends.py +144 -4
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_claude_code_agent.py +14 -3
- thunc-0.2.3/tests/test_codex_agent.py +208 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_execution.py +34 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_native.py +190 -7
- thunc-0.2.3/tests/test_text_protocol.py +272 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/__init__.py +1 -1
- {thunc-0.2.2 → thunc-0.2.3}/thunc/agent.py +59 -25
- {thunc-0.2.2 → thunc-0.2.3}/thunc/backends.py +129 -33
- {thunc-0.2.2 → thunc-0.2.3}/thunc/claude_code.py +65 -130
- thunc-0.2.3/thunc/codex.py +276 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/core.py +15 -1
- {thunc-0.2.2 → thunc-0.2.3}/thunc/execution.py +23 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/native.py +275 -40
- thunc-0.2.3/thunc/relay.py +179 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/activities.py +193 -10
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/effects.py +11 -3
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/registry.py +37 -7
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/workflows.py +1 -1
- {thunc-0.2.2 → thunc-0.2.3}/thunc/tools.py +132 -25
- {thunc-0.2.2 → thunc-0.2.3}/uv.lock +1 -1
- thunc-0.2.2/tests/temporal/test_contract.py +0 -118
- {thunc-0.2.2 → thunc-0.2.3}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/demo-agent.gif +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/demo-function.gif +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/social-preview.png +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/workflows/ci.yml +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.github/workflows/publish.yml +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/.gitignore +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/LICENSE +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/README.md +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/__init__.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/__main__.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_agent.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_backends.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_cache.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_calls.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_schema.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_startup.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/bench_tools.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/fakes.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/benchmarks/harness.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/design/og-card.html +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/.nojekyll +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/apple-touch-icon.png +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/favicon-96.png +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/favicon.ico +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/favicon.svg +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/google2cd177e5e85b3c3e.html +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/jev.html +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/og.png +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/site.js +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/docs/temporal.html +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/dynamic_prompts.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/hello.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/jev_hello.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/jev_inbox.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/jev_with_claude.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/log_triage.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/repo_guide.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/support_inbox.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/application.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/client.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/pipeline.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/examples/temporal/worker.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/conftest.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/eval_prompts.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_agent.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_bool_decision.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_dict_output.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_hello.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/live_tests/test_literal_choice.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/pytest-temporal.ini +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/future_types.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/conftest.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/process_worker.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/temporal/test_storage.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_benchmarks.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_cache.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_calls.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_cli.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_jev.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_permissions.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_profiling.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/tests/test_schema.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/__main__.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/cache.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/config.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/decorator.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/errors.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/mcp_relay.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/permissions.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/profiling.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/prompts.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/py.typed +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/runs.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/schema.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/store.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/__init__.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/adapters.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/client.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/models.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/storage.py +0 -0
- {thunc-0.2.2 → thunc-0.2.3}/thunc/temporal/worker.py +0 -0
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
thunc 0.2.2 makes **agents use their tools reliably**, especially on Claude Code, and makes calls faster. Agents on Claude Code now make native tool calls instead of writing each action as JSON text, a failed step is retried instead of ending the run, and the agent's tools fill gaps that a new tool-use benchmark found.
|
|
2
|
+
|
|
3
|
+
| 8 tool-heavy tasks, Claude Sonnet 5.5 | Passed | Seconds per task | $ per task |
|
|
4
|
+
|---|---|---|---|
|
|
5
|
+
| thunc 0.2.1 agents on Claude Code | 12/24 | 99 | 0.084 |
|
|
6
|
+
| **thunc 0.2.2 agents on Claude Code** | **24/24** | **10** | **0.022** |
|
|
7
|
+
| Claude Code itself, for reference | 24/24 | 11 | 0.069 |
|
|
8
|
+
|
|
9
|
+
On Claude Opus 5.5 both versions passed every task, but 0.2.2 took 15 seconds and $0.051 a task instead of 44 seconds and $0.264.
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install --upgrade thunc
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
import thunc
|
|
17
|
+
|
|
18
|
+
thunc.configure(backend="claude-code") # agents now make native tool calls here
|
|
19
|
+
fixer = thunc.Agent("fixer", workdir=".", permissions=["write:src/**", "run:pytest"])
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@fixer.task
|
|
23
|
+
def make_tests_pass() -> str:
|
|
24
|
+
"""Run the tests in services/api and fix what fails. Say what you changed."""
|
|
25
|
+
...
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
run = fixer.run(make_tests_pass) # run.commands, run.files_changed, and any retries in run.session
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
thunc run --profile my_script.py # where the time went: model time, thunc's own, per tool
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
**Agents on Claude Code make native tool calls** ([#46](https://github.com/Eltarras/thunc/pull/46))
|
|
36
|
+
- The agent's tools are an MCP server that one `claude -p` process per run calls, and thunc carries out each call with its own tools, permissions and run record. Before, the model wrote each action as JSON text. Current models slip back into their trained tool calls: they invented tool results and ran on to the timeout, or the CLI refused a tool call it couldn't parse.
|
|
37
|
+
- Calls from one model reply count as one step. A turn without a tool call gets a nudge, and a turn that ends in an error gets a second chance, both in the same session.
|
|
38
|
+
- **Fallback:** with a `claude` CLI too old for these options, or with MCP servers turned off by a policy, a run falls back to the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
|
|
39
|
+
- The CLI runs in the agent's `workdir`, so the environment details it adds to the prompt name the right folder.
|
|
40
|
+
|
|
41
|
+
**Steadier runs** ([#46](https://github.com/Eltarras/thunc/pull/46))
|
|
42
|
+
- **A failed step is retried** twice before the run fails, and each retry is logged in the run record. This covers timeouts, lost connections, rate limits, server errors and CLI calls that end in an error (`thunc.errors.TransientError`). Before, one bad call ended the run.
|
|
43
|
+
- A text-protocol step on Claude Code or Codex now times out after 120 seconds instead of the whole 300-second `timeout`.
|
|
44
|
+
|
|
45
|
+
**Better tools for agents** ([#46](https://github.com/Eltarras/thunc/pull/46))
|
|
46
|
+
- **`run` takes `cwd`**: a folder inside `workdir` to run the command in.
|
|
47
|
+
- **The `shell` permission** (opt-in) runs command lines through the system shell, so pipes, `&&`, `cd` and redirects work. It can't be combined with `!run:` rules, since a shell command line can't be checked word by word.
|
|
48
|
+
- **`search` takes `glob`**: `*.py` matches by file name, `src/**/*.ts` by path.
|
|
49
|
+
- **Git-aware `list` and `search`**: in a git repository they leave out what git ignores, so a stale `build/` copy no longer crowds out the real source.
|
|
50
|
+
- **Long command output keeps its start and its end**, so an agent sees the first error as well as the summary.
|
|
51
|
+
|
|
52
|
+
**Faster calls** ([#45](https://github.com/Eltarras/thunc/pull/45))
|
|
53
|
+
- **The `anthropic` and `openai` backends reuse their connections**: one SDK client per process instead of a new TCP and TLS handshake each call. In a local benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
|
|
54
|
+
- **Text-protocol agents can act several times per reply** (Codex and `protocol="text"`), with a JSON array of independent actions. On Codex, the prompt eval took fewer replies (fix / review / analysis: 6.0 / 4.8 / 5.0 → 5.0 / 3.0 / 3.6) and less time (37 / 27 / 27 s → 29 / 19 / 21 s).
|
|
55
|
+
- **The `codex` backend returns as soon as the answer arrives**, without waiting about 0.4 s for Codex to shut down.
|
|
56
|
+
- **`thunc run --profile SCRIPT`** (or `-m module`) reports where a program's time went, per function: calls, cache hits, retries, failures, model time against thunc's own, and time per tool for agents. `python -m benchmarks` runs a dependency-free benchmark suite.
|
|
57
|
+
|
|
58
|
+
**Tested**
|
|
59
|
+
- Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows. Temporal integration tests run on Python 3.10 and 3.14.
|
|
60
|
+
- The MCP path has an end-to-end test with a fake `claude` that starts the real tool server, covering parallel and serial calls, permissions, errors and the fallback.
|
|
61
|
+
- Live: the tool-use benchmark (`live_tests/bench_tooluse.py`, with its report in `live_tests/bench_tooluse_report.md`) on Sonnet 5.5 and Opus 5.5, the prompt eval on Claude Code (45/45), and the live tests on Claude Code.
|
|
62
|
+
- The agent loops on the Claude and OpenAI APIs haven't had a live tool-use benchmark yet.
|
|
63
|
+
|
|
64
|
+
**Behaviour changes**
|
|
65
|
+
- Agents on `claude-code` make native tool calls by default, and fall back to the text protocol if they can't.
|
|
66
|
+
- In a git repository, `list` and `search` leave out what git ignores. A folder named explicitly is still listed and searched.
|
|
67
|
+
- Long command output keeps its start as well as its end.
|
|
68
|
+
- The `claude-code` backend loads none of your Claude Code settings (`--setting-sources ""`). No `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so a folder you point an agent at can't instruct it unless you pass `follow=`.
|
|
69
|
+
- Durable runs on Claude Code still use the text protocol.
|
|
70
|
+
- Nothing is removed.
|
|
71
|
+
|
|
72
|
+
Still beta: expect bugs, and the API may change. Known issues are listed in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
|
|
73
|
+
|
|
74
|
+
**Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.1...v0.2.2
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
thunc 0.2.3 is the last 0.2 release. It makes **native tool calls the way agents work wherever they run**: on Codex now, as on Claude Code, and in durable runs on Claude Code, which pick up after a crash mid-call. Durable agents can take `tools=`, the Claude API agent path no longer loses runs to `max_tokens` or a stalled reply, and `edit` can change many places at once.
|
|
2
|
+
|
|
3
|
+
| 8 tool-heavy tasks, Claude Sonnet 5.5, 3 runs each | Passed | Seconds per task | $ per task |
|
|
4
|
+
|---|---|---|---|
|
|
5
|
+
| thunc 0.2.2 agents, text protocol (Codex, durable runs, fallbacks) | 20/24 | 47 | 0.084 |
|
|
6
|
+
| **thunc 0.2.3 agents, text protocol** | **24/24** | **27** | 0.082 |
|
|
7
|
+
| thunc 0.2.3 agents on Claude Code, native calls | 24/24 | 8 | 0.021 |
|
|
8
|
+
| Claude Code itself, for reference | 24/24 | 10 | 0.078 |
|
|
9
|
+
|
|
10
|
+
Through the Claude API, thunc's agents passed 8 of 8 on Sonnet 5.5 ($0.023 a task) and on Opus 5.5 ($0.049 a task). On Codex, native calls took 30 seconds a task, against 45 for the text protocol 0.2.2 used there and 27 for Codex itself (16 of 16 each).
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
pip install --upgrade thunc
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
```python
|
|
17
|
+
import thunc
|
|
18
|
+
|
|
19
|
+
thunc.configure(backend="codex") # agents now make native tool calls here too
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def find_issue(title: str) -> int: # a tool is your own code, with a docstring for the model
|
|
23
|
+
"""Find an open issue by title. Returns its number, or 0."""
|
|
24
|
+
return tracker.find(title)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
triage = thunc.Agent("triage", workdir=".", tools=[find_issue], effort="high")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@triage.task
|
|
31
|
+
def flaky_tests() -> list[str]:
|
|
32
|
+
"""Run the tests three times and list the ones that fail only sometimes."""
|
|
33
|
+
...
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
# On a Temporal worker: durable runs take the agent's own tools now, and name those that may run
|
|
38
|
+
# again after a crash.
|
|
39
|
+
registry.agent_task("triage", flaky_tests, version="1", workspace_id="repo", retry_safe_tools=["find_issue"])
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
**Native tool calls on Codex** ([#55](https://github.com/Eltarras/thunc/pull/55))
|
|
43
|
+
- The agent's tools are an MCP server that Codex calls, through the same relay as Claude Code: thunc carries out each call with its own tools, permissions and run record. Codex's own tools and your `~/.codex` config stay out, its sandbox stays read-only, and only thunc's server is approved to run without asking.
|
|
44
|
+
- A turn that ends without `finish`, or fails, is continued with `codex exec resume`. The run's Codex session is deleted when the run ends.
|
|
45
|
+
- **Fallback:** when Codex can't start them, a run uses the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
|
|
46
|
+
|
|
47
|
+
**Durable runs** ([#56](https://github.com/Eltarras/thunc/pull/56), [#57](https://github.com/Eltarras/thunc/pull/57))
|
|
48
|
+
- **On Claude Code, native calls in segments.** One activity keeps one `claude -p` process for many model replies, and checkpoints the run and Claude Code's session before each reply's calls. If the worker stops or the CLI dies, the retried activity restores the session and continues it with `--resume`. The call that was in flight, asked for again, gets its old journal entry: it's replayed, waits for `resolve()`, or runs. Never twice.
|
|
49
|
+
- **`tools=`.** Each call of the agent's own functions is journaled like a command: recorded before it runs and after, replayed when completed, and waiting for `resolve()` if a worker stopped mid-call. `retry_safe_tools=` names the ones that may simply run again.
|
|
50
|
+
|
|
51
|
+
**The Claude API agent path** ([#52](https://github.com/Eltarras/thunc/pull/52))
|
|
52
|
+
- **`Agent(effort=...)`** on every backend. Claude 4.6 and later on the API get `"high"` by default; Claude Opus 5.5's own default is `"medium"`, which is low for agentic coding.
|
|
53
|
+
- Replies are streamed with room for 64,000 tokens. A reply cut off at `max_tokens` gets an error result for its calls instead of ending the run, `pause_turn` carries on, and a reply that sends nothing for `timeout` seconds is stopped and asked again (one Opus reply in the benchmark sent nothing for an hour).
|
|
54
|
+
- Old tool results are cleared on long runs (context editing), tools are strict where the model and schema allow, and connection errors, rate limits and server errors are retried as a step.
|
|
55
|
+
|
|
56
|
+
**Steadier agents everywhere** ([#51](https://github.com/Eltarras/thunc/pull/51), [#52](https://github.com/Eltarras/thunc/pull/52), [#53](https://github.com/Eltarras/thunc/pull/53))
|
|
57
|
+
- **The text protocol reads replies that aren't only the action**: the first complete action amid prose, a fence, `<invoke>` markup or made-up results, with a note to the model to keep to JSON. On Claude Code, a text step stops as soon as its action has arrived.
|
|
58
|
+
- **`finish` in the same reply as other calls is refused** (except beside `remember`): in the benchmark, a run returned a guess written before its own read came back.
|
|
59
|
+
- **The model is told when few steps are left**, in its last three replies before `max_steps`.
|
|
60
|
+
|
|
61
|
+
**Better tools** ([#50](https://github.com/Eltarras/thunc/pull/50), [#51](https://github.com/Eltarras/thunc/pull/51), [#49](https://github.com/Eltarras/thunc/pull/49))
|
|
62
|
+
- **`edit` takes `replace_all`, and `edits`** for several changes to one file in one call, all made or none.
|
|
63
|
+
- **`edit` no longer needs a prior `read`**: it only changes text the agent quotes exactly. `write` still replaces only a file read with `read`.
|
|
64
|
+
- **Large system prompts work on Claude Code**: the prompt goes in a file, not on the command line, where large `follow=` files could pass the operating system's limit.
|
|
65
|
+
|
|
66
|
+
**Tested**
|
|
67
|
+
- Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows; Temporal integration tests on Python 3.10 and 3.14, including a Claude Code run whose CLI dies mid-run. Tests can't reach a real `claude`, `codex` or `jev`.
|
|
68
|
+
- Live: the tool-use benchmark on Claude Code (Sonnet 5.5), the Claude API (Sonnet 5.5 and Opus 5.5) and Codex, with the results in `live_tests/bench_tooluse_report.md`; a durable run on real Claude Code with the CLI killed after its first tool result, which finished with each effect done once.
|
|
69
|
+
|
|
70
|
+
**Behaviour changes**
|
|
71
|
+
- Agents on `codex` make native tool calls by default, and fall back to the text protocol if they can't.
|
|
72
|
+
- Durable runs on `claude-code` make native tool calls. Runs already in progress keep the text protocol.
|
|
73
|
+
- Agents on the Claude API think at effort `high` on Claude 4.6 and later.
|
|
74
|
+
- `edit` no longer needs a prior `read`; `write` still does.
|
|
75
|
+
- `finish` in the same reply as other calls is refused, except beside `remember`.
|
|
76
|
+
- Nothing is removed.
|
|
77
|
+
|
|
78
|
+
**Known issues:** on Codex, native calls count each tool call as a step, so a long task reaches `max_steps` sooner, and their tokens aren't reported. Durable runs on Codex use the text protocol. On a fix in a long file, native calls take about twice Claude Code's steps. The full list is in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
|
|
79
|
+
|
|
80
|
+
**Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.2...v0.2.3
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
name: Release notes
|
|
2
|
+
|
|
3
|
+
# Run by hand (Actions > Release notes > Run workflow) with a version like 0.2.2. It writes the
|
|
4
|
+
# notes in .github/release-notes/v<version>.md to that release, and sets its title to
|
|
5
|
+
# "v<version> (beta)". A release that doesn't exist yet is created as a draft pre-release on the
|
|
6
|
+
# commit the workflow runs from, for you to review and publish. Editing or drafting a release
|
|
7
|
+
# doesn't trigger publish.yml, which runs only when you publish one.
|
|
8
|
+
|
|
9
|
+
on:
|
|
10
|
+
workflow_dispatch:
|
|
11
|
+
inputs:
|
|
12
|
+
version:
|
|
13
|
+
description: "Version, like 0.2.2"
|
|
14
|
+
required: true
|
|
15
|
+
|
|
16
|
+
permissions:
|
|
17
|
+
contents: write # create and edit releases
|
|
18
|
+
|
|
19
|
+
jobs:
|
|
20
|
+
notes:
|
|
21
|
+
runs-on: ubuntu-latest
|
|
22
|
+
steps:
|
|
23
|
+
- uses: actions/checkout@v4
|
|
24
|
+
- name: Write the notes to the release
|
|
25
|
+
env:
|
|
26
|
+
GH_TOKEN: ${{ github.token }}
|
|
27
|
+
VERSION: ${{ inputs.version }}
|
|
28
|
+
run: |
|
|
29
|
+
if [[ ! "$VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
|
|
30
|
+
echo "::error::version must look like 0.2.2, not '$VERSION'"
|
|
31
|
+
exit 1
|
|
32
|
+
fi
|
|
33
|
+
tag="v$VERSION"
|
|
34
|
+
notes=".github/release-notes/$tag.md"
|
|
35
|
+
if [[ ! -f "$notes" ]]; then
|
|
36
|
+
echo "::error::$notes doesn't exist; add it first"
|
|
37
|
+
exit 1
|
|
38
|
+
fi
|
|
39
|
+
if gh release view "$tag" > /dev/null 2>&1; then
|
|
40
|
+
gh release edit "$tag" --title "$tag (beta)" --notes-file "$notes"
|
|
41
|
+
echo "Updated the notes of $tag."
|
|
42
|
+
else
|
|
43
|
+
gh release create "$tag" --draft --prerelease --target "$GITHUB_SHA" \
|
|
44
|
+
--title "$tag (beta)" --notes-file "$notes"
|
|
45
|
+
echo "Created $tag as a draft pre-release."
|
|
46
|
+
fi
|
|
@@ -3,6 +3,123 @@
|
|
|
3
3
|
All notable changes to thunc. The full notes for each release are on the
|
|
4
4
|
[releases page](https://github.com/Eltarras/thunc/releases).
|
|
5
5
|
|
|
6
|
+
## 0.2.3 (beta)
|
|
7
|
+
|
|
8
|
+
**Native tool calls everywhere agents run, durable runs that keep them, and the last of 0.2.**
|
|
9
|
+
Agents on Codex make native tool calls through the same MCP relay as Claude Code, durable runs on
|
|
10
|
+
Claude Code do too and pick up after a crash mid-call, and durable agents can take `tools=`. On the
|
|
11
|
+
Claude API, agents stream their replies, think at effort `high`, and no longer lose a run to
|
|
12
|
+
`max_tokens` or a stalled reply. `edit` can replace every occurrence or make several changes at
|
|
13
|
+
once, and no longer needs a prior `read`. In the tool-use benchmark (`live_tests/bench_tooluse.py`,
|
|
14
|
+
8 tasks, Claude Sonnet 5.5, 3 runs each), every harness passed 24 of 24, the text protocol included
|
|
15
|
+
(20 of 24 in 0.2.2) in about half the time; through the Claude API, Sonnet 5.5 and Opus 5.5 passed
|
|
16
|
+
8 of 8; on Codex, native calls took 30 seconds a task against 45 for the text protocol.
|
|
17
|
+
|
|
18
|
+
### Behavior changes
|
|
19
|
+
|
|
20
|
+
- **Agents on `codex` make native tool calls**, as on Claude Code since 0.2.2 (see Added). When
|
|
21
|
+
Codex can't start them, a run falls back to the text protocol with a warning; `protocol="text"`
|
|
22
|
+
keeps the old way, and durable runs on Codex still use it.
|
|
23
|
+
- **`finish` called in the same reply as other calls is refused** (except beside `remember`): its
|
|
24
|
+
value can't account for results the model hasn't seen yet. The other calls run, and the model is
|
|
25
|
+
told to call `finish` on its own. In the tool-use benchmark, a run on the text protocol batched
|
|
26
|
+
`[search, read, finish 0.0]` and returned the guess. Every protocol and durable runs get it.
|
|
27
|
+
- **Agents on the Claude API think at effort `high` by default** on Claude 4.6 and later. Claude
|
|
28
|
+
Opus 5.5's own default is `medium`, which is low for agentic coding. `effort=` changes it (below).
|
|
29
|
+
- **`edit` no longer needs the file to have been read first.** It only changes text the agent
|
|
30
|
+
quotes exactly, so it can't overwrite what the agent hasn't seen. With the `shell` permission,
|
|
31
|
+
agents often read files with `cat`, and `edit` refused them until they read the file again with
|
|
32
|
+
`read`: 7 times in 24 runs of the tool-use benchmark. A file the agent did read must still not
|
|
33
|
+
have changed on disk since, and `write` still replaces only a file read with `read` (a file
|
|
34
|
+
changed by an edit alone still counts as unread). The `run` tool's description, with `shell`, now
|
|
35
|
+
says to read files with `read`.
|
|
36
|
+
|
|
37
|
+
### Added
|
|
38
|
+
|
|
39
|
+
- **Durable runs on Claude Code make native tool calls.** A run goes in segments: one activity keeps
|
|
40
|
+
one `claude -p` process for many model replies, and before each reply's calls are carried out it
|
|
41
|
+
saves a checkpoint of the run and of Claude Code's session. If the worker stops or the CLI dies,
|
|
42
|
+
the retried activity restores the session and continues it with `--resume`; Claude Code marks the
|
|
43
|
+
call that was in flight as interrupted, the model asks for it again, and it gets the journal entry
|
|
44
|
+
it had, so it's replayed, waits for `resolve()`, or runs (one the model doesn't ask for again
|
|
45
|
+
keeps its entry). The session is removed when the run ends. When Claude Code can't start native
|
|
46
|
+
calls, the run goes on with the text protocol. Runs already in progress keep the text protocol.
|
|
47
|
+
- **Durable agents can have `tools=`.** Each call of one of the agent's own functions is journaled
|
|
48
|
+
like a command: the intent is recorded before it runs and its result after, so a retried activity
|
|
49
|
+
replays the result instead of calling it again, and a call interrupted by a worker stopping waits
|
|
50
|
+
for `resolve()`. `registry.agent_task(..., retry_safe_tools=["find_issue"])` names the tools that
|
|
51
|
+
may run again instead. A tool's description, arguments and retry marking are part of the task's
|
|
52
|
+
fingerprint, so changing one needs a new version; tasks without tools keep their fingerprint.
|
|
53
|
+
- **Native calls on `codex`**: the agent's tools are an MCP server (thunc's relay, given with
|
|
54
|
+
`-c mcp_servers.thunc.*`) that Codex calls; thunc carries out each call with its own tools,
|
|
55
|
+
permissions and run record. Each `codex exec` is a turn: one that ends without `finish` is
|
|
56
|
+
continued with `codex exec resume`, as is one that fails (twice at most). Codex's own tools and
|
|
57
|
+
your `~/.codex` config stay out and its sandbox stays read-only; only thunc's server is approved
|
|
58
|
+
to run without asking, with a tool timeout above `command_timeout`. Resuming needs the session
|
|
59
|
+
saved, so thunc deletes the run's Codex session (`codex delete --force`) when the run ends.
|
|
60
|
+
- **`Agent(effort=...)`**: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`, on every backend
|
|
61
|
+
(`output_config.effort` on the Claude API, `reasoning.effort` on OpenAI, `--effort` on Claude Code,
|
|
62
|
+
`model_reasoning_effort` on Codex; the last two go up to `"xhigh"`). Recorded in `agent.json` only
|
|
63
|
+
when set, so durable tasks registered without it keep their fingerprint.
|
|
64
|
+
- **The model is told when few steps are left.** In its last three replies before `max_steps`, the
|
|
65
|
+
last tool result says how many replies remain, so the model can finish with what it has instead
|
|
66
|
+
of being cut off. Every protocol and durable runs get it; the run record keeps each tool's output.
|
|
67
|
+
- **`edit` can replace every occurrence, and make several changes in one call.** With
|
|
68
|
+
`"replace_all": true`, every occurrence of `old` is replaced and the result gives the count. With
|
|
69
|
+
`"edits": [{"old": ..., "new": ..., "replace_all"?: ...}, ...]` (at most 50) instead of `old` and
|
|
70
|
+
`new`, the changes apply in order, each to the text the ones before it left; if one fails, none is
|
|
71
|
+
made, and the error names it. In the tool-use benchmark, models renamed a symbol by writing a
|
|
72
|
+
throwaway script instead of making 26 separate edits, and took twice Claude Code's turns on a
|
|
73
|
+
multi-spot fix. In a durable run, a multi-edit is one effect, recovered as a whole.
|
|
74
|
+
|
|
75
|
+
### Changed
|
|
76
|
+
|
|
77
|
+
- **The text protocol reads replies that aren't only the action** (finding 1 of the tool-use
|
|
78
|
+
benchmark report). Models trained for native tool calls often wrap the action in prose, a code
|
|
79
|
+
fence or `<invoke>` markup, or carry on past it with results they make up: on Sonnet 5.5, 25% of
|
|
80
|
+
text-protocol replies were sent back as "not valid JSON" with a correct action inside. Now the
|
|
81
|
+
first complete action (or array of them) in the reply is used, with literal newlines in its
|
|
82
|
+
strings accepted, and a reply written only as `<invoke name="...">` markup is read as its calls,
|
|
83
|
+
each argument in its tool's type (or as one JSON `args` parameter). A reply with no action in it
|
|
84
|
+
is still sent back, as before. A reply read this way runs, but its results carry a note to reply
|
|
85
|
+
with the JSON action alone: without it, a model that slipped into markup was never corrected, and
|
|
86
|
+
on Sonnet 5.5 fell into repeating empty markup until the step timed out.
|
|
87
|
+
This is the text protocol on Codex, `protocol="text"`, durable runs on Codex, and Claude
|
|
88
|
+
Code's fallback from native calls.
|
|
89
|
+
- **A text-protocol step on Claude Code stops once its action is complete.** The reply is streamed
|
|
90
|
+
(`--output-format stream-json --include-partial-messages`) and the CLI is stopped as soon as a
|
|
91
|
+
complete action has arrived, rather than left to make up the tool's result until the step times
|
|
92
|
+
out (7 of 8 replayed first steps on Sonnet ran past 120 seconds that way). A batch that has begun
|
|
93
|
+
is waited for, and two blocks of `<invoke>` markup end the step too (a model repeating itself).
|
|
94
|
+
Plain `@thunc.function` calls on Claude Code aren't streamed.
|
|
95
|
+
- **The Claude API agent path keeps runs going** (finding 7 of the tool-use benchmark report):
|
|
96
|
+
- Replies are streamed with `max_tokens=64000` (was 16,000 without streaming), so a large write
|
|
97
|
+
fits.
|
|
98
|
+
- A reply cut off at `max_tokens` no longer ends the run: its tool calls aren't run and get an
|
|
99
|
+
error result saying so, and the model is asked again. Two in a row end the run.
|
|
100
|
+
- `pause_turn` is asked to carry on (up to 6 times in a row) instead of ending the run.
|
|
101
|
+
`model_context_window_exceeded` ends it with a message that says so.
|
|
102
|
+
- On Claude 4.6 and later, the API clears old tool results on long runs (context editing, beta
|
|
103
|
+
`context-management-2025-06-27`).
|
|
104
|
+
- Tools are `strict` (arguments guaranteed to match their schema) on the models that support it,
|
|
105
|
+
when the schema allows it: the built-in tools except `edit`, `remember`, and `finish` and custom
|
|
106
|
+
tools whose schema is closed.
|
|
107
|
+
- A reply that sends nothing for `timeout` seconds (300 by default) is stopped and asked again. The
|
|
108
|
+
SDK's read timeout doesn't catch it, as the API's keep-alive pings count as reading: in the
|
|
109
|
+
tool-use benchmark, one reply on Claude Opus 5.5 sent nothing for an hour.
|
|
110
|
+
- A lost connection, a rate limit or a server error on the Claude and OpenAI APIs is retried as a
|
|
111
|
+
step, like the CLI backends' errors, instead of ending the run.
|
|
112
|
+
|
|
113
|
+
### Fixed
|
|
114
|
+
|
|
115
|
+
- **A large system prompt no longer stops the `claude-code` backend from starting.** It went on the
|
|
116
|
+
command line, so large `follow=` files and memory could pass the operating system's limit on its
|
|
117
|
+
length (128 KB for one argument on Linux, 32,767 characters for the whole line on Windows), and
|
|
118
|
+
starting `claude` failed with a raw `OSError: Argument list too long`. The prompt now goes in a
|
|
119
|
+
temporary file (`--system-prompt-file`), as it already did for agents' native calls and on Codex.
|
|
120
|
+
This covers `@thunc.function` and `thunc.call`, agents on the text protocol, and durable runs on
|
|
121
|
+
Claude Code. A command line that is still too long raises a `ThuncError` that gives its size.
|
|
122
|
+
|
|
6
123
|
## 0.2.2 (beta)
|
|
7
124
|
|
|
8
125
|
**Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
|
|
@@ -86,8 +86,9 @@ THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
|
|
|
86
86
|
5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
|
|
87
87
|
`Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
|
|
88
88
|
the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
|
|
89
|
-
A CLI that can call MCP tools can get native calls the way Claude Code
|
|
90
|
-
(`thunc/claude_code.py`, tested with
|
|
89
|
+
A CLI that can call MCP tools can get native calls the way Claude Code and Codex do: through the
|
|
90
|
+
relay in `thunc/relay.py` (see `thunc/claude_code.py` and `thunc/codex.py`, tested with fake CLIs
|
|
91
|
+
in `tests/test_claude_code_agent.py` and `tests/test_codex_agent.py`).
|
|
91
92
|
Raise `TransientError` for failures worth asking again, so agent runs retry them.
|
|
92
93
|
Agents and durable runs refuse typed backends.
|
|
93
94
|
6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
|
|
@@ -153,20 +154,20 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
|
|
|
153
154
|
|
|
154
155
|
**Open: agents and durable runs.**
|
|
155
156
|
|
|
156
|
-
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
-
|
|
169
|
-
|
|
157
|
+
- On Codex, native calls count each tool call as a step: Codex's events don't say which calls came
|
|
158
|
+
from the same model reply. A long task reaches `max_steps` sooner than on Claude Code (22 steps on
|
|
159
|
+
`rename` in the tool-use benchmark, against `max_steps=40`).
|
|
160
|
+
- Tokens aren't reported for native runs on Codex: `codex exec --json` reports usage at the end of
|
|
161
|
+
a turn, and a run ends inside one when the model calls `finish`.
|
|
162
|
+
- Durable runs on `codex` use the JSON text protocol, not native calls.
|
|
163
|
+
- On a fix in a long file (`deep_fix` in the tool-use benchmark), native calls take about twice
|
|
164
|
+
Claude Code's steps (8 against 4.3), reading around the file in pages. A larger `read` limit is
|
|
165
|
+
the next thing to measure.
|
|
166
|
+
- Durable runs on Claude Code save the whole session file at each checkpoint, so a long run's saved
|
|
167
|
+
state grows with every reply, and counts toward the 16 MiB limit. A CLI that dies leaves Claude
|
|
168
|
+
Code's own `~/.claude/sessions/<pid>.json` behind.
|
|
169
|
+
- The `shell` permission's own test (pipes, `cd`, redirects) is skipped on Windows; only a single
|
|
170
|
+
command line runs through `cmd /c` in CI.
|
|
170
171
|
- Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
|
|
171
172
|
tombstones are kept forever. There's no context summarization either, so a long run fails once
|
|
172
173
|
its saved state passes 16 MiB.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thunc
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: think + function: call an LLM like a typed Python function.
|
|
5
5
|
Project-URL: Homepage, https://eltarras.github.io/thunc/
|
|
6
6
|
Project-URL: Documentation, https://eltarras.github.io/thunc/docs/
|
|
@@ -203,8 +203,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
|
|
|
203
203
|
`pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
|
|
204
204
|
any server that speaks the OpenAI Responses API.
|
|
205
205
|
- `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
|
|
206
|
-
Both run with their own tools turned off, so the model can only answer; an agent on
|
|
207
|
-
|
|
206
|
+
Both run with their own tools turned off, so the model can only answer; an agent on either
|
|
207
|
+
gets only its thunc tools, as native calls (see Agents). `codex` also ignores
|
|
208
208
|
`~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
|
|
209
209
|
login still works. Pick the model with `configure(model=...)` or `model=`.
|
|
210
210
|
- `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
|
|
@@ -305,10 +305,15 @@ calls are native too: the agent's tools are an MCP server that one `claude -p` p
|
|
|
305
305
|
calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
|
|
306
306
|
can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
|
|
307
307
|
text protocol below instead, with a warning, and so do later runs in the process;
|
|
308
|
-
`protocol="native"` fails instead. On Codex the
|
|
309
|
-
|
|
310
|
-
`
|
|
311
|
-
|
|
308
|
+
`protocol="native"` fails instead. On Codex it's the same: each `codex exec` is a turn in which
|
|
309
|
+
the model calls the agent's tools through the MCP server, a turn that ends without `finish` is
|
|
310
|
+
continued with `codex exec resume`, Codex's own tools stay off and its sandbox read-only, and the
|
|
311
|
+
run's Codex session is deleted when the run ends. With `protocol="text"` the model replies with
|
|
312
|
+
JSON actions as text instead: one at a time, or several independent ones (reading three files) as
|
|
313
|
+
a JSON array, which saves turns. That works on any backend, for example with a server behind
|
|
314
|
+
`OPENAI_BASE_URL` that has no function calling. (Durable runs on Codex use it.) A reply that wraps its
|
|
315
|
+
action in prose or tool-call markup, or carries on past it, is read for its first complete action,
|
|
316
|
+
and on Claude Code the step stops as soon as that action has arrived.
|
|
312
317
|
The `jev` backend only answers typed questions and cannot run agents, even for a task returning
|
|
313
318
|
`bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
|
|
314
319
|
or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
@@ -333,14 +338,18 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
333
338
|
the run carries on. Bad rules fail when the agent is declared.
|
|
334
339
|
- **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
|
|
335
340
|
such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
|
|
336
|
-
exactly once
|
|
341
|
+
exactly once, or every occurrence with `replace_all`; several changes to one file can go in one
|
|
342
|
+
call as `edits`, all made or none) when a write rule allows it; `run` when a run or shell rule
|
|
343
|
+
allows it; and
|
|
337
344
|
`remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
|
|
338
345
|
outside are refused, and the rules are checked on where a link really leads. Files the agent may
|
|
339
346
|
not read are left out of `list` and `search`, and so is what git ignores, in a git repository
|
|
340
347
|
(build output, caches, vendored code); a folder named explicitly is still listed and searched.
|
|
341
|
-
- **No blind overwrites.** A file is only replaced
|
|
342
|
-
run, and only if it hasn't changed on disk since.
|
|
343
|
-
|
|
348
|
+
- **No blind overwrites.** A file is only replaced (`write`) after the agent read it with `read` in
|
|
349
|
+
the same run, and only if it hasn't changed on disk since. An `edit` needs no read, because it
|
|
350
|
+
only changes text the agent quotes exactly, but a file the agent did read must not have changed
|
|
351
|
+
since. There is no undo, so run agents that write in a git repository with a clean tree, and
|
|
352
|
+
review their changes with `git diff`.
|
|
344
353
|
- **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
|
|
345
354
|
permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
|
|
346
355
|
agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
|
|
@@ -379,9 +388,15 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
379
388
|
`system=thunc.prompts.CODING + "\n\nTarget Python 3.10."`.
|
|
380
389
|
- **Time:** `max_steps=40` bounds the model replies in a run, and `timeout=` (seconds) bounds the
|
|
381
390
|
run's time. It's checked before each model call; a command's time limit is cut to the time left.
|
|
391
|
+
In its last three replies before `max_steps`, the model is told how many are left, so it can
|
|
392
|
+
finish with what it has.
|
|
393
|
+
- **`effort=`** sets how hard the model thinks: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`
|
|
394
|
+
(`openai` and `codex` go up to `"xhigh"`). By default it's `"high"` on the `anthropic` backend for
|
|
395
|
+
Claude 4.6 and later (Claude Opus 5.5's own default is `"medium"`, low for agentic coding), and
|
|
396
|
+
each backend's own default elsewhere.
|
|
382
397
|
- **Options:** `thunc.Agent(name, *, workdir, system=None, permissions=(), env=None,
|
|
383
398
|
command_timeout=120, follow=False, protocol=None, tools=(), timeout=None, max_steps=40,
|
|
384
|
-
retries=2, backend=None, model=None)`, and `@agent.task(instructions=..., ensure=...)`.
|
|
399
|
+
retries=2, backend=None, model=None, effort=None)`, and `@agent.task(instructions=..., ensure=...)`.
|
|
385
400
|
`@thunc.agent(name, workdir=..., instructions=..., ensure=..., **options)` takes the same options.
|
|
386
401
|
`async def` tasks work.
|
|
387
402
|
- **What happened in a run.** Calling a task returns its value. `agent.run(task, *args)` runs it
|
|
@@ -401,7 +416,9 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
401
416
|
A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
|
|
402
417
|
a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
|
|
403
418
|
each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
|
|
404
|
-
seconds before it's retried.
|
|
419
|
+
seconds before it's retried. On the Claude API, a reply cut off at `max_tokens` doesn't end the
|
|
420
|
+
run: its tool calls get an error result saying so (twice in a row does). With tracing on, each run
|
|
421
|
+
is also one line with every model reply.
|
|
405
422
|
|
|
406
423
|
**How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
|
|
407
424
|
small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
|
|
@@ -458,8 +475,9 @@ print(run.value)
|
|
|
458
475
|
```
|
|
459
476
|
|
|
460
477
|
Durable mode requires a Temporal service, a worker, and persistent storage on the
|
|
461
|
-
same volume. File changes and memory updates use recovery receipts. Commands
|
|
462
|
-
uncertain outcomes pause for operator resolution instead of blindly
|
|
478
|
+
same volume. File changes and memory updates use recovery receipts. Commands, and the agent's
|
|
479
|
+
own `tools=` functions, with uncertain outcomes pause for operator resolution instead of blindly
|
|
480
|
+
running twice (`retry_safe_tools=` names functions that may run again).
|
|
463
481
|
Temporal does not back up your workspace or guarantee exactly-once external effects.
|
|
464
482
|
|
|
465
483
|
The [Temporal guide and runnable example](examples/temporal/README.md) cover service
|
|
@@ -491,6 +509,8 @@ thunc/
|
|
|
491
509
|
runs.py thunc.Run and AgentError: what a run did
|
|
492
510
|
native.py how a run talks to its backend: native tool calls or the text protocol
|
|
493
511
|
claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
|
|
512
|
+
codex.py native tool calls on Codex, the same way
|
|
513
|
+
relay.py the run's end of the MCP server: the connection the calls come through
|
|
494
514
|
store.py the agent's folder: memory, settings, run records, the lock
|
|
495
515
|
prompts.py the agent's system prompt
|
|
496
516
|
__main__.py the thunc command: thunc run [--profile], thunc cache list / clear
|
|
@@ -168,8 +168,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
|
|
|
168
168
|
`pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
|
|
169
169
|
any server that speaks the OpenAI Responses API.
|
|
170
170
|
- `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
|
|
171
|
-
Both run with their own tools turned off, so the model can only answer; an agent on
|
|
172
|
-
|
|
171
|
+
Both run with their own tools turned off, so the model can only answer; an agent on either
|
|
172
|
+
gets only its thunc tools, as native calls (see Agents). `codex` also ignores
|
|
173
173
|
`~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
|
|
174
174
|
login still works. Pick the model with `configure(model=...)` or `model=`.
|
|
175
175
|
- `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
|
|
@@ -270,10 +270,15 @@ calls are native too: the agent's tools are an MCP server that one `claude -p` p
|
|
|
270
270
|
calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
|
|
271
271
|
can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
|
|
272
272
|
text protocol below instead, with a warning, and so do later runs in the process;
|
|
273
|
-
`protocol="native"` fails instead. On Codex the
|
|
274
|
-
|
|
275
|
-
`
|
|
276
|
-
|
|
273
|
+
`protocol="native"` fails instead. On Codex it's the same: each `codex exec` is a turn in which
|
|
274
|
+
the model calls the agent's tools through the MCP server, a turn that ends without `finish` is
|
|
275
|
+
continued with `codex exec resume`, Codex's own tools stay off and its sandbox read-only, and the
|
|
276
|
+
run's Codex session is deleted when the run ends. With `protocol="text"` the model replies with
|
|
277
|
+
JSON actions as text instead: one at a time, or several independent ones (reading three files) as
|
|
278
|
+
a JSON array, which saves turns. That works on any backend, for example with a server behind
|
|
279
|
+
`OPENAI_BASE_URL` that has no function calling. (Durable runs on Codex use it.) A reply that wraps its
|
|
280
|
+
action in prose or tool-call markup, or carries on past it, is read for its first complete action,
|
|
281
|
+
and on Claude Code the step stops as soon as that action has arrived.
|
|
277
282
|
The `jev` backend only answers typed questions and cannot run agents, even for a task returning
|
|
278
283
|
`bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
|
|
279
284
|
or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
@@ -298,14 +303,18 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
298
303
|
the run carries on. Bad rules fail when the agent is declared.
|
|
299
304
|
- **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
|
|
300
305
|
such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
|
|
301
|
-
exactly once
|
|
306
|
+
exactly once, or every occurrence with `replace_all`; several changes to one file can go in one
|
|
307
|
+
call as `edits`, all made or none) when a write rule allows it; `run` when a run or shell rule
|
|
308
|
+
allows it; and
|
|
302
309
|
`remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
|
|
303
310
|
outside are refused, and the rules are checked on where a link really leads. Files the agent may
|
|
304
311
|
not read are left out of `list` and `search`, and so is what git ignores, in a git repository
|
|
305
312
|
(build output, caches, vendored code); a folder named explicitly is still listed and searched.
|
|
306
|
-
- **No blind overwrites.** A file is only replaced
|
|
307
|
-
run, and only if it hasn't changed on disk since.
|
|
308
|
-
|
|
313
|
+
- **No blind overwrites.** A file is only replaced (`write`) after the agent read it with `read` in
|
|
314
|
+
the same run, and only if it hasn't changed on disk since. An `edit` needs no read, because it
|
|
315
|
+
only changes text the agent quotes exactly, but a file the agent did read must not have changed
|
|
316
|
+
since. There is no undo, so run agents that write in a git repository with a clean tree, and
|
|
317
|
+
review their changes with `git diff`.
|
|
309
318
|
- **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
|
|
310
319
|
permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
|
|
311
320
|
agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
|
|
@@ -344,9 +353,15 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
344
353
|
`system=thunc.prompts.CODING + "\n\nTarget Python 3.10."`.
|
|
345
354
|
- **Time:** `max_steps=40` bounds the model replies in a run, and `timeout=` (seconds) bounds the
|
|
346
355
|
run's time. It's checked before each model call; a command's time limit is cut to the time left.
|
|
356
|
+
In its last three replies before `max_steps`, the model is told how many are left, so it can
|
|
357
|
+
finish with what it has.
|
|
358
|
+
- **`effort=`** sets how hard the model thinks: `"low"`, `"medium"`, `"high"`, `"xhigh"` or `"max"`
|
|
359
|
+
(`openai` and `codex` go up to `"xhigh"`). By default it's `"high"` on the `anthropic` backend for
|
|
360
|
+
Claude 4.6 and later (Claude Opus 5.5's own default is `"medium"`, low for agentic coding), and
|
|
361
|
+
each backend's own default elsewhere.
|
|
347
362
|
- **Options:** `thunc.Agent(name, *, workdir, system=None, permissions=(), env=None,
|
|
348
363
|
command_timeout=120, follow=False, protocol=None, tools=(), timeout=None, max_steps=40,
|
|
349
|
-
retries=2, backend=None, model=None)`, and `@agent.task(instructions=..., ensure=...)`.
|
|
364
|
+
retries=2, backend=None, model=None, effort=None)`, and `@agent.task(instructions=..., ensure=...)`.
|
|
350
365
|
`@thunc.agent(name, workdir=..., instructions=..., ensure=..., **options)` takes the same options.
|
|
351
366
|
`async def` tasks work.
|
|
352
367
|
- **What happened in a run.** Calling a task returns its value. `agent.run(task, *args)` runs it
|
|
@@ -366,7 +381,9 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
366
381
|
A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
|
|
367
382
|
a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
|
|
368
383
|
each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
|
|
369
|
-
seconds before it's retried.
|
|
384
|
+
seconds before it's retried. On the Claude API, a reply cut off at `max_tokens` doesn't end the
|
|
385
|
+
run: its tool calls get an error result saying so (twice in a row does). With tracing on, each run
|
|
386
|
+
is also one line with every model reply.
|
|
370
387
|
|
|
371
388
|
**How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
|
|
372
389
|
small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
|
|
@@ -423,8 +440,9 @@ print(run.value)
|
|
|
423
440
|
```
|
|
424
441
|
|
|
425
442
|
Durable mode requires a Temporal service, a worker, and persistent storage on the
|
|
426
|
-
same volume. File changes and memory updates use recovery receipts. Commands
|
|
427
|
-
uncertain outcomes pause for operator resolution instead of blindly
|
|
443
|
+
same volume. File changes and memory updates use recovery receipts. Commands, and the agent's
|
|
444
|
+
own `tools=` functions, with uncertain outcomes pause for operator resolution instead of blindly
|
|
445
|
+
running twice (`retry_safe_tools=` names functions that may run again).
|
|
428
446
|
Temporal does not back up your workspace or guarantee exactly-once external effects.
|
|
429
447
|
|
|
430
448
|
The [Temporal guide and runnable example](examples/temporal/README.md) cover service
|
|
@@ -456,6 +474,8 @@ thunc/
|
|
|
456
474
|
runs.py thunc.Run and AgentError: what a run did
|
|
457
475
|
native.py how a run talks to its backend: native tool calls or the text protocol
|
|
458
476
|
claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
|
|
477
|
+
codex.py native tool calls on Codex, the same way
|
|
478
|
+
relay.py the run's end of the MCP server: the connection the calls come through
|
|
459
479
|
store.py the agent's folder: memory, settings, run records, the lock
|
|
460
480
|
prompts.py the agent's system prompt
|
|
461
481
|
__main__.py the thunc command: thunc run [--profile], thunc cache list / clear
|