thunc 0.2.2__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (151) hide show
  1. thunc-0.3.0/.github/release-notes/v0.2.2.md +74 -0
  2. thunc-0.3.0/.github/release-notes/v0.2.3.md +80 -0
  3. thunc-0.3.0/.github/release-notes/v0.3.0.md +86 -0
  4. {thunc-0.2.2 → thunc-0.3.0}/.github/workflows/ci.yml +26 -0
  5. thunc-0.3.0/.github/workflows/publish-watch.yml +81 -0
  6. {thunc-0.2.2 → thunc-0.3.0}/.github/workflows/publish.yml +2 -0
  7. thunc-0.3.0/.github/workflows/release-notes.yml +46 -0
  8. {thunc-0.2.2 → thunc-0.3.0}/.gitignore +2 -0
  9. thunc-0.3.0/CHANGELOG.md +344 -0
  10. {thunc-0.2.2 → thunc-0.3.0}/CONTRIBUTING.md +46 -16
  11. {thunc-0.2.2 → thunc-0.3.0}/PKG-INFO +107 -20
  12. {thunc-0.2.2 → thunc-0.3.0}/README.md +104 -19
  13. thunc-0.3.0/docs/demo-agent.gif +0 -0
  14. thunc-0.3.0/docs/demo-function.gif +0 -0
  15. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/agents.html +20 -13
  16. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/api.html +19 -11
  17. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/backends.html +8 -5
  18. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/caching.html +28 -9
  19. thunc-0.3.0/docs/docs/changelog.html +229 -0
  20. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/functions.html +7 -4
  21. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/index.html +14 -6
  22. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/jev.html +6 -3
  23. {thunc-0.2.2 → thunc-0.3.0}/docs/docs/temporal.html +12 -4
  24. thunc-0.3.0/docs/docs/watch.html +190 -0
  25. thunc-0.3.0/docs/docs/write.html +154 -0
  26. {thunc-0.2.2 → thunc-0.3.0}/docs/index.html +3 -3
  27. {thunc-0.2.2 → thunc-0.3.0}/docs/site.css +7 -0
  28. {thunc-0.2.2 → thunc-0.3.0}/docs/sitemap.xml +12 -9
  29. {thunc-0.2.2 → thunc-0.3.0}/examples/temporal/README.md +14 -0
  30. thunc-0.3.0/examples/thunc_write.py +46 -0
  31. {thunc-0.2.2 → thunc-0.3.0}/live_tests/bench_tooluse.py +167 -9
  32. {thunc-0.2.2 → thunc-0.3.0}/live_tests/bench_tooluse_report.md +76 -0
  33. thunc-0.3.0/live_tests/test_writing.py +49 -0
  34. {thunc-0.2.2 → thunc-0.3.0}/pyproject.toml +5 -1
  35. {thunc-0.2.2 → thunc-0.3.0}/tests/conftest.py +18 -0
  36. {thunc-0.2.2 → thunc-0.3.0}/tests/fake_claude.py +28 -1
  37. thunc-0.3.0/tests/fake_codex_mcp.py +142 -0
  38. thunc-0.3.0/tests/temporal/test_contract.py +250 -0
  39. {thunc-0.2.2 → thunc-0.3.0}/tests/temporal/test_runtime.py +110 -0
  40. thunc-0.3.0/tests/temporal/test_segments.py +160 -0
  41. {thunc-0.2.2 → thunc-0.3.0}/tests/test_agent.py +228 -11
  42. {thunc-0.2.2 → thunc-0.3.0}/tests/test_backends.py +144 -4
  43. {thunc-0.2.2 → thunc-0.3.0}/tests/test_claude_code_agent.py +14 -3
  44. {thunc-0.2.2 → thunc-0.3.0}/tests/test_cli.py +72 -1
  45. thunc-0.3.0/tests/test_codex_agent.py +208 -0
  46. thunc-0.3.0/tests/test_events.py +149 -0
  47. {thunc-0.2.2 → thunc-0.3.0}/tests/test_execution.py +34 -0
  48. {thunc-0.2.2 → thunc-0.3.0}/tests/test_native.py +190 -7
  49. thunc-0.3.0/tests/test_source.py +235 -0
  50. thunc-0.3.0/tests/test_text_protocol.py +272 -0
  51. thunc-0.3.0/tests/test_writing.py +461 -0
  52. {thunc-0.2.2 → thunc-0.3.0}/thunc/__init__.py +1 -1
  53. thunc-0.3.0/thunc/__main__.py +312 -0
  54. {thunc-0.2.2 → thunc-0.3.0}/thunc/agent.py +126 -25
  55. {thunc-0.2.2 → thunc-0.3.0}/thunc/backends.py +129 -33
  56. {thunc-0.2.2 → thunc-0.3.0}/thunc/claude_code.py +65 -130
  57. thunc-0.3.0/thunc/codex.py +276 -0
  58. {thunc-0.2.2 → thunc-0.3.0}/thunc/core.py +53 -1
  59. {thunc-0.2.2 → thunc-0.3.0}/thunc/decorator.py +46 -5
  60. thunc-0.3.0/thunc/events.py +87 -0
  61. {thunc-0.2.2 → thunc-0.3.0}/thunc/execution.py +23 -0
  62. {thunc-0.2.2 → thunc-0.3.0}/thunc/native.py +275 -40
  63. {thunc-0.2.2 → thunc-0.3.0}/thunc/prompts.py +14 -0
  64. thunc-0.3.0/thunc/relay.py +179 -0
  65. thunc-0.3.0/thunc/source.py +326 -0
  66. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/activities.py +193 -10
  67. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/effects.py +11 -3
  68. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/registry.py +37 -7
  69. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/workflows.py +1 -1
  70. {thunc-0.2.2 → thunc-0.3.0}/thunc/tools.py +132 -25
  71. thunc-0.3.0/thunc/writing.py +890 -0
  72. {thunc-0.2.2 → thunc-0.3.0}/uv.lock +21 -2
  73. thunc-0.2.2/CHANGELOG.md +0 -186
  74. thunc-0.2.2/tests/temporal/test_contract.py +0 -118
  75. thunc-0.2.2/thunc/__main__.py +0 -160
  76. {thunc-0.2.2 → thunc-0.3.0}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
  77. {thunc-0.2.2 → thunc-0.3.0}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
  78. {thunc-0.2.2 → thunc-0.3.0}/.github/demo-agent.gif +0 -0
  79. {thunc-0.2.2 → thunc-0.3.0}/.github/demo-function.gif +0 -0
  80. {thunc-0.2.2 → thunc-0.3.0}/.github/social-preview.png +0 -0
  81. {thunc-0.2.2 → thunc-0.3.0}/LICENSE +0 -0
  82. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/README.md +0 -0
  83. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/__init__.py +0 -0
  84. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/__main__.py +0 -0
  85. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_agent.py +0 -0
  86. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_backends.py +0 -0
  87. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_cache.py +0 -0
  88. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_calls.py +0 -0
  89. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_schema.py +0 -0
  90. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_startup.py +0 -0
  91. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/bench_tools.py +0 -0
  92. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/fakes.py +0 -0
  93. {thunc-0.2.2 → thunc-0.3.0}/benchmarks/harness.py +0 -0
  94. {thunc-0.2.2 → thunc-0.3.0}/design/og-card.html +0 -0
  95. {thunc-0.2.2 → thunc-0.3.0}/docs/.nojekyll +0 -0
  96. {thunc-0.2.2 → thunc-0.3.0}/docs/apple-touch-icon.png +0 -0
  97. {thunc-0.2.2 → thunc-0.3.0}/docs/favicon-96.png +0 -0
  98. {thunc-0.2.2 → thunc-0.3.0}/docs/favicon.ico +0 -0
  99. {thunc-0.2.2 → thunc-0.3.0}/docs/favicon.svg +0 -0
  100. {thunc-0.2.2 → thunc-0.3.0}/docs/google2cd177e5e85b3c3e.html +0 -0
  101. {thunc-0.2.2 → thunc-0.3.0}/docs/jev.html +0 -0
  102. {thunc-0.2.2 → thunc-0.3.0}/docs/og.png +0 -0
  103. {thunc-0.2.2 → thunc-0.3.0}/docs/site.js +0 -0
  104. {thunc-0.2.2 → thunc-0.3.0}/docs/temporal.html +0 -0
  105. {thunc-0.2.2 → thunc-0.3.0}/examples/dynamic_prompts.py +0 -0
  106. {thunc-0.2.2 → thunc-0.3.0}/examples/hello.py +0 -0
  107. {thunc-0.2.2 → thunc-0.3.0}/examples/jev_hello.py +0 -0
  108. {thunc-0.2.2 → thunc-0.3.0}/examples/jev_inbox.py +0 -0
  109. {thunc-0.2.2 → thunc-0.3.0}/examples/jev_with_claude.py +0 -0
  110. {thunc-0.2.2 → thunc-0.3.0}/examples/log_triage.py +0 -0
  111. {thunc-0.2.2 → thunc-0.3.0}/examples/repo_guide.py +0 -0
  112. {thunc-0.2.2 → thunc-0.3.0}/examples/support_inbox.py +0 -0
  113. {thunc-0.2.2 → thunc-0.3.0}/examples/temporal/application.py +0 -0
  114. {thunc-0.2.2 → thunc-0.3.0}/examples/temporal/client.py +0 -0
  115. {thunc-0.2.2 → thunc-0.3.0}/examples/temporal/pipeline.py +0 -0
  116. {thunc-0.2.2 → thunc-0.3.0}/examples/temporal/worker.py +0 -0
  117. {thunc-0.2.2 → thunc-0.3.0}/live_tests/conftest.py +0 -0
  118. {thunc-0.2.2 → thunc-0.3.0}/live_tests/eval_prompts.py +0 -0
  119. {thunc-0.2.2 → thunc-0.3.0}/live_tests/test_agent.py +0 -0
  120. {thunc-0.2.2 → thunc-0.3.0}/live_tests/test_bool_decision.py +0 -0
  121. {thunc-0.2.2 → thunc-0.3.0}/live_tests/test_dict_output.py +0 -0
  122. {thunc-0.2.2 → thunc-0.3.0}/live_tests/test_hello.py +0 -0
  123. {thunc-0.2.2 → thunc-0.3.0}/live_tests/test_literal_choice.py +0 -0
  124. {thunc-0.2.2 → thunc-0.3.0}/pytest-temporal.ini +0 -0
  125. {thunc-0.2.2 → thunc-0.3.0}/tests/future_types.py +0 -0
  126. {thunc-0.2.2 → thunc-0.3.0}/tests/temporal/conftest.py +0 -0
  127. {thunc-0.2.2 → thunc-0.3.0}/tests/temporal/process_worker.py +0 -0
  128. {thunc-0.2.2 → thunc-0.3.0}/tests/temporal/test_storage.py +0 -0
  129. {thunc-0.2.2 → thunc-0.3.0}/tests/test_benchmarks.py +0 -0
  130. {thunc-0.2.2 → thunc-0.3.0}/tests/test_cache.py +0 -0
  131. {thunc-0.2.2 → thunc-0.3.0}/tests/test_calls.py +0 -0
  132. {thunc-0.2.2 → thunc-0.3.0}/tests/test_jev.py +0 -0
  133. {thunc-0.2.2 → thunc-0.3.0}/tests/test_permissions.py +0 -0
  134. {thunc-0.2.2 → thunc-0.3.0}/tests/test_profiling.py +0 -0
  135. {thunc-0.2.2 → thunc-0.3.0}/tests/test_schema.py +0 -0
  136. {thunc-0.2.2 → thunc-0.3.0}/thunc/cache.py +0 -0
  137. {thunc-0.2.2 → thunc-0.3.0}/thunc/config.py +0 -0
  138. {thunc-0.2.2 → thunc-0.3.0}/thunc/errors.py +0 -0
  139. {thunc-0.2.2 → thunc-0.3.0}/thunc/mcp_relay.py +0 -0
  140. {thunc-0.2.2 → thunc-0.3.0}/thunc/permissions.py +0 -0
  141. {thunc-0.2.2 → thunc-0.3.0}/thunc/profiling.py +0 -0
  142. {thunc-0.2.2 → thunc-0.3.0}/thunc/py.typed +0 -0
  143. {thunc-0.2.2 → thunc-0.3.0}/thunc/runs.py +0 -0
  144. {thunc-0.2.2 → thunc-0.3.0}/thunc/schema.py +0 -0
  145. {thunc-0.2.2 → thunc-0.3.0}/thunc/store.py +0 -0
  146. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/__init__.py +0 -0
  147. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/adapters.py +0 -0
  148. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/client.py +0 -0
  149. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/models.py +0 -0
  150. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/storage.py +0 -0
  151. {thunc-0.2.2 → thunc-0.3.0}/thunc/temporal/worker.py +0 -0
@@ -0,0 +1,74 @@
1
+ thunc 0.2.2 makes **agents use their tools reliably**, especially on Claude Code, and makes calls faster. Agents on Claude Code now make native tool calls instead of writing each action as JSON text, a failed step is retried instead of ending the run, and the agent's tools fill gaps that a new tool-use benchmark found.
2
+
3
+ | 8 tool-heavy tasks, Claude Sonnet 5.5 | Passed | Seconds per task | $ per task |
4
+ |---|---|---|---|
5
+ | thunc 0.2.1 agents on Claude Code | 12/24 | 99 | 0.084 |
6
+ | **thunc 0.2.2 agents on Claude Code** | **24/24** | **10** | **0.022** |
7
+ | Claude Code itself, for reference | 24/24 | 11 | 0.069 |
8
+
9
+ On Claude Opus 5.5 both versions passed every task, but 0.2.2 took 15 seconds and $0.051 a task instead of 44 seconds and $0.264.
10
+
11
+ ```bash
12
+ pip install --upgrade thunc
13
+ ```
14
+
15
+ ```python
16
+ import thunc
17
+
18
+ thunc.configure(backend="claude-code") # agents now make native tool calls here
19
+ fixer = thunc.Agent("fixer", workdir=".", permissions=["write:src/**", "run:pytest"])
20
+
21
+
22
+ @fixer.task
23
+ def make_tests_pass() -> str:
24
+ """Run the tests in services/api and fix what fails. Say what you changed."""
25
+ ...
26
+
27
+
28
+ run = fixer.run(make_tests_pass) # run.commands, run.files_changed, and any retries in run.session
29
+ ```
30
+
31
+ ```bash
32
+ thunc run --profile my_script.py # where the time went: model time, thunc's own, per tool
33
+ ```
34
+
35
+ **Agents on Claude Code make native tool calls** ([#46](https://github.com/Eltarras/thunc/pull/46))
36
+ - The agent's tools are an MCP server that one `claude -p` process per run calls, and thunc carries out each call with its own tools, permissions and run record. Before, the model wrote each action as JSON text. Current models slip back into their trained tool calls: they invented tool results and ran on to the timeout, or the CLI refused a tool call it couldn't parse.
37
+ - Calls from one model reply count as one step. A turn without a tool call gets a nudge, and a turn that ends in an error gets a second chance, both in the same session.
38
+ - **Fallback:** with a `claude` CLI too old for these options, or with MCP servers turned off by a policy, a run falls back to the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
39
+ - The CLI runs in the agent's `workdir`, so the environment details it adds to the prompt name the right folder.
40
+
41
+ **Steadier runs** ([#46](https://github.com/Eltarras/thunc/pull/46))
42
+ - **A failed step is retried** twice before the run fails, and each retry is logged in the run record. This covers timeouts, lost connections, rate limits, server errors and CLI calls that end in an error (`thunc.errors.TransientError`). Before, one bad call ended the run.
43
+ - A text-protocol step on Claude Code or Codex now times out after 120 seconds instead of the whole 300-second `timeout`.
44
+
45
+ **Better tools for agents** ([#46](https://github.com/Eltarras/thunc/pull/46))
46
+ - **`run` takes `cwd`**: a folder inside `workdir` to run the command in.
47
+ - **The `shell` permission** (opt-in) runs command lines through the system shell, so pipes, `&&`, `cd` and redirects work. It can't be combined with `!run:` rules, since a shell command line can't be checked word by word.
48
+ - **`search` takes `glob`**: `*.py` matches by file name, `src/**/*.ts` by path.
49
+ - **Git-aware `list` and `search`**: in a git repository they leave out what git ignores, so a stale `build/` copy no longer crowds out the real source.
50
+ - **Long command output keeps its start and its end**, so an agent sees the first error as well as the summary.
51
+
52
+ **Faster calls** ([#45](https://github.com/Eltarras/thunc/pull/45))
53
+ - **The `anthropic` and `openai` backends reuse their connections**: one SDK client per process instead of a new TCP and TLS handshake each call. In a local benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
54
+ - **Text-protocol agents can act several times per reply** (Codex and `protocol="text"`), with a JSON array of independent actions. On Codex, the prompt eval took fewer replies (fix / review / analysis: 6.0 / 4.8 / 5.0 → 5.0 / 3.0 / 3.6) and less time (37 / 27 / 27 s → 29 / 19 / 21 s).
55
+ - **The `codex` backend returns as soon as the answer arrives**, without waiting about 0.4 s for Codex to shut down.
56
+ - **`thunc run --profile SCRIPT`** (or `-m module`) reports where a program's time went, per function: calls, cache hits, retries, failures, model time against thunc's own, and time per tool for agents. `python -m benchmarks` runs a dependency-free benchmark suite.
57
+
58
+ **Tested**
59
+ - Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows. Temporal integration tests run on Python 3.10 and 3.14.
60
+ - The MCP path has an end-to-end test with a fake `claude` that starts the real tool server, covering parallel and serial calls, permissions, errors and the fallback.
61
+ - Live: the tool-use benchmark (`live_tests/bench_tooluse.py`, with its report in `live_tests/bench_tooluse_report.md`) on Sonnet 5.5 and Opus 5.5, the prompt eval on Claude Code (45/45), and the live tests on Claude Code.
62
+ - The agent loops on the Claude and OpenAI APIs haven't had a live tool-use benchmark yet.
63
+
64
+ **Behavior changes**
65
+ - Agents on `claude-code` make native tool calls by default, and fall back to the text protocol if they can't.
66
+ - In a git repository, `list` and `search` leave out what git ignores. A folder named explicitly is still listed and searched.
67
+ - Long command output keeps its start as well as its end.
68
+ - The `claude-code` backend loads none of your Claude Code settings (`--setting-sources ""`). No `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so a folder you point an agent at can't instruct it unless you pass `follow=`.
69
+ - Durable runs on Claude Code still use the text protocol.
70
+ - Nothing is removed.
71
+
72
+ Still beta: expect bugs, and the API may change. Known issues are listed in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
73
+
74
+ **Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.1...v0.2.2
@@ -0,0 +1,80 @@
1
+ thunc 0.2.3 is the last 0.2 release. It makes **native tool calls the way agents work wherever they run**: on Codex now, as on Claude Code, and in durable runs on Claude Code, which pick up after a crash mid-call. Durable agents can take `tools=`, the Claude API agent path no longer loses runs to `max_tokens` or a stalled reply, and `edit` can change many places at once.
2
+
3
+ | 8 tool-heavy tasks, Claude Sonnet 5.5, 3 runs each | Passed | Seconds per task | $ per task |
4
+ |---|---|---|---|
5
+ | thunc 0.2.2 agents, text protocol (Codex, durable runs, fallbacks) | 20/24 | 47 | 0.084 |
6
+ | **thunc 0.2.3 agents, text protocol** | **24/24** | **27** | 0.082 |
7
+ | thunc 0.2.3 agents on Claude Code, native calls | 24/24 | 8 | 0.021 |
8
+ | Claude Code itself, for reference | 24/24 | 10 | 0.078 |
9
+
10
+ Through the Claude API, thunc's agents passed 8 of 8 on Sonnet 5.5 ($0.023 a task) and on Opus 5.5 ($0.049 a task). On Codex, native calls took 30 seconds a task, against 45 for the text protocol 0.2.2 used there and 27 for Codex itself (16 of 16 each).
11
+
12
+ ```bash
13
+ pip install --upgrade thunc
14
+ ```
15
+
16
+ ```python
17
+ import thunc
18
+
19
+ thunc.configure(backend="codex") # agents now make native tool calls here too
20
+
21
+
22
+ def find_issue(title: str) -> int: # a tool is your own code, with a docstring for the model
23
+ """Find an open issue by title. Returns its number, or 0."""
24
+ return tracker.find(title)
25
+
26
+
27
+ triage = thunc.Agent("triage", workdir=".", tools=[find_issue], effort="high")
28
+
29
+
30
+ @triage.task
31
+ def flaky_tests() -> list[str]:
32
+ """Run the tests three times and list the ones that fail only sometimes."""
33
+ ...
34
+ ```
35
+
36
+ ```python
37
+ # On a Temporal worker: durable runs take the agent's own tools now, and name those that may run
38
+ # again after a crash.
39
+ registry.agent_task("triage", flaky_tests, version="1", workspace_id="repo", retry_safe_tools=["find_issue"])
40
+ ```
41
+
42
+ **Native tool calls on Codex** ([#55](https://github.com/Eltarras/thunc/pull/55))
43
+ - The agent's tools are an MCP server that Codex calls, through the same relay as Claude Code: thunc carries out each call with its own tools, permissions and run record. Codex's own tools and your `~/.codex` config stay out, its sandbox stays read-only, and only thunc's server is approved to run without asking.
44
+ - A turn that ends without `finish`, or fails, is continued with `codex exec resume`. The run's Codex session is deleted when the run ends.
45
+ - **Fallback:** when Codex can't start them, a run uses the text protocol with a warning. `protocol="native"` raises instead, and `protocol="text"` keeps the old way.
46
+
47
+ **Durable runs** ([#56](https://github.com/Eltarras/thunc/pull/56), [#57](https://github.com/Eltarras/thunc/pull/57))
48
+ - **On Claude Code, native calls in segments.** One activity keeps one `claude -p` process for many model replies, and checkpoints the run and Claude Code's session before each reply's calls. If the worker stops or the CLI dies, the retried activity restores the session and continues it with `--resume`. The call that was in flight, asked for again, gets its old journal entry: it's replayed, waits for `resolve()`, or runs. Never twice.
49
+ - **`tools=`.** Each call of the agent's own functions is journaled like a command: recorded before it runs and after, replayed when completed, and waiting for `resolve()` if a worker stopped mid-call. `retry_safe_tools=` names the ones that may simply run again.
50
+
51
+ **The Claude API agent path** ([#52](https://github.com/Eltarras/thunc/pull/52))
52
+ - **`Agent(effort=...)`** on every backend. Claude 4.6 and later on the API get `"high"` by default; Claude Opus 5.5's own default is `"medium"`, which is low for agentic coding.
53
+ - Replies are streamed with room for 64,000 tokens. A reply cut off at `max_tokens` gets an error result for its calls instead of ending the run, `pause_turn` carries on, and a reply that sends nothing for `timeout` seconds is stopped and asked again (one Opus reply in the benchmark sent nothing for an hour).
54
+ - Old tool results are cleared on long runs (context editing), tools are strict where the model and schema allow, and connection errors, rate limits and server errors are retried as a step.
55
+
56
+ **Steadier agents everywhere** ([#51](https://github.com/Eltarras/thunc/pull/51), [#52](https://github.com/Eltarras/thunc/pull/52), [#53](https://github.com/Eltarras/thunc/pull/53))
57
+ - **The text protocol reads replies that aren't only the action**: the first complete action amid prose, a fence, `<invoke>` markup or made-up results, with a note to the model to keep to JSON. On Claude Code, a text step stops as soon as its action has arrived.
58
+ - **`finish` in the same reply as other calls is refused** (except beside `remember`): in the benchmark, a run returned a guess written before its own read came back.
59
+ - **The model is told when few steps are left**, in its last three replies before `max_steps`.
60
+
61
+ **Better tools** ([#50](https://github.com/Eltarras/thunc/pull/50), [#51](https://github.com/Eltarras/thunc/pull/51), [#49](https://github.com/Eltarras/thunc/pull/49))
62
+ - **`edit` takes `replace_all`, and `edits`** for several changes to one file in one call, all made or none.
63
+ - **`edit` no longer needs a prior `read`**: it only changes text the agent quotes exactly. `write` still replaces only a file read with `read`.
64
+ - **Large system prompts work on Claude Code**: the prompt goes in a file, not on the command line, where large `follow=` files could pass the operating system's limit.
65
+
66
+ **Tested**
67
+ - Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows; Temporal integration tests on Python 3.10 and 3.14, including a Claude Code run whose CLI dies mid-run. Tests can't reach a real `claude`, `codex` or `jev`.
68
+ - Live: the tool-use benchmark on Claude Code (Sonnet 5.5), the Claude API (Sonnet 5.5 and Opus 5.5) and Codex, with the results in `live_tests/bench_tooluse_report.md`; a durable run on real Claude Code with the CLI killed after its first tool result, which finished with each effect done once.
69
+
70
+ **Behavior changes**
71
+ - Agents on `codex` make native tool calls by default, and fall back to the text protocol if they can't.
72
+ - Durable runs on `claude-code` make native tool calls. Runs already in progress keep the text protocol.
73
+ - Agents on the Claude API think at effort `high` on Claude 4.6 and later.
74
+ - `edit` no longer needs a prior `read`; `write` still does.
75
+ - `finish` in the same reply as other calls is refused, except beside `remember`.
76
+ - Nothing is removed.
77
+
78
+ **Known issues:** on Codex, native calls count each tool call as a step, so a long task reaches `max_steps` sooner, and their tokens aren't reported. Durable runs on Codex use the text protocol. On a fix in a long file, native calls take about twice Claude Code's steps. The full list is in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
79
+
80
+ **Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.2...v0.2.3
@@ -0,0 +1,86 @@
1
+ thunc 0.3.0 lets you **watch a program's calls and agent runs live**. `thunc watch app.py` runs your program with a dashboard in the terminal: the calls waiting on a model, retries and why each reply was rejected, timings for each function, each agent's steps as they happen, and the `--profile` report when the program ends. A program needs nothing added to be watched.
2
+
3
+ It also brings an experiment, **thunc write**: a function declared with `@thunc.function(write=True)` writes its own body into your file on its first call, and runs as plain Python from then on. A program that uses neither behaves exactly as in 0.2.3.
4
+
5
+ ```bash
6
+ pip install --upgrade "thunc[watch]"
7
+ ```
8
+
9
+ ```bash
10
+ thunc watch support_inbox.py --limit 20 # a script and its arguments, as with thunc run
11
+ thunc watch -m myapp.triage # a module
12
+ thunc watch -- uv run app.py # any command
13
+ thunc watch --agents # agent runs in ./.thunc_agents, from any process
14
+ thunc watch --plain app.py # one line per event, for CI logs and pipes
15
+ ```
16
+
17
+ ```
18
+ thunc watch support_inbox.py claude-code/sonnet 00:18.0 ● running
19
+ 1 Overview 2 Agents 3 Calls 4 Summary
20
+ ────────────────────────────────────────────────────────────────────────────────────────────
21
+ IN FLIGHT 4 running
22
+ ⠙ draft_reply ticket="Password reset email never… attempt 1 4.2s ████████████
23
+ ⠙ urgency ticket="Refund still not showing a… attempt 2 4.1s ██████████░░
24
+ FUNCTION CALLS CACHED RETRIES FAILED MEAN P95 MODEL RECENT
25
+ category 8 0 0 0 1.5s 2.5s 11.7s ▃▄▅█▆▇▃▄
26
+ urgency 7 0 1 0 2.4s 4.7s 16.9s ▂▄▄▆▃█▅
27
+ AGENTS 0 running
28
+ ✓ repo-guide tests_for(feature="caching") finished · 6 steps 12.4s
29
+ EVENTS
30
+ 17:38:13 ✓ find_order → Order(id='A-1043') 1.8s
31
+ 17:38:15 ↻ urgency attempt 1: not valid JSON: 'high' 2.8s
32
+ ```
33
+
34
+ **The dashboard** ([#59](https://github.com/Eltarras/thunc/pull/59))
35
+ - **Five screens:** Overview (calls in flight, a table per function, running agents, results per second, recent events), Agents (each run's steps, time split, changed files and denials), Calls (each attempt of a call: the reply and why it was rejected), Summary (the `thunc run --profile` tables, opened when the program ends) and Output (what the program printed).
36
+ - **Mouse or keys.** Click a row to select it and again to open it, or use `↑` `↓` and `⏎`. `f` shows only retries, failures and denials; `p` pauses the display while the program keeps running; `?` lists every key.
37
+ - **`--agents [DIR]`** follows the agent runs recorded in a folder, from another terminal, a web app or a worker. A run whose process died partway through shows as interrupted.
38
+ - **`--save-events` and `--replay`** keep a run and play it back, which is a good thing to attach to a bug report. `--replay` also plays one agent run's session record.
39
+ - **`--plain`** prints one line per result, retry and agent step to stderr, then the report. It's what you get when the output isn't a terminal.
40
+
41
+ **How it's packaged**
42
+ - The dashboard is a Rust binary in its own package, [thunc-watch](https://pypi.org/project/thunc-watch/) 0.1, with wheels for macOS, Linux (glibc and musl) and Windows. `thunc[watch]` installs it, so thunc itself stays pure Python with no dependencies. It's also on your `PATH` as `thunc-watch`.
43
+ - Without it, `thunc watch` says how to install it.
44
+
45
+ **`THUNC_EVENTS`**
46
+ - When `THUNC_EVENTS` names a file, thunc appends one JSON line to it per call start, model reply, call end, agent tool call and agent end; `thunc watch` sets it for the program it runs. To watch a program started some other way, set it yourself and run `thunc watch --events FILE`.
47
+ - Inputs, replies and values are cut to 120-character previews; `--capture` (or `THUNC_EVENTS_CAPTURE=1`) sends them whole. Like a trace, they can contain personal data.
48
+ - When it isn't set, nothing is written.
49
+
50
+ **thunc write, experimental** ([#63](https://github.com/Eltarras/thunc/pull/63), [#64](https://github.com/Eltarras/thunc/pull/64))
51
+
52
+ > **Experimental.** Its behavior, options and the code it writes may change, or it may be removed, in a later release without a deprecation period. It edits your source files, only while you develop; review each change as a diff before you commit it.
53
+
54
+ ```python
55
+ @thunc.function(write=True)
56
+ def minutes(duration: str) -> int:
57
+ """Convert a duration like '1h 30m', '90 min' or '2 hours' to whole minutes."""
58
+ ...
59
+
60
+
61
+ minutes("1h 30m") # -> 90, and minutes() is now Python in your file
62
+ ```
63
+
64
+ ```
65
+ thunc: writing minutes() in durations.py (first call)
66
+ thunc: checked against 6 model answers: all agree
67
+ thunc: wrote durations.py lines 5-27 in 14s (answer 4s, draft 11s, test calls 9s; side by side). Removed @thunc.function. Review: git diff durations.py
68
+ ```
69
+
70
+ - **On the first call**, three requests start side by side: the call's answer, a draft of the body from the docstring and signature, and five test calls, each answered by the model on its own. The draft is linted, run on the call and the test calls, and has to match every answer; one that doesn't goes back with the failing calls, up to three drafts.
71
+ - **A passing draft goes into the file** in place of `...`, the decorator is removed, the checked calls become doctest examples, and the call runs the new code.
72
+ - **When it isn't written**, because the task needs judgment or no draft passed, the call returns the model's answer and the file is untouched. The reason is kept in `.thunc_write/` until the docstring or signature changes.
73
+ - **It's refused, with a warning,** in CI or with `THUNC_WRITE=0` (set it in production), outside your project, and for read-only, installed or changed files and nested functions.
74
+ - **`thunc write FILE::FUNCTION [--dry-run]`** writes one ahead of its first call, or shows the diff.
75
+
76
+ **Tested**
77
+ - Offline tests run on Python 3.10–3.14 on Ubuntu, and on Windows, including the events and the `thunc watch` command. CI runs the dashboard's `cargo fmt`, `clippy` and tests on Linux, macOS and Windows, and on Linux builds the PyPI wheel and runs `thunc watch` from it.
78
+ - thunc write: 45 offline tests of the source editing and the writing, end to end against a scripted model, including CRLF files; live, `live_tests/test_writing.py` on Codex wrote a function whose doctests pass.
79
+ - `watch/demo/demo_app.py` runs the support-inbox functions and an agent against a scripted backend, with delays and bad replies, and makes no model calls.
80
+
81
+ **Behavior changes**
82
+ - None. `thunc watch`, `THUNC_EVENTS` and `write=` are new; nothing is written unless `THUNC_EVENTS` is set, and no file is edited unless a function has `write=True`. Nothing is removed.
83
+
84
+ **Known issues:** the dashboard only watches: it can't stop a single agent run or approve `ask:` actions yet. `--agents` sees one folder, not every project on the machine. On Windows, Ctrl+C and quitting end the program at once, without the `KeyboardInterrupt` it gets on macOS and Linux. thunc write's checks are only as good as the model's answers, so review the code it writes; `thunc write` can't write methods ahead of their first call, and it doesn't run on `jev`. The full list is in [CONTRIBUTING.md](https://github.com/Eltarras/thunc/blob/main/CONTRIBUTING.md#known-issues), and every change is in [CHANGELOG.md](https://github.com/Eltarras/thunc/blob/main/CHANGELOG.md).
85
+
86
+ **Full changelog:** https://github.com/Eltarras/thunc/compare/v0.2.3...v0.3.0
@@ -36,6 +36,32 @@ jobs:
36
36
  env:
37
37
  PYTHONIOENCODING: utf-8 # the console's own code page can't print every test id and message
38
38
 
39
+ # thunc-watch, the terminal dashboard: a Rust crate in watch/, separate from the Python package.
40
+ watch:
41
+ strategy:
42
+ fail-fast: false # each platform reports its own result
43
+ matrix:
44
+ os: [ubuntu-latest, macos-latest, windows-latest]
45
+ runs-on: ${{ matrix.os }}
46
+ defaults:
47
+ run:
48
+ working-directory: watch
49
+ steps:
50
+ - uses: actions/checkout@v4
51
+ - uses: dtolnay/rust-toolchain@stable
52
+ with:
53
+ components: rustfmt, clippy
54
+ - run: cargo fmt --check
55
+ - run: cargo clippy --all-targets --locked -- -D warnings
56
+ - run: cargo test --locked
57
+ - name: The PyPI wheel builds and runs
58
+ if: matrix.os == 'ubuntu-latest'
59
+ run: |
60
+ pipx run maturin build --release --locked --out dist
61
+ python -m venv /tmp/wheel-check
62
+ /tmp/wheel-check/bin/pip install --quiet dist/*.whl ..
63
+ /tmp/wheel-check/bin/thunc watch --version
64
+
39
65
  temporal-integration:
40
66
  runs-on: ubuntu-latest
41
67
  strategy:
@@ -0,0 +1,81 @@
1
+ name: Publish thunc-watch to PyPI
2
+
3
+ # Runs when you publish a GitHub release tagged watch-vX.Y.Z (thunc's own vX.Y.Z releases are
4
+ # publish.yml's). Builds the dashboard's wheels for each platform and publishes them as the
5
+ # thunc-watch package, with PyPI Trusted Publishing: no API token is stored anywhere.
6
+
7
+ on:
8
+ release:
9
+ types: [published]
10
+
11
+ jobs:
12
+ version:
13
+ if: startsWith(github.event.release.tag_name, 'watch-v')
14
+ runs-on: ubuntu-latest
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - name: The tag matches watch/Cargo.toml
18
+ env:
19
+ TAG: ${{ github.event.release.tag_name }}
20
+ run: |
21
+ version=$(grep -m1 '^version' watch/Cargo.toml | cut -d'"' -f2)
22
+ if [ "${TAG#watch-v}" != "$version" ]; then
23
+ echo "The tag is $TAG but watch/Cargo.toml says $version" >&2
24
+ exit 1
25
+ fi
26
+
27
+ wheels:
28
+ needs: version
29
+ strategy:
30
+ fail-fast: false
31
+ matrix:
32
+ include:
33
+ - { os: ubuntu-latest, target: x86_64, manylinux: auto }
34
+ - { os: ubuntu-latest, target: aarch64, manylinux: auto }
35
+ - { os: ubuntu-latest, target: x86_64, manylinux: musllinux_1_2 }
36
+ - { os: ubuntu-latest, target: aarch64, manylinux: musllinux_1_2 }
37
+ - { os: macos-latest, target: aarch64, manylinux: "" }
38
+ - { os: macos-latest, target: x86_64, manylinux: "" }
39
+ - { os: windows-latest, target: x64, manylinux: "" }
40
+ runs-on: ${{ matrix.os }}
41
+ steps:
42
+ - uses: actions/checkout@v4
43
+ - uses: PyO3/maturin-action@v1
44
+ with:
45
+ working-directory: watch
46
+ target: ${{ matrix.target }}
47
+ manylinux: ${{ matrix.manylinux || 'auto' }}
48
+ args: --release --locked --out dist
49
+ - uses: actions/upload-artifact@v4
50
+ with:
51
+ name: wheels-${{ matrix.os }}-${{ matrix.target }}-${{ matrix.manylinux }}
52
+ path: watch/dist
53
+
54
+ sdist:
55
+ needs: version
56
+ runs-on: ubuntu-latest
57
+ steps:
58
+ - uses: actions/checkout@v4
59
+ - uses: PyO3/maturin-action@v1
60
+ with:
61
+ working-directory: watch
62
+ command: sdist
63
+ args: --out dist
64
+ - uses: actions/upload-artifact@v4
65
+ with:
66
+ name: wheels-sdist
67
+ path: watch/dist
68
+
69
+ publish:
70
+ needs: [wheels, sdist]
71
+ runs-on: ubuntu-latest
72
+ environment: pypi
73
+ permissions:
74
+ id-token: write # lets PyPI verify this workflow; required for Trusted Publishing
75
+ steps:
76
+ - uses: actions/download-artifact@v4
77
+ with:
78
+ pattern: wheels-*
79
+ merge-multiple: true
80
+ path: dist/
81
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -2,6 +2,7 @@ name: Publish to PyPI
2
2
 
3
3
  # Runs when you publish a GitHub release (e.g. tag v0.1.0). Uses PyPI Trusted Publishing:
4
4
  # no API token is stored anywhere; PyPI trusts this workflow in this repo.
5
+ # Releases tagged watch-vX.Y.Z are the dashboard's, published by publish-watch.yml instead.
5
6
 
6
7
  on:
7
8
  release:
@@ -9,6 +10,7 @@ on:
9
10
 
10
11
  jobs:
11
12
  build:
13
+ if: ${{ !startsWith(github.event.release.tag_name, 'watch-v') }}
12
14
  runs-on: ubuntu-latest
13
15
  steps:
14
16
  - uses: actions/checkout@v4
@@ -0,0 +1,46 @@
1
+ name: Release notes
2
+
3
+ # Run by hand (Actions > Release notes > Run workflow) with a version like 0.2.2. It writes the
4
+ # notes in .github/release-notes/v<version>.md to that release, and sets its title to
5
+ # "v<version> (beta)". A release that doesn't exist yet is created as a draft pre-release on the
6
+ # commit the workflow runs from, for you to review and publish. Editing or drafting a release
7
+ # doesn't trigger publish.yml, which runs only when you publish one.
8
+
9
+ on:
10
+ workflow_dispatch:
11
+ inputs:
12
+ version:
13
+ description: "Version, like 0.2.2"
14
+ required: true
15
+
16
+ permissions:
17
+ contents: write # create and edit releases
18
+
19
+ jobs:
20
+ notes:
21
+ runs-on: ubuntu-latest
22
+ steps:
23
+ - uses: actions/checkout@v4
24
+ - name: Write the notes to the release
25
+ env:
26
+ GH_TOKEN: ${{ github.token }}
27
+ VERSION: ${{ inputs.version }}
28
+ run: |
29
+ if [[ ! "$VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
30
+ echo "::error::version must look like 0.2.2, not '$VERSION'"
31
+ exit 1
32
+ fi
33
+ tag="v$VERSION"
34
+ notes=".github/release-notes/$tag.md"
35
+ if [[ ! -f "$notes" ]]; then
36
+ echo "::error::$notes doesn't exist; add it first"
37
+ exit 1
38
+ fi
39
+ if gh release view "$tag" > /dev/null 2>&1; then
40
+ gh release edit "$tag" --title "$tag (beta)" --notes-file "$notes"
41
+ echo "Updated the notes of $tag."
42
+ else
43
+ gh release create "$tag" --draft --prerelease --target "$GITHUB_SHA" \
44
+ --title "$tag (beta)" --notes-file "$notes"
45
+ echo "Created $tag as a draft pre-release."
46
+ fi
@@ -13,3 +13,5 @@ dist/
13
13
  .temporal-dev/
14
14
  examples/temporal/temporal-state/
15
15
  examples/temporal/temporal-workspace/
16
+ watch/target/
17
+ watch/dist/