thunc 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thunc-0.2.2/.github/demo-agent.gif +0 -0
- thunc-0.2.2/.github/demo-function.gif +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/CHANGELOG.md +69 -0
- {thunc-0.2.1 → thunc-0.2.2}/CONTRIBUTING.md +76 -13
- {thunc-0.2.1 → thunc-0.2.2}/PKG-INFO +119 -53
- {thunc-0.2.1 → thunc-0.2.2}/README.md +117 -52
- thunc-0.2.2/benchmarks/README.md +32 -0
- thunc-0.2.2/benchmarks/__init__.py +1 -0
- thunc-0.2.2/benchmarks/__main__.py +60 -0
- thunc-0.2.2/benchmarks/bench_agent.py +62 -0
- thunc-0.2.2/benchmarks/bench_backends.py +84 -0
- thunc-0.2.2/benchmarks/bench_cache.py +34 -0
- thunc-0.2.2/benchmarks/bench_calls.py +127 -0
- thunc-0.2.2/benchmarks/bench_schema.py +85 -0
- thunc-0.2.2/benchmarks/bench_startup.py +30 -0
- thunc-0.2.2/benchmarks/bench_tools.py +102 -0
- thunc-0.2.2/benchmarks/fakes.py +101 -0
- thunc-0.2.2/benchmarks/harness.py +190 -0
- thunc-0.2.2/design/og-card.html +61 -0
- thunc-0.2.2/docs/docs/agents.html +189 -0
- thunc-0.2.2/docs/docs/api.html +150 -0
- thunc-0.2.2/docs/docs/backends.html +122 -0
- thunc-0.2.2/docs/docs/caching.html +109 -0
- thunc-0.2.2/docs/docs/functions.html +172 -0
- thunc-0.2.2/docs/docs/index.html +139 -0
- thunc-0.2.2/docs/docs/jev.html +182 -0
- thunc-0.2.2/docs/docs/temporal.html +147 -0
- thunc-0.2.2/docs/index.html +215 -0
- thunc-0.2.2/docs/jev.html +14 -0
- thunc-0.2.2/docs/og.png +0 -0
- thunc-0.2.2/docs/site.css +218 -0
- thunc-0.2.2/docs/site.js +94 -0
- thunc-0.2.2/docs/sitemap.xml +12 -0
- thunc-0.2.2/docs/temporal.html +14 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/repo_guide.py +4 -2
- thunc-0.2.2/live_tests/bench_tooluse.py +903 -0
- thunc-0.2.2/live_tests/bench_tooluse_report.md +385 -0
- {thunc-0.2.1 → thunc-0.2.2}/pyproject.toml +2 -1
- thunc-0.2.2/tests/fake_claude.py +135 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_runtime.py +37 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_agent.py +252 -6
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_backends.py +206 -69
- thunc-0.2.2/tests/test_benchmarks.py +13 -0
- thunc-0.2.2/tests/test_claude_code_agent.py +267 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_native.py +6 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_permissions.py +19 -0
- thunc-0.2.2/tests/test_profiling.py +119 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/__init__.py +1 -1
- {thunc-0.2.1 → thunc-0.2.2}/thunc/__main__.py +50 -2
- {thunc-0.2.1 → thunc-0.2.2}/thunc/agent.py +115 -26
- {thunc-0.2.1 → thunc-0.2.2}/thunc/backends.py +150 -30
- thunc-0.2.2/thunc/claude_code.py +351 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/core.py +25 -4
- thunc-0.2.2/thunc/errors.py +15 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/execution.py +4 -0
- thunc-0.2.2/thunc/mcp_relay.py +77 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/native.py +54 -18
- {thunc-0.2.1 → thunc-0.2.2}/thunc/permissions.py +23 -7
- thunc-0.2.2/thunc/profiling.py +261 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/tools.py +140 -41
- {thunc-0.2.1 → thunc-0.2.2}/uv.lock +1 -1
- thunc-0.2.1/docs/index.html +0 -279
- thunc-0.2.1/docs/jev.html +0 -325
- thunc-0.2.1/docs/sitemap.xml +0 -15
- thunc-0.2.1/docs/temporal.html +0 -40
- thunc-0.2.1/thunc/errors.py +0 -2
- {thunc-0.2.1 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/.github/social-preview.png +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/.github/workflows/ci.yml +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/.github/workflows/publish.yml +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/.gitignore +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/LICENSE +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/.nojekyll +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/apple-touch-icon.png +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/favicon-96.png +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/favicon.ico +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/favicon.svg +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/docs/google2cd177e5e85b3c3e.html +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/dynamic_prompts.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/hello.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/jev_hello.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/jev_inbox.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/jev_with_claude.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/log_triage.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/support_inbox.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/README.md +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/application.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/client.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/pipeline.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/examples/temporal/worker.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/conftest.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/eval_prompts.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_agent.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_bool_decision.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_dict_output.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_hello.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/live_tests/test_literal_choice.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/pytest-temporal.ini +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/conftest.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/future_types.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/conftest.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/process_worker.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_contract.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/temporal/test_storage.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_cache.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_calls.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_cli.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_execution.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_jev.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/tests/test_schema.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/cache.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/config.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/decorator.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/prompts.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/py.typed +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/runs.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/schema.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/store.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/__init__.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/activities.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/adapters.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/client.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/effects.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/models.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/registry.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/storage.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/worker.py +0 -0
- {thunc-0.2.1 → thunc-0.2.2}/thunc/temporal/workflows.py +0 -0
|
Binary file
|
|
Binary file
|
|
@@ -3,6 +3,75 @@
|
|
|
3
3
|
All notable changes to thunc. The full notes for each release are on the
|
|
4
4
|
[releases page](https://github.com/Eltarras/thunc/releases).
|
|
5
5
|
|
|
6
|
+
## 0.2.2 (beta)
|
|
7
|
+
|
|
8
|
+
**Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
|
|
9
|
+
calls instead of writing each action as JSON text, a failed step is retried instead of ending the
|
|
10
|
+
run, and the agent's tools fill gaps a benchmark found. In that tool-use benchmark
|
|
11
|
+
(`live_tests/bench_tooluse.py`, 8 tasks, Claude Sonnet 5.5, 3 runs each), agents on Claude Code went
|
|
12
|
+
from 12 of 24 runs passing to 24 of 24, from 99 to 10 seconds a task, and from $0.084 to $0.022 a
|
|
13
|
+
task; Claude Code itself took 11 seconds and $0.069. The API backends also reuse their connections,
|
|
14
|
+
Codex answers return sooner, and `thunc run --profile` shows where a program's time goes.
|
|
15
|
+
|
|
16
|
+
### Behavior changes
|
|
17
|
+
|
|
18
|
+
Nothing is removed, but these defaults change:
|
|
19
|
+
|
|
20
|
+
- **Agents on `claude-code` make native tool calls** (see Added). With a `claude` CLI too old for
|
|
21
|
+
them, or with MCP servers turned off by a policy, a run falls back to the text protocol with a
|
|
22
|
+
warning. `protocol="text"` keeps the old way.
|
|
23
|
+
- **`list` and `search` leave out what git ignores** in a git repository (build output, caches,
|
|
24
|
+
vendored code). A folder named explicitly is still listed and searched.
|
|
25
|
+
- **Long command output keeps its start and its end** (the first error and the summary), not only
|
|
26
|
+
the end.
|
|
27
|
+
- **The `claude-code` backend loads none of your Claude Code settings** (`--setting-sources ""`):
|
|
28
|
+
no `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so an
|
|
29
|
+
agent's `workdir` can't give it instructions unless `follow=` asks for them.
|
|
30
|
+
|
|
31
|
+
### Added
|
|
32
|
+
|
|
33
|
+
- **Native calls on `claude-code`**: the agent's tools are an MCP server that one `claude -p`
|
|
34
|
+
process per run calls; thunc carries out each call with its own tools, permissions and run
|
|
35
|
+
record. The CLI runs in the agent's `workdir`. `protocol="text"` keeps the old way, and durable
|
|
36
|
+
runs on Claude Code still use it. When Claude Code can't start native calls (an older CLI, or MCP
|
|
37
|
+
servers turned off by a policy), a run falls back to the text protocol with a warning and a
|
|
38
|
+
`fallback` entry in its record; `protocol="native"` raises instead.
|
|
39
|
+
- **`run` takes `cwd`**, a folder inside `workdir` to run the command in.
|
|
40
|
+
- **The `shell` permission** runs command lines through the system shell, so pipes, `&&`, `cd`
|
|
41
|
+
and redirects work. Off by default; it can't be combined with `!run:` rules.
|
|
42
|
+
- **`search` takes `glob`** (`*.py` by file name, `src/**/*.ts` by path) to limit the files
|
|
43
|
+
searched.
|
|
44
|
+
- **`thunc run --profile`**: runs a script (or `-m module`) and prints a performance report to
|
|
45
|
+
stderr when it ends: per function, calls, cache hits, retries, failures, total/mean/p95/max time
|
|
46
|
+
and the split between model time and thunc's own; for agents, steps and time in each tool; and
|
|
47
|
+
the share of wall time spent in thunc, with the overlap from `thunc.map`.
|
|
48
|
+
|
|
49
|
+
### Changed
|
|
50
|
+
|
|
51
|
+
- **A failed step is retried.** A timeout, lost connection, rate limit, server error or CLI call
|
|
52
|
+
that ended in an error (`thunc.errors.TransientError`) is retried twice in an agent run, with a
|
|
53
|
+
note in the run record, before the run fails. A text-protocol step on Claude Code or Codex may
|
|
54
|
+
take 120 seconds before it's retried, instead of the whole `timeout`.
|
|
55
|
+
- **The `anthropic` and `openai` backends reuse their connections.** One SDK client is shared by
|
|
56
|
+
every call in the process (`thunc.map`'s threads and agent runs included), instead of a new
|
|
57
|
+
client, and so a new TCP and TLS handshake, for each call. A new client is made when the API key,
|
|
58
|
+
the SDK's environment variables (`ANTHROPIC_*`, `OPENAI_*`) or the process change. In a local
|
|
59
|
+
benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
|
|
60
|
+
- **Agents on the text protocol can act several times per reply.** On Codex and `protocol="text"`
|
|
61
|
+
(and on Claude Code when it falls back to the text protocol), a reply can be a JSON array of
|
|
62
|
+
independent actions (reading three files) instead of one. They run in order, at most 16 per reply,
|
|
63
|
+
and every result comes back together, as with native tool calls. Each turn resends the whole
|
|
64
|
+
transcript and, on the CLI backends, starts the CLI, so fewer turns save both. A single JSON
|
|
65
|
+
action works as before. On Codex, with `live_tests/eval_prompts.py` (default prompt, 5 runs of
|
|
66
|
+
each task), every run batched its first reads: replies went from 6.0 / 4.8 / 5.0 to 5.0 / 3.0 /
|
|
67
|
+
3.6 (fix / review / analysis) and the mean time from 37 / 27 / 27 s to 29 / 19 / 21 s, with the
|
|
68
|
+
same work done and 30/30 passing.
|
|
69
|
+
- **The `codex` backend returns as soon as the answer arrives.** It reads Codex's JSON events as
|
|
70
|
+
they come (`codex exec --json`) instead of waiting for the process to exit and reading the answer
|
|
71
|
+
from a file. Codex takes about 0.4 s to shut down after answering; that now happens in the
|
|
72
|
+
background. Over 8 alternating pairs of real calls the new way was faster every time, by a median
|
|
73
|
+
of 0.67 s on a call of about 4 s. Every agent turn on Codex is one call, so the saving repeats.
|
|
74
|
+
|
|
6
75
|
## 0.2.1 (beta)
|
|
7
76
|
|
|
8
77
|
**Durable agents with Temporal.** An optional `thunc[temporal]` runtime records each model turn
|
|
@@ -16,34 +16,59 @@ Thanks for helping. thunc is in beta, so feedback on the API is as useful as cod
|
|
|
16
16
|
|
|
17
17
|
```bash
|
|
18
18
|
git clone https://github.com/Eltarras/thunc && cd thunc
|
|
19
|
-
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
|
|
19
|
+
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
`temporal-test` installs the pinned Temporal SDK the durable-run tests use. Without it the tests in
|
|
23
|
+
`tests/temporal/` are skipped and everything else still runs. The README's
|
|
24
|
+
[Development](README.md#development) section lists the commands for live tests.
|
|
23
25
|
|
|
24
26
|
## Before you open a PR
|
|
25
27
|
|
|
26
|
-
CI runs these on Python 3.10 to 3.14, and all must pass
|
|
28
|
+
CI runs these on Python 3.10 to 3.14 on Linux, and all must pass. Run each on its own and check
|
|
29
|
+
its exit code:
|
|
27
30
|
|
|
28
31
|
```bash
|
|
29
32
|
.venv/bin/pytest
|
|
30
|
-
.venv/bin/ruff check .
|
|
33
|
+
.venv/bin/ruff check .
|
|
34
|
+
.venv/bin/ruff format --check .
|
|
31
35
|
.venv/bin/mypy --strict thunc
|
|
36
|
+
.venv/bin/mypy --strict --platform win32 thunc
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
CI also runs the offline tests on Windows (Python 3.13), since the agent lock, command handling and
|
|
40
|
+
paths have Windows-only code. If you touch `thunc/temporal/`, run the integration tests against a
|
|
41
|
+
real local Temporal service too (they download it once and make no model calls):
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
|
|
32
45
|
```
|
|
33
46
|
|
|
34
47
|
## Ground rules
|
|
35
48
|
|
|
36
49
|
- **Standard library only in the core.** A provider SDK is an optional extra in `pyproject.toml`,
|
|
37
|
-
imported inside its backend function, never at module level.
|
|
50
|
+
imported inside its backend function, never at module level. The same goes for Temporal:
|
|
51
|
+
only `thunc/temporal/` imports `temporalio`, and `thunc.call`, `@thunc.function` and
|
|
52
|
+
`agent.run()` must keep working without it.
|
|
38
53
|
- **Tests in `tests/` never call a real model.** Use the `fake` fixture in `tests/conftest.py`,
|
|
39
|
-
which returns scripted replies and records the prompts.
|
|
40
|
-
|
|
54
|
+
which returns scripted replies and records the prompts. Tests of native tool calls replace only
|
|
55
|
+
the SDK client and use the SDK's real types. Real calls belong in `live_tests/`, which CI
|
|
56
|
+
doesn't run.
|
|
57
|
+
- **Agent permissions are a safety boundary.** A change to `permissions.py`, `tools.py` or the
|
|
58
|
+
effect journal in `thunc/temporal/effects.py` needs a test that fails when the rule is broken.
|
|
59
|
+
Check it by breaking the rule on purpose and watching the test fail.
|
|
60
|
+
- **Durable runs replay.** Workflow code in `thunc/temporal/workflows.py` stays deterministic;
|
|
61
|
+
model calls, tools and file access happen in activities. A file or memory change goes through
|
|
62
|
+
the intent and receipt journal, and a command whose outcome is uncertain is never rerun
|
|
63
|
+
automatically. Before changing orchestration or the saved state format, version it and replay
|
|
64
|
+
saved histories (see the [Temporal guide](examples/temporal/README.md)).
|
|
41
65
|
- **User data stays out of the instructions.** Inputs are sent separately from the prompt
|
|
42
66
|
(see `_build_prompt` in `thunc/core.py`). Don't add code paths that paste inputs into
|
|
43
67
|
instructions.
|
|
44
68
|
- **Failures are loud.** When no valid answer arrives, raise `ThuncError`; never return a
|
|
45
69
|
default value.
|
|
46
|
-
- **Update the README** when you change the public API or the supported return types
|
|
70
|
+
- **Update the README** when you change the public API or the supported return types, and the
|
|
71
|
+
[Temporal guide](examples/temporal/README.md) when you change durable runs.
|
|
47
72
|
- **One change per PR**, with a description of what it does and how you tested it.
|
|
48
73
|
|
|
49
74
|
## Adding a backend
|
|
@@ -58,7 +83,14 @@ CI runs these on Python 3.10 to 3.14, and all must pass:
|
|
|
58
83
|
`thunc/config.py` if needed.
|
|
59
84
|
3. Add an optional extra for its SDK in `pyproject.toml`, and add the extra to the CI install.
|
|
60
85
|
4. Add offline tests with a stubbed SDK module, like the existing ones in `tests/test_backends.py`.
|
|
61
|
-
5.
|
|
86
|
+
5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
|
|
87
|
+
`Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
|
|
88
|
+
the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
|
|
89
|
+
A CLI that can call MCP tools can get native calls the way Claude Code does
|
|
90
|
+
(`thunc/claude_code.py`, tested with a fake CLI in `tests/test_claude_code_agent.py`).
|
|
91
|
+
Raise `TransientError` for failures worth asking again, so agent runs retry them.
|
|
92
|
+
Agents and durable runs refuse typed backends.
|
|
93
|
+
6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
|
|
62
94
|
the PR that you did.
|
|
63
95
|
|
|
64
96
|
## Known issues
|
|
@@ -73,8 +105,7 @@ model is asked again instead of thunc guessing:
|
|
|
73
105
|
- Quoted numbers (`"4"` for an `int`), labels in the wrong case (`Bug` for `bug`), prose around
|
|
74
106
|
JSON, Python-style values (`['a']`, `None`), trailing commas, curly quotes, non-ASCII digits.
|
|
75
107
|
- Unquoted text for `str | None`, and exclamations like `Yes!` for a `bool`.
|
|
76
|
-
- Several `<think>` blocks in a row,
|
|
77
|
-
four-backtick fences.
|
|
108
|
+
- Several `<think>` blocks in a row, `~~~`, indented or four-backtick fences.
|
|
78
109
|
- A one-key object whose key is a field of the expected dataclass: `{"customer": {...}}` for an
|
|
79
110
|
`Order` with a `customer` field is an `Order` with a bad customer, not a wrapper.
|
|
80
111
|
- A reply after the closing fence (for example a reasoning paragraph): it could be a second answer.
|
|
@@ -86,7 +117,11 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
|
|
|
86
117
|
**Open: behavior that might change.**
|
|
87
118
|
|
|
88
119
|
- `thunc.map` loses every result when one item fails, and returns coroutines for `async` functions.
|
|
89
|
-
- Backend errors (timeouts, connection failures) are never retried
|
|
120
|
+
- Backend errors (timeouts, connection failures) are never retried by `thunc.call`; only bad
|
|
121
|
+
replies are. Local agent runs retry a step that failed with a `TransientError` twice, and durable
|
|
122
|
+
runs retry transient provider failures.
|
|
123
|
+
- Several lines of prose before a code fence are read as a preamble, though `_unfence` in
|
|
124
|
+
`thunc/schema.py` documents one line, and no test pins either behavior.
|
|
90
125
|
- A union takes the first option that accepts the reply as JSON, in the order written: `3` for
|
|
91
126
|
`float | int` is `3.0`. An unquoted label is only tried after that, so `2` for
|
|
92
127
|
`Literal["2"] | int` is the int `2`, not the label.
|
|
@@ -109,10 +144,38 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
|
|
|
109
144
|
- A list of dataclasses passed as an input is shown to the model as Python reprs, not JSON.
|
|
110
145
|
- A string return annotation naming a class defined after the function raises `NameError` when
|
|
111
146
|
the decorator runs.
|
|
112
|
-
-
|
|
147
|
+
- On Python 3.10 and 3.11, a `make_dataclass` class whose field types are strings naming anything
|
|
148
|
+
but builtins (`"Item"`, `"Annotated[int, 'x']"`) raises `NameError`: before 3.12 its module is
|
|
149
|
+
`types`, not the caller's.
|
|
150
|
+
- `Annotated[...]` works on dataclass fields but not as a return type.
|
|
113
151
|
- The Python 3.10 fallback for string annotations (`_hints` and `resolve_strings` in
|
|
114
152
|
`thunc/schema.py`) can be removed when 3.10 support is dropped.
|
|
115
153
|
|
|
154
|
+
**Open: agents and durable runs.**
|
|
155
|
+
|
|
156
|
+
- Durable runs on `claude-code` use the JSON text protocol, not native calls: the MCP path keeps
|
|
157
|
+
one CLI process for the whole run, which a durable step can't snapshot. The text protocol is the
|
|
158
|
+
weaker one: in `live_tests/bench_tooluse.py` on Sonnet 5.5 it passed 12 of 24 runs, and 20 of 24
|
|
159
|
+
with step retries (measured before action arrays), against 24 of 24 for native calls.
|
|
160
|
+
- The `shell` permission isn't tested on Windows: its test is skipped there, so `cmd /c` has never
|
|
161
|
+
run in CI.
|
|
162
|
+
- With `shell`, a file read with `cat` doesn't count as read for `edit`, which refuses until the
|
|
163
|
+
agent reads it with `read` (the no-blind-overwrite rule). Agents work around it with a short
|
|
164
|
+
`read`, at the cost of a step.
|
|
165
|
+
- The agent loops on the Claude and OpenAI APIs have no live tool-use benchmark yet (only
|
|
166
|
+
`live_tests/eval_prompts.py`); their `max_tokens`, effort and stop-reason handling were reviewed
|
|
167
|
+
from the code only (`live_tests/bench_tooluse_report.md`, finding 7).
|
|
168
|
+
- Durable agents can't use `tools=` yet: the effects of the program's own functions can't be
|
|
169
|
+
journaled.
|
|
170
|
+
- Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
|
|
171
|
+
tombstones are kept forever. There's no context summarization either, so a long run fails once
|
|
172
|
+
its saved state passes 16 MiB.
|
|
173
|
+
- A durable workspace can only restart on the same volume and absolute paths; nothing moves it
|
|
174
|
+
between hosts.
|
|
175
|
+
- A hard worker kill can leave a command's subprocesses running.
|
|
176
|
+
- Durable runs aren't tested on Windows (the code is only type-checked for it), and there are no
|
|
177
|
+
live provider tests for durable runs yet.
|
|
178
|
+
|
|
116
179
|
## Reporting bugs
|
|
117
180
|
|
|
118
181
|
Open an [issue](https://github.com/Eltarras/thunc/issues) with:
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: thunc
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: think + function: call an LLM like a typed Python function.
|
|
5
5
|
Project-URL: Homepage, https://eltarras.github.io/thunc/
|
|
6
|
+
Project-URL: Documentation, https://eltarras.github.io/thunc/docs/
|
|
6
7
|
Project-URL: Repository, https://github.com/Eltarras/thunc
|
|
7
8
|
Project-URL: Issues, https://github.com/Eltarras/thunc/issues
|
|
8
9
|
Author: Hussein Eltarras
|
|
@@ -32,20 +33,24 @@ Requires-Dist: pytest-asyncio>=0.24; extra == 'temporal-test'
|
|
|
32
33
|
Requires-Dist: temporalio==1.34.0; extra == 'temporal-test'
|
|
33
34
|
Description-Content-Type: text/markdown
|
|
34
35
|
|
|
35
|
-
](https://github.com/Eltarras/thunc/actions/workflows/ci.yml)
|
|
38
|
-
[](https://pypi.org/project/thunc/)
|
|
39
|
-
[](https://eltarras.github.io/thunc/)
|
|
36
|
+
![A thunc function returning list[Item] is called with "2 oat lattes and a croissant pls. oh, one more latte!" and returns two Item dataclasses: oat latte ×3 and croissant ×1.](https://raw.githubusercontent.com/Eltarras/thunc/main/.github/demo-function.gif)
|
|
40
37
|
|
|
41
38
|
**think + function.** Call an LLM like a typed Python function.
|
|
42
39
|
|
|
43
|
-
|
|
40
|
+
[](https://pypi.org/project/thunc/)
|
|
41
|
+
[](https://pypi.org/project/thunc/)
|
|
42
|
+
[](https://github.com/Eltarras/thunc/actions/workflows/ci.yml)
|
|
43
|
+
[](https://github.com/Eltarras/thunc/blob/main/LICENSE)
|
|
44
|
+
[](https://eltarras.github.io/thunc/docs/)
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
pip install thunc
|
|
48
|
+
```
|
|
44
49
|
|
|
45
50
|
```python
|
|
46
51
|
import thunc
|
|
47
52
|
|
|
48
|
-
thunc.configure(backend="claude-code")
|
|
53
|
+
thunc.configure(backend="claude-code") # or "codex", "anthropic", "openai"
|
|
49
54
|
|
|
50
55
|
|
|
51
56
|
@thunc.function
|
|
@@ -57,26 +62,43 @@ def urgency(ticket: str) -> int:
|
|
|
57
62
|
urgency("I was charged twice!") # -> 4, a checked int
|
|
58
63
|
```
|
|
59
64
|
|
|
60
|
-
The
|
|
61
|
-
after that `thunc.ThuncError` is raised.
|
|
62
|
-
Python 3.10+.
|
|
65
|
+
The docstring is the prompt and the return annotation is the type. The answer is parsed into that
|
|
66
|
+
type; if it doesn't fit, the model is asked again, and after that `thunc.ThuncError` is raised. No
|
|
67
|
+
dependencies, Python 3.10+.
|
|
68
|
+
|
|
69
|
+
It also runs [agents](#agents): typed functions that can read, edit and test your code before they
|
|
70
|
+
answer.
|
|
63
71
|
|
|
64
|
-
|
|
65
|
-
OpenAI API), or your Claude Code or Codex login.
|
|
72
|
+
The [docs](https://eltarras.github.io/thunc/docs/) cover everything below, a page per topic.
|
|
66
73
|
|
|
67
|
-
##
|
|
74
|
+
## Quickstart
|
|
75
|
+
|
|
76
|
+
No API key needed if you have Claude Code or Codex installed: thunc can use their login.
|
|
68
77
|
|
|
69
78
|
```bash
|
|
70
|
-
pip install thunc
|
|
71
|
-
|
|
72
|
-
pip install "thunc[openai]" # adds the OpenAI API backend
|
|
73
|
-
pip install "thunc[temporal]" # adds durable agents on Temporal
|
|
79
|
+
pip install thunc
|
|
80
|
+
THUNC_BACKEND=claude-code python3 -c 'import thunc; print(thunc.call("Say hello in five words or fewer."))'
|
|
74
81
|
```
|
|
75
82
|
|
|
76
|
-
|
|
83
|
+
Swap in the backend you have:
|
|
84
|
+
|
|
85
|
+
| You have | Set | Install |
|
|
86
|
+
|---|---|---|
|
|
87
|
+
| [Claude Code](https://claude.com/claude-code), logged in | `THUNC_BACKEND=claude-code` | `pip install thunc` |
|
|
88
|
+
| [Codex](https://github.com/openai/codex), logged in | `THUNC_BACKEND=codex` | `pip install thunc` |
|
|
89
|
+
| An Anthropic API key | `ANTHROPIC_API_KEY` | `pip install "thunc[anthropic]"` |
|
|
90
|
+
| An OpenAI API key | `OPENAI_API_KEY` | `pip install "thunc[openai]"` |
|
|
91
|
+
| A local model (LM Studio) | `OPENAI_BASE_URL` (see **Local models** below) | `pip install "thunc[openai]"` |
|
|
92
|
+
|
|
93
|
+
With an API key set, thunc picks that backend on its own, so `THUNC_BACKEND` isn't needed. In code,
|
|
94
|
+
`thunc.configure(backend=...)` does the same. Durable agents on Temporal add
|
|
95
|
+
`pip install "thunc[temporal]"`.
|
|
96
|
+
|
|
97
|
+
> **Beta (v0.2).** The API may still change. Bug reports and feedback are welcome in
|
|
98
|
+
> [issues](https://github.com/Eltarras/thunc/issues).
|
|
77
99
|
|
|
78
|
-
Clone the repo and run
|
|
79
|
-
|
|
100
|
+
**More examples.** Clone the repo and run them from its root, with no install, through your
|
|
101
|
+
Claude Code login:
|
|
80
102
|
|
|
81
103
|
```bash
|
|
82
104
|
git clone https://github.com/Eltarras/thunc && cd thunc
|
|
@@ -85,9 +107,6 @@ python3 -m examples.support_inbox
|
|
|
85
107
|
THUNC_BACKEND=codex python3 -m examples.log_triage
|
|
86
108
|
```
|
|
87
109
|
|
|
88
|
-
With an API key instead, install the SDK and pick the backend: `THUNC_BACKEND=openai` with
|
|
89
|
-
`OPENAI_API_KEY`, or `THUNC_BACKEND=anthropic` with `ANTHROPIC_API_KEY`.
|
|
90
|
-
|
|
91
110
|
## Two ways to write a prompt
|
|
92
111
|
|
|
93
112
|
| | When | |
|
|
@@ -184,7 +203,8 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
|
|
|
184
203
|
`pip install "thunc[openai]"`. The default model is `gpt-5.5`. `OPENAI_BASE_URL` points it at
|
|
185
204
|
any server that speaks the OpenAI Responses API.
|
|
186
205
|
- `claude-code` and `codex` call your local CLI login, and are meant for cheap testing.
|
|
187
|
-
Both run with their own tools turned off, so the model can only answer
|
|
206
|
+
Both run with their own tools turned off, so the model can only answer; an agent on
|
|
207
|
+
`claude-code` gets only its thunc tools, as native calls (see Agents). `codex` also ignores
|
|
188
208
|
`~/.codex/config.toml` (your MCP servers, plugins, `notify` command and model settings); your
|
|
189
209
|
login still works. Pick the model with `configure(model=...)` or `model=`.
|
|
190
210
|
- `jev` is TypeSafe's [Jev](https://typesafe.ai/blog/introducing-system-one-models-and-jev)
|
|
@@ -198,7 +218,7 @@ or takes `--cache-dir`; it can't see a `configure(cache_dir=...)` in your code.
|
|
|
198
218
|
CLI always uses `jev-latest`), and an answer that fails `ensure=` isn't retried, since Jev would
|
|
199
219
|
give the same one. It's only used when you choose it: `backend="jev"` or `THUNC_BACKEND=jev`.
|
|
200
220
|
Setup (install the CLI, log in, check it works): the
|
|
201
|
-
[Jev guide](https://eltarras.github.io/thunc/jev.html).
|
|
221
|
+
[Jev guide](https://eltarras.github.io/thunc/docs/jev.html).
|
|
202
222
|
|
|
203
223
|
**Local models:** the `openai` backend works with a local server through `OPENAI_BASE_URL`. This
|
|
204
224
|
has been tested with [LM Studio](https://lmstudio.ai) running `openai/gpt-oss-20b`:
|
|
@@ -214,6 +234,31 @@ giving it alone.
|
|
|
214
234
|
The backend can also be set with `THUNC_BACKEND`. With none set, `ANTHROPIC_API_KEY` (or a
|
|
215
235
|
`configure(api_key=...)` alone) selects `anthropic`, and otherwise `OPENAI_API_KEY` selects `openai`.
|
|
216
236
|
|
|
237
|
+
**Profiling:** run your program with `thunc run --profile` to see where the time went when it ends:
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
thunc run --profile support_inbox.py --limit 20 # a script and its arguments
|
|
241
|
+
thunc run --profile -m myapp.triage # a module, as with python -m
|
|
242
|
+
```
|
|
243
|
+
|
|
244
|
+
The report goes to stderr: per function, the calls, cache hits, retries and failures, the total,
|
|
245
|
+
mean, p95 and slowest time, and how much of it was the model and how much thunc's own work
|
|
246
|
+
(building the prompt, parsing, the cache). Agent runs get their steps, model time and time in each
|
|
247
|
+
tool. It also says what share of the program's wall time was spent in thunc, and how much calls
|
|
248
|
+
overlapped under `thunc.map`. Without `--profile`, `thunc run` just runs the program, and nothing
|
|
249
|
+
is recorded. The program's exit code is passed through.
|
|
250
|
+
|
|
251
|
+
```
|
|
252
|
+
CALLS
|
|
253
|
+
FUNCTION CALLS CACHED RETRIES FAILED TOTAL MEAN P95 MAX MODEL LOCAL
|
|
254
|
+
urgency 11 1 1 0 1.70s 155ms 309ms 309ms 1.69s 12ms
|
|
255
|
+
|
|
256
|
+
In thunc: 774ms of 980ms wall time (79%); the rest was the program's own code
|
|
257
|
+
Model time: 1.69s, 99% of the time in calls (anthropic/default model 1.69s)
|
|
258
|
+
Concurrency: calls overlapped 2.2x on average (thunc.map or threads)
|
|
259
|
+
Slowest: urgency took 309ms
|
|
260
|
+
```
|
|
261
|
+
|
|
217
262
|
**Type checking:** signatures and return types are visible to mypy and Pyright. mypy reports
|
|
218
263
|
empty bodies; turn that off with `disable_error_code = ["empty-body"]`.
|
|
219
264
|
|
|
@@ -221,6 +266,8 @@ empty bodies; turn that off with `disable_error_code = ["empty-body"]`.
|
|
|
221
266
|
|
|
222
267
|
> **New in 0.2.** Agents are new; their API may change in a later release as feedback comes in.
|
|
223
268
|
|
|
269
|
+

|
|
270
|
+
|
|
224
271
|
An agent is a typed function that can look around before it answers. Give it a name and a working
|
|
225
272
|
directory, declare its tasks the way you write `@thunc.function`, and call them from Python:
|
|
226
273
|
|
|
@@ -253,9 +300,15 @@ repo.call(f"Where is {setting} set?", returns=str)
|
|
|
253
300
|
Each call is one run. The model takes one step at a time (list a folder, search, read or edit a
|
|
254
301
|
file) and ends by calling `finish` with a value of the return type, which is checked like any thunc
|
|
255
302
|
result. It runs on the Claude and OpenAI APIs through their own tool calls (the
|
|
256
|
-
model can make several at once, and the fixed part of the prompt is cached)
|
|
257
|
-
|
|
258
|
-
|
|
303
|
+
model can make several at once, and the fixed part of the prompt is cached). On Claude Code the
|
|
304
|
+
calls are native too: the agent's tools are an MCP server that one `claude -p` process per run
|
|
305
|
+
calls, while thunc carries out each call with its own tools, permissions and records. If Claude Code
|
|
306
|
+
can't start them (an older `claude` CLI, or MCP servers turned off by a policy), the run uses the
|
|
307
|
+
text protocol below instead, with a warning, and so do later runs in the process;
|
|
308
|
+
`protocol="native"` fails instead. On Codex the model replies with JSON actions as text: one at a
|
|
309
|
+
time, or several independent ones (reading three files) as a JSON array, which saves turns.
|
|
310
|
+
`protocol="text"` uses that way on any backend, for example with a server behind `OPENAI_BASE_URL`
|
|
311
|
+
that has no function calling. (Durable runs on Claude Code use it too.)
|
|
259
312
|
The `jev` backend only answers typed questions and cannot run agents, even for a task returning
|
|
260
313
|
`bool` or `Literal[...]`. An agent run using it raises `ThuncError` before creating any run files
|
|
261
314
|
or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
@@ -272,25 +325,29 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
272
325
|
| `write:docs/**`, `write` | create and edit matching files (all files with no path); also lets it read them |
|
|
273
326
|
| `read:src/**` | read only these; any `read:` rule replaces the read-everything default |
|
|
274
327
|
| `run:pytest`, `run:git log`, `run` | run commands that start with these words (`run:git log` allows `git log --oneline`, not `git push`); `run` alone allows any |
|
|
328
|
+
| `shell` | run any command line in a shell (`sh -c`, or `cmd /c` on Windows), so pipes, `&&`, `cd` and redirects work; off by default, and it can't be combined with `!run:` rules |
|
|
275
329
|
| `!read:.env*`, `!write:...`, `!run:git push`, `!memory` | deny; a deny always wins, and `!read` also stops writing |
|
|
276
330
|
|
|
277
331
|
`*` stays within one folder, `**` crosses folders, and paths are relative to `workdir`. The agent
|
|
278
332
|
is told its permissions, and an action they don't allow is refused with the reason, after which
|
|
279
333
|
the run carries on. Bad rules fail when the agent is declared.
|
|
280
|
-
- **Tools:** `list`, `read` and `search
|
|
281
|
-
|
|
282
|
-
allows it;
|
|
283
|
-
must stay inside `workdir`: `..`, absolute paths and symlinks that point
|
|
284
|
-
the rules are checked on where a link really leads. Files the agent may
|
|
285
|
-
`list` and `search
|
|
334
|
+
- **Tools:** `list`, `read` and `search` (a regular expression, optionally limited with a `glob`
|
|
335
|
+
such as `*.py`); `write` (create a file, or replace one) and `edit` (replace text that appears
|
|
336
|
+
exactly once) when a write rule allows it; `run` when a run or shell rule allows it; and
|
|
337
|
+
`remember`. Every path must stay inside `workdir`: `..`, absolute paths and symlinks that point
|
|
338
|
+
outside are refused, and the rules are checked on where a link really leads. Files the agent may
|
|
339
|
+
not read are left out of `list` and `search`, and so is what git ignores, in a git repository
|
|
340
|
+
(build output, caches, vendored code); a folder named explicitly is still listed and searched.
|
|
286
341
|
- **No blind overwrites.** A file is only replaced or edited after the agent read it in the same
|
|
287
342
|
run, and only if it hasn't changed on disk since. There is no undo, so run agents that write in a
|
|
288
343
|
git repository with a clean tree, and review their changes with `git diff`.
|
|
289
|
-
- **Commands** run in `workdir
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
344
|
+
- **Commands** run in `workdir`, or in a folder inside it given as `cwd`. Without the `shell`
|
|
345
|
+
permission there's no shell, so `&&`, pipes, `cd`, redirects and `$VARIABLES` don't work (the
|
|
346
|
+
agent is told). They get a minimal environment: `PATH`, `HOME`, the locale and temp-folder
|
|
347
|
+
variables, and whatever you pass in `env=`, so your API keys don't reach them. Each has a time
|
|
348
|
+
limit (`command_timeout=120` seconds) that also stops the processes it started, and the agent
|
|
349
|
+
sees the exit code and the output: the start and the end when it's long, since the first error is
|
|
350
|
+
often at the start and the summary at the end.
|
|
294
351
|
- **A permitted command can do anything its program can.** `run:pytest` runs the project's code,
|
|
295
352
|
which can read or change any file your user account can, whatever the read and write rules say.
|
|
296
353
|
Permissions limit which tools the model uses; they aren't a sandbox. For untrusted input, run the
|
|
@@ -341,24 +398,29 @@ or calling a backend. Use `@thunc.function` or `thunc.call` for Jev questions.
|
|
|
341
398
|
|
|
342
399
|
- **Failures are loud.** A run that hits `max_steps`, never gives a valid value, or loses its
|
|
343
400
|
backend raises `thunc.AgentError` (a `ThuncError`), whose `.run` is the record up to that point.
|
|
344
|
-
|
|
401
|
+
A step that fails for a reason asking again may fix (a timeout, a lost connection, a rate limit,
|
|
402
|
+
a server error, a CLI call that ended in an error) is retried twice, after 2 and 4 seconds, and
|
|
403
|
+
each retry is in the run's record. A text-protocol step on Claude Code or Codex may take 120
|
|
404
|
+
seconds before it's retried. With tracing on, each run is also one line with every model reply.
|
|
345
405
|
|
|
346
406
|
**How the prompt was tested.** `python -m live_tests.eval_prompts --backend anthropic` runs three
|
|
347
407
|
small tasks (fix a bug, review a diff, answer a question about a repo) with three versions of the
|
|
348
|
-
system prompt: bare (no working method), the default, and the task's preset. Five runs of each on
|
|
349
|
-
4 October 2026:
|
|
408
|
+
system prompt: bare (no working method), the default, and the task's preset. Five runs of each, on
|
|
409
|
+
the Claude API on 4 October 2026 and on Claude Code on 5 October 2026:
|
|
350
410
|
|
|
351
|
-
| | Claude API (Opus 5.5, native calls) | Claude Code (
|
|
411
|
+
| | Claude API (Opus 5.5, native calls) | Claude Code (Sonnet 5.5, native calls) |
|
|
352
412
|
|---|---|---|
|
|
353
413
|
| Passed | 45/45: every task, every version | 45/45 |
|
|
354
|
-
| Steps (bare / default / preset) | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.4 / 2.8, analysis 3.0 / 3.0 / 3.0 | fix
|
|
414
|
+
| Steps (bare / default / preset) | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.4 / 2.8, analysis 3.0 / 3.0 / 3.0 | fix 4.0 / 4.0 / 4.0, review 2.0 / 2.0 / 2.0, analysis 3.0 / 2.8 / 2.6 |
|
|
355
415
|
| Cost | $0.76 for all 45 runs (cache reads were 257,553 of 312,294 input tokens) | |
|
|
356
416
|
|
|
357
417
|
Every version passed every time, so these tasks are too easy to tell the versions apart: the result
|
|
358
|
-
says the prompt does no harm, not that it helps.
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
more.
|
|
418
|
+
says the prompt does no harm, not that it helps. On the API, the review preset read more of the code
|
|
419
|
+
before answering, and each review flagged the renamed function as a minor issue (outside code
|
|
420
|
+
importing the old name breaks), never as blocking. On Claude Code, only the review preset flagged
|
|
421
|
+
it (2 of 5 runs, as minor). On the text protocol these tasks took more steps (fix 5.2 / 6.0 / 6.0
|
|
422
|
+
on Claude Code before native calls). For harder tasks that do tell harnesses apart, see the tool-use
|
|
423
|
+
benchmark in `live_tests/bench_tooluse.py` and its report.
|
|
362
424
|
|
|
363
425
|
## Durable agents with Temporal
|
|
364
426
|
|
|
@@ -428,15 +490,17 @@ thunc/
|
|
|
428
490
|
permissions.py the agent's permission rules
|
|
429
491
|
runs.py thunc.Run and AgentError: what a run did
|
|
430
492
|
native.py how a run talks to its backend: native tool calls or the text protocol
|
|
493
|
+
claude_code.py native tool calls on Claude Code, through an MCP server (mcp_relay.py)
|
|
431
494
|
store.py the agent's folder: memory, settings, run records, the lock
|
|
432
495
|
prompts.py the agent's system prompt
|
|
433
|
-
__main__.py the thunc command: thunc cache list / clear
|
|
496
|
+
__main__.py the thunc command: thunc run [--profile], thunc cache list / clear
|
|
497
|
+
profiling.py thunc run --profile: timing records and the report
|
|
434
498
|
core.py thunc.call, thunc.map, tracing
|
|
435
499
|
cache.py the answer cache: saving, listing, clearing
|
|
436
500
|
schema.py return types: describe, parse, validate
|
|
437
501
|
config.py settings and backend selection
|
|
438
502
|
backends.py anthropic, openai, claude-code, codex, jev
|
|
439
|
-
errors.py ThuncError
|
|
503
|
+
errors.py ThuncError, and TransientError for failures worth asking again
|
|
440
504
|
tests/ offline: a fake backend, never a real model
|
|
441
505
|
live_tests/ against a real model: hello, a yes/no decision, labels and ratings, messy text to a dict
|
|
442
506
|
examples/
|
|
@@ -452,12 +516,14 @@ examples/
|
|
|
452
516
|
## Development
|
|
453
517
|
|
|
454
518
|
```bash
|
|
455
|
-
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
|
|
519
|
+
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
|
|
456
520
|
.venv/bin/pytest # offline tests (these run in CI)
|
|
521
|
+
THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal # a real local Temporal service; no model calls
|
|
457
522
|
.venv/bin/pytest live_tests # real model calls through your Claude Code login; costs quota
|
|
458
523
|
THUNC_BACKEND=anthropic .venv/bin/pytest live_tests # the same, through the Claude API (needs ANTHROPIC_API_KEY)
|
|
459
524
|
THUNC_BACKEND=openai .venv/bin/pytest live_tests # the same, through the OpenAI API (needs OPENAI_API_KEY)
|
|
460
|
-
.venv/bin/ruff check . && .venv/bin/
|
|
525
|
+
.venv/bin/ruff check . && .venv/bin/ruff format --check .
|
|
526
|
+
.venv/bin/mypy --strict thunc && .venv/bin/mypy --strict --platform win32 thunc
|
|
461
527
|
```
|
|
462
528
|
|
|
463
529
|
## License
|