thunc 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- thunc-0.2.2/.github/demo-agent.gif +0 -0
- thunc-0.2.2/.github/demo-function.gif +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/.github/workflows/ci.yml +18 -1
- {thunc-0.2.0 → thunc-0.2.2}/.gitignore +4 -0
- thunc-0.2.2/CHANGELOG.md +186 -0
- {thunc-0.2.0 → thunc-0.2.2}/CONTRIBUTING.md +76 -13
- {thunc-0.2.0 → thunc-0.2.2}/PKG-INFO +168 -52
- {thunc-0.2.0 → thunc-0.2.2}/README.md +161 -51
- thunc-0.2.2/benchmarks/README.md +32 -0
- thunc-0.2.2/benchmarks/__init__.py +1 -0
- thunc-0.2.2/benchmarks/__main__.py +60 -0
- thunc-0.2.2/benchmarks/bench_agent.py +62 -0
- thunc-0.2.2/benchmarks/bench_backends.py +84 -0
- thunc-0.2.2/benchmarks/bench_cache.py +34 -0
- thunc-0.2.2/benchmarks/bench_calls.py +127 -0
- thunc-0.2.2/benchmarks/bench_schema.py +85 -0
- thunc-0.2.2/benchmarks/bench_startup.py +30 -0
- thunc-0.2.2/benchmarks/bench_tools.py +102 -0
- thunc-0.2.2/benchmarks/fakes.py +101 -0
- thunc-0.2.2/benchmarks/harness.py +190 -0
- thunc-0.2.2/design/og-card.html +61 -0
- thunc-0.2.2/docs/docs/agents.html +189 -0
- thunc-0.2.2/docs/docs/api.html +150 -0
- thunc-0.2.2/docs/docs/backends.html +122 -0
- thunc-0.2.2/docs/docs/caching.html +109 -0
- thunc-0.2.2/docs/docs/functions.html +172 -0
- thunc-0.2.2/docs/docs/index.html +139 -0
- thunc-0.2.2/docs/docs/jev.html +182 -0
- thunc-0.2.2/docs/docs/temporal.html +147 -0
- thunc-0.2.2/docs/index.html +215 -0
- thunc-0.2.2/docs/jev.html +14 -0
- thunc-0.2.2/docs/og.png +0 -0
- thunc-0.2.2/docs/site.css +218 -0
- thunc-0.2.2/docs/site.js +94 -0
- thunc-0.2.2/docs/sitemap.xml +12 -0
- thunc-0.2.2/docs/temporal.html +14 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/repo_guide.py +4 -2
- thunc-0.2.2/examples/temporal/README.md +202 -0
- thunc-0.2.2/examples/temporal/application.py +28 -0
- thunc-0.2.2/examples/temporal/client.py +22 -0
- thunc-0.2.2/examples/temporal/pipeline.py +67 -0
- thunc-0.2.2/examples/temporal/worker.py +23 -0
- thunc-0.2.2/live_tests/bench_tooluse.py +903 -0
- thunc-0.2.2/live_tests/bench_tooluse_report.md +385 -0
- {thunc-0.2.0 → thunc-0.2.2}/pyproject.toml +5 -1
- thunc-0.2.2/pytest-temporal.ini +4 -0
- thunc-0.2.2/tests/fake_claude.py +135 -0
- thunc-0.2.2/tests/temporal/conftest.py +5 -0
- thunc-0.2.2/tests/temporal/process_worker.py +60 -0
- thunc-0.2.2/tests/temporal/test_contract.py +118 -0
- thunc-0.2.2/tests/temporal/test_runtime.py +492 -0
- thunc-0.2.2/tests/temporal/test_storage.py +58 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_agent.py +252 -6
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_backends.py +206 -69
- thunc-0.2.2/tests/test_benchmarks.py +13 -0
- thunc-0.2.2/tests/test_claude_code_agent.py +267 -0
- thunc-0.2.2/tests/test_execution.py +82 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_native.py +35 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_permissions.py +19 -0
- thunc-0.2.2/tests/test_profiling.py +119 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/__init__.py +1 -1
- {thunc-0.2.0 → thunc-0.2.2}/thunc/__main__.py +50 -2
- {thunc-0.2.0 → thunc-0.2.2}/thunc/agent.py +150 -93
- {thunc-0.2.0 → thunc-0.2.2}/thunc/backends.py +150 -26
- thunc-0.2.2/thunc/claude_code.py +351 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/config.py +3 -1
- {thunc-0.2.0 → thunc-0.2.2}/thunc/core.py +25 -4
- {thunc-0.2.0 → thunc-0.2.2}/thunc/decorator.py +2 -0
- thunc-0.2.2/thunc/errors.py +15 -0
- thunc-0.2.2/thunc/execution.py +135 -0
- thunc-0.2.2/thunc/mcp_relay.py +77 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/native.py +79 -14
- {thunc-0.2.0 → thunc-0.2.2}/thunc/permissions.py +23 -7
- thunc-0.2.2/thunc/profiling.py +261 -0
- thunc-0.2.2/thunc/temporal/__init__.py +37 -0
- thunc-0.2.2/thunc/temporal/activities.py +357 -0
- thunc-0.2.2/thunc/temporal/adapters.py +48 -0
- thunc-0.2.2/thunc/temporal/client.py +130 -0
- thunc-0.2.2/thunc/temporal/effects.py +166 -0
- thunc-0.2.2/thunc/temporal/models.py +67 -0
- thunc-0.2.2/thunc/temporal/registry.py +165 -0
- thunc-0.2.2/thunc/temporal/storage.py +158 -0
- thunc-0.2.2/thunc/temporal/worker.py +86 -0
- thunc-0.2.2/thunc/temporal/workflows.py +245 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/tools.py +160 -42
- {thunc-0.2.0 → thunc-0.2.2}/uv.lock +115 -3
- thunc-0.2.0/CHANGELOG.md +0 -89
- thunc-0.2.0/docs/index.html +0 -267
- thunc-0.2.0/docs/jev.html +0 -325
- thunc-0.2.0/docs/sitemap.xml +0 -11
- thunc-0.2.0/thunc/errors.py +0 -2
- {thunc-0.2.0 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/.github/social-preview.png +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/.github/workflows/publish.yml +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/LICENSE +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/.nojekyll +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/apple-touch-icon.png +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/favicon-96.png +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/favicon.ico +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/favicon.svg +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/docs/google2cd177e5e85b3c3e.html +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/dynamic_prompts.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/hello.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/jev_hello.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/jev_inbox.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/jev_with_claude.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/log_triage.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/examples/support_inbox.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/conftest.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/eval_prompts.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_agent.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_bool_decision.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_dict_output.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_hello.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_literal_choice.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/conftest.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/future_types.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_cache.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_calls.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_cli.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_jev.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/tests/test_schema.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/cache.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/prompts.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/py.typed +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/runs.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/schema.py +0 -0
- {thunc-0.2.0 → thunc-0.2.2}/thunc/store.py +0 -0
|
Binary file
|
|
Binary file
|
|
@@ -16,11 +16,12 @@ jobs:
|
|
|
16
16
|
- uses: actions/setup-python@v5
|
|
17
17
|
with:
|
|
18
18
|
python-version: ${{ matrix.python-version }}
|
|
19
|
-
- run: pip install -e ".[anthropic,openai,dev]"
|
|
19
|
+
- run: pip install -e ".[anthropic,openai,temporal-test,dev]"
|
|
20
20
|
- run: pytest # offline tests only; live_tests need a model and are not run in CI
|
|
21
21
|
- run: ruff check .
|
|
22
22
|
- run: ruff format --check .
|
|
23
23
|
- run: mypy --strict thunc
|
|
24
|
+
- run: mypy --strict --platform win32 thunc
|
|
24
25
|
|
|
25
26
|
# The lock, process stopping and path handling have code just for Windows.
|
|
26
27
|
test-windows:
|
|
@@ -34,3 +35,19 @@ jobs:
|
|
|
34
35
|
- run: pytest
|
|
35
36
|
env:
|
|
36
37
|
PYTHONIOENCODING: utf-8 # the console's own code page can't print every test id and message
|
|
38
|
+
|
|
39
|
+
temporal-integration:
|
|
40
|
+
runs-on: ubuntu-latest
|
|
41
|
+
strategy:
|
|
42
|
+
matrix:
|
|
43
|
+
python-version: ["3.10", "3.14"]
|
|
44
|
+
steps:
|
|
45
|
+
- uses: actions/checkout@v4
|
|
46
|
+
- uses: actions/setup-python@v5
|
|
47
|
+
with:
|
|
48
|
+
python-version: ${{ matrix.python-version }}
|
|
49
|
+
- run: pip install -e ".[anthropic,openai,temporal-test,dev]"
|
|
50
|
+
- name: Real service, restart and replay tests (no providers)
|
|
51
|
+
run: pytest -c pytest-temporal.ini tests/temporal -q
|
|
52
|
+
env:
|
|
53
|
+
THUNC_TEMPORAL_TESTS: "1"
|
thunc-0.2.2/CHANGELOG.md
ADDED
|
@@ -0,0 +1,186 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to thunc. The full notes for each release are on the
|
|
4
|
+
[releases page](https://github.com/Eltarras/thunc/releases).
|
|
5
|
+
|
|
6
|
+
## 0.2.2 (beta)
|
|
7
|
+
|
|
8
|
+
**Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
|
|
9
|
+
calls instead of writing each action as JSON text, a failed step is retried instead of ending the
|
|
10
|
+
run, and the agent's tools fill gaps a benchmark found. In that tool-use benchmark
|
|
11
|
+
(`live_tests/bench_tooluse.py`, 8 tasks, Claude Sonnet 5.5, 3 runs each), agents on Claude Code went
|
|
12
|
+
from 12 of 24 runs passing to 24 of 24, from 99 to 10 seconds a task, and from $0.084 to $0.022 a
|
|
13
|
+
task; Claude Code itself took 11 seconds and $0.069. The API backends also reuse their connections,
|
|
14
|
+
Codex answers return sooner, and `thunc run --profile` shows where a program's time goes.
|
|
15
|
+
|
|
16
|
+
### Behavior changes
|
|
17
|
+
|
|
18
|
+
Nothing is removed, but these defaults change:
|
|
19
|
+
|
|
20
|
+
- **Agents on `claude-code` make native tool calls** (see Added). With a `claude` CLI too old for
|
|
21
|
+
them, or with MCP servers turned off by a policy, a run falls back to the text protocol with a
|
|
22
|
+
warning. `protocol="text"` keeps the old way.
|
|
23
|
+
- **`list` and `search` leave out what git ignores** in a git repository (build output, caches,
|
|
24
|
+
vendored code). A folder named explicitly is still listed and searched.
|
|
25
|
+
- **Long command output keeps its start and its end** (the first error and the summary), not only
|
|
26
|
+
the end.
|
|
27
|
+
- **The `claude-code` backend loads none of your Claude Code settings** (`--setting-sources ""`):
|
|
28
|
+
no `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so an
|
|
29
|
+
agent's `workdir` can't give it instructions unless `follow=` asks for them.
|
|
30
|
+
|
|
31
|
+
### Added
|
|
32
|
+
|
|
33
|
+
- **Native calls on `claude-code`**: the agent's tools are an MCP server that one `claude -p`
|
|
34
|
+
process per run calls; thunc carries out each call with its own tools, permissions and run
|
|
35
|
+
record. The CLI runs in the agent's `workdir`. `protocol="text"` keeps the old way, and durable
|
|
36
|
+
runs on Claude Code still use it. When Claude Code can't start native calls (an older CLI, or MCP
|
|
37
|
+
servers turned off by a policy), a run falls back to the text protocol with a warning and a
|
|
38
|
+
`fallback` entry in its record; `protocol="native"` raises instead.
|
|
39
|
+
- **`run` takes `cwd`**, a folder inside `workdir` to run the command in.
|
|
40
|
+
- **The `shell` permission** runs command lines through the system shell, so pipes, `&&`, `cd`
|
|
41
|
+
and redirects work. Off by default; it can't be combined with `!run:` rules.
|
|
42
|
+
- **`search` takes `glob`** (`*.py` by file name, `src/**/*.ts` by path) to limit the files
|
|
43
|
+
searched.
|
|
44
|
+
- **`thunc run --profile`**: runs a script (or `-m module`) and prints a performance report to
|
|
45
|
+
stderr when it ends: per function, calls, cache hits, retries, failures, total/mean/p95/max time
|
|
46
|
+
and the split between model time and thunc's own; for agents, steps and time in each tool; and
|
|
47
|
+
the share of wall time spent in thunc, with the overlap from `thunc.map`.
|
|
48
|
+
|
|
49
|
+
### Changed
|
|
50
|
+
|
|
51
|
+
- **A failed step is retried.** A timeout, lost connection, rate limit, server error or CLI call
|
|
52
|
+
that ended in an error (`thunc.errors.TransientError`) is retried twice in an agent run, with a
|
|
53
|
+
note in the run record, before the run fails. A text-protocol step on Claude Code or Codex may
|
|
54
|
+
take 120 seconds before it's retried, instead of the whole `timeout`.
|
|
55
|
+
- **The `anthropic` and `openai` backends reuse their connections.** One SDK client is shared by
|
|
56
|
+
every call in the process (`thunc.map`'s threads and agent runs included), instead of a new
|
|
57
|
+
client, and so a new TCP and TLS handshake, for each call. A new client is made when the API key,
|
|
58
|
+
the SDK's environment variables (`ANTHROPIC_*`, `OPENAI_*`) or the process change. In a local
|
|
59
|
+
benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
|
|
60
|
+
- **Agents on the text protocol can act several times per reply.** On Codex and `protocol="text"`
|
|
61
|
+
(and on Claude Code when it falls back to the text protocol), a reply can be a JSON array of
|
|
62
|
+
independent actions (reading three files) instead of one. They run in order, at most 16 per reply,
|
|
63
|
+
and every result comes back together, as with native tool calls. Each turn resends the whole
|
|
64
|
+
transcript and, on the CLI backends, starts the CLI, so fewer turns save both. A single JSON
|
|
65
|
+
action works as before. On Codex, with `live_tests/eval_prompts.py` (default prompt, 5 runs of
|
|
66
|
+
each task), every run batched its first reads: replies went from 6.0 / 4.8 / 5.0 to 5.0 / 3.0 /
|
|
67
|
+
3.6 (fix / review / analysis) and the mean time from 37 / 27 / 27 s to 29 / 19 / 21 s, with the
|
|
68
|
+
same work done and 30/30 passing.
|
|
69
|
+
- **The `codex` backend returns as soon as the answer arrives.** It reads Codex's JSON events as
|
|
70
|
+
they come (`codex exec --json`) instead of waiting for the process to exit and reading the answer
|
|
71
|
+
from a file. Codex takes about 0.4 s to shut down after answering; that now happens in the
|
|
72
|
+
background. Over 8 alternating pairs of real calls the new way was faster every time, by a median
|
|
73
|
+
of 0.67 s on a call of about 4 s. Every agent turn on Codex is one call, so the saving repeats.
|
|
74
|
+
|
|
75
|
+
## 0.2.1 (beta)
|
|
76
|
+
|
|
77
|
+
**Durable agents with Temporal.** An optional `thunc[temporal]` runtime records each model turn
|
|
78
|
+
and tool call in a Temporal workflow, so a run survives worker restarts and can be reattached
|
|
79
|
+
from another process. Local thunc stays dependency-free. See the
|
|
80
|
+
[Temporal guide](https://github.com/Eltarras/thunc/blob/main/examples/temporal/README.md).
|
|
81
|
+
|
|
82
|
+
### Added
|
|
83
|
+
|
|
84
|
+
- **`thunc.temporal`**: `Registry`, `Worker`, `Runtime` and `Handle` to register versioned tasks
|
|
85
|
+
and start, reattach to, inspect, cancel and resolve durable runs. One coordinator per
|
|
86
|
+
workspace runs requests in order; a repeated request ID reattaches to the same run. Agents
|
|
87
|
+
with `tools=` or `timeout=` can't be registered for durable runs yet
|
|
88
|
+
([#37](https://github.com/Eltarras/thunc/pull/37)).
|
|
89
|
+
- **Recoverable tool effects**: file writes and memory notes go through an intent and receipt
|
|
90
|
+
journal with atomic replacement and content hashes. A command whose outcome is uncertain is
|
|
91
|
+
never rerun automatically: the run waits for an operator's `resolve()`
|
|
92
|
+
([#37](https://github.com/Eltarras/thunc/pull/37)).
|
|
93
|
+
- **`thunc.temporal.adapters.execute_task`** composes registered tasks from native Temporal
|
|
94
|
+
workflows, with a classify → agent analysis → typed summary example
|
|
95
|
+
([#38](https://github.com/Eltarras/thunc/pull/38)).
|
|
96
|
+
|
|
97
|
+
### Changed
|
|
98
|
+
|
|
99
|
+
- The agent loop's decisions moved into a shared engine (`thunc/execution.py`) that local and
|
|
100
|
+
durable runs both use. A `remember` call that fails no longer appears in `Run.notes`
|
|
101
|
+
([#36](https://github.com/Eltarras/thunc/pull/36)).
|
|
102
|
+
|
|
103
|
+
## 0.2.0 (beta)
|
|
104
|
+
|
|
105
|
+
**Agents.** An agent is a typed function that can look around before it answers: it lists,
|
|
106
|
+
reads and searches files in a working directory, and, when its permissions allow, writes, edits
|
|
107
|
+
and runs commands, then returns a checked value of the task's return type.
|
|
108
|
+
See the [Agents section](https://github.com/Eltarras/thunc#agents) of the README.
|
|
109
|
+
|
|
110
|
+
### Added
|
|
111
|
+
|
|
112
|
+
- **`thunc.Agent(name, workdir=...)` and `@agent.task`**: declare tasks like `@thunc.function`.
|
|
113
|
+
Read-only by default, with `list`, `read`, `search` and `remember` tools. Sync and async tasks
|
|
114
|
+
([#23](https://github.com/Eltarras/thunc/pull/23)).
|
|
115
|
+
- **Memory and run files** in `.thunc_agents/<name>/`: `memory.md` (notes kept between runs),
|
|
116
|
+
`agent.json`, and one JSONL record per run. Runs of one agent take turns through an OS file
|
|
117
|
+
lock ([#23](https://github.com/Eltarras/thunc/pull/23)).
|
|
118
|
+
- **Permission rules**: `write:`, `read:`, `run:` and `!` denies with globs, plus the `write` and
|
|
119
|
+
`edit` tools. Edits need a fresh read of the file in the same run
|
|
120
|
+
([#26](https://github.com/Eltarras/thunc/pull/26)).
|
|
121
|
+
- **The `run` tool**: commands allowed by `run:` rules run without a shell, with a minimal
|
|
122
|
+
environment and a time limit that also stops their child processes
|
|
123
|
+
([#28](https://github.com/Eltarras/thunc/pull/28)).
|
|
124
|
+
- **`agent.run(task, ...)`** returns a `thunc.Run` with the value, files changed, commands,
|
|
125
|
+
denials, notes and steps. A failed run raises `thunc.AgentError` with the partial record
|
|
126
|
+
([#29](https://github.com/Eltarras/thunc/pull/29)).
|
|
127
|
+
- **`follow=`** gives the agent `AGENTS.md` / `CLAUDE.md`, or files you name, as instructions.
|
|
128
|
+
Off by default ([#31](https://github.com/Eltarras/thunc/pull/31)).
|
|
129
|
+
- **Native tool calls** on the Claude and OpenAI APIs, with the fixed part of the prompt cached.
|
|
130
|
+
Claude Code and Codex use a JSON text protocol; `protocol="text"` picks it on an API too
|
|
131
|
+
([#32](https://github.com/Eltarras/thunc/pull/32)).
|
|
132
|
+
- **System prompt presets**: `thunc.prompts.CODING`, `CODE_REVIEW` and `ANALYSIS`, for `system=`
|
|
133
|
+
([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
134
|
+
- **`tools=`**: your own typed, documented Python functions as agent tools
|
|
135
|
+
([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
136
|
+
- **`agent.call(...)`**, the agent version of `thunc.call`, and the **`@thunc.agent(...)`**
|
|
137
|
+
shorthand for a one-task agent ([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
138
|
+
- **`timeout=`** bounds a run's time; command limits are cut to the time left
|
|
139
|
+
([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
140
|
+
- **Files changed by commands** are included in `Run.files_changed`
|
|
141
|
+
([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
142
|
+
- `live_tests/eval_prompts.py`, an evaluation of the agent system prompt (bare / default /
|
|
143
|
+
preset). 90/90 runs passed on the Claude API and Claude Code
|
|
144
|
+
([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
145
|
+
- CI now also runs the offline tests on Windows ([#34](https://github.com/Eltarras/thunc/pull/34)).
|
|
146
|
+
|
|
147
|
+
### Changed
|
|
148
|
+
|
|
149
|
+
- **`thunc.agent` is the decorator** for one-task agents. `from thunc.agent import Agent` still
|
|
150
|
+
works; only `import thunc.agent as m` now gives the decorator rather than the module.
|
|
151
|
+
- The `codex` backend leaves out Codex's own permission notes, which made it refuse allowed edits
|
|
152
|
+
([#26](https://github.com/Eltarras/thunc/pull/26)).
|
|
153
|
+
- Agents refuse the `jev` backend with a `ThuncError` before a run starts; it only answers typed
|
|
154
|
+
questions ([#33](https://github.com/Eltarras/thunc/pull/33)).
|
|
155
|
+
|
|
156
|
+
Nothing changes for `@thunc.function` and `thunc.call`.
|
|
157
|
+
|
|
158
|
+
## 0.1.3 (beta)
|
|
159
|
+
|
|
160
|
+
- **`jev` backend** for TypeSafe's Jev judgment model: `bool` and `Literal` answers in about
|
|
161
|
+
0.3 s ([#27](https://github.com/Eltarras/thunc/pull/27)).
|
|
162
|
+
- The **`codex` backend** runs with Codex's own tools off and ignores `~/.codex/config.toml`
|
|
163
|
+
([#25](https://github.com/Eltarras/thunc/pull/25)).
|
|
164
|
+
- `@thunc.function` bodies like `return 1` now raise `TypeError` at definition
|
|
165
|
+
([#24](https://github.com/Eltarras/thunc/pull/24)).
|
|
166
|
+
|
|
167
|
+
## 0.1.2 (beta)
|
|
168
|
+
|
|
169
|
+
- **`cache=True`** saves valid answers on disk; `thunc.clear_cache()`, `thunc.cache_info()` and
|
|
170
|
+
the `thunc cache` command manage them ([#18](https://github.com/Eltarras/thunc/pull/18),
|
|
171
|
+
[#19](https://github.com/Eltarras/thunc/pull/19)).
|
|
172
|
+
- **`system=`** replaces the opening of the default system prompt; Codex gets it as its
|
|
173
|
+
instructions file ([#21](https://github.com/Eltarras/thunc/pull/21)).
|
|
174
|
+
- **Sturdier parsing**: common near-misses are read, and wrong values are retried instead of
|
|
175
|
+
returned. Only `ThuncError` escapes ([#20](https://github.com/Eltarras/thunc/pull/20)).
|
|
176
|
+
- An empty reply is no longer a valid `str`.
|
|
177
|
+
|
|
178
|
+
## 0.1.1 (beta)
|
|
179
|
+
|
|
180
|
+
- **`openai` backend** on the Responses API, and local models through `OPENAI_BASE_URL`.
|
|
181
|
+
- The Claude API backend tested live; CONTRIBUTING.md and Discussions added.
|
|
182
|
+
|
|
183
|
+
## 0.1.0 (beta)
|
|
184
|
+
|
|
185
|
+
- First release: `@thunc.function`, `thunc.call`, typed and validated results with retries and
|
|
186
|
+
`ensure=`, `thunc.map`, JSONL tracing; the `anthropic`, `claude-code` and `codex` backends.
|
|
@@ -16,34 +16,59 @@ Thanks for helping. thunc is in beta, so feedback on the API is as useful as cod
|
|
|
16
16
|
|
|
17
17
|
```bash
|
|
18
18
|
git clone https://github.com/Eltarras/thunc && cd thunc
|
|
19
|
-
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
|
|
19
|
+
python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
|
|
20
20
|
```
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
`temporal-test` installs the pinned Temporal SDK the durable-run tests use. Without it the tests in
|
|
23
|
+
`tests/temporal/` are skipped and everything else still runs. The README's
|
|
24
|
+
[Development](README.md#development) section lists the commands for live tests.
|
|
23
25
|
|
|
24
26
|
## Before you open a PR
|
|
25
27
|
|
|
26
|
-
CI runs these on Python 3.10 to 3.14, and all must pass
|
|
28
|
+
CI runs these on Python 3.10 to 3.14 on Linux, and all must pass. Run each on its own and check
|
|
29
|
+
its exit code:
|
|
27
30
|
|
|
28
31
|
```bash
|
|
29
32
|
.venv/bin/pytest
|
|
30
|
-
.venv/bin/ruff check .
|
|
33
|
+
.venv/bin/ruff check .
|
|
34
|
+
.venv/bin/ruff format --check .
|
|
31
35
|
.venv/bin/mypy --strict thunc
|
|
36
|
+
.venv/bin/mypy --strict --platform win32 thunc
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
CI also runs the offline tests on Windows (Python 3.13), since the agent lock, command handling and
|
|
40
|
+
paths have Windows-only code. If you touch `thunc/temporal/`, run the integration tests against a
|
|
41
|
+
real local Temporal service too (they download it once and make no model calls):
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
|
|
32
45
|
```
|
|
33
46
|
|
|
34
47
|
## Ground rules
|
|
35
48
|
|
|
36
49
|
- **Standard library only in the core.** A provider SDK is an optional extra in `pyproject.toml`,
|
|
37
|
-
imported inside its backend function, never at module level.
|
|
50
|
+
imported inside its backend function, never at module level. The same goes for Temporal:
|
|
51
|
+
only `thunc/temporal/` imports `temporalio`, and `thunc.call`, `@thunc.function` and
|
|
52
|
+
`agent.run()` must keep working without it.
|
|
38
53
|
- **Tests in `tests/` never call a real model.** Use the `fake` fixture in `tests/conftest.py`,
|
|
39
|
-
which returns scripted replies and records the prompts.
|
|
40
|
-
|
|
54
|
+
which returns scripted replies and records the prompts. Tests of native tool calls replace only
|
|
55
|
+
the SDK client and use the SDK's real types. Real calls belong in `live_tests/`, which CI
|
|
56
|
+
doesn't run.
|
|
57
|
+
- **Agent permissions are a safety boundary.** A change to `permissions.py`, `tools.py` or the
|
|
58
|
+
effect journal in `thunc/temporal/effects.py` needs a test that fails when the rule is broken.
|
|
59
|
+
Check it by breaking the rule on purpose and watching the test fail.
|
|
60
|
+
- **Durable runs replay.** Workflow code in `thunc/temporal/workflows.py` stays deterministic;
|
|
61
|
+
model calls, tools and file access happen in activities. A file or memory change goes through
|
|
62
|
+
the intent and receipt journal, and a command whose outcome is uncertain is never rerun
|
|
63
|
+
automatically. Before changing orchestration or the saved state format, version it and replay
|
|
64
|
+
saved histories (see the [Temporal guide](examples/temporal/README.md)).
|
|
41
65
|
- **User data stays out of the instructions.** Inputs are sent separately from the prompt
|
|
42
66
|
(see `_build_prompt` in `thunc/core.py`). Don't add code paths that paste inputs into
|
|
43
67
|
instructions.
|
|
44
68
|
- **Failures are loud.** When no valid answer arrives, raise `ThuncError`; never return a
|
|
45
69
|
default value.
|
|
46
|
-
- **Update the README** when you change the public API or the supported return types
|
|
70
|
+
- **Update the README** when you change the public API or the supported return types, and the
|
|
71
|
+
[Temporal guide](examples/temporal/README.md) when you change durable runs.
|
|
47
72
|
- **One change per PR**, with a description of what it does and how you tested it.
|
|
48
73
|
|
|
49
74
|
## Adding a backend
|
|
@@ -58,7 +83,14 @@ CI runs these on Python 3.10 to 3.14, and all must pass:
|
|
|
58
83
|
`thunc/config.py` if needed.
|
|
59
84
|
3. Add an optional extra for its SDK in `pyproject.toml`, and add the extra to the CI install.
|
|
60
85
|
4. Add offline tests with a stubbed SDK module, like the existing ones in `tests/test_backends.py`.
|
|
61
|
-
5.
|
|
86
|
+
5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
|
|
87
|
+
`Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
|
|
88
|
+
the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
|
|
89
|
+
A CLI that can call MCP tools can get native calls the way Claude Code does
|
|
90
|
+
(`thunc/claude_code.py`, tested with a fake CLI in `tests/test_claude_code_agent.py`).
|
|
91
|
+
Raise `TransientError` for failures worth asking again, so agent runs retry them.
|
|
92
|
+
Agents and durable runs refuse typed backends.
|
|
93
|
+
6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
|
|
62
94
|
the PR that you did.
|
|
63
95
|
|
|
64
96
|
## Known issues
|
|
@@ -73,8 +105,7 @@ model is asked again instead of thunc guessing:
|
|
|
73
105
|
- Quoted numbers (`"4"` for an `int`), labels in the wrong case (`Bug` for `bug`), prose around
|
|
74
106
|
JSON, Python-style values (`['a']`, `None`), trailing commas, curly quotes, non-ASCII digits.
|
|
75
107
|
- Unquoted text for `str | None`, and exclamations like `Yes!` for a `bool`.
|
|
76
|
-
- Several `<think>` blocks in a row,
|
|
77
|
-
four-backtick fences.
|
|
108
|
+
- Several `<think>` blocks in a row, `~~~`, indented or four-backtick fences.
|
|
78
109
|
- A one-key object whose key is a field of the expected dataclass: `{"customer": {...}}` for an
|
|
79
110
|
`Order` with a `customer` field is an `Order` with a bad customer, not a wrapper.
|
|
80
111
|
- A reply after the closing fence (for example a reasoning paragraph): it could be a second answer.
|
|
@@ -86,7 +117,11 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
|
|
|
86
117
|
**Open: behavior that might change.**
|
|
87
118
|
|
|
88
119
|
- `thunc.map` loses every result when one item fails, and returns coroutines for `async` functions.
|
|
89
|
-
- Backend errors (timeouts, connection failures) are never retried
|
|
120
|
+
- Backend errors (timeouts, connection failures) are never retried by `thunc.call`; only bad
|
|
121
|
+
replies are. Local agent runs retry a step that failed with a `TransientError` twice, and durable
|
|
122
|
+
runs retry transient provider failures.
|
|
123
|
+
- Several lines of prose before a code fence are read as a preamble, though `_unfence` in
|
|
124
|
+
`thunc/schema.py` documents one line, and no test pins either behavior.
|
|
90
125
|
- A union takes the first option that accepts the reply as JSON, in the order written: `3` for
|
|
91
126
|
`float | int` is `3.0`. An unquoted label is only tried after that, so `2` for
|
|
92
127
|
`Literal["2"] | int` is the int `2`, not the label.
|
|
@@ -109,10 +144,38 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
|
|
|
109
144
|
- A list of dataclasses passed as an input is shown to the model as Python reprs, not JSON.
|
|
110
145
|
- A string return annotation naming a class defined after the function raises `NameError` when
|
|
111
146
|
the decorator runs.
|
|
112
|
-
-
|
|
147
|
+
- On Python 3.10 and 3.11, a `make_dataclass` class whose field types are strings naming anything
|
|
148
|
+
but builtins (`"Item"`, `"Annotated[int, 'x']"`) raises `NameError`: before 3.12 its module is
|
|
149
|
+
`types`, not the caller's.
|
|
150
|
+
- `Annotated[...]` works on dataclass fields but not as a return type.
|
|
113
151
|
- The Python 3.10 fallback for string annotations (`_hints` and `resolve_strings` in
|
|
114
152
|
`thunc/schema.py`) can be removed when 3.10 support is dropped.
|
|
115
153
|
|
|
154
|
+
**Open: agents and durable runs.**
|
|
155
|
+
|
|
156
|
+
- Durable runs on `claude-code` use the JSON text protocol, not native calls: the MCP path keeps
|
|
157
|
+
one CLI process for the whole run, which a durable step can't snapshot. The text protocol is the
|
|
158
|
+
weaker one: in `live_tests/bench_tooluse.py` on Sonnet 5.5 it passed 12 of 24 runs, and 20 of 24
|
|
159
|
+
with step retries (measured before action arrays), against 24 of 24 for native calls.
|
|
160
|
+
- The `shell` permission isn't tested on Windows: its test is skipped there, so `cmd /c` has never
|
|
161
|
+
run in CI.
|
|
162
|
+
- With `shell`, a file read with `cat` doesn't count as read for `edit`, which refuses until the
|
|
163
|
+
agent reads it with `read` (the no-blind-overwrite rule). Agents work around it with a short
|
|
164
|
+
`read`, at the cost of a step.
|
|
165
|
+
- The agent loops on the Claude and OpenAI APIs have no live tool-use benchmark yet (only
|
|
166
|
+
`live_tests/eval_prompts.py`); their `max_tokens`, effort and stop-reason handling were reviewed
|
|
167
|
+
from the code only (`live_tests/bench_tooluse_report.md`, finding 7).
|
|
168
|
+
- Durable agents can't use `tools=` yet: the effects of the program's own functions can't be
|
|
169
|
+
journaled.
|
|
170
|
+
- Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
|
|
171
|
+
tombstones are kept forever. There's no context summarization either, so a long run fails once
|
|
172
|
+
its saved state passes 16 MiB.
|
|
173
|
+
- A durable workspace can only restart on the same volume and absolute paths; nothing moves it
|
|
174
|
+
between hosts.
|
|
175
|
+
- A hard worker kill can leave a command's subprocesses running.
|
|
176
|
+
- Durable runs aren't tested on Windows (the code is only type-checked for it), and there are no
|
|
177
|
+
live provider tests for durable runs yet.
|
|
178
|
+
|
|
116
179
|
## Reporting bugs
|
|
117
180
|
|
|
118
181
|
Open an [issue](https://github.com/Eltarras/thunc/issues) with:
|