thunc 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (129) hide show
  1. thunc-0.2.2/.github/demo-agent.gif +0 -0
  2. thunc-0.2.2/.github/demo-function.gif +0 -0
  3. {thunc-0.2.0 → thunc-0.2.2}/.github/workflows/ci.yml +18 -1
  4. {thunc-0.2.0 → thunc-0.2.2}/.gitignore +4 -0
  5. thunc-0.2.2/CHANGELOG.md +186 -0
  6. {thunc-0.2.0 → thunc-0.2.2}/CONTRIBUTING.md +76 -13
  7. {thunc-0.2.0 → thunc-0.2.2}/PKG-INFO +168 -52
  8. {thunc-0.2.0 → thunc-0.2.2}/README.md +161 -51
  9. thunc-0.2.2/benchmarks/README.md +32 -0
  10. thunc-0.2.2/benchmarks/__init__.py +1 -0
  11. thunc-0.2.2/benchmarks/__main__.py +60 -0
  12. thunc-0.2.2/benchmarks/bench_agent.py +62 -0
  13. thunc-0.2.2/benchmarks/bench_backends.py +84 -0
  14. thunc-0.2.2/benchmarks/bench_cache.py +34 -0
  15. thunc-0.2.2/benchmarks/bench_calls.py +127 -0
  16. thunc-0.2.2/benchmarks/bench_schema.py +85 -0
  17. thunc-0.2.2/benchmarks/bench_startup.py +30 -0
  18. thunc-0.2.2/benchmarks/bench_tools.py +102 -0
  19. thunc-0.2.2/benchmarks/fakes.py +101 -0
  20. thunc-0.2.2/benchmarks/harness.py +190 -0
  21. thunc-0.2.2/design/og-card.html +61 -0
  22. thunc-0.2.2/docs/docs/agents.html +189 -0
  23. thunc-0.2.2/docs/docs/api.html +150 -0
  24. thunc-0.2.2/docs/docs/backends.html +122 -0
  25. thunc-0.2.2/docs/docs/caching.html +109 -0
  26. thunc-0.2.2/docs/docs/functions.html +172 -0
  27. thunc-0.2.2/docs/docs/index.html +139 -0
  28. thunc-0.2.2/docs/docs/jev.html +182 -0
  29. thunc-0.2.2/docs/docs/temporal.html +147 -0
  30. thunc-0.2.2/docs/index.html +215 -0
  31. thunc-0.2.2/docs/jev.html +14 -0
  32. thunc-0.2.2/docs/og.png +0 -0
  33. thunc-0.2.2/docs/site.css +218 -0
  34. thunc-0.2.2/docs/site.js +94 -0
  35. thunc-0.2.2/docs/sitemap.xml +12 -0
  36. thunc-0.2.2/docs/temporal.html +14 -0
  37. {thunc-0.2.0 → thunc-0.2.2}/examples/repo_guide.py +4 -2
  38. thunc-0.2.2/examples/temporal/README.md +202 -0
  39. thunc-0.2.2/examples/temporal/application.py +28 -0
  40. thunc-0.2.2/examples/temporal/client.py +22 -0
  41. thunc-0.2.2/examples/temporal/pipeline.py +67 -0
  42. thunc-0.2.2/examples/temporal/worker.py +23 -0
  43. thunc-0.2.2/live_tests/bench_tooluse.py +903 -0
  44. thunc-0.2.2/live_tests/bench_tooluse_report.md +385 -0
  45. {thunc-0.2.0 → thunc-0.2.2}/pyproject.toml +5 -1
  46. thunc-0.2.2/pytest-temporal.ini +4 -0
  47. thunc-0.2.2/tests/fake_claude.py +135 -0
  48. thunc-0.2.2/tests/temporal/conftest.py +5 -0
  49. thunc-0.2.2/tests/temporal/process_worker.py +60 -0
  50. thunc-0.2.2/tests/temporal/test_contract.py +118 -0
  51. thunc-0.2.2/tests/temporal/test_runtime.py +492 -0
  52. thunc-0.2.2/tests/temporal/test_storage.py +58 -0
  53. {thunc-0.2.0 → thunc-0.2.2}/tests/test_agent.py +252 -6
  54. {thunc-0.2.0 → thunc-0.2.2}/tests/test_backends.py +206 -69
  55. thunc-0.2.2/tests/test_benchmarks.py +13 -0
  56. thunc-0.2.2/tests/test_claude_code_agent.py +267 -0
  57. thunc-0.2.2/tests/test_execution.py +82 -0
  58. {thunc-0.2.0 → thunc-0.2.2}/tests/test_native.py +35 -0
  59. {thunc-0.2.0 → thunc-0.2.2}/tests/test_permissions.py +19 -0
  60. thunc-0.2.2/tests/test_profiling.py +119 -0
  61. {thunc-0.2.0 → thunc-0.2.2}/thunc/__init__.py +1 -1
  62. {thunc-0.2.0 → thunc-0.2.2}/thunc/__main__.py +50 -2
  63. {thunc-0.2.0 → thunc-0.2.2}/thunc/agent.py +150 -93
  64. {thunc-0.2.0 → thunc-0.2.2}/thunc/backends.py +150 -26
  65. thunc-0.2.2/thunc/claude_code.py +351 -0
  66. {thunc-0.2.0 → thunc-0.2.2}/thunc/config.py +3 -1
  67. {thunc-0.2.0 → thunc-0.2.2}/thunc/core.py +25 -4
  68. {thunc-0.2.0 → thunc-0.2.2}/thunc/decorator.py +2 -0
  69. thunc-0.2.2/thunc/errors.py +15 -0
  70. thunc-0.2.2/thunc/execution.py +135 -0
  71. thunc-0.2.2/thunc/mcp_relay.py +77 -0
  72. {thunc-0.2.0 → thunc-0.2.2}/thunc/native.py +79 -14
  73. {thunc-0.2.0 → thunc-0.2.2}/thunc/permissions.py +23 -7
  74. thunc-0.2.2/thunc/profiling.py +261 -0
  75. thunc-0.2.2/thunc/temporal/__init__.py +37 -0
  76. thunc-0.2.2/thunc/temporal/activities.py +357 -0
  77. thunc-0.2.2/thunc/temporal/adapters.py +48 -0
  78. thunc-0.2.2/thunc/temporal/client.py +130 -0
  79. thunc-0.2.2/thunc/temporal/effects.py +166 -0
  80. thunc-0.2.2/thunc/temporal/models.py +67 -0
  81. thunc-0.2.2/thunc/temporal/registry.py +165 -0
  82. thunc-0.2.2/thunc/temporal/storage.py +158 -0
  83. thunc-0.2.2/thunc/temporal/worker.py +86 -0
  84. thunc-0.2.2/thunc/temporal/workflows.py +245 -0
  85. {thunc-0.2.0 → thunc-0.2.2}/thunc/tools.py +160 -42
  86. {thunc-0.2.0 → thunc-0.2.2}/uv.lock +115 -3
  87. thunc-0.2.0/CHANGELOG.md +0 -89
  88. thunc-0.2.0/docs/index.html +0 -267
  89. thunc-0.2.0/docs/jev.html +0 -325
  90. thunc-0.2.0/docs/sitemap.xml +0 -11
  91. thunc-0.2.0/thunc/errors.py +0 -2
  92. {thunc-0.2.0 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/ideas.yml +0 -0
  93. {thunc-0.2.0 → thunc-0.2.2}/.github/DISCUSSION_TEMPLATE/q-a.yml +0 -0
  94. {thunc-0.2.0 → thunc-0.2.2}/.github/social-preview.png +0 -0
  95. {thunc-0.2.0 → thunc-0.2.2}/.github/workflows/publish.yml +0 -0
  96. {thunc-0.2.0 → thunc-0.2.2}/LICENSE +0 -0
  97. {thunc-0.2.0 → thunc-0.2.2}/docs/.nojekyll +0 -0
  98. {thunc-0.2.0 → thunc-0.2.2}/docs/apple-touch-icon.png +0 -0
  99. {thunc-0.2.0 → thunc-0.2.2}/docs/favicon-96.png +0 -0
  100. {thunc-0.2.0 → thunc-0.2.2}/docs/favicon.ico +0 -0
  101. {thunc-0.2.0 → thunc-0.2.2}/docs/favicon.svg +0 -0
  102. {thunc-0.2.0 → thunc-0.2.2}/docs/google2cd177e5e85b3c3e.html +0 -0
  103. {thunc-0.2.0 → thunc-0.2.2}/examples/dynamic_prompts.py +0 -0
  104. {thunc-0.2.0 → thunc-0.2.2}/examples/hello.py +0 -0
  105. {thunc-0.2.0 → thunc-0.2.2}/examples/jev_hello.py +0 -0
  106. {thunc-0.2.0 → thunc-0.2.2}/examples/jev_inbox.py +0 -0
  107. {thunc-0.2.0 → thunc-0.2.2}/examples/jev_with_claude.py +0 -0
  108. {thunc-0.2.0 → thunc-0.2.2}/examples/log_triage.py +0 -0
  109. {thunc-0.2.0 → thunc-0.2.2}/examples/support_inbox.py +0 -0
  110. {thunc-0.2.0 → thunc-0.2.2}/live_tests/conftest.py +0 -0
  111. {thunc-0.2.0 → thunc-0.2.2}/live_tests/eval_prompts.py +0 -0
  112. {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_agent.py +0 -0
  113. {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_bool_decision.py +0 -0
  114. {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_dict_output.py +0 -0
  115. {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_hello.py +0 -0
  116. {thunc-0.2.0 → thunc-0.2.2}/live_tests/test_literal_choice.py +0 -0
  117. {thunc-0.2.0 → thunc-0.2.2}/tests/conftest.py +0 -0
  118. {thunc-0.2.0 → thunc-0.2.2}/tests/future_types.py +0 -0
  119. {thunc-0.2.0 → thunc-0.2.2}/tests/test_cache.py +0 -0
  120. {thunc-0.2.0 → thunc-0.2.2}/tests/test_calls.py +0 -0
  121. {thunc-0.2.0 → thunc-0.2.2}/tests/test_cli.py +0 -0
  122. {thunc-0.2.0 → thunc-0.2.2}/tests/test_jev.py +0 -0
  123. {thunc-0.2.0 → thunc-0.2.2}/tests/test_schema.py +0 -0
  124. {thunc-0.2.0 → thunc-0.2.2}/thunc/cache.py +0 -0
  125. {thunc-0.2.0 → thunc-0.2.2}/thunc/prompts.py +0 -0
  126. {thunc-0.2.0 → thunc-0.2.2}/thunc/py.typed +0 -0
  127. {thunc-0.2.0 → thunc-0.2.2}/thunc/runs.py +0 -0
  128. {thunc-0.2.0 → thunc-0.2.2}/thunc/schema.py +0 -0
  129. {thunc-0.2.0 → thunc-0.2.2}/thunc/store.py +0 -0
Binary file
Binary file
@@ -16,11 +16,12 @@ jobs:
16
16
  - uses: actions/setup-python@v5
17
17
  with:
18
18
  python-version: ${{ matrix.python-version }}
19
- - run: pip install -e ".[anthropic,openai,dev]"
19
+ - run: pip install -e ".[anthropic,openai,temporal-test,dev]"
20
20
  - run: pytest # offline tests only; live_tests need a model and are not run in CI
21
21
  - run: ruff check .
22
22
  - run: ruff format --check .
23
23
  - run: mypy --strict thunc
24
+ - run: mypy --strict --platform win32 thunc
24
25
 
25
26
  # The lock, process stopping and path handling have code just for Windows.
26
27
  test-windows:
@@ -34,3 +35,19 @@ jobs:
34
35
  - run: pytest
35
36
  env:
36
37
  PYTHONIOENCODING: utf-8 # the console's own code page can't print every test id and message
38
+
39
+ temporal-integration:
40
+ runs-on: ubuntu-latest
41
+ strategy:
42
+ matrix:
43
+ python-version: ["3.10", "3.14"]
44
+ steps:
45
+ - uses: actions/checkout@v4
46
+ - uses: actions/setup-python@v5
47
+ with:
48
+ python-version: ${{ matrix.python-version }}
49
+ - run: pip install -e ".[anthropic,openai,temporal-test,dev]"
50
+ - name: Real service, restart and replay tests (no providers)
51
+ run: pytest -c pytest-temporal.ini tests/temporal -q
52
+ env:
53
+ THUNC_TEMPORAL_TESTS: "1"
@@ -9,3 +9,7 @@ dist/
9
9
  *.jsonl
10
10
  .thunc_cache/
11
11
  .thunc_agents/
12
+ .thunc_temporal/
13
+ .temporal-dev/
14
+ examples/temporal/temporal-state/
15
+ examples/temporal/temporal-workspace/
@@ -0,0 +1,186 @@
1
+ # Changelog
2
+
3
+ All notable changes to thunc. The full notes for each release are on the
4
+ [releases page](https://github.com/Eltarras/thunc/releases).
5
+
6
+ ## 0.2.2 (beta)
7
+
8
+ **Agents that use their tools reliably, and faster calls.** Agents on Claude Code make native tool
9
+ calls instead of writing each action as JSON text, a failed step is retried instead of ending the
10
+ run, and the agent's tools fill gaps a benchmark found. In that tool-use benchmark
11
+ (`live_tests/bench_tooluse.py`, 8 tasks, Claude Sonnet 5.5, 3 runs each), agents on Claude Code went
12
+ from 12 of 24 runs passing to 24 of 24, from 99 to 10 seconds a task, and from $0.084 to $0.022 a
13
+ task; Claude Code itself took 11 seconds and $0.069. The API backends also reuse their connections,
14
+ Codex answers return sooner, and `thunc run --profile` shows where a program's time goes.
15
+
16
+ ### Behavior changes
17
+
18
+ Nothing is removed, but these defaults change:
19
+
20
+ - **Agents on `claude-code` make native tool calls** (see Added). With a `claude` CLI too old for
21
+ them, or with MCP servers turned off by a policy, a run falls back to the text protocol with a
22
+ warning. `protocol="text"` keeps the old way.
23
+ - **`list` and `search` leave out what git ignores** in a git repository (build output, caches,
24
+ vendored code). A folder named explicitly is still listed and searched.
25
+ - **Long command output keeps its start and its end** (the first error and the summary), not only
26
+ the end.
27
+ - **The `claude-code` backend loads none of your Claude Code settings** (`--setting-sources ""`):
28
+ no `CLAUDE.md`, settings or hooks reach thunc's calls, plain function calls included, so an
29
+ agent's `workdir` can't give it instructions unless `follow=` asks for them.
30
+
31
+ ### Added
32
+
33
+ - **Native calls on `claude-code`**: the agent's tools are an MCP server that one `claude -p`
34
+ process per run calls; thunc carries out each call with its own tools, permissions and run
35
+ record. The CLI runs in the agent's `workdir`. `protocol="text"` keeps the old way, and durable
36
+ runs on Claude Code still use it. When Claude Code can't start native calls (an older CLI, or MCP
37
+ servers turned off by a policy), a run falls back to the text protocol with a warning and a
38
+ `fallback` entry in its record; `protocol="native"` raises instead.
39
+ - **`run` takes `cwd`**, a folder inside `workdir` to run the command in.
40
+ - **The `shell` permission** runs command lines through the system shell, so pipes, `&&`, `cd`
41
+ and redirects work. Off by default; it can't be combined with `!run:` rules.
42
+ - **`search` takes `glob`** (`*.py` by file name, `src/**/*.ts` by path) to limit the files
43
+ searched.
44
+ - **`thunc run --profile`**: runs a script (or `-m module`) and prints a performance report to
45
+ stderr when it ends: per function, calls, cache hits, retries, failures, total/mean/p95/max time
46
+ and the split between model time and thunc's own; for agents, steps and time in each tool; and
47
+ the share of wall time spent in thunc, with the overlap from `thunc.map`.
48
+
49
+ ### Changed
50
+
51
+ - **A failed step is retried.** A timeout, lost connection, rate limit, server error or CLI call
52
+ that ended in an error (`thunc.errors.TransientError`) is retried twice in an agent run, with a
53
+ note in the run record, before the run fails. A text-protocol step on Claude Code or Codex may
54
+ take 120 seconds before it's retried, instead of the whole `timeout`.
55
+ - **The `anthropic` and `openai` backends reuse their connections.** One SDK client is shared by
56
+ every call in the process (`thunc.map`'s threads and agent runs included), instead of a new
57
+ client, and so a new TCP and TLS handshake, for each call. A new client is made when the API key,
58
+ the SDK's environment variables (`ANTHROPIC_*`, `OPENAI_*`) or the process change. In a local
59
+ benchmark with 60 ms of connection setup, 20 calls in a row went from 1.47 s to 68 ms.
60
+ - **Agents on the text protocol can act several times per reply.** On Codex and `protocol="text"`
61
+ (and on Claude Code when it falls back to the text protocol), a reply can be a JSON array of
62
+ independent actions (reading three files) instead of one. They run in order, at most 16 per reply,
63
+ and every result comes back together, as with native tool calls. Each turn resends the whole
64
+ transcript and, on the CLI backends, starts the CLI, so fewer turns save both. A single JSON
65
+ action works as before. On Codex, with `live_tests/eval_prompts.py` (default prompt, 5 runs of
66
+ each task), every run batched its first reads: replies went from 6.0 / 4.8 / 5.0 to 5.0 / 3.0 /
67
+ 3.6 (fix / review / analysis) and the mean time from 37 / 27 / 27 s to 29 / 19 / 21 s, with the
68
+ same work done and 30/30 passing.
69
+ - **The `codex` backend returns as soon as the answer arrives.** It reads Codex's JSON events as
70
+ they come (`codex exec --json`) instead of waiting for the process to exit and reading the answer
71
+ from a file. Codex takes about 0.4 s to shut down after answering; that now happens in the
72
+ background. Over 8 alternating pairs of real calls the new way was faster every time, by a median
73
+ of 0.67 s on a call of about 4 s. Every agent turn on Codex is one call, so the saving repeats.
74
+
75
+ ## 0.2.1 (beta)
76
+
77
+ **Durable agents with Temporal.** An optional `thunc[temporal]` runtime records each model turn
78
+ and tool call in a Temporal workflow, so a run survives worker restarts and can be reattached
79
+ from another process. Local thunc stays dependency-free. See the
80
+ [Temporal guide](https://github.com/Eltarras/thunc/blob/main/examples/temporal/README.md).
81
+
82
+ ### Added
83
+
84
+ - **`thunc.temporal`**: `Registry`, `Worker`, `Runtime` and `Handle` to register versioned tasks
85
+ and start, reattach to, inspect, cancel and resolve durable runs. One coordinator per
86
+ workspace runs requests in order; a repeated request ID reattaches to the same run. Agents
87
+ with `tools=` or `timeout=` can't be registered for durable runs yet
88
+ ([#37](https://github.com/Eltarras/thunc/pull/37)).
89
+ - **Recoverable tool effects**: file writes and memory notes go through an intent and receipt
90
+ journal with atomic replacement and content hashes. A command whose outcome is uncertain is
91
+ never rerun automatically: the run waits for an operator's `resolve()`
92
+ ([#37](https://github.com/Eltarras/thunc/pull/37)).
93
+ - **`thunc.temporal.adapters.execute_task`** composes registered tasks from native Temporal
94
+ workflows, with a classify → agent analysis → typed summary example
95
+ ([#38](https://github.com/Eltarras/thunc/pull/38)).
96
+
97
+ ### Changed
98
+
99
+ - The agent loop's decisions moved into a shared engine (`thunc/execution.py`) that local and
100
+ durable runs both use. A `remember` call that fails no longer appears in `Run.notes`
101
+ ([#36](https://github.com/Eltarras/thunc/pull/36)).
102
+
103
+ ## 0.2.0 (beta)
104
+
105
+ **Agents.** An agent is a typed function that can look around before it answers: it lists,
106
+ reads and searches files in a working directory, and, when its permissions allow, writes, edits
107
+ and runs commands, then returns a checked value of the task's return type.
108
+ See the [Agents section](https://github.com/Eltarras/thunc#agents) of the README.
109
+
110
+ ### Added
111
+
112
+ - **`thunc.Agent(name, workdir=...)` and `@agent.task`**: declare tasks like `@thunc.function`.
113
+ Read-only by default, with `list`, `read`, `search` and `remember` tools. Sync and async tasks
114
+ ([#23](https://github.com/Eltarras/thunc/pull/23)).
115
+ - **Memory and run files** in `.thunc_agents/<name>/`: `memory.md` (notes kept between runs),
116
+ `agent.json`, and one JSONL record per run. Runs of one agent take turns through an OS file
117
+ lock ([#23](https://github.com/Eltarras/thunc/pull/23)).
118
+ - **Permission rules**: `write:`, `read:`, `run:` and `!` denies with globs, plus the `write` and
119
+ `edit` tools. Edits need a fresh read of the file in the same run
120
+ ([#26](https://github.com/Eltarras/thunc/pull/26)).
121
+ - **The `run` tool**: commands allowed by `run:` rules run without a shell, with a minimal
122
+ environment and a time limit that also stops their child processes
123
+ ([#28](https://github.com/Eltarras/thunc/pull/28)).
124
+ - **`agent.run(task, ...)`** returns a `thunc.Run` with the value, files changed, commands,
125
+ denials, notes and steps. A failed run raises `thunc.AgentError` with the partial record
126
+ ([#29](https://github.com/Eltarras/thunc/pull/29)).
127
+ - **`follow=`** gives the agent `AGENTS.md` / `CLAUDE.md`, or files you name, as instructions.
128
+ Off by default ([#31](https://github.com/Eltarras/thunc/pull/31)).
129
+ - **Native tool calls** on the Claude and OpenAI APIs, with the fixed part of the prompt cached.
130
+ Claude Code and Codex use a JSON text protocol; `protocol="text"` picks it on an API too
131
+ ([#32](https://github.com/Eltarras/thunc/pull/32)).
132
+ - **System prompt presets**: `thunc.prompts.CODING`, `CODE_REVIEW` and `ANALYSIS`, for `system=`
133
+ ([#34](https://github.com/Eltarras/thunc/pull/34)).
134
+ - **`tools=`**: your own typed, documented Python functions as agent tools
135
+ ([#34](https://github.com/Eltarras/thunc/pull/34)).
136
+ - **`agent.call(...)`**, the agent version of `thunc.call`, and the **`@thunc.agent(...)`**
137
+ shorthand for a one-task agent ([#34](https://github.com/Eltarras/thunc/pull/34)).
138
+ - **`timeout=`** bounds a run's time; command limits are cut to the time left
139
+ ([#34](https://github.com/Eltarras/thunc/pull/34)).
140
+ - **Files changed by commands** are included in `Run.files_changed`
141
+ ([#34](https://github.com/Eltarras/thunc/pull/34)).
142
+ - `live_tests/eval_prompts.py`, an evaluation of the agent system prompt (bare / default /
143
+ preset). 90/90 runs passed on the Claude API and Claude Code
144
+ ([#34](https://github.com/Eltarras/thunc/pull/34)).
145
+ - CI now also runs the offline tests on Windows ([#34](https://github.com/Eltarras/thunc/pull/34)).
146
+
147
+ ### Changed
148
+
149
+ - **`thunc.agent` is the decorator** for one-task agents. `from thunc.agent import Agent` still
150
+ works; only `import thunc.agent as m` now gives the decorator rather than the module.
151
+ - The `codex` backend leaves out Codex's own permission notes, which made it refuse allowed edits
152
+ ([#26](https://github.com/Eltarras/thunc/pull/26)).
153
+ - Agents refuse the `jev` backend with a `ThuncError` before a run starts; it only answers typed
154
+ questions ([#33](https://github.com/Eltarras/thunc/pull/33)).
155
+
156
+ Nothing changes for `@thunc.function` and `thunc.call`.
157
+
158
+ ## 0.1.3 (beta)
159
+
160
+ - **`jev` backend** for TypeSafe's Jev judgment model: `bool` and `Literal` answers in about
161
+ 0.3 s ([#27](https://github.com/Eltarras/thunc/pull/27)).
162
+ - The **`codex` backend** runs with Codex's own tools off and ignores `~/.codex/config.toml`
163
+ ([#25](https://github.com/Eltarras/thunc/pull/25)).
164
+ - `@thunc.function` bodies like `return 1` now raise `TypeError` at definition
165
+ ([#24](https://github.com/Eltarras/thunc/pull/24)).
166
+
167
+ ## 0.1.2 (beta)
168
+
169
+ - **`cache=True`** saves valid answers on disk; `thunc.clear_cache()`, `thunc.cache_info()` and
170
+ the `thunc cache` command manage them ([#18](https://github.com/Eltarras/thunc/pull/18),
171
+ [#19](https://github.com/Eltarras/thunc/pull/19)).
172
+ - **`system=`** replaces the opening of the default system prompt; Codex gets it as its
173
+ instructions file ([#21](https://github.com/Eltarras/thunc/pull/21)).
174
+ - **Sturdier parsing**: common near-misses are read, and wrong values are retried instead of
175
+ returned. Only `ThuncError` escapes ([#20](https://github.com/Eltarras/thunc/pull/20)).
176
+ - An empty reply is no longer a valid `str`.
177
+
178
+ ## 0.1.1 (beta)
179
+
180
+ - **`openai` backend** on the Responses API, and local models through `OPENAI_BASE_URL`.
181
+ - The Claude API backend tested live; CONTRIBUTING.md and Discussions added.
182
+
183
+ ## 0.1.0 (beta)
184
+
185
+ - First release: `@thunc.function`, `thunc.call`, typed and validated results with retries and
186
+ `ensure=`, `thunc.map`, JSONL tracing; the `anthropic`, `claude-code` and `codex` backends.
@@ -16,34 +16,59 @@ Thanks for helping. thunc is in beta, so feedback on the API is as useful as cod
16
16
 
17
17
  ```bash
18
18
  git clone https://github.com/Eltarras/thunc && cd thunc
19
- python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,dev]"
19
+ python3 -m venv .venv && .venv/bin/pip install -e ".[anthropic,openai,temporal-test,dev]"
20
20
  ```
21
21
 
22
- The README's [Development](README.md#development) section lists the commands for tests and checks.
22
+ `temporal-test` installs the pinned Temporal SDK the durable-run tests use. Without it the tests in
23
+ `tests/temporal/` are skipped and everything else still runs. The README's
24
+ [Development](README.md#development) section lists the commands for live tests.
23
25
 
24
26
  ## Before you open a PR
25
27
 
26
- CI runs these on Python 3.10 to 3.14, and all must pass:
28
+ CI runs these on Python 3.10 to 3.14 on Linux, and all must pass. Run each on its own and check
29
+ its exit code:
27
30
 
28
31
  ```bash
29
32
  .venv/bin/pytest
30
- .venv/bin/ruff check . && .venv/bin/ruff format --check .
33
+ .venv/bin/ruff check .
34
+ .venv/bin/ruff format --check .
31
35
  .venv/bin/mypy --strict thunc
36
+ .venv/bin/mypy --strict --platform win32 thunc
37
+ ```
38
+
39
+ CI also runs the offline tests on Windows (Python 3.13), since the agent lock, command handling and
40
+ paths have Windows-only code. If you touch `thunc/temporal/`, run the integration tests against a
41
+ real local Temporal service too (they download it once and make no model calls):
42
+
43
+ ```bash
44
+ THUNC_TEMPORAL_TESTS=1 .venv/bin/pytest -c pytest-temporal.ini tests/temporal
32
45
  ```
33
46
 
34
47
  ## Ground rules
35
48
 
36
49
  - **Standard library only in the core.** A provider SDK is an optional extra in `pyproject.toml`,
37
- imported inside its backend function, never at module level.
50
+ imported inside its backend function, never at module level. The same goes for Temporal:
51
+ only `thunc/temporal/` imports `temporalio`, and `thunc.call`, `@thunc.function` and
52
+ `agent.run()` must keep working without it.
38
53
  - **Tests in `tests/` never call a real model.** Use the `fake` fixture in `tests/conftest.py`,
39
- which returns scripted replies and records the prompts. Real calls belong in `live_tests/`,
40
- which CI doesn't run.
54
+ which returns scripted replies and records the prompts. Tests of native tool calls replace only
55
+ the SDK client and use the SDK's real types. Real calls belong in `live_tests/`, which CI
56
+ doesn't run.
57
+ - **Agent permissions are a safety boundary.** A change to `permissions.py`, `tools.py` or the
58
+ effect journal in `thunc/temporal/effects.py` needs a test that fails when the rule is broken.
59
+ Check it by breaking the rule on purpose and watching the test fail.
60
+ - **Durable runs replay.** Workflow code in `thunc/temporal/workflows.py` stays deterministic;
61
+ model calls, tools and file access happen in activities. A file or memory change goes through
62
+ the intent and receipt journal, and a command whose outcome is uncertain is never rerun
63
+ automatically. Before changing orchestration or the saved state format, version it and replay
64
+ saved histories (see the [Temporal guide](examples/temporal/README.md)).
41
65
  - **User data stays out of the instructions.** Inputs are sent separately from the prompt
42
66
  (see `_build_prompt` in `thunc/core.py`). Don't add code paths that paste inputs into
43
67
  instructions.
44
68
  - **Failures are loud.** When no valid answer arrives, raise `ThuncError`; never return a
45
69
  default value.
46
- - **Update the README** when you change the public API or the supported return types.
70
+ - **Update the README** when you change the public API or the supported return types, and the
71
+ [Temporal guide](examples/temporal/README.md) when you change durable runs.
47
72
  - **One change per PR**, with a description of what it does and how you tested it.
48
73
 
49
74
  ## Adding a backend
@@ -58,7 +83,14 @@ CI runs these on Python 3.10 to 3.14, and all must pass:
58
83
  `thunc/config.py` if needed.
59
84
  3. Add an optional extra for its SDK in `pyproject.toml`, and add the extra to the CI install.
60
85
  4. Add offline tests with a stubbed SDK module, like the existing ones in `tests/test_backends.py`.
61
- 5. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
86
+ 5. Agents work on any text backend through the JSON text protocol. For native tool calls, add a
87
+ `Conversation` for the API in `thunc/native.py`, add the backend to `native.NATIVE`, and pick
88
+ the class where `thunc/agent.py` builds the conversation. Test it like `tests/test_native.py`.
89
+ A CLI that can call MCP tools can get native calls the way Claude Code does
90
+ (`thunc/claude_code.py`, tested with a fake CLI in `tests/test_claude_code_agent.py`).
91
+ Raise `TransientError` for failures worth asking again, so agent runs retry them.
92
+ Agents and durable runs refuse typed backends.
93
+ 6. Run `THUNC_BACKEND=<name> .venv/bin/pytest live_tests` against the real service, and say in
62
94
  the PR that you did.
63
95
 
64
96
  ## Known issues
@@ -73,8 +105,7 @@ model is asked again instead of thunc guessing:
73
105
  - Quoted numbers (`"4"` for an `int`), labels in the wrong case (`Bug` for `bug`), prose around
74
106
  JSON, Python-style values (`['a']`, `None`), trailing commas, curly quotes, non-ASCII digits.
75
107
  - Unquoted text for `str | None`, and exclamations like `Yes!` for a `bool`.
76
- - Several `<think>` blocks in a row, several lines of prose before a fence, `~~~`, indented or
77
- four-backtick fences.
108
+ - Several `<think>` blocks in a row, `~~~`, indented or four-backtick fences.
78
109
  - A one-key object whose key is a field of the expected dataclass: `{"customer": {...}}` for an
79
110
  `Order` with a `customer` field is an `Order` with a bad customer, not a wrapper.
80
111
  - A reply after the closing fence (for example a reasoning paragraph): it could be a second answer.
@@ -86,7 +117,11 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
86
117
  **Open: behavior that might change.**
87
118
 
88
119
  - `thunc.map` loses every result when one item fails, and returns coroutines for `async` functions.
89
- - Backend errors (timeouts, connection failures) are never retried; only bad replies are.
120
+ - Backend errors (timeouts, connection failures) are never retried by `thunc.call`; only bad
121
+ replies are. Local agent runs retry a step that failed with a `TransientError` twice, and durable
122
+ runs retry transient provider failures.
123
+ - Several lines of prose before a code fence are read as a preamble, though `_unfence` in
124
+ `thunc/schema.py` documents one line, and no test pins either behavior.
90
125
  - A union takes the first option that accepts the reply as JSON, in the order written: `3` for
91
126
  `float | int` is `3.0`. An unquoted label is only tried after that, so `2` for
92
127
  `Literal["2"] | int` is the int `2`, not the label.
@@ -109,10 +144,38 @@ no reliable way to tell them from a real one. Only an empty reply is retried.
109
144
  - A list of dataclasses passed as an input is shown to the model as Python reprs, not JSON.
110
145
  - A string return annotation naming a class defined after the function raises `NameError` when
111
146
  the decorator runs.
112
- - `make_dataclass` classes, and `Annotated` fields on Python 3.10, aren't supported.
147
+ - On Python 3.10 and 3.11, a `make_dataclass` class whose field types are strings naming anything
148
+ but builtins (`"Item"`, `"Annotated[int, 'x']"`) raises `NameError`: before 3.12 its module is
149
+ `types`, not the caller's.
150
+ - `Annotated[...]` works on dataclass fields but not as a return type.
113
151
  - The Python 3.10 fallback for string annotations (`_hints` and `resolve_strings` in
114
152
  `thunc/schema.py`) can be removed when 3.10 support is dropped.
115
153
 
154
+ **Open: agents and durable runs.**
155
+
156
+ - Durable runs on `claude-code` use the JSON text protocol, not native calls: the MCP path keeps
157
+ one CLI process for the whole run, which a durable step can't snapshot. The text protocol is the
158
+ weaker one: in `live_tests/bench_tooluse.py` on Sonnet 5.5 it passed 12 of 24 runs, and 20 of 24
159
+ with step retries (measured before action arrays), against 24 of 24 for native calls.
160
+ - The `shell` permission isn't tested on Windows: its test is skipped there, so `cmd /c` has never
161
+ run in CI.
162
+ - With `shell`, a file read with `cat` doesn't count as read for `edit`, which refuses until the
163
+ agent reads it with `read` (the no-blind-overwrite rule). Agents work around it with a short
164
+ `read`, at the cost of a step.
165
+ - The agent loops on the Claude and OpenAI APIs have no live tool-use benchmark yet (only
166
+ `live_tests/eval_prompts.py`); their `max_tokens`, effort and stop-reason handling were reviewed
167
+ from the code only (`live_tests/bench_tooluse_report.md`, finding 7).
168
+ - Durable agents can't use `tools=` yet: the effects of the program's own functions can't be
169
+ journaled.
170
+ - Durable runs have no garbage collection: the journal, transcript artifacts and request-ID
171
+ tombstones are kept forever. There's no context summarization either, so a long run fails once
172
+ its saved state passes 16 MiB.
173
+ - A durable workspace can only restart on the same volume and absolute paths; nothing moves it
174
+ between hosts.
175
+ - A hard worker kill can leave a command's subprocesses running.
176
+ - Durable runs aren't tested on Windows (the code is only type-checked for it), and there are no
177
+ live provider tests for durable runs yet.
178
+
116
179
  ## Reporting bugs
117
180
 
118
181
  Open an [issue](https://github.com/Eltarras/thunc/issues) with: