wrencode 0.1.4.7__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: wrencode
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: A minimal
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: A minimal agent harness for coding, in a single Python file
|
|
5
5
|
Project-URL: Homepage, https://github.com/almostly/wrencode
|
|
6
6
|
Project-URL: Repository, https://github.com/almostly/wrencode
|
|
7
7
|
Project-URL: Issues, https://github.com/almostly/wrencode/issues
|
|
@@ -23,7 +23,7 @@ Description-Content-Type: text/markdown
|
|
|
23
23
|
|
|
24
24
|
# 🐦 WrenCode
|
|
25
25
|
|
|
26
|
-
A minimal
|
|
26
|
+
A minimal agent harness for coding, in a single Python file.
|
|
27
27
|
|
|
28
28
|
Named after Harold Wren - the alias of a genius who built a superintelligent AI and operated quietly in the background.
|
|
29
29
|
|
|
@@ -31,9 +31,9 @@ Named after Harold Wren - the alias of a genius who built a superintelligent AI
|
|
|
31
31
|
|
|
32
32
|
## What it is
|
|
33
33
|
|
|
34
|
-
WrenCode is a
|
|
34
|
+
WrenCode is a coding agent harness: everything around the model that turns it into an agent. It runs the tool-calling loop, executes tools, builds the system prompt, and manages context, locally or via API, giving an LLM the ability to read, write, and edit files, search codebases, and run shell commands - enough to autonomously navigate and modify a real project.
|
|
35
35
|
|
|
36
|
-
Where Claude Code is the batteries-included
|
|
36
|
+
Where Claude Code is the batteries-included harness, WrenCode is the **"understand and own your agent" harness**: the entire agent loop fits in one readable file, runs against local or hosted models, and is yours to hack.
|
|
37
37
|
|
|
38
38
|
## Backends
|
|
39
39
|
|
|
@@ -47,7 +47,9 @@ saved choice, e.g. for CI.
|
|
|
47
47
|
|`anthropic` |Claude via Anthropic API |binary + source |
|
|
48
48
|
|`openai` |GPT models via OpenAI API |binary + source |
|
|
49
49
|
|`openrouter` |Any model via OpenRouter |binary + source |
|
|
50
|
+
|`nanogpt` |Any model via NanoGPT |binary + source |
|
|
50
51
|
|`ollama` |Local models via a running `ollama serve`|binary + source |
|
|
52
|
+
|`openai-compatible`|vLLM, llama.cpp, Hugging Face, any OpenAI-compatible server|binary + source|
|
|
51
53
|
|`local` |Local proxy via Anthropic-compatible API|binary + source |
|
|
52
54
|
|`transformers`|HuggingFace Transformers (CPU/MPS/GPU) |source install only |
|
|
53
55
|
|`mlx` |Apple Silicon via MLX |source install, macOS |
|
|
@@ -61,13 +63,37 @@ The default local models are
|
|
|
61
63
|
[`deburky/gpt-oss-claude-mlx`](https://huggingface.co/deburky/gpt-oss-claude-mlx)
|
|
62
64
|
(MLX) — override either with `MODEL=...`.
|
|
63
65
|
|
|
66
|
+
### OpenAI-compatible servers
|
|
67
|
+
|
|
68
|
+
`openai-compatible` talks to any server that implements OpenAI chat completions,
|
|
69
|
+
using native tool calls. Point it at the server with `OPENAI_COMPATIBLE_BASE_URL`
|
|
70
|
+
(default `http://localhost:8000/v1`). If the server serves exactly one model,
|
|
71
|
+
WrenCode uses it; otherwise set `MODEL`.
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
# vLLM (tool calling needs these flags; pick the parser for your model)
|
|
75
|
+
vllm serve Qwen/Qwen2.5-Coder-7B-Instruct --enable-auto-tool-choice --tool-call-parser hermes
|
|
76
|
+
BACKEND=openai-compatible wrencode
|
|
77
|
+
|
|
78
|
+
# llama.cpp (--jinja enables tool calling)
|
|
79
|
+
llama-server -m qwen2.5-coder-7b-instruct-q4_k_m.gguf --jinja --port 8080
|
|
80
|
+
BACKEND=openai-compatible OPENAI_COMPATIBLE_BASE_URL=http://localhost:8080/v1 wrencode
|
|
81
|
+
|
|
82
|
+
# Hugging Face Inference Providers
|
|
83
|
+
BACKEND=openai-compatible OPENAI_COMPATIBLE_BASE_URL=https://router.huggingface.co/v1 \
|
|
84
|
+
OPENAI_COMPATIBLE_API_KEY=$HF_TOKEN MODEL=Qwen/Qwen2.5-Coder-32B-Instruct wrencode
|
|
85
|
+
```
|
|
86
|
+
|
|
64
87
|
## Tools
|
|
65
88
|
|
|
66
89
|
The agent has access to seven tools:
|
|
67
90
|
|
|
68
91
|
- **read** - read a file with line numbers, or list a directory
|
|
69
92
|
- **write** - write content to a file
|
|
70
|
-
- **edit** - replace a unique string in a file
|
|
93
|
+
- **edit** - replace a unique string in a file. If the text only matches with
|
|
94
|
+
its indentation shifted by a consistent amount (a common slip when quoting a
|
|
95
|
+
method), the edit is applied with the replacement shifted to match; otherwise
|
|
96
|
+
the error shows the closest lines in the file
|
|
71
97
|
- **glob** - find files by pattern, sorted by modification time
|
|
72
98
|
- **grep** - search files for a regex pattern using `rg` when available, falling back to `grep`
|
|
73
99
|
- **bash** - run a shell command with timeout and streaming output
|
|
@@ -83,6 +109,77 @@ subtasks. Recursion is capped by `WRENCODE_MAX_SUBAGENT_DEPTH` (default 2), and
|
|
|
83
109
|
each subagent round is bounded. For autonomous subagent runs, enable
|
|
84
110
|
`--yes` / `WRENCODE_AUTO_APPROVE` so sub-tool calls don't block on confirmation.
|
|
85
111
|
|
|
112
|
+
## Project instructions (AGENTS.md)
|
|
113
|
+
|
|
114
|
+
WrenCode reads [`AGENTS.md`](https://agents.md) files and adds them to the
|
|
115
|
+
system prompt, so conventions you've written for other agents apply here too.
|
|
116
|
+
It looks in `~/.wrencode/`, then in every directory from the git root down to
|
|
117
|
+
the workspace (outside a git repo, only the workspace). A directory without an
|
|
118
|
+
`AGENTS.md` falls back to `CLAUDE.md`. Files closer to the workspace come later
|
|
119
|
+
and take precedence. The total is capped at 32,000 characters, and the files
|
|
120
|
+
loaded are listed at startup.
|
|
121
|
+
|
|
122
|
+
## Headless mode
|
|
123
|
+
|
|
124
|
+
`-p` / `--print` runs a single prompt without the interactive UI, for scripts,
|
|
125
|
+
CI, and evals:
|
|
126
|
+
|
|
127
|
+
```bash
|
|
128
|
+
wrencode -p "Why is test_parse failing?"
|
|
129
|
+
git diff | wrencode -p "Review this diff" # prompt from stdin
|
|
130
|
+
wrencode --yes -p "Fix the lint errors" --max-turns 20
|
|
131
|
+
wrencode -p "List the TODOs" --output-format json | jq -r .result
|
|
132
|
+
wrencode --yes -p "Make the tests pass" --verify "python3 -m unittest -q"
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
- stdout carries only the final answer (or one JSON object with
|
|
136
|
+
`--output-format json`: `result`, `is_error`, `stop_reason`, `num_turns`,
|
|
137
|
+
`backend`, `model`); progress and tool output go to stderr.
|
|
138
|
+
- Each run starts from a fresh history and doesn't touch the saved one.
|
|
139
|
+
- Without `--yes`, writes and shell commands are declined (the model is told
|
|
140
|
+
why) instead of waiting for approval. Read-only tools always work.
|
|
141
|
+
- The exit code is `0` when the agent finishes, `1` if it errors, hits
|
|
142
|
+
`--max-turns`, or stops on repeated tool errors, and `2` for bad arguments.
|
|
143
|
+
- `--verify CMD` checks the agent's claim of being done: WrenCode runs `CMD`
|
|
144
|
+
in the workspace when the agent finishes, and if it fails, sends the output
|
|
145
|
+
back and lets the agent continue (up to 3 attempts in all). The result says
|
|
146
|
+
`verified: true/false`, and a final failure exits `1` with
|
|
147
|
+
`stop_reason: "verify_failed"`. `--max-turns` applies to each attempt.
|
|
148
|
+
|
|
149
|
+
### Structured output
|
|
150
|
+
|
|
151
|
+
`--json-schema` makes the answer a JSON value that matches a schema, given as
|
|
152
|
+
a file or inline:
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
wrencode -p "Review this repo for bugs" --json-schema bugs.schema.json
|
|
156
|
+
wrencode -p "Is the build green?" --json-schema '{"type": "object", "properties": {"green": {"type": "boolean"}}, "required": ["green"]}'
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
The agent gets a `respond` tool whose arguments are your schema, and the run
|
|
160
|
+
ends when it calls `respond` with a valid answer. If the answer doesn't match,
|
|
161
|
+
the validation errors go back to the model so it can fix them; if it never
|
|
162
|
+
calls `respond`, the run fails with `stop_reason: "no_structured_output"`.
|
|
163
|
+
stdout is the JSON value (or, with `--output-format json`, the usual object
|
|
164
|
+
with a `structured_output` field). Validation is built in and covers the
|
|
165
|
+
common keywords: `type`, `enum`, `const`, `properties`, `required`,
|
|
166
|
+
`additionalProperties`, `items`, length and numeric bounds, `pattern`, and
|
|
167
|
+
`anyOf`/`oneOf`/`allOf`.
|
|
168
|
+
|
|
169
|
+
## Context management
|
|
170
|
+
|
|
171
|
+
Long sessions are compacted automatically. Before each model call WrenCode
|
|
172
|
+
estimates the prompt size (about 4 characters per token), and once it passes
|
|
173
|
+
`WRENCODE_COMPACT_AT` (default 75%) of `WRENCODE_CONTEXT_TOKENS` (default
|
|
174
|
+
128,000) it has the model summarize the older messages: the request, files
|
|
175
|
+
touched, commands and results, decisions, and what's left to do. The most recent
|
|
176
|
+
messages, about a quarter of the window, are kept verbatim, along with the
|
|
177
|
+
user's latest request, so it works mid-task, between tool calls. If a request still
|
|
178
|
+
fails with a context-length error, WrenCode compacts and retries once.
|
|
179
|
+
|
|
180
|
+
Set `WRENCODE_CONTEXT_TOKENS` to your model's window, especially for local
|
|
181
|
+
models with small ones. `/compact` summarizes on demand.
|
|
182
|
+
|
|
86
183
|
## Installation
|
|
87
184
|
|
|
88
185
|
### Option 1: Standalone binary (recommended)
|
|
@@ -170,6 +267,12 @@ For OpenRouter:
|
|
|
170
267
|
export OPENROUTER_API_KEY=your_key
|
|
171
268
|
```
|
|
172
269
|
|
|
270
|
+
For NanoGPT:
|
|
271
|
+
|
|
272
|
+
```bash
|
|
273
|
+
export NANOGPT_API_KEY=your_key
|
|
274
|
+
```
|
|
275
|
+
|
|
173
276
|
For HuggingFace Transformers:
|
|
174
277
|
|
|
175
278
|
```bash
|
|
@@ -197,6 +300,9 @@ BACKEND=openai MODEL=gpt-4o python3 wrencode.py
|
|
|
197
300
|
# OpenRouter
|
|
198
301
|
BACKEND=openrouter MODEL=anthropic/claude-3-haiku python3 wrencode.py
|
|
199
302
|
|
|
303
|
+
# NanoGPT
|
|
304
|
+
BACKEND=nanogpt MODEL=z-ai/glm-5.3-flash-uncensored python3 wrencode.py
|
|
305
|
+
|
|
200
306
|
# Ollama (needs `ollama serve` running and the model pulled)
|
|
201
307
|
BACKEND=ollama MODEL=llama3.2 python3 wrencode.py
|
|
202
308
|
|
|
@@ -207,15 +313,22 @@ BACKEND=transformers MODEL=deburky/gpt-oss-claude-code python3 wrencode.py
|
|
|
207
313
|
BACKEND=local LOCAL_PORT=8082 python3 wrencode.py
|
|
208
314
|
```
|
|
209
315
|
|
|
210
|
-
## Releasing
|
|
316
|
+
## Releasing
|
|
211
317
|
|
|
212
|
-
|
|
318
|
+
Versions and [`CHANGELOG.md`](CHANGELOG.md) are managed with
|
|
319
|
+
[commitizen](https://commitizen-tools.github.io/commitizen/), so write commit
|
|
320
|
+
messages as [conventional commits](https://www.conventionalcommits.org/)
|
|
321
|
+
(`feat: ...`, `fix(edit): ...`, `refactor: ...`). To cut a release:
|
|
213
322
|
|
|
214
323
|
```bash
|
|
215
|
-
|
|
216
|
-
git push origin
|
|
324
|
+
uvx --from commitizen cz bump # bumps WRENCODE_VERSION, updates CHANGELOG.md, tags
|
|
325
|
+
git push origin main --tags
|
|
217
326
|
```
|
|
218
327
|
|
|
328
|
+
Preview the next changelog entry with `uvx --from commitizen cz changelog --dry-run`.
|
|
329
|
+
|
|
330
|
+
Binaries are built automatically by GitHub Actions when a version tag is pushed.
|
|
331
|
+
|
|
219
332
|
This publishes release assets:
|
|
220
333
|
- `wrencode-linux-x64`
|
|
221
334
|
- `wrencode-macos-x64`
|
|
@@ -227,7 +340,7 @@ This publishes release assets:
|
|
|
227
340
|
|--------------|----------------------------------------------|
|
|
228
341
|
|`/help` |Show available commands |
|
|
229
342
|
|`/c` |Clear conversation history |
|
|
230
|
-
|`/compact` |Summarize history to reduce context
|
|
343
|
+
|`/compact` |Summarize history to reduce context |
|
|
231
344
|
|`/q` or `exit`|Quit |
|
|
232
345
|
|
|
233
346
|
## Environment Variables
|
|
@@ -242,7 +355,11 @@ This publishes release assets:
|
|
|
242
355
|
|`WRENCODE_UNRESTRICTED_PATHS`|`0` |Allow paths outside workspace |
|
|
243
356
|
|`WRENCODE_AUTO_APPROVE` |`0` |Skip y/N confirmation for writes/commands (headless; also `--yes`)|
|
|
244
357
|
|`WRENCODE_MAX_SUBAGENT_DEPTH`|`2` |Max nested subagent recursion depth (`task` tool)|
|
|
245
|
-
|`MAX_TOKENS` |`
|
|
358
|
+
|`MAX_TOKENS` |`8192` |Max tokens per response |
|
|
359
|
+
|`WRENCODE_HTTP_TIMEOUT` |`600` |Seconds to wait for a model response|
|
|
360
|
+
|`WRENCODE_HTTP_RETRIES` |`2` |Retries on HTTP 429/5xx, with backoff|
|
|
361
|
+
|`WRENCODE_CONTEXT_TOKENS` |`128000` |Model context window, for auto-compaction|
|
|
362
|
+
|`WRENCODE_COMPACT_AT` |`0.75` |Compact at this fraction of the window (`0` disables)|
|
|
246
363
|
|`MAX_READ_BYTES` |`4MB` |Max file size to read |
|
|
247
364
|
|`MAX_READ_LINES` |`800` |Max lines returned per read |
|
|
248
365
|
|`GREP_MAX_MATCHES` |`80` |Max grep results |
|
|
@@ -250,11 +367,14 @@ This publishes release assets:
|
|
|
250
367
|
|`MAX_TOOL_OUTPUT_CHARS` |`48000` |Max tool output before truncation |
|
|
251
368
|
|`GLOB_SKIP_DIRS` |`.git,node_modules,...`|Directories to skip in glob |
|
|
252
369
|
|`OPENROUTER_API_KEY` |- |OpenRouter API key |
|
|
370
|
+
|`NANOGPT_API_KEY` |- |NanoGPT API key |
|
|
253
371
|
|`OPENAI_API_KEY` |- |OpenAI API key |
|
|
254
372
|
|`ANTHROPIC_API_KEY` |- |Anthropic API key |
|
|
255
373
|
|`LOCAL_API_KEY` |`local` |Local proxy API key |
|
|
256
374
|
|`LOCAL_PORT` |`8082` |Local proxy port |
|
|
257
375
|
|`OLLAMA_HOST` |`http://localhost:11434`|Ollama server base URL |
|
|
376
|
+
|`OPENAI_COMPATIBLE_BASE_URL` |`http://localhost:8000/v1`|OpenAI-compatible server base URL|
|
|
377
|
+
|`OPENAI_COMPATIBLE_API_KEY` |- |Key for that server, if it needs one|
|
|
258
378
|
|
|
259
379
|
## History
|
|
260
380
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# 🐦 WrenCode
|
|
2
2
|
|
|
3
|
-
A minimal
|
|
3
|
+
A minimal agent harness for coding, in a single Python file.
|
|
4
4
|
|
|
5
5
|
Named after Harold Wren - the alias of a genius who built a superintelligent AI and operated quietly in the background.
|
|
6
6
|
|
|
@@ -8,9 +8,9 @@ Named after Harold Wren - the alias of a genius who built a superintelligent AI
|
|
|
8
8
|
|
|
9
9
|
## What it is
|
|
10
10
|
|
|
11
|
-
WrenCode is a
|
|
11
|
+
WrenCode is a coding agent harness: everything around the model that turns it into an agent. It runs the tool-calling loop, executes tools, builds the system prompt, and manages context, locally or via API, giving an LLM the ability to read, write, and edit files, search codebases, and run shell commands - enough to autonomously navigate and modify a real project.
|
|
12
12
|
|
|
13
|
-
Where Claude Code is the batteries-included
|
|
13
|
+
Where Claude Code is the batteries-included harness, WrenCode is the **"understand and own your agent" harness**: the entire agent loop fits in one readable file, runs against local or hosted models, and is yours to hack.
|
|
14
14
|
|
|
15
15
|
## Backends
|
|
16
16
|
|
|
@@ -24,7 +24,9 @@ saved choice, e.g. for CI.
|
|
|
24
24
|
|`anthropic` |Claude via Anthropic API |binary + source |
|
|
25
25
|
|`openai` |GPT models via OpenAI API |binary + source |
|
|
26
26
|
|`openrouter` |Any model via OpenRouter |binary + source |
|
|
27
|
+
|`nanogpt` |Any model via NanoGPT |binary + source |
|
|
27
28
|
|`ollama` |Local models via a running `ollama serve`|binary + source |
|
|
29
|
+
|`openai-compatible`|vLLM, llama.cpp, Hugging Face, any OpenAI-compatible server|binary + source|
|
|
28
30
|
|`local` |Local proxy via Anthropic-compatible API|binary + source |
|
|
29
31
|
|`transformers`|HuggingFace Transformers (CPU/MPS/GPU) |source install only |
|
|
30
32
|
|`mlx` |Apple Silicon via MLX |source install, macOS |
|
|
@@ -38,13 +40,37 @@ The default local models are
|
|
|
38
40
|
[`deburky/gpt-oss-claude-mlx`](https://huggingface.co/deburky/gpt-oss-claude-mlx)
|
|
39
41
|
(MLX) — override either with `MODEL=...`.
|
|
40
42
|
|
|
43
|
+
### OpenAI-compatible servers
|
|
44
|
+
|
|
45
|
+
`openai-compatible` talks to any server that implements OpenAI chat completions,
|
|
46
|
+
using native tool calls. Point it at the server with `OPENAI_COMPATIBLE_BASE_URL`
|
|
47
|
+
(default `http://localhost:8000/v1`). If the server serves exactly one model,
|
|
48
|
+
WrenCode uses it; otherwise set `MODEL`.
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
# vLLM (tool calling needs these flags; pick the parser for your model)
|
|
52
|
+
vllm serve Qwen/Qwen2.5-Coder-7B-Instruct --enable-auto-tool-choice --tool-call-parser hermes
|
|
53
|
+
BACKEND=openai-compatible wrencode
|
|
54
|
+
|
|
55
|
+
# llama.cpp (--jinja enables tool calling)
|
|
56
|
+
llama-server -m qwen2.5-coder-7b-instruct-q4_k_m.gguf --jinja --port 8080
|
|
57
|
+
BACKEND=openai-compatible OPENAI_COMPATIBLE_BASE_URL=http://localhost:8080/v1 wrencode
|
|
58
|
+
|
|
59
|
+
# Hugging Face Inference Providers
|
|
60
|
+
BACKEND=openai-compatible OPENAI_COMPATIBLE_BASE_URL=https://router.huggingface.co/v1 \
|
|
61
|
+
OPENAI_COMPATIBLE_API_KEY=$HF_TOKEN MODEL=Qwen/Qwen2.5-Coder-32B-Instruct wrencode
|
|
62
|
+
```
|
|
63
|
+
|
|
41
64
|
## Tools
|
|
42
65
|
|
|
43
66
|
The agent has access to seven tools:
|
|
44
67
|
|
|
45
68
|
- **read** - read a file with line numbers, or list a directory
|
|
46
69
|
- **write** - write content to a file
|
|
47
|
-
- **edit** - replace a unique string in a file
|
|
70
|
+
- **edit** - replace a unique string in a file. If the text only matches with
|
|
71
|
+
its indentation shifted by a consistent amount (a common slip when quoting a
|
|
72
|
+
method), the edit is applied with the replacement shifted to match; otherwise
|
|
73
|
+
the error shows the closest lines in the file
|
|
48
74
|
- **glob** - find files by pattern, sorted by modification time
|
|
49
75
|
- **grep** - search files for a regex pattern using `rg` when available, falling back to `grep`
|
|
50
76
|
- **bash** - run a shell command with timeout and streaming output
|
|
@@ -60,6 +86,77 @@ subtasks. Recursion is capped by `WRENCODE_MAX_SUBAGENT_DEPTH` (default 2), and
|
|
|
60
86
|
each subagent round is bounded. For autonomous subagent runs, enable
|
|
61
87
|
`--yes` / `WRENCODE_AUTO_APPROVE` so sub-tool calls don't block on confirmation.
|
|
62
88
|
|
|
89
|
+
## Project instructions (AGENTS.md)
|
|
90
|
+
|
|
91
|
+
WrenCode reads [`AGENTS.md`](https://agents.md) files and adds them to the
|
|
92
|
+
system prompt, so conventions you've written for other agents apply here too.
|
|
93
|
+
It looks in `~/.wrencode/`, then in every directory from the git root down to
|
|
94
|
+
the workspace (outside a git repo, only the workspace). A directory without an
|
|
95
|
+
`AGENTS.md` falls back to `CLAUDE.md`. Files closer to the workspace come later
|
|
96
|
+
and take precedence. The total is capped at 32,000 characters, and the files
|
|
97
|
+
loaded are listed at startup.
|
|
98
|
+
|
|
99
|
+
## Headless mode
|
|
100
|
+
|
|
101
|
+
`-p` / `--print` runs a single prompt without the interactive UI, for scripts,
|
|
102
|
+
CI, and evals:
|
|
103
|
+
|
|
104
|
+
```bash
|
|
105
|
+
wrencode -p "Why is test_parse failing?"
|
|
106
|
+
git diff | wrencode -p "Review this diff" # prompt from stdin
|
|
107
|
+
wrencode --yes -p "Fix the lint errors" --max-turns 20
|
|
108
|
+
wrencode -p "List the TODOs" --output-format json | jq -r .result
|
|
109
|
+
wrencode --yes -p "Make the tests pass" --verify "python3 -m unittest -q"
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
- stdout carries only the final answer (or one JSON object with
|
|
113
|
+
`--output-format json`: `result`, `is_error`, `stop_reason`, `num_turns`,
|
|
114
|
+
`backend`, `model`); progress and tool output go to stderr.
|
|
115
|
+
- Each run starts from a fresh history and doesn't touch the saved one.
|
|
116
|
+
- Without `--yes`, writes and shell commands are declined (the model is told
|
|
117
|
+
why) instead of waiting for approval. Read-only tools always work.
|
|
118
|
+
- The exit code is `0` when the agent finishes, `1` if it errors, hits
|
|
119
|
+
`--max-turns`, or stops on repeated tool errors, and `2` for bad arguments.
|
|
120
|
+
- `--verify CMD` checks the agent's claim of being done: WrenCode runs `CMD`
|
|
121
|
+
in the workspace when the agent finishes, and if it fails, sends the output
|
|
122
|
+
back and lets the agent continue (up to 3 attempts in all). The result says
|
|
123
|
+
`verified: true/false`, and a final failure exits `1` with
|
|
124
|
+
`stop_reason: "verify_failed"`. `--max-turns` applies to each attempt.
|
|
125
|
+
|
|
126
|
+
### Structured output
|
|
127
|
+
|
|
128
|
+
`--json-schema` makes the answer a JSON value that matches a schema, given as
|
|
129
|
+
a file or inline:
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
wrencode -p "Review this repo for bugs" --json-schema bugs.schema.json
|
|
133
|
+
wrencode -p "Is the build green?" --json-schema '{"type": "object", "properties": {"green": {"type": "boolean"}}, "required": ["green"]}'
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The agent gets a `respond` tool whose arguments are your schema, and the run
|
|
137
|
+
ends when it calls `respond` with a valid answer. If the answer doesn't match,
|
|
138
|
+
the validation errors go back to the model so it can fix them; if it never
|
|
139
|
+
calls `respond`, the run fails with `stop_reason: "no_structured_output"`.
|
|
140
|
+
stdout is the JSON value (or, with `--output-format json`, the usual object
|
|
141
|
+
with a `structured_output` field). Validation is built in and covers the
|
|
142
|
+
common keywords: `type`, `enum`, `const`, `properties`, `required`,
|
|
143
|
+
`additionalProperties`, `items`, length and numeric bounds, `pattern`, and
|
|
144
|
+
`anyOf`/`oneOf`/`allOf`.
|
|
145
|
+
|
|
146
|
+
## Context management
|
|
147
|
+
|
|
148
|
+
Long sessions are compacted automatically. Before each model call WrenCode
|
|
149
|
+
estimates the prompt size (about 4 characters per token), and once it passes
|
|
150
|
+
`WRENCODE_COMPACT_AT` (default 75%) of `WRENCODE_CONTEXT_TOKENS` (default
|
|
151
|
+
128,000) it has the model summarize the older messages: the request, files
|
|
152
|
+
touched, commands and results, decisions, and what's left to do. The most recent
|
|
153
|
+
messages, about a quarter of the window, are kept verbatim, along with the
|
|
154
|
+
user's latest request, so it works mid-task, between tool calls. If a request still
|
|
155
|
+
fails with a context-length error, WrenCode compacts and retries once.
|
|
156
|
+
|
|
157
|
+
Set `WRENCODE_CONTEXT_TOKENS` to your model's window, especially for local
|
|
158
|
+
models with small ones. `/compact` summarizes on demand.
|
|
159
|
+
|
|
63
160
|
## Installation
|
|
64
161
|
|
|
65
162
|
### Option 1: Standalone binary (recommended)
|
|
@@ -147,6 +244,12 @@ For OpenRouter:
|
|
|
147
244
|
export OPENROUTER_API_KEY=your_key
|
|
148
245
|
```
|
|
149
246
|
|
|
247
|
+
For NanoGPT:
|
|
248
|
+
|
|
249
|
+
```bash
|
|
250
|
+
export NANOGPT_API_KEY=your_key
|
|
251
|
+
```
|
|
252
|
+
|
|
150
253
|
For HuggingFace Transformers:
|
|
151
254
|
|
|
152
255
|
```bash
|
|
@@ -174,6 +277,9 @@ BACKEND=openai MODEL=gpt-4o python3 wrencode.py
|
|
|
174
277
|
# OpenRouter
|
|
175
278
|
BACKEND=openrouter MODEL=anthropic/claude-3-haiku python3 wrencode.py
|
|
176
279
|
|
|
280
|
+
# NanoGPT
|
|
281
|
+
BACKEND=nanogpt MODEL=z-ai/glm-5.3-flash-uncensored python3 wrencode.py
|
|
282
|
+
|
|
177
283
|
# Ollama (needs `ollama serve` running and the model pulled)
|
|
178
284
|
BACKEND=ollama MODEL=llama3.2 python3 wrencode.py
|
|
179
285
|
|
|
@@ -184,15 +290,22 @@ BACKEND=transformers MODEL=deburky/gpt-oss-claude-code python3 wrencode.py
|
|
|
184
290
|
BACKEND=local LOCAL_PORT=8082 python3 wrencode.py
|
|
185
291
|
```
|
|
186
292
|
|
|
187
|
-
## Releasing
|
|
293
|
+
## Releasing
|
|
188
294
|
|
|
189
|
-
|
|
295
|
+
Versions and [`CHANGELOG.md`](CHANGELOG.md) are managed with
|
|
296
|
+
[commitizen](https://commitizen-tools.github.io/commitizen/), so write commit
|
|
297
|
+
messages as [conventional commits](https://www.conventionalcommits.org/)
|
|
298
|
+
(`feat: ...`, `fix(edit): ...`, `refactor: ...`). To cut a release:
|
|
190
299
|
|
|
191
300
|
```bash
|
|
192
|
-
|
|
193
|
-
git push origin
|
|
301
|
+
uvx --from commitizen cz bump # bumps WRENCODE_VERSION, updates CHANGELOG.md, tags
|
|
302
|
+
git push origin main --tags
|
|
194
303
|
```
|
|
195
304
|
|
|
305
|
+
Preview the next changelog entry with `uvx --from commitizen cz changelog --dry-run`.
|
|
306
|
+
|
|
307
|
+
Binaries are built automatically by GitHub Actions when a version tag is pushed.
|
|
308
|
+
|
|
196
309
|
This publishes release assets:
|
|
197
310
|
- `wrencode-linux-x64`
|
|
198
311
|
- `wrencode-macos-x64`
|
|
@@ -204,7 +317,7 @@ This publishes release assets:
|
|
|
204
317
|
|--------------|----------------------------------------------|
|
|
205
318
|
|`/help` |Show available commands |
|
|
206
319
|
|`/c` |Clear conversation history |
|
|
207
|
-
|`/compact` |Summarize history to reduce context
|
|
320
|
+
|`/compact` |Summarize history to reduce context |
|
|
208
321
|
|`/q` or `exit`|Quit |
|
|
209
322
|
|
|
210
323
|
## Environment Variables
|
|
@@ -219,7 +332,11 @@ This publishes release assets:
|
|
|
219
332
|
|`WRENCODE_UNRESTRICTED_PATHS`|`0` |Allow paths outside workspace |
|
|
220
333
|
|`WRENCODE_AUTO_APPROVE` |`0` |Skip y/N confirmation for writes/commands (headless; also `--yes`)|
|
|
221
334
|
|`WRENCODE_MAX_SUBAGENT_DEPTH`|`2` |Max nested subagent recursion depth (`task` tool)|
|
|
222
|
-
|`MAX_TOKENS` |`
|
|
335
|
+
|`MAX_TOKENS` |`8192` |Max tokens per response |
|
|
336
|
+
|`WRENCODE_HTTP_TIMEOUT` |`600` |Seconds to wait for a model response|
|
|
337
|
+
|`WRENCODE_HTTP_RETRIES` |`2` |Retries on HTTP 429/5xx, with backoff|
|
|
338
|
+
|`WRENCODE_CONTEXT_TOKENS` |`128000` |Model context window, for auto-compaction|
|
|
339
|
+
|`WRENCODE_COMPACT_AT` |`0.75` |Compact at this fraction of the window (`0` disables)|
|
|
223
340
|
|`MAX_READ_BYTES` |`4MB` |Max file size to read |
|
|
224
341
|
|`MAX_READ_LINES` |`800` |Max lines returned per read |
|
|
225
342
|
|`GREP_MAX_MATCHES` |`80` |Max grep results |
|
|
@@ -227,11 +344,14 @@ This publishes release assets:
|
|
|
227
344
|
|`MAX_TOOL_OUTPUT_CHARS` |`48000` |Max tool output before truncation |
|
|
228
345
|
|`GLOB_SKIP_DIRS` |`.git,node_modules,...`|Directories to skip in glob |
|
|
229
346
|
|`OPENROUTER_API_KEY` |- |OpenRouter API key |
|
|
347
|
+
|`NANOGPT_API_KEY` |- |NanoGPT API key |
|
|
230
348
|
|`OPENAI_API_KEY` |- |OpenAI API key |
|
|
231
349
|
|`ANTHROPIC_API_KEY` |- |Anthropic API key |
|
|
232
350
|
|`LOCAL_API_KEY` |`local` |Local proxy API key |
|
|
233
351
|
|`LOCAL_PORT` |`8082` |Local proxy port |
|
|
234
352
|
|`OLLAMA_HOST` |`http://localhost:11434`|Ollama server base URL |
|
|
353
|
+
|`OPENAI_COMPATIBLE_BASE_URL` |`http://localhost:8000/v1`|OpenAI-compatible server base URL|
|
|
354
|
+
|`OPENAI_COMPATIBLE_API_KEY` |- |Key for that server, if it needs one|
|
|
235
355
|
|
|
236
356
|
## History
|
|
237
357
|
|
|
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "wrencode"
|
|
7
7
|
dynamic = ["version"]
|
|
8
|
-
description = "A minimal
|
|
8
|
+
description = "A minimal agent harness for coding, in a single Python file"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
11
11
|
license = "MIT"
|
|
@@ -52,3 +52,15 @@ include = ["wrencode.py"]
|
|
|
52
52
|
|
|
53
53
|
[tool.hatch.build.targets.sdist]
|
|
54
54
|
include = ["wrencode.py", "README.md"]
|
|
55
|
+
|
|
56
|
+
[tool.commitizen]
|
|
57
|
+
name = "cz_conventional_commits"
|
|
58
|
+
version = "0.2.0"
|
|
59
|
+
version_scheme = "pep440"
|
|
60
|
+
tag_format = "$version"
|
|
61
|
+
legacy_tag_formats = ["v$version"]
|
|
62
|
+
version_files = ["wrencode.py:WRENCODE_VERSION"]
|
|
63
|
+
update_changelog_on_bump = true
|
|
64
|
+
changelog_incremental = true
|
|
65
|
+
# Pre-commitizen four-part tags (0.1.4.x); their history is in CHANGELOG.md.
|
|
66
|
+
ignored_tag_formats = ["$major.$minor.$patch.*"]
|