corecoder 0.4.0__tar.gz → 0.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {corecoder-0.4.0 → corecoder-0.4.2}/PKG-INFO +36 -23
- {corecoder-0.4.0 → corecoder-0.4.2}/README.md +34 -21
- {corecoder-0.4.0 → corecoder-0.4.2}/README_CN.md +34 -21
- {corecoder-0.4.0 → corecoder-0.4.2}/article/00-index.md +1 -1
- {corecoder-0.4.0 → corecoder-0.4.2}/article/00-index_EN.md +1 -1
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/__init__.py +3 -3
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/agent.py +17 -5
- corecoder-0.4.2/corecoder/checkpoints.py +38 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/cli.py +27 -11
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/context.py +9 -9
- corecoder-0.4.2/corecoder/demo.py +84 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/llm.py +26 -1
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/__init__.py +5 -3
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/agent.py +5 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/bash.py +6 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/edit.py +6 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/glob_tool.py +7 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/grep.py +19 -5
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/read.py +5 -2
- corecoder-0.4.2/corecoder/tools/todo.py +79 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/write.py +7 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/pyproject.toml +1 -1
- corecoder-0.4.2/tests/test_checkpoints.py +54 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_core.py +84 -6
- corecoder-0.4.2/tests/test_demo.py +39 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_litellm.py +1 -2
- {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_session.py +1 -1
- {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_tools.py +106 -3
- {corecoder-0.4.0 → corecoder-0.4.2}/.github/workflows/ci.yml +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/.github/workflows/publish.yml +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/.gitignore +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/LICENSE +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/01-the-loop.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/01-the-loop_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/02-tools.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/02-tools_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/03-llm-and-cost.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/03-llm-and-cost_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/04-context.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/04-context_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/05-parallel-and-subagents.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/05-parallel-and-subagents_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/06-session-and-cli.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/06-session-and-cli_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/07-build-your-own.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/article/07-build-your-own_EN.md +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/assets/demo.png +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/assets/demo_en.png +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/__main__.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/config.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/prompt.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/session.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/base.py +0 -0
- {corecoder-0.4.0 → corecoder-0.4.2}/tests/__init__.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: corecoder
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.2
|
|
4
4
|
Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
|
|
5
5
|
Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
|
|
6
6
|
Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
|
|
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
|
|
|
37
37
|
|
|
38
38
|
# CoreCoder
|
|
39
39
|
|
|
40
|
-
**The nanoGPT of coding agents. 1,
|
|
40
|
+
**The nanoGPT of coding agents. 1,201 lines of pure Python — understand how a coding agent actually works, then fork your own.**
|
|
41
41
|
|
|
42
42
|
*learn from it · fork it · ship something better*
|
|
43
43
|
|
|
@@ -47,22 +47,33 @@ Description-Content-Type: text/markdown
|
|
|
47
47
|
[](https://python.org)
|
|
48
48
|
[](LICENSE)
|
|
49
49
|
[](https://github.com/he-yufeng/CoreCoder/actions)
|
|
50
|
-
[](article/00-index_EN.md)
|
|
51
51
|
[](article/00-index_EN.md)
|
|
52
52
|
|
|
53
53
|
</div>
|
|
54
54
|
|
|
55
|
-
- **Readable end to end.** Read the whole engine in an afternoon
|
|
56
|
-
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
55
|
+
- **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
|
|
56
|
+
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
57
57
|
- **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
|
|
58
58
|
|
|
59
|
+
## How it compares
|
|
60
|
+
|
|
61
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
62
|
+
|---|---|---|---|---|
|
|
63
|
+
| Lines of code | ~1,201 engine / 2,008 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
64
|
+
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
65
|
+
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
66
|
+
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
67
|
+
|
|
68
|
+
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
69
|
+
|
|
59
70
|
## What this is
|
|
60
71
|
|
|
61
72
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
62
73
|
|
|
63
|
-
The engine (loop, model interface, context, tools, sessions) is 1,
|
|
74
|
+
The engine (loop, model interface, context, tools, sessions) is 1,201 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 21 files: 2,008 physical lines, 1,614 net, every one short enough to read in a single sitting.
|
|
64
75
|
|
|
65
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask.
|
|
76
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
66
77
|
|
|
67
78
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
68
79
|
|
|
@@ -93,6 +104,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
93
104
|
|---|---|
|
|
94
105
|
| OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
95
106
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
107
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
96
108
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
97
109
|
|
|
98
110
|
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
@@ -108,7 +120,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
108
120
|
|
|
109
121
|
```
|
|
110
122
|
corecoder/
|
|
111
|
-
├── agent.py agent loop + parallel tool exec
|
|
123
|
+
├── agent.py agent loop + parallel tool exec 162 lines ← start here
|
|
112
124
|
├── llm.py streaming client + retry + cost 336 lines
|
|
113
125
|
├── context.py three-tier context compaction 210 lines
|
|
114
126
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
@@ -122,11 +134,12 @@ corecoder/
|
|
|
122
134
|
├── glob_tool.py filename matching 47 lines
|
|
123
135
|
├── read.py file read 53 lines
|
|
124
136
|
├── write.py file write 38 lines
|
|
137
|
+
├── todo.py agent-maintained task checklist 79 lines
|
|
125
138
|
├── agent.py sub-agent spawning 58 lines
|
|
126
139
|
└── base.py tool base class 27 lines
|
|
127
140
|
```
|
|
128
141
|
|
|
129
|
-
|
|
142
|
+
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
|
|
130
143
|
|
|
131
144
|
## A `while` loop is the whole agent
|
|
132
145
|
|
|
@@ -147,7 +160,7 @@ def chat(self, user_input):
|
|
|
147
160
|
return "(hit the round limit)"
|
|
148
161
|
```
|
|
149
162
|
|
|
150
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess
|
|
163
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
151
164
|
|
|
152
165
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
153
166
|
|
|
@@ -197,17 +210,6 @@ Going deeper, the directions are out in the open too. None of the following is i
|
|
|
197
210
|
|
|
198
211
|
The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
|
|
199
212
|
|
|
200
|
-
## How it compares
|
|
201
|
-
|
|
202
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
203
|
-
|---|---|---|---|---|
|
|
204
|
-
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
205
|
-
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
206
|
-
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
207
|
-
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
208
|
-
|
|
209
|
-
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
210
|
-
|
|
211
213
|
## Commands
|
|
212
214
|
|
|
213
215
|
Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
@@ -217,15 +219,26 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
|
217
219
|
/compact compact the context by hand
|
|
218
220
|
/tokens token usage and cost estimate
|
|
219
221
|
/diff files changed this session
|
|
222
|
+
/undo revert the most recent file change
|
|
220
223
|
/save /sessions save / list sessions
|
|
221
224
|
quit / exit exit (Ctrl+C cancels the current round)
|
|
222
225
|
```
|
|
223
226
|
|
|
224
227
|
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
225
228
|
|
|
229
|
+
## Related Projects
|
|
230
|
+
|
|
231
|
+
If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
|
|
232
|
+
|
|
233
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
|
|
234
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
|
|
235
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
|
|
236
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
|
|
237
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
|
|
238
|
+
|
|
226
239
|
## Contributing / License
|
|
227
240
|
|
|
228
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
241
|
+
Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
229
242
|
|
|
230
243
|
---
|
|
231
244
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
# CoreCoder
|
|
4
4
|
|
|
5
|
-
**The nanoGPT of coding agents. 1,
|
|
5
|
+
**The nanoGPT of coding agents. 1,201 lines of pure Python — understand how a coding agent actually works, then fork your own.**
|
|
6
6
|
|
|
7
7
|
*learn from it · fork it · ship something better*
|
|
8
8
|
|
|
@@ -12,22 +12,33 @@
|
|
|
12
12
|
[](https://python.org)
|
|
13
13
|
[](LICENSE)
|
|
14
14
|
[](https://github.com/he-yufeng/CoreCoder/actions)
|
|
15
|
-
[](article/00-index_EN.md)
|
|
16
16
|
[](article/00-index_EN.md)
|
|
17
17
|
|
|
18
18
|
</div>
|
|
19
19
|
|
|
20
|
-
- **Readable end to end.** Read the whole engine in an afternoon
|
|
21
|
-
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
20
|
+
- **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
|
|
21
|
+
- **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
|
|
22
22
|
- **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
|
|
23
23
|
|
|
24
|
+
## How it compares
|
|
25
|
+
|
|
26
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
27
|
+
|---|---|---|---|---|
|
|
28
|
+
| Lines of code | ~1,201 engine / 2,008 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
29
|
+
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
30
|
+
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
31
|
+
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
32
|
+
|
|
33
|
+
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
34
|
+
|
|
24
35
|
## What this is
|
|
25
36
|
|
|
26
37
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
27
38
|
|
|
28
|
-
The engine (loop, model interface, context, tools, sessions) is 1,
|
|
39
|
+
The engine (loop, model interface, context, tools, sessions) is 1,201 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 21 files: 2,008 physical lines, 1,614 net, every one short enough to read in a single sitting.
|
|
29
40
|
|
|
30
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask.
|
|
41
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
31
42
|
|
|
32
43
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
33
44
|
|
|
@@ -58,6 +69,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
58
69
|
|---|---|
|
|
59
70
|
| OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
60
71
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
72
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
61
73
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
62
74
|
|
|
63
75
|
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
@@ -73,7 +85,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
73
85
|
|
|
74
86
|
```
|
|
75
87
|
corecoder/
|
|
76
|
-
├── agent.py agent loop + parallel tool exec
|
|
88
|
+
├── agent.py agent loop + parallel tool exec 162 lines ← start here
|
|
77
89
|
├── llm.py streaming client + retry + cost 336 lines
|
|
78
90
|
├── context.py three-tier context compaction 210 lines
|
|
79
91
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
@@ -87,11 +99,12 @@ corecoder/
|
|
|
87
99
|
├── glob_tool.py filename matching 47 lines
|
|
88
100
|
├── read.py file read 53 lines
|
|
89
101
|
├── write.py file write 38 lines
|
|
102
|
+
├── todo.py agent-maintained task checklist 79 lines
|
|
90
103
|
├── agent.py sub-agent spawning 58 lines
|
|
91
104
|
└── base.py tool base class 27 lines
|
|
92
105
|
```
|
|
93
106
|
|
|
94
|
-
|
|
107
|
+
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
|
|
95
108
|
|
|
96
109
|
## A `while` loop is the whole agent
|
|
97
110
|
|
|
@@ -112,7 +125,7 @@ def chat(self, user_input):
|
|
|
112
125
|
return "(hit the round limit)"
|
|
113
126
|
```
|
|
114
127
|
|
|
115
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess
|
|
128
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
116
129
|
|
|
117
130
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
118
131
|
|
|
@@ -162,17 +175,6 @@ Going deeper, the directions are out in the open too. None of the following is i
|
|
|
162
175
|
|
|
163
176
|
The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
|
|
164
177
|
|
|
165
|
-
## How it compares
|
|
166
|
-
|
|
167
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
168
|
-
|---|---|---|---|---|
|
|
169
|
-
| Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
170
|
-
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
171
|
-
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
172
|
-
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
173
|
-
|
|
174
|
-
The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
|
|
175
|
-
|
|
176
178
|
## Commands
|
|
177
179
|
|
|
178
180
|
Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
@@ -182,15 +184,26 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
|
182
184
|
/compact compact the context by hand
|
|
183
185
|
/tokens token usage and cost estimate
|
|
184
186
|
/diff files changed this session
|
|
187
|
+
/undo revert the most recent file change
|
|
185
188
|
/save /sessions save / list sessions
|
|
186
189
|
quit / exit exit (Ctrl+C cancels the current round)
|
|
187
190
|
```
|
|
188
191
|
|
|
189
192
|
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
190
193
|
|
|
194
|
+
## Related Projects
|
|
195
|
+
|
|
196
|
+
If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
|
|
197
|
+
|
|
198
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
|
|
199
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
|
|
200
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
|
|
201
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
|
|
202
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
|
|
203
|
+
|
|
191
204
|
## Contributing / License
|
|
192
205
|
|
|
193
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
206
|
+
Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
194
207
|
|
|
195
208
|
---
|
|
196
209
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
# CoreCoder
|
|
4
4
|
|
|
5
|
-
**编程 agent 里的 nanoGPT。
|
|
5
|
+
**编程 agent 里的 nanoGPT。1201 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
|
|
6
6
|
|
|
7
7
|
*learn from it · fork it · ship something better*
|
|
8
8
|
|
|
@@ -12,22 +12,33 @@
|
|
|
12
12
|
[](https://python.org)
|
|
13
13
|
[](LICENSE)
|
|
14
14
|
[](https://github.com/he-yufeng/CoreCoder/actions)
|
|
15
|
-
[](article/)
|
|
16
16
|
[](article/)
|
|
17
17
|
|
|
18
18
|
</div>
|
|
19
19
|
|
|
20
|
-
- **读得完。**
|
|
21
|
-
- **改得动。**
|
|
20
|
+
- **读得完。** 一个下午读完整个引擎,没有一处藏着你看不懂的魔法。
|
|
21
|
+
- **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图。
|
|
22
22
|
- **留白即起点。** 刻意只留最小核心,没做的那些不是半成品,是留给你 fork 出更好东西的地方。
|
|
23
23
|
|
|
24
|
+
## 和谁比
|
|
25
|
+
|
|
26
|
+
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
27
|
+
|---|---|---|---|---|
|
|
28
|
+
| 代码量 | 引擎约 1201 行 / 整包 2008 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
|
|
29
|
+
| 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
|
|
30
|
+
| 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
|
|
31
|
+
| 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
|
|
32
|
+
|
|
33
|
+
nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
|
|
34
|
+
|
|
24
35
|
## 这是什么
|
|
25
36
|
|
|
26
37
|
我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
|
|
27
38
|
|
|
28
|
-
引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是
|
|
39
|
+
引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1201 行。连最外层的 CLI、配置、打包一起算,整个包 21 个文件、物理 2008 行、净 1614 行,每个文件都短到能一口气读完。
|
|
29
40
|
|
|
30
|
-
它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,
|
|
41
|
+
它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,103 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
|
|
31
42
|
|
|
32
43
|
代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
|
|
33
44
|
|
|
@@ -58,6 +69,7 @@ pip install -e .
|
|
|
58
69
|
|---|---|
|
|
59
70
|
| OpenAI(默认 `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
|
|
60
71
|
| DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
|
|
72
|
+
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
61
73
|
| 本地 Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
62
74
|
|
|
63
75
|
Kimi、Qwen 这些同样是改这两个变量;连 OpenAI 兼容接口都不给的 provider,装上可选的 LiteLLM 后端(`pip install "corecoder[litellm]"`)能路由一百多家。第三篇文章把这块讲得更细。key 可以直接 `export`,也可以在项目根目录扔个 `.env`,启动时自动加载。然后:
|
|
@@ -73,7 +85,7 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
|
|
|
73
85
|
|
|
74
86
|
```
|
|
75
87
|
corecoder/
|
|
76
|
-
├── agent.py agent 主循环 + 并行工具执行
|
|
88
|
+
├── agent.py agent 主循环 + 并行工具执行 162 行 ← 从这里开始读
|
|
77
89
|
├── llm.py 流式客户端 + 重试 + 成本统计 336 行
|
|
78
90
|
├── context.py 三层上下文压缩 210 行
|
|
79
91
|
├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
|
|
@@ -87,11 +99,12 @@ corecoder/
|
|
|
87
99
|
├── glob_tool.py 文件名匹配 47 行
|
|
88
100
|
├── read.py 文件读取 53 行
|
|
89
101
|
├── write.py 文件写入 38 行
|
|
102
|
+
├── todo.py agent 自维护的任务清单 79 行
|
|
90
103
|
├── agent.py 子 agent 派生 58 行
|
|
91
104
|
└── base.py 工具基类 27 行
|
|
92
105
|
```
|
|
93
106
|
|
|
94
|
-
|
|
107
|
+
八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。
|
|
95
108
|
|
|
96
109
|
## 一个 while 循环就是 agent 的本体
|
|
97
110
|
|
|
@@ -112,7 +125,7 @@ def chat(self, user_input):
|
|
|
112
125
|
return "(已达轮次上限)"
|
|
113
126
|
```
|
|
114
127
|
|
|
115
|
-
就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py`
|
|
128
|
+
就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 最后成了全项目最大的文件,不是因为调模型有多难,而是流式返回里一个工具调用的参数会被切成好几段先后送到、得按顺序拼回去,provider 偶尔吐半截 JSON 或把 usage 填成 null,限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。这些不起眼的脏活,而不是那个循环,才是一个 agent 从能演示走到能交付真正吃工程功夫的地方;第三篇文章顺着它拆到每一行。
|
|
116
129
|
|
|
117
130
|
有三个决定值得单独看,因为它们是「先读懂别人怎么做」之后才做得出的取舍,也是你 fork 自己 agent 时可以直接抄走的判断。
|
|
118
131
|
|
|
@@ -162,17 +175,6 @@ print(Agent(llm=llm).chat("找出项目里所有 TODO 注释并列出来"))
|
|
|
162
175
|
|
|
163
176
|
README 只给方向,每条的代码细节第七篇接着讲。挑一个动手,就是把它做得更好的开始。
|
|
164
177
|
|
|
165
|
-
## 和谁比
|
|
166
|
-
|
|
167
|
-
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
168
|
-
|---|---|---|---|---|
|
|
169
|
-
| 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
|
|
170
|
-
| 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
|
|
171
|
-
| 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
|
|
172
|
-
| 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
|
|
173
|
-
|
|
174
|
-
nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
|
|
175
|
-
|
|
176
178
|
## 命令
|
|
177
179
|
|
|
178
180
|
进了 REPL,`/help` 列全部,常用的这几个:
|
|
@@ -182,15 +184,26 @@ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个
|
|
|
182
184
|
/compact 手动压缩上下文
|
|
183
185
|
/tokens 查看 token 用量和费用估算
|
|
184
186
|
/diff 查看本次会话改过的文件
|
|
187
|
+
/undo 撤销最近一次文件改动
|
|
185
188
|
/save /sessions 保存 / 列出会话
|
|
186
189
|
quit / exit 退出(Ctrl+C 取消当前回合)
|
|
187
190
|
```
|
|
188
191
|
|
|
189
192
|
会话 ID 会先清洗成安全字符再拿去当文件名,存档统统落在 `~/.corecoder/sessions` 里,恶意会话名穿越不出去。
|
|
190
193
|
|
|
194
|
+
## 相关项目
|
|
195
|
+
|
|
196
|
+
如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
|
|
197
|
+
|
|
198
|
+
- **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — 被丢进一个陌生代码库?它给你一份带「从哪读起」路径的 wiki,一个可自托管的 DeepWiki 替代。
|
|
199
|
+
- **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — 别再手动刷招聘网站:它按你的简历给岗位排序,还能跑模拟面试。
|
|
200
|
+
- **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — 签字前先把有风险的条款挑出来:它读合同、标出危险点。
|
|
201
|
+
- **[GitSense](https://github.com/he-yufeng/GitSense)** — 想给开源做贡献?它帮你找到值得做的 issue,还能估你的 PR 多大概率被合。
|
|
202
|
+
- **[CodeABC](https://github.com/he-yufeng/CodeABC)** — 不会写代码也能看懂一个项目,专给小白做的。
|
|
203
|
+
|
|
191
204
|
## 贡献 / License
|
|
192
205
|
|
|
193
|
-
动手之前先跑一遍 `pytest tests/ -q`(
|
|
206
|
+
动手之前先跑一遍 `pytest tests/ -q`(103 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
|
|
194
207
|
|
|
195
208
|
---
|
|
196
209
|
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
|
|
9
9
|
因为 Claude Code 本体太大了。它的代码量在几十万行的量级,光一个负责跑 shell 命令的 BashTool 就上千行,真要逐行读完,多数人会在第三个文件就关掉编辑器。可 agent 的核心其实没那么复杂,复杂的是工程化之后那些为真实世界兜底的边角:模型半路被打断怎么办、上下文塞满了怎么办、十个工具调用同时回来怎么办、provider 偶尔抽风返回 500 又怎么办。这些东西摊在几十万行里,你很难一眼看到骨架。
|
|
10
10
|
|
|
11
|
-
CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是
|
|
11
|
+
CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1201 行;连最外层的 CLI 终端、配置和打包都算上,整个包去掉空行和注释 1614 行,物理代码 2008 行。每个文件都短到能一口气读完,跑起来又确实是个能改代码、能跑命令、能自己开子任务、能压缩上下文、能算钱的 agent。它不是玩具。它是把生产级 agent 的每个设计决策,拣出最核心的那版,用最少的代码诚实地写一遍。
|
|
12
12
|
|
|
13
13
|
读它,约等于读一份 Claude Code 的「可运行注释版」。区别在于,CoreCoder 的每一行你都能在自己机器上断点、改、跑、看它出什么效果。本系列里我引用的每一个行数、每一段代码,都是从仓库里现读现核的,不是凭印象。这点我后面会反复较真,因为编码 agent 这个领域,太多文章在凭感觉编数字。
|
|
14
14
|
|
|
@@ -8,7 +8,7 @@ First, use CoreCoder, an open source project whose core runs to just over a thou
|
|
|
8
8
|
|
|
9
9
|
Because Claude Code itself is too big. Its codebase runs into the hundreds of thousands of lines, and the BashTool alone, the part that runs shell commands, is over a thousand. Read it line by line and most people close the editor by the third file. Yet the core of an agent isn't really that complicated. What's complicated is everything engineered around it to survive the real world: what to do when the model gets interrupted mid-stream, when the context window fills up, when ten tool calls come back at once, when a provider hiccups and returns a 500. Spread across hundreds of thousands of lines, that skeleton is hard to see.
|
|
10
10
|
|
|
11
|
-
What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is
|
|
11
|
+
What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is 1201 lines once you drop blank lines and comments; add the outermost CLI terminal, config, and packaging and the whole package is 1614 lines without blanks and comments, 2008 physical. Every file is short enough to read in one sitting, and what runs is genuinely an agent: it edits code, runs commands, spins up its own subtasks, compresses context, tracks cost. It is not a toy. It takes every design decision a production agent makes, picks the most essential version of each, and writes it out honestly in the least code it can.
|
|
12
12
|
|
|
13
13
|
Reading it is close to reading a runnable, annotated edition of Claude Code. The difference is that every line of CoreCoder is something you can breakpoint, change, run, and watch on your own machine. Every line count and every snippet I quote in this series is read straight out of the repository, not recalled from memory. I'll keep being pedantic about that, because in the coding-agent space far too many write-ups just make the numbers up.
|
|
14
14
|
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.4.
|
|
3
|
+
__version__ = "0.4.2"
|
|
4
4
|
|
|
5
5
|
from corecoder.agent import Agent
|
|
6
|
-
from corecoder.llm import LLM
|
|
7
6
|
from corecoder.config import Config
|
|
7
|
+
from corecoder.llm import LLM
|
|
8
8
|
from corecoder.tools import ALL_TOOLS
|
|
9
9
|
|
|
10
|
-
__all__ = ["
|
|
10
|
+
__all__ = ["ALL_TOOLS", "LLM", "Agent", "Config", "__version__"]
|
|
@@ -11,12 +11,14 @@ which means it's done working and ready to report back.
|
|
|
11
11
|
|
|
12
12
|
import concurrent.futures
|
|
13
13
|
import inspect
|
|
14
|
+
|
|
15
|
+
from .context import ContextManager
|
|
14
16
|
from .llm import LLM
|
|
17
|
+
from .prompt import system_prompt
|
|
15
18
|
from .tools import ALL_TOOLS
|
|
16
|
-
from .tools.base import Tool
|
|
17
19
|
from .tools.agent import AgentTool
|
|
18
|
-
from .
|
|
19
|
-
from .
|
|
20
|
+
from .tools.base import Tool
|
|
21
|
+
from .tools.todo import TodoWriteTool
|
|
20
22
|
|
|
21
23
|
|
|
22
24
|
class Agent:
|
|
@@ -40,8 +42,17 @@ class Agent:
|
|
|
40
42
|
if isinstance(t, AgentTool):
|
|
41
43
|
t._parent_agent = self
|
|
42
44
|
|
|
45
|
+
self._todo = next((t for t in self.tools if isinstance(t, TodoWriteTool)), None)
|
|
46
|
+
|
|
43
47
|
def _full_messages(self) -> list[dict]:
|
|
44
|
-
|
|
48
|
+
system = self._system
|
|
49
|
+
# the task list is re-injected every round, so the model always sees the
|
|
50
|
+
# current state rather than a stale copy buried in old tool results
|
|
51
|
+
if self._todo is not None:
|
|
52
|
+
rendered = self._todo.render()
|
|
53
|
+
if rendered:
|
|
54
|
+
system += "\n\n# Current task list\n" + rendered
|
|
55
|
+
return [{"role": "system", "content": system}] + self.messages
|
|
45
56
|
|
|
46
57
|
def _tool_schemas(self) -> list[dict]:
|
|
47
58
|
return [t.schema() for t in self.tools]
|
|
@@ -109,9 +120,10 @@ class Agent:
|
|
|
109
120
|
inspect.signature(tool.execute).bind(**tc.arguments)
|
|
110
121
|
except TypeError as e:
|
|
111
122
|
return f"Error: bad arguments for {tc.name}: {e}"
|
|
123
|
+
# a tool that blows up gets reported back as text, never kills the loop
|
|
112
124
|
try:
|
|
113
125
|
return tool.execute(**tc.arguments)
|
|
114
|
-
except Exception as e:
|
|
126
|
+
except Exception as e: # noqa: BLE001
|
|
115
127
|
return f"Error executing {tc.name}: {e}"
|
|
116
128
|
|
|
117
129
|
def _exec_tools_parallel(self, tool_calls, on_tool=None) -> list[str]:
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Session-scoped undo for file mutations.
|
|
2
|
+
|
|
3
|
+
edit_file and write_file record a checkpoint before touching a file; /undo
|
|
4
|
+
pops the latest one and restores the previous bytes (or removes the file if
|
|
5
|
+
it did not exist). In-memory only: undo history dies with the process, and
|
|
6
|
+
bash side effects are not tracked, only the two file-writing tools.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
# (path, prior bytes or None if the file did not exist)
|
|
12
|
+
_stack: list[tuple[str, bytes | None]] = []
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def record(path: Path) -> None:
|
|
16
|
+
"""Capture the pre-mutation state of path. Call right before writing."""
|
|
17
|
+
_stack.append((str(path), path.read_bytes() if path.exists() else None))
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def undo() -> str:
|
|
21
|
+
"""Restore the most recent checkpoint."""
|
|
22
|
+
if not _stack:
|
|
23
|
+
return "Nothing to undo."
|
|
24
|
+
path_str, prior = _stack.pop()
|
|
25
|
+
p = Path(path_str)
|
|
26
|
+
if prior is None:
|
|
27
|
+
p.unlink(missing_ok=True)
|
|
28
|
+
return f"Removed {path_str} (created this session)."
|
|
29
|
+
p.write_bytes(prior)
|
|
30
|
+
return f"Restored {path_str}."
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def pending() -> int:
|
|
34
|
+
return len(_stack)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def clear() -> None:
|
|
38
|
+
_stack.clear()
|