corecoder 0.4.0__tar.gz → 0.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. {corecoder-0.4.0 → corecoder-0.4.2}/PKG-INFO +36 -23
  2. {corecoder-0.4.0 → corecoder-0.4.2}/README.md +34 -21
  3. {corecoder-0.4.0 → corecoder-0.4.2}/README_CN.md +34 -21
  4. {corecoder-0.4.0 → corecoder-0.4.2}/article/00-index.md +1 -1
  5. {corecoder-0.4.0 → corecoder-0.4.2}/article/00-index_EN.md +1 -1
  6. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/__init__.py +3 -3
  7. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/agent.py +17 -5
  8. corecoder-0.4.2/corecoder/checkpoints.py +38 -0
  9. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/cli.py +27 -11
  10. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/context.py +9 -9
  11. corecoder-0.4.2/corecoder/demo.py +84 -0
  12. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/llm.py +26 -1
  13. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/__init__.py +5 -3
  14. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/agent.py +5 -2
  15. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/bash.py +6 -2
  16. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/edit.py +6 -2
  17. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/glob_tool.py +7 -2
  18. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/grep.py +19 -5
  19. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/read.py +5 -2
  20. corecoder-0.4.2/corecoder/tools/todo.py +79 -0
  21. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/write.py +7 -2
  22. {corecoder-0.4.0 → corecoder-0.4.2}/pyproject.toml +1 -1
  23. corecoder-0.4.2/tests/test_checkpoints.py +54 -0
  24. {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_core.py +84 -6
  25. corecoder-0.4.2/tests/test_demo.py +39 -0
  26. {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_litellm.py +1 -2
  27. {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_session.py +1 -1
  28. {corecoder-0.4.0 → corecoder-0.4.2}/tests/test_tools.py +106 -3
  29. {corecoder-0.4.0 → corecoder-0.4.2}/.github/workflows/ci.yml +0 -0
  30. {corecoder-0.4.0 → corecoder-0.4.2}/.github/workflows/publish.yml +0 -0
  31. {corecoder-0.4.0 → corecoder-0.4.2}/.gitignore +0 -0
  32. {corecoder-0.4.0 → corecoder-0.4.2}/LICENSE +0 -0
  33. {corecoder-0.4.0 → corecoder-0.4.2}/article/01-the-loop.md +0 -0
  34. {corecoder-0.4.0 → corecoder-0.4.2}/article/01-the-loop_EN.md +0 -0
  35. {corecoder-0.4.0 → corecoder-0.4.2}/article/02-tools.md +0 -0
  36. {corecoder-0.4.0 → corecoder-0.4.2}/article/02-tools_EN.md +0 -0
  37. {corecoder-0.4.0 → corecoder-0.4.2}/article/03-llm-and-cost.md +0 -0
  38. {corecoder-0.4.0 → corecoder-0.4.2}/article/03-llm-and-cost_EN.md +0 -0
  39. {corecoder-0.4.0 → corecoder-0.4.2}/article/04-context.md +0 -0
  40. {corecoder-0.4.0 → corecoder-0.4.2}/article/04-context_EN.md +0 -0
  41. {corecoder-0.4.0 → corecoder-0.4.2}/article/05-parallel-and-subagents.md +0 -0
  42. {corecoder-0.4.0 → corecoder-0.4.2}/article/05-parallel-and-subagents_EN.md +0 -0
  43. {corecoder-0.4.0 → corecoder-0.4.2}/article/06-session-and-cli.md +0 -0
  44. {corecoder-0.4.0 → corecoder-0.4.2}/article/06-session-and-cli_EN.md +0 -0
  45. {corecoder-0.4.0 → corecoder-0.4.2}/article/07-build-your-own.md +0 -0
  46. {corecoder-0.4.0 → corecoder-0.4.2}/article/07-build-your-own_EN.md +0 -0
  47. {corecoder-0.4.0 → corecoder-0.4.2}/assets/demo.png +0 -0
  48. {corecoder-0.4.0 → corecoder-0.4.2}/assets/demo_en.png +0 -0
  49. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/__main__.py +0 -0
  50. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/config.py +0 -0
  51. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/prompt.py +0 -0
  52. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/session.py +0 -0
  53. {corecoder-0.4.0 → corecoder-0.4.2}/corecoder/tools/base.py +0 -0
  54. {corecoder-0.4.0 → corecoder-0.4.2}/tests/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: corecoder
3
- Version: 0.4.0
3
+ Version: 0.4.2
4
4
  Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
5
5
  Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
6
6
  Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
37
37
 
38
38
  # CoreCoder
39
39
 
40
- **The nanoGPT of coding agents. 1,081 lines of pure Python — understand how a coding agent actually works, then fork your own.**
40
+ **The nanoGPT of coding agents. 1,201 lines of pure Python — understand how a coding agent actually works, then fork your own.**
41
41
 
42
42
  *learn from it · fork it · ship something better*
43
43
 
@@ -47,22 +47,33 @@ Description-Content-Type: text/markdown
47
47
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
48
48
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
49
49
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
50
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/00-index_EN.md)
50
+ [![engine](https://img.shields.io/badge/engine-1201_LoC-blue)](article/00-index_EN.md)
51
51
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
52
52
 
53
53
  </div>
54
54
 
55
- - **Readable end to end.** Read the whole engine in an afternoon: 1,081 lines of pure Python, with no magic hidden anywhere you can't follow it.
56
- - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram. It just isn't meant to be your daily driver.
55
+ - **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
56
+ - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
57
57
  - **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
58
58
 
59
+ ## How it compares
60
+
61
+ | | CoreCoder | Claude Code | aider | nanoGPT |
62
+ |---|---|---|---|---|
63
+ | Lines of code | ~1,201 engine / 2,008 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
64
+ | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
65
+ | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
66
+ | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
67
+
68
+ The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
69
+
59
70
  ## What this is
60
71
 
61
72
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
62
73
 
63
- The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
74
+ The engine (loop, model interface, context, tools, sessions) is 1,201 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 21 files: 2,008 physical lines, 1,614 net, every one short enough to read in a single sitting.
64
75
 
65
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 86 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
76
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
66
77
 
67
78
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
68
79
 
@@ -93,6 +104,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
93
104
  |---|---|
94
105
  | OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
95
106
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
107
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
96
108
  | Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
97
109
 
98
110
  Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
@@ -108,7 +120,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
108
120
 
109
121
  ```
110
122
  corecoder/
111
- ├── agent.py agent loop + parallel tool exec 150 lines ← start here
123
+ ├── agent.py agent loop + parallel tool exec 162 lines ← start here
112
124
  ├── llm.py streaming client + retry + cost 336 lines
113
125
  ├── context.py three-tier context compaction 210 lines
114
126
  ├── session.py save / resume + path-traversal guard 97 lines
@@ -122,11 +134,12 @@ corecoder/
122
134
  ├── glob_tool.py filename matching 47 lines
123
135
  ├── read.py file read 53 lines
124
136
  ├── write.py file write 38 lines
137
+ ├── todo.py agent-maintained task checklist 79 lines
125
138
  ├── agent.py sub-agent spawning 58 lines
126
139
  └── base.py tool base class 27 lines
127
140
  ```
128
141
 
129
- Seven tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, and `agent` (which spawns a sub-agent). Add the packaging files like `__init__` and `__main__` and the package is 18 files; strip the CLI shell and config and the engine itself is about 1,081 lines.
142
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
130
143
 
131
144
  ## A `while` loop is the whole agent
132
145
 
@@ -147,7 +160,7 @@ def chat(self, user_input):
147
160
  return "(hit the round limit)"
148
161
  ```
149
162
 
150
- That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess that shows up once the loop runs against the real world. `llm.py` is the biggest file in the project, not because calling a model is hard, but because in a streamed response a single tool call's arguments arrive in several fragments, one after another, and you have to stitch them back together in order. A provider will occasionally hand you half a JSON object, or fill the `usage` field with null; rate limits (429), timeouts, dropped connections and 5xx all need backoff-and-retry, while the other 4xx should just raise instead of being retried into the ground. Even an OpenAI extension like `stream_options` gets handled: some providers reject it outright with a 400, so the code strips it and resends exactly once, only on a 400, and never stacks that on top of the retry backoff. Lay all this grunt work out and it's the genuinely hard engineering part of taking an agent from demo to delivery. The unglamorous part turns out to be the loop itself; the thousand lines of fallback around it are where the real work lives.
163
+ That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
151
164
 
152
165
  Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
153
166
 
@@ -197,17 +210,6 @@ Going deeper, the directions are out in the open too. None of the following is i
197
210
 
198
211
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
199
212
 
200
- ## How it compares
201
-
202
- | | CoreCoder | Claude Code | aider | nanoGPT |
203
- |---|---|---|---|---|
204
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
205
- | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
206
- | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
207
- | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
208
-
209
- The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
210
-
211
213
  ## Commands
212
214
 
213
215
  Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
@@ -217,15 +219,26 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
217
219
  /compact compact the context by hand
218
220
  /tokens token usage and cost estimate
219
221
  /diff files changed this session
222
+ /undo revert the most recent file change
220
223
  /save /sessions save / list sessions
221
224
  quit / exit exit (Ctrl+C cancels the current round)
222
225
  ```
223
226
 
224
227
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
225
228
 
229
+ ## Related Projects
230
+
231
+ If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
232
+
233
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
234
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
235
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
236
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
237
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
238
+
226
239
  ## Contributing / License
227
240
 
228
- Before you send anything, run `pytest tests/ -q` (86 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
241
+ Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
229
242
 
230
243
  ---
231
244
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **The nanoGPT of coding agents. 1,081 lines of pure Python — understand how a coding agent actually works, then fork your own.**
5
+ **The nanoGPT of coding agents. 1,201 lines of pure Python — understand how a coding agent actually works, then fork your own.**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,22 +12,33 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/00-index_EN.md)
15
+ [![engine](https://img.shields.io/badge/engine-1201_LoC-blue)](article/00-index_EN.md)
16
16
  [![essays](https://img.shields.io/badge/source--reading-8_bilingual-orange)](article/00-index_EN.md)
17
17
 
18
18
  </div>
19
19
 
20
- - **Readable end to end.** Read the whole engine in an afternoon: 1,081 lines of pure Python, with no magic hidden anywhere you can't follow it.
21
- - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram. It just isn't meant to be your daily driver.
20
+ - **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
21
+ - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
22
22
  - **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
23
23
 
24
+ ## How it compares
25
+
26
+ | | CoreCoder | Claude Code | aider | nanoGPT |
27
+ |---|---|---|---|---|
28
+ | Lines of code | ~1,201 engine / 2,008 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
29
+ | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
30
+ | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
31
+ | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
32
+
33
+ The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
34
+
24
35
  ## What this is
25
36
 
26
37
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
27
38
 
28
- The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
39
+ The engine (loop, model interface, context, tools, sessions) is 1,201 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 21 files: 2,008 physical lines, 1,614 net, every one short enough to read in a single sitting.
29
40
 
30
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 86 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
41
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
31
42
 
32
43
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
33
44
 
@@ -58,6 +69,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
58
69
  |---|---|
59
70
  | OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
60
71
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
72
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
61
73
  | Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
62
74
 
63
75
  Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
@@ -73,7 +85,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
73
85
 
74
86
  ```
75
87
  corecoder/
76
- ├── agent.py agent loop + parallel tool exec 150 lines ← start here
88
+ ├── agent.py agent loop + parallel tool exec 162 lines ← start here
77
89
  ├── llm.py streaming client + retry + cost 336 lines
78
90
  ├── context.py three-tier context compaction 210 lines
79
91
  ├── session.py save / resume + path-traversal guard 97 lines
@@ -87,11 +99,12 @@ corecoder/
87
99
  ├── glob_tool.py filename matching 47 lines
88
100
  ├── read.py file read 53 lines
89
101
  ├── write.py file write 38 lines
102
+ ├── todo.py agent-maintained task checklist 79 lines
90
103
  ├── agent.py sub-agent spawning 58 lines
91
104
  └── base.py tool base class 27 lines
92
105
  ```
93
106
 
94
- Seven tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, and `agent` (which spawns a sub-agent). Add the packaging files like `__init__` and `__main__` and the package is 18 files; strip the CLI shell and config and the engine itself is about 1,081 lines.
107
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
95
108
 
96
109
  ## A `while` loop is the whole agent
97
110
 
@@ -112,7 +125,7 @@ def chat(self, user_input):
112
125
  return "(hit the round limit)"
113
126
  ```
114
127
 
115
- That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess that shows up once the loop runs against the real world. `llm.py` is the biggest file in the project, not because calling a model is hard, but because in a streamed response a single tool call's arguments arrive in several fragments, one after another, and you have to stitch them back together in order. A provider will occasionally hand you half a JSON object, or fill the `usage` field with null; rate limits (429), timeouts, dropped connections and 5xx all need backoff-and-retry, while the other 4xx should just raise instead of being retried into the ground. Even an OpenAI extension like `stream_options` gets handled: some providers reject it outright with a 400, so the code strips it and resends exactly once, only on a 400, and never stacks that on top of the retry backoff. Lay all this grunt work out and it's the genuinely hard engineering part of taking an agent from demo to delivery. The unglamorous part turns out to be the loop itself; the thousand lines of fallback around it are where the real work lives.
128
+ That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
116
129
 
117
130
  Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
118
131
 
@@ -162,17 +175,6 @@ Going deeper, the directions are out in the open too. None of the following is i
162
175
 
163
176
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
164
177
 
165
- ## How it compares
166
-
167
- | | CoreCoder | Claude Code | aider | nanoGPT |
168
- |---|---|---|---|---|
169
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
170
- | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
171
- | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
172
- | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
173
-
174
- The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
175
-
176
178
  ## Commands
177
179
 
178
180
  Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
@@ -182,15 +184,26 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
182
184
  /compact compact the context by hand
183
185
  /tokens token usage and cost estimate
184
186
  /diff files changed this session
187
+ /undo revert the most recent file change
185
188
  /save /sessions save / list sessions
186
189
  quit / exit exit (Ctrl+C cancels the current round)
187
190
  ```
188
191
 
189
192
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
190
193
 
194
+ ## Related Projects
195
+
196
+ If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
197
+
198
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
199
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
200
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
201
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
202
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
203
+
191
204
  ## Contributing / License
192
205
 
193
- Before you send anything, run `pytest tests/ -q` (86 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
206
+ Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
194
207
 
195
208
  ---
196
209
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  # CoreCoder
4
4
 
5
- **编程 agent 里的 nanoGPT。1081 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
5
+ **编程 agent 里的 nanoGPT。1201 行纯 Python,读懂一个 coding agent 到底怎么运作,再 fork 出你自己的。**
6
6
 
7
7
  *learn from it · fork it · ship something better*
8
8
 
@@ -12,22 +12,33 @@
12
12
  [![Python](https://img.shields.io/badge/python-3.10+-blue)](https://python.org)
13
13
  [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
14
14
  [![Tests](https://github.com/he-yufeng/CoreCoder/actions/workflows/ci.yml/badge.svg)](https://github.com/he-yufeng/CoreCoder/actions)
15
- [![engine](https://img.shields.io/badge/engine-1081_LoC-blue)](article/)
15
+ [![engine](https://img.shields.io/badge/engine-1201_LoC-blue)](article/)
16
16
  [![源码导读](https://img.shields.io/badge/源码导读-8篇双语-orange)](article/)
17
17
 
18
18
  </div>
19
19
 
20
- - **读得完。** 一个下午读完整个引擎,1081 行纯 Python,没有一处藏着你看不懂的魔法。
21
- - **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图;但它不是拿来当日常工具的。
20
+ - **读得完。** 一个下午读完整个引擎,没有一处藏着你看不懂的魔法。
21
+ - **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图。
22
22
  - **留白即起点。** 刻意只留最小核心,没做的那些不是半成品,是留给你 fork 出更好东西的地方。
23
23
 
24
+ ## 和谁比
25
+
26
+ | | CoreCoder | Claude Code | aider | nanoGPT |
27
+ |---|---|---|---|---|
28
+ | 代码量 | 引擎约 1201 行 / 整包 2008 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
29
+ | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
30
+ | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
31
+ | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
32
+
33
+ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
34
+
24
35
  ## 这是什么
25
36
 
26
37
  我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
27
38
 
28
- 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行。连最外层的 CLI、配置、打包一起算,整个包 18 个文件、物理 1714 行、净 1385 行,每个文件都短到能一口气读完。
39
+ 引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1201 行。连最外层的 CLI、配置、打包一起算,整个包 21 个文件、物理 2008 行、净 1614 行,每个文件都短到能一口气读完。
29
40
 
30
- 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,86 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
41
+ 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,103 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
31
42
 
32
43
  代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
33
44
 
@@ -58,6 +69,7 @@ pip install -e .
58
69
  |---|---|
59
70
  | OpenAI(默认 `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
60
71
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
72
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
61
73
  | 本地 Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
62
74
 
63
75
  Kimi、Qwen 这些同样是改这两个变量;连 OpenAI 兼容接口都不给的 provider,装上可选的 LiteLLM 后端(`pip install "corecoder[litellm]"`)能路由一百多家。第三篇文章把这块讲得更细。key 可以直接 `export`,也可以在项目根目录扔个 `.env`,启动时自动加载。然后:
@@ -73,7 +85,7 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
73
85
 
74
86
  ```
75
87
  corecoder/
76
- ├── agent.py agent 主循环 + 并行工具执行 150 行 ← 从这里开始读
88
+ ├── agent.py agent 主循环 + 并行工具执行 162 行 ← 从这里开始读
77
89
  ├── llm.py 流式客户端 + 重试 + 成本统计 336 行
78
90
  ├── context.py 三层上下文压缩 210 行
79
91
  ├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
@@ -87,11 +99,12 @@ corecoder/
87
99
  ├── glob_tool.py 文件名匹配 47 行
88
100
  ├── read.py 文件读取 53 行
89
101
  ├── write.py 文件写入 38 行
102
+ ├── todo.py agent 自维护的任务清单 79 行
90
103
  ├── agent.py 子 agent 派生 58 行
91
104
  └── base.py 工具基类 27 行
92
105
  ```
93
106
 
94
- 七个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`agent`(派子 agent)。再加上 `__init__`、`__main__` 这些打包文件,整包 18 个文件;剥掉 CLI 外壳和配置,引擎本身约 1081 行。
107
+ 八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。
95
108
 
96
109
  ## 一个 while 循环就是 agent 的本体
97
110
 
@@ -112,7 +125,7 @@ def chat(self, user_input):
112
125
  return "(已达轮次上限)"
113
126
  ```
114
127
 
115
- 就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 是全项目最大的文件,不是因为调模型有多难,而是流式返回里,一个工具调用的参数会被切成好几段先后送到,你得按顺序拼回去;provider 偶尔吐半截 JSON,或者把 usage 字段填成 null;限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。连 `stream_options` 这种 OpenAI 扩展都照顾到了:有的 provider 直接回 400 拒收,代码就只在 400 时把它摘掉重发一次,绝不和退避重试叠上去。这些脏活摊开来,恰恰是一个 agent 从能演示走到能交付,工程上最费劲的那部分。祛魅祛的是那个循环,不是这一千行兜底。
128
+ 就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 最后成了全项目最大的文件,不是因为调模型有多难,而是流式返回里一个工具调用的参数会被切成好几段先后送到、得按顺序拼回去,provider 偶尔吐半截 JSON 或把 usage 填成 null,限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。这些不起眼的脏活,而不是那个循环,才是一个 agent 从能演示走到能交付真正吃工程功夫的地方;第三篇文章顺着它拆到每一行。
116
129
 
117
130
  有三个决定值得单独看,因为它们是「先读懂别人怎么做」之后才做得出的取舍,也是你 fork 自己 agent 时可以直接抄走的判断。
118
131
 
@@ -162,17 +175,6 @@ print(Agent(llm=llm).chat("找出项目里所有 TODO 注释并列出来"))
162
175
 
163
176
  README 只给方向,每条的代码细节第七篇接着讲。挑一个动手,就是把它做得更好的开始。
164
177
 
165
- ## 和谁比
166
-
167
- | | CoreCoder | Claude Code | aider | nanoGPT |
168
- |---|---|---|---|---|
169
- | 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
170
- | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
171
- | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
172
- | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
173
-
174
- nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
175
-
176
178
  ## 命令
177
179
 
178
180
  进了 REPL,`/help` 列全部,常用的这几个:
@@ -182,15 +184,26 @@ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个
182
184
  /compact 手动压缩上下文
183
185
  /tokens 查看 token 用量和费用估算
184
186
  /diff 查看本次会话改过的文件
187
+ /undo 撤销最近一次文件改动
185
188
  /save /sessions 保存 / 列出会话
186
189
  quit / exit 退出(Ctrl+C 取消当前回合)
187
190
  ```
188
191
 
189
192
  会话 ID 会先清洗成安全字符再拿去当文件名,存档统统落在 `~/.corecoder/sessions` 里,恶意会话名穿越不出去。
190
193
 
194
+ ## 相关项目
195
+
196
+ 如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
197
+
198
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — 被丢进一个陌生代码库?它给你一份带「从哪读起」路径的 wiki,一个可自托管的 DeepWiki 替代。
199
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — 别再手动刷招聘网站:它按你的简历给岗位排序,还能跑模拟面试。
200
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — 签字前先把有风险的条款挑出来:它读合同、标出危险点。
201
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — 想给开源做贡献?它帮你找到值得做的 issue,还能估你的 PR 多大概率被合。
202
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — 不会写代码也能看懂一个项目,专给小白做的。
203
+
191
204
  ## 贡献 / License
192
205
 
193
- 动手之前先跑一遍 `pytest tests/ -q`(86 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
206
+ 动手之前先跑一遍 `pytest tests/ -q`(103 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
194
207
 
195
208
  ---
196
209
 
@@ -8,7 +8,7 @@
8
8
 
9
9
  因为 Claude Code 本体太大了。它的代码量在几十万行的量级,光一个负责跑 shell 命令的 BashTool 就上千行,真要逐行读完,多数人会在第三个文件就关掉编辑器。可 agent 的核心其实没那么复杂,复杂的是工程化之后那些为真实世界兜底的边角:模型半路被打断怎么办、上下文塞满了怎么办、十个工具调用同时回来怎么办、provider 偶尔抽风返回 500 又怎么办。这些东西摊在几十万行里,你很难一眼看到骨架。
10
10
 
11
- CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行;连最外层的 CLI 终端、配置和打包都算上,整个包去掉空行和注释 1385 行,物理代码 1714 行。每个文件都短到能一口气读完,跑起来又确实是个能改代码、能跑命令、能自己开子任务、能压缩上下文、能算钱的 agent。它不是玩具。它是把生产级 agent 的每个设计决策,拣出最核心的那版,用最少的代码诚实地写一遍。
11
+ CoreCoder 做的事,是把这套骨架压到一千行出头的纯 Python。准确说,agent 引擎(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1201 行;连最外层的 CLI 终端、配置和打包都算上,整个包去掉空行和注释 1614 行,物理代码 2008 行。每个文件都短到能一口气读完,跑起来又确实是个能改代码、能跑命令、能自己开子任务、能压缩上下文、能算钱的 agent。它不是玩具。它是把生产级 agent 的每个设计决策,拣出最核心的那版,用最少的代码诚实地写一遍。
12
12
 
13
13
  读它,约等于读一份 Claude Code 的「可运行注释版」。区别在于,CoreCoder 的每一行你都能在自己机器上断点、改、跑、看它出什么效果。本系列里我引用的每一个行数、每一段代码,都是从仓库里现读现核的,不是凭印象。这点我后面会反复较真,因为编码 agent 这个领域,太多文章在凭感觉编数字。
14
14
 
@@ -8,7 +8,7 @@ First, use CoreCoder, an open source project whose core runs to just over a thou
8
8
 
9
9
  Because Claude Code itself is too big. Its codebase runs into the hundreds of thousands of lines, and the BashTool alone, the part that runs shell commands, is over a thousand. Read it line by line and most people close the editor by the third file. Yet the core of an agent isn't really that complicated. What's complicated is everything engineered around it to survive the real world: what to do when the model gets interrupted mid-stream, when the context window fills up, when ten tool calls come back at once, when a provider hiccups and returns a 500. Spread across hundreds of thousands of lines, that skeleton is hard to see.
10
10
 
11
- What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is 1081 lines once you drop blank lines and comments; add the outermost CLI terminal, config, and packaging and the whole package is 1385 lines without blanks and comments, 1714 physical. Every file is short enough to read in one sitting, and what runs is genuinely an agent: it edits code, runs commands, spins up its own subtasks, compresses context, tracks cost. It is not a toy. It takes every design decision a production agent makes, picks the most essential version of each, and writes it out honestly in the least code it can.
11
+ What CoreCoder does is squeeze the skeleton down to just over a thousand lines of pure Python. To be precise, the agent engine (the loop, model interface, context, tools, session) is 1201 lines once you drop blank lines and comments; add the outermost CLI terminal, config, and packaging and the whole package is 1614 lines without blanks and comments, 2008 physical. Every file is short enough to read in one sitting, and what runs is genuinely an agent: it edits code, runs commands, spins up its own subtasks, compresses context, tracks cost. It is not a toy. It takes every design decision a production agent makes, picks the most essential version of each, and writes it out honestly in the least code it can.
12
12
 
13
13
  Reading it is close to reading a runnable, annotated edition of Claude Code. The difference is that every line of CoreCoder is something you can breakpoint, change, run, and watch on your own machine. Every line count and every snippet I quote in this series is read straight out of the repository, not recalled from memory. I'll keep being pedantic about that, because in the coding-agent space far too many write-ups just make the numbers up.
14
14
 
@@ -1,10 +1,10 @@
1
1
  """CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.4.2"
4
4
 
5
5
  from corecoder.agent import Agent
6
- from corecoder.llm import LLM
7
6
  from corecoder.config import Config
7
+ from corecoder.llm import LLM
8
8
  from corecoder.tools import ALL_TOOLS
9
9
 
10
- __all__ = ["Agent", "LLM", "Config", "ALL_TOOLS", "__version__"]
10
+ __all__ = ["ALL_TOOLS", "LLM", "Agent", "Config", "__version__"]
@@ -11,12 +11,14 @@ which means it's done working and ready to report back.
11
11
 
12
12
  import concurrent.futures
13
13
  import inspect
14
+
15
+ from .context import ContextManager
14
16
  from .llm import LLM
17
+ from .prompt import system_prompt
15
18
  from .tools import ALL_TOOLS
16
- from .tools.base import Tool
17
19
  from .tools.agent import AgentTool
18
- from .prompt import system_prompt
19
- from .context import ContextManager
20
+ from .tools.base import Tool
21
+ from .tools.todo import TodoWriteTool
20
22
 
21
23
 
22
24
  class Agent:
@@ -40,8 +42,17 @@ class Agent:
40
42
  if isinstance(t, AgentTool):
41
43
  t._parent_agent = self
42
44
 
45
+ self._todo = next((t for t in self.tools if isinstance(t, TodoWriteTool)), None)
46
+
43
47
  def _full_messages(self) -> list[dict]:
44
- return [{"role": "system", "content": self._system}] + self.messages
48
+ system = self._system
49
+ # the task list is re-injected every round, so the model always sees the
50
+ # current state rather than a stale copy buried in old tool results
51
+ if self._todo is not None:
52
+ rendered = self._todo.render()
53
+ if rendered:
54
+ system += "\n\n# Current task list\n" + rendered
55
+ return [{"role": "system", "content": system}] + self.messages
45
56
 
46
57
  def _tool_schemas(self) -> list[dict]:
47
58
  return [t.schema() for t in self.tools]
@@ -109,9 +120,10 @@ class Agent:
109
120
  inspect.signature(tool.execute).bind(**tc.arguments)
110
121
  except TypeError as e:
111
122
  return f"Error: bad arguments for {tc.name}: {e}"
123
+ # a tool that blows up gets reported back as text, never kills the loop
112
124
  try:
113
125
  return tool.execute(**tc.arguments)
114
- except Exception as e:
126
+ except Exception as e: # noqa: BLE001
115
127
  return f"Error executing {tc.name}: {e}"
116
128
 
117
129
  def _exec_tools_parallel(self, tool_calls, on_tool=None) -> list[str]:
@@ -0,0 +1,38 @@
1
+ """Session-scoped undo for file mutations.
2
+
3
+ edit_file and write_file record a checkpoint before touching a file; /undo
4
+ pops the latest one and restores the previous bytes (or removes the file if
5
+ it did not exist). In-memory only: undo history dies with the process, and
6
+ bash side effects are not tracked, only the two file-writing tools.
7
+ """
8
+
9
+ from pathlib import Path
10
+
11
+ # (path, prior bytes or None if the file did not exist)
12
+ _stack: list[tuple[str, bytes | None]] = []
13
+
14
+
15
+ def record(path: Path) -> None:
16
+ """Capture the pre-mutation state of path. Call right before writing."""
17
+ _stack.append((str(path), path.read_bytes() if path.exists() else None))
18
+
19
+
20
+ def undo() -> str:
21
+ """Restore the most recent checkpoint."""
22
+ if not _stack:
23
+ return "Nothing to undo."
24
+ path_str, prior = _stack.pop()
25
+ p = Path(path_str)
26
+ if prior is None:
27
+ p.unlink(missing_ok=True)
28
+ return f"Removed {path_str} (created this session)."
29
+ p.write_bytes(prior)
30
+ return f"Restored {path_str}."
31
+
32
+
33
+ def pending() -> int:
34
+ return len(_stack)
35
+
36
+
37
+ def clear() -> None:
38
+ _stack.clear()