corecoder 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {corecoder-0.6.0 → corecoder-0.8.0}/.github/workflows/ci.yml +6 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/PKG-INFO +62 -29
- {corecoder-0.6.0 → corecoder-0.8.0}/README.md +60 -27
- {corecoder-0.6.0 → corecoder-0.8.0}/README_CN.md +59 -27
- {corecoder-0.6.0 → corecoder-0.8.0}/article/00-index.md +5 -4
- {corecoder-0.6.0 → corecoder-0.8.0}/article/00-index_EN.md +5 -4
- {corecoder-0.6.0 → corecoder-0.8.0}/article/01-the-loop.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/01-the-loop_EN.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/02-tools.md +16 -4
- {corecoder-0.6.0 → corecoder-0.8.0}/article/02-tools_EN.md +16 -4
- {corecoder-0.6.0 → corecoder-0.8.0}/article/03-llm-and-cost.md +2 -2
- {corecoder-0.6.0 → corecoder-0.8.0}/article/03-llm-and-cost_EN.md +2 -2
- {corecoder-0.6.0 → corecoder-0.8.0}/article/04-context.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/04-context_EN.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/05-parallel-and-subagents.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/05-parallel-and-subagents_EN.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/06-session-and-cli.md +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/article/06-session-and-cli_EN.md +1 -1
- corecoder-0.8.0/article/08-extensibility.md +98 -0
- corecoder-0.8.0/article/08-extensibility_EN.md +98 -0
- corecoder-0.8.0/assets/demo-plan-hooks.gif +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/__init__.py +1 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/agent.py +30 -3
- corecoder-0.8.0/corecoder/checkpoints.py +93 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/cli.py +16 -4
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/context.py +12 -2
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/demo.py +2 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/hooks.py +4 -2
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/llm.py +77 -12
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/permissions.py +31 -4
- corecoder-0.8.0/corecoder/shell.py +61 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/agent.py +8 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/base.py +5 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/bash.py +86 -14
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/edit.py +26 -23
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/write.py +8 -5
- corecoder-0.8.0/examples/plan_hooks_demo.py +140 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/pyproject.toml +2 -2
- corecoder-0.8.0/tests/conftest.py +32 -0
- corecoder-0.8.0/tests/test_checkpoints.py +105 -0
- corecoder-0.8.0/tests/test_core.py +774 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_demo.py +18 -1
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_permissions.py +26 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_plan_mode.py +5 -12
- corecoder-0.8.0/tests/test_repl.py +140 -0
- corecoder-0.8.0/tests/test_safety_matrix.py +208 -0
- corecoder-0.8.0/tests/test_shell.py +71 -0
- corecoder-0.8.0/tests/test_slash_commands.py +130 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_tools.py +76 -0
- corecoder-0.6.0/corecoder/checkpoints.py +0 -38
- corecoder-0.6.0/tests/conftest.py +0 -11
- corecoder-0.6.0/tests/test_checkpoints.py +0 -54
- corecoder-0.6.0/tests/test_core.py +0 -311
- {corecoder-0.6.0 → corecoder-0.8.0}/.github/workflows/publish.yml +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/.gitignore +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/LICENSE +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/article/07-build-your-own.md +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/article/07-build-your-own_EN.md +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/assets/demo.png +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/assets/demo_en.png +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/__main__.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/config.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/mcp.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/prompt.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/session.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/__init__.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/glob_tool.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/grep.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/read.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/corecoder/tools/todo.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/__init__.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_hooks.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_litellm.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_mcp.py +0 -0
- {corecoder-0.6.0 → corecoder-0.8.0}/tests/test_session.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: corecoder
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Minimal AI coding agent (
|
|
3
|
+
Version: 0.8.0
|
|
4
|
+
Summary: Minimal AI coding agent (2,735 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
|
|
5
5
|
Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
|
|
6
6
|
Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
|
|
7
7
|
Project-URL: Issues, https://github.com/he-yufeng/CoreCoder/issues
|
|
@@ -37,7 +37,7 @@ Description-Content-Type: text/markdown
|
|
|
37
37
|
|
|
38
38
|
# CoreCoder
|
|
39
39
|
|
|
40
|
-
**The nanoGPT of coding agents. 1,
|
|
40
|
+
**The nanoGPT of coding agents. A 1.3k-line engine inside 2,735 readable lines of pure Python: understand how a coding agent actually works, then fork your own.**
|
|
41
41
|
|
|
42
42
|
*learn from it · fork it · ship something better*
|
|
43
43
|
|
|
@@ -47,7 +47,7 @@ Description-Content-Type: text/markdown
|
|
|
47
47
|
[](https://python.org)
|
|
48
48
|
[](LICENSE)
|
|
49
49
|
[](https://github.com/he-yufeng/CoreCoder/actions)
|
|
50
|
-
[](article/00-index_EN.md)
|
|
51
51
|
[](article/00-index_EN.md)
|
|
52
52
|
|
|
53
53
|
</div>
|
|
@@ -60,7 +60,7 @@ Description-Content-Type: text/markdown
|
|
|
60
60
|
|
|
61
61
|
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
62
62
|
|---|---|---|---|---|
|
|
63
|
-
| Lines of code | ~1,
|
|
63
|
+
| Lines of code | ~1,309 engine / 2,735 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
64
64
|
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
65
65
|
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
66
66
|
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
@@ -71,15 +71,15 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
|
|
|
71
71
|
|
|
72
72
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
73
73
|
|
|
74
|
-
The engine (loop, model interface, context, tools, sessions) is 1,
|
|
74
|
+
The engine (loop, model interface, context, tools, sessions) is 1,309 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 25 files: 2,735 physical lines, 2,205 net, every one short enough to read in a single sitting. The growth since the original 1,161-line snapshot went into visible features: plan mode, hooks and checkpoints, each documented below.
|
|
75
75
|
|
|
76
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first.
|
|
76
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 215 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
77
77
|
|
|
78
78
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
79
79
|
|
|
80
80
|
<p align="center">
|
|
81
|
-
<img src="https://raw.githubusercontent.com/he-yufeng/CoreCoder/main/assets/
|
|
82
|
-
alt="
|
|
81
|
+
<img src="https://raw.githubusercontent.com/he-yufeng/CoreCoder/main/assets/demo-plan-hooks.gif" width="760"
|
|
82
|
+
alt="Plan mode in action: the agent reads fib.py, gets its edit refused while plan mode is on, presents a plan, and only after approval edits, tests, and reports — with Pre/PostToolUse hooks firing around every call.">
|
|
83
83
|
</p>
|
|
84
84
|
|
|
85
85
|
<p align="center"><sub><i>These thousand lines really do run a full loop end to end: ask it to fix buggy.py and it reads the file, edits the code, runs it once to confirm, then reports back on its own. Watch it, then come back and read the code.</i></sub></p>
|
|
@@ -107,7 +107,9 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
107
107
|
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
108
108
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
109
109
|
|
|
110
|
-
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
110
|
+
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. Thinking models are first-class too: deepseek-reasoner, kimi-k3 and friends stream their chain-of-thought, and CoreCoder shows it dimmed as it works, kept out of the conversation history so providers never see it come back. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
111
|
+
|
|
112
|
+
Smoke-tested end to end (read the file, edit it, run it, report back) against DeepSeek, Qwen3 and Kimi K2 via a single OpenRouter-compatible endpoint; each completed the full loop. One note for one-shot scripts: `-p` refuses mutating tools unless you pass `--yes`, by design.
|
|
111
113
|
|
|
112
114
|
```bash
|
|
113
115
|
corecoder # interactive REPL
|
|
@@ -120,26 +122,31 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
120
122
|
|
|
121
123
|
```
|
|
122
124
|
corecoder/
|
|
123
|
-
├── agent.py agent loop + parallel tool exec
|
|
124
|
-
├── llm.py streaming client + retry + cost
|
|
125
|
-
├── context.py three-tier context compaction
|
|
125
|
+
├── agent.py agent loop + parallel tool exec 240 lines ← start here
|
|
126
|
+
├── llm.py streaming client + retry + cost 332 lines
|
|
127
|
+
├── context.py three-tier context compaction 220 lines
|
|
126
128
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
127
|
-
├── permissions.py consent for mutating tools
|
|
128
|
-
├── hooks.py Pre/PostToolUse shell hooks
|
|
129
|
+
├── permissions.py consent for mutating tools 75 lines
|
|
130
|
+
├── hooks.py Pre/PostToolUse shell hooks 87 lines
|
|
131
|
+
├── shell.py POSIX shell routing (Git Bash on Windows) 61 lines
|
|
129
132
|
├── mcp.py MCP stdio client for external tools 208 lines
|
|
130
133
|
├── prompt.py system prompt 41 lines
|
|
131
|
-
├── cli.py REPL + slash commands + one-shot
|
|
134
|
+
├── cli.py REPL + slash commands + one-shot 358 lines
|
|
132
135
|
├── config.py env-var config 55 lines
|
|
136
|
+
├── checkpoints.py /undo snapshot and restore 93 lines
|
|
137
|
+
├── demo.py offline end-to-end demo 101 lines
|
|
133
138
|
└── tools/
|
|
134
|
-
├── bash.py shell + dangerous-command gate + cd
|
|
135
|
-
├── edit.py unique-match search/replace + diff
|
|
139
|
+
├── bash.py shell + dangerous-command gate + cd 203 lines
|
|
140
|
+
├── edit.py unique-match search/replace + diff 99 lines
|
|
136
141
|
├── grep.py content search 93 lines
|
|
137
142
|
├── glob_tool.py filename matching 52 lines
|
|
138
143
|
├── read.py file read 56 lines
|
|
139
|
-
├── write.py file write
|
|
144
|
+
├── write.py file write 46 lines
|
|
140
145
|
├── todo.py agent-maintained task checklist 79 lines
|
|
141
|
-
├── agent.py sub-agent spawning
|
|
142
|
-
└── base.py tool base class
|
|
146
|
+
├── agent.py sub-agent spawning 72 lines
|
|
147
|
+
└── base.py tool base class 32 lines
|
|
148
|
+
examples/
|
|
149
|
+
└── plan_hooks_demo.py offline plan mode + hooks demo (no API key)
|
|
143
150
|
```
|
|
144
151
|
|
|
145
152
|
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core. If `~/.corecoder/mcp.json` exists, its MCP servers join the eight as extra `mcp__*` tools; the MCP section below covers it.
|
|
@@ -163,7 +170,7 @@ def chat(self, user_input):
|
|
|
163
170
|
return "(hit the round limit)"
|
|
164
171
|
```
|
|
165
172
|
|
|
166
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the
|
|
173
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the engine, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. A stream can also die after it has already started, so the retry wraps the whole request, not just the connect. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
167
174
|
|
|
168
175
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
169
176
|
|
|
@@ -177,23 +184,24 @@ Every one of these *whys* is traced down to the actual lines of code in the seri
|
|
|
177
184
|
|
|
178
185
|
## The source-reading series · 8 bilingual essays
|
|
179
186
|
|
|
180
|
-
I also wrote a bilingual source-reading series, one intro plus
|
|
187
|
+
I also wrote a bilingual source-reading series, one intro plus eight parts, each in Chinese with an English mirror. Against CoreCoder's actual code, it walks through how agents like Claude Code work under the hood. One hard rule I set myself: every line count and every snippet is re-read and re-checked from the repo, never written from memory. The first six get you reading, the seventh gets you forking, and the eighth is about extending it without touching the loop; read them in any order.
|
|
181
188
|
|
|
182
189
|
- **[Intro · Read Claude Code through CoreCoder, then build your own](article/00-index_EN.md)**
|
|
183
190
|
- **[01 · An agent, at its core, is a `while` loop](article/01-the-loop_EN.md)** — the main loop in `agent.py`, interrupts, and the round limit
|
|
184
|
-
- **[02 · The tool system: letting the model act, safely](article/02-tools_EN.md)** — the
|
|
191
|
+
- **[02 · The tool system: letting the model act, safely](article/02-tools_EN.md)** — the eight tools in `tools/` and the bash safety gate
|
|
185
192
|
- **[03 · Plug in any LLM, and keep the bill honest](article/03-llm-and-cost_EN.md)** — `llm.py`'s provider wrapper, retries, and cost accounting
|
|
186
193
|
- **[04 · Surviving a long task on a finite window](article/04-context_EN.md)** — `context.py`'s three-tier compaction and orphaned tool messages
|
|
187
194
|
- **[05 · Parallel execution and sub-agents](article/05-parallel-and-subagents_EN.md)** — thread-pool concurrency and sub-agent isolation
|
|
188
195
|
- **[06 · Turning it into a real command-line tool](article/06-session-and-cli_EN.md)** — `session.py` and path-traversal defense
|
|
189
196
|
- **[07 · Fork CoreCoder into your own coding agent](article/07-build-your-own_EN.md)** — from fork to custom tools to swapping models
|
|
197
|
+
- **[08 · Three ways to extend without touching the loop: MCP, hooks, and plan mode](article/08-extensibility_EN.md)** — the v0.6.0 extensibility trio and the contract that makes them safe
|
|
190
198
|
|
|
191
199
|
## Fork it, build something better
|
|
192
200
|
|
|
193
201
|
Once you understand it, the natural next step is to fork. Getting started doesn't take much:
|
|
194
202
|
|
|
195
|
-
- **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (
|
|
196
|
-
- **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (
|
|
203
|
+
- **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (332 lines) is the entry point for all provider adaptation.
|
|
204
|
+
- **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (32 lines): run tests, fetch a page, call an LSP, whatever. The end of the second essay walks you through your first one by hand.
|
|
197
205
|
- **Rewrite the system prompt.** `prompt.py` is all of 41 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
|
|
198
206
|
- **Import it as a library.** The top level exports `Agent`, `LLM`, and `Config`, ready to embed in your own program:
|
|
199
207
|
|
|
@@ -228,13 +236,13 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
|
228
236
|
quit / exit exit (Ctrl+C cancels the current round)
|
|
229
237
|
```
|
|
230
238
|
|
|
231
|
-
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
239
|
+
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out. Undo history persists the same way: checkpoints land in `~/.corecoder/checkpoints.json`, so `/undo` still reaches back after a restart.
|
|
232
240
|
|
|
233
241
|
## Permissions
|
|
234
242
|
|
|
235
243
|
Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the model asks. The mutating ones (`edit_file`, `write_file`, `bash`, and spawning a sub-agent) stop for consent first, and the REPL banner shows which mode you're in:
|
|
236
244
|
|
|
237
|
-
- In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool
|
|
245
|
+
- In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool and persists in `~/.corecoder/permissions.json` across restarts; a sub-agent inherits the same layer, so consent follows the work wherever it happens.
|
|
238
246
|
- In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
|
|
239
247
|
- The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
|
|
240
248
|
|
|
@@ -255,6 +263,31 @@ Drop a `hooks.json` under `~/.corecoder` and your own shell commands run around
|
|
|
255
263
|
|
|
256
264
|
Each hook gets the call as JSON on stdin (`tool_name`, `tool_input`; post hooks also get `tool_response`). The matcher is an exact tool name; empty or `*` fires on every tool. A pre hook can veto the call with exit code 2, and its stderr travels back to the model as the reason so it can route around the block. Post hooks only observe and can never block. A hook that errors or runs past ten seconds is skipped with a warning: hooks assist the loop, they never get to kill it. The whole mechanism is `hooks.py`, and the REPL banner shows how many hooks loaded.
|
|
257
265
|
|
|
266
|
+
Two worth stealing (the commands lean on `jq`, the usual suspect):
|
|
267
|
+
|
|
268
|
+
```bash
|
|
269
|
+
# 1. lint gate: after every edit/write, run the project's fast linter on the
|
|
270
|
+
# touched file. The model sees the output and fixes its own mistakes in
|
|
271
|
+
# the same turn instead of waiting for CI.
|
|
272
|
+
{
|
|
273
|
+
"PostToolUse": [{
|
|
274
|
+
"matcher": "edit_file",
|
|
275
|
+
"command": "f=$(jq -r .tool_input.file_path); ruff check \"$f\" 2>&1 | head -20"
|
|
276
|
+
}]
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
# 2. write protect: refuse edits under paths you never want an agent to
|
|
280
|
+
# touch. Exit code 2 vetoes the call and the message reaches the model.
|
|
281
|
+
{
|
|
282
|
+
"PreToolUse": [{
|
|
283
|
+
"matcher": "edit_file",
|
|
284
|
+
"command": "case \"$(jq -r .tool_input.file_path)\" in .env*|*/secrets/*|*.pem) echo 'that path is off-limits' >&2; exit 2;; esac"
|
|
285
|
+
}]
|
|
286
|
+
}
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
Both are plain shell; nothing here is CoreCoder-specific syntax beyond the JSON shape and the exit-code-2 veto.
|
|
290
|
+
|
|
258
291
|
## MCP servers
|
|
259
292
|
|
|
260
293
|
Drop a `mcp.json` under `~/.corecoder` and tools from any MCP server join the agent over stdio, the same config shape as Claude Code's:
|
|
@@ -281,7 +314,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
|
|
|
281
314
|
|
|
282
315
|
## Contributing / License
|
|
283
316
|
|
|
284
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
317
|
+
Before you send anything, run `pytest tests/ -q` (215 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
285
318
|
|
|
286
319
|
---
|
|
287
320
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
# CoreCoder
|
|
4
4
|
|
|
5
|
-
**The nanoGPT of coding agents. 1,
|
|
5
|
+
**The nanoGPT of coding agents. A 1.3k-line engine inside 2,735 readable lines of pure Python: understand how a coding agent actually works, then fork your own.**
|
|
6
6
|
|
|
7
7
|
*learn from it · fork it · ship something better*
|
|
8
8
|
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
[](https://python.org)
|
|
13
13
|
[](LICENSE)
|
|
14
14
|
[](https://github.com/he-yufeng/CoreCoder/actions)
|
|
15
|
-
[](article/00-index_EN.md)
|
|
16
16
|
[](article/00-index_EN.md)
|
|
17
17
|
|
|
18
18
|
</div>
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
|
|
26
26
|
| | CoreCoder | Claude Code | aider | nanoGPT |
|
|
27
27
|
|---|---|---|---|---|
|
|
28
|
-
| Lines of code | ~1,
|
|
28
|
+
| Lines of code | ~1,309 engine / 2,735 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
|
|
29
29
|
| Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
|
|
30
30
|
| Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
|
|
31
31
|
| What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
|
|
@@ -36,15 +36,15 @@ The nanoGPT column is there as a reference point: minimal, readable, but it teac
|
|
|
36
36
|
|
|
37
37
|
I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
|
|
38
38
|
|
|
39
|
-
The engine (loop, model interface, context, tools, sessions) is 1,
|
|
39
|
+
The engine (loop, model interface, context, tools, sessions) is 1,309 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 25 files: 2,735 physical lines, 2,205 net, every one short enough to read in a single sitting. The growth since the original 1,161-line snapshot went into visible features: plan mode, hooks and checkpoints, each documented below.
|
|
40
40
|
|
|
41
|
-
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first.
|
|
41
|
+
And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. Anything that would mutate your disk or run a command stops for your consent first. 215 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
|
|
42
42
|
|
|
43
43
|
The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
|
|
44
44
|
|
|
45
45
|
<p align="center">
|
|
46
|
-
<img src="https://raw.githubusercontent.com/he-yufeng/CoreCoder/main/assets/
|
|
47
|
-
alt="
|
|
46
|
+
<img src="https://raw.githubusercontent.com/he-yufeng/CoreCoder/main/assets/demo-plan-hooks.gif" width="760"
|
|
47
|
+
alt="Plan mode in action: the agent reads fib.py, gets its edit refused while plan mode is on, presents a plan, and only after approval edits, tests, and reports — with Pre/PostToolUse hooks firing around every call.">
|
|
48
48
|
</p>
|
|
49
49
|
|
|
50
50
|
<p align="center"><sub><i>These thousand lines really do run a full loop end to end: ask it to fix buggy.py and it reads the file, edits the code, runs it once to confirm, then reports back on its own. Watch it, then come back and read the code.</i></sub></p>
|
|
@@ -72,7 +72,9 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
|
|
|
72
72
|
| OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
|
|
73
73
|
| Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
|
|
74
74
|
|
|
75
|
-
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
75
|
+
Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. Thinking models are first-class too: deepseek-reasoner, kimi-k3 and friends stream their chain-of-thought, and CoreCoder shows it dimmed as it works, kept out of the conversation history so providers never see it come back. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
|
|
76
|
+
|
|
77
|
+
Smoke-tested end to end (read the file, edit it, run it, report back) against DeepSeek, Qwen3 and Kimi K2 via a single OpenRouter-compatible endpoint; each completed the full loop. One note for one-shot scripts: `-p` refuses mutating tools unless you pass `--yes`, by design.
|
|
76
78
|
|
|
77
79
|
```bash
|
|
78
80
|
corecoder # interactive REPL
|
|
@@ -85,26 +87,31 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
|
|
|
85
87
|
|
|
86
88
|
```
|
|
87
89
|
corecoder/
|
|
88
|
-
├── agent.py agent loop + parallel tool exec
|
|
89
|
-
├── llm.py streaming client + retry + cost
|
|
90
|
-
├── context.py three-tier context compaction
|
|
90
|
+
├── agent.py agent loop + parallel tool exec 240 lines ← start here
|
|
91
|
+
├── llm.py streaming client + retry + cost 332 lines
|
|
92
|
+
├── context.py three-tier context compaction 220 lines
|
|
91
93
|
├── session.py save / resume + path-traversal guard 97 lines
|
|
92
|
-
├── permissions.py consent for mutating tools
|
|
93
|
-
├── hooks.py Pre/PostToolUse shell hooks
|
|
94
|
+
├── permissions.py consent for mutating tools 75 lines
|
|
95
|
+
├── hooks.py Pre/PostToolUse shell hooks 87 lines
|
|
96
|
+
├── shell.py POSIX shell routing (Git Bash on Windows) 61 lines
|
|
94
97
|
├── mcp.py MCP stdio client for external tools 208 lines
|
|
95
98
|
├── prompt.py system prompt 41 lines
|
|
96
|
-
├── cli.py REPL + slash commands + one-shot
|
|
99
|
+
├── cli.py REPL + slash commands + one-shot 358 lines
|
|
97
100
|
├── config.py env-var config 55 lines
|
|
101
|
+
├── checkpoints.py /undo snapshot and restore 93 lines
|
|
102
|
+
├── demo.py offline end-to-end demo 101 lines
|
|
98
103
|
└── tools/
|
|
99
|
-
├── bash.py shell + dangerous-command gate + cd
|
|
100
|
-
├── edit.py unique-match search/replace + diff
|
|
104
|
+
├── bash.py shell + dangerous-command gate + cd 203 lines
|
|
105
|
+
├── edit.py unique-match search/replace + diff 99 lines
|
|
101
106
|
├── grep.py content search 93 lines
|
|
102
107
|
├── glob_tool.py filename matching 52 lines
|
|
103
108
|
├── read.py file read 56 lines
|
|
104
|
-
├── write.py file write
|
|
109
|
+
├── write.py file write 46 lines
|
|
105
110
|
├── todo.py agent-maintained task checklist 79 lines
|
|
106
|
-
├── agent.py sub-agent spawning
|
|
107
|
-
└── base.py tool base class
|
|
111
|
+
├── agent.py sub-agent spawning 72 lines
|
|
112
|
+
└── base.py tool base class 32 lines
|
|
113
|
+
examples/
|
|
114
|
+
└── plan_hooks_demo.py offline plan mode + hooks demo (no API key)
|
|
108
115
|
```
|
|
109
116
|
|
|
110
117
|
Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core. If `~/.corecoder/mcp.json` exists, its MCP servers join the eight as extra `mcp__*` tools; the MCP section below covers it.
|
|
@@ -128,7 +135,7 @@ def chat(self, user_input):
|
|
|
128
135
|
return "(hit the round limit)"
|
|
129
136
|
```
|
|
130
137
|
|
|
131
|
-
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the
|
|
138
|
+
That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the engine, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. A stream can also die after it has already started, so the retry wraps the whole request, not just the connect. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
|
|
132
139
|
|
|
133
140
|
Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
|
|
134
141
|
|
|
@@ -142,23 +149,24 @@ Every one of these *whys* is traced down to the actual lines of code in the seri
|
|
|
142
149
|
|
|
143
150
|
## The source-reading series · 8 bilingual essays
|
|
144
151
|
|
|
145
|
-
I also wrote a bilingual source-reading series, one intro plus
|
|
152
|
+
I also wrote a bilingual source-reading series, one intro plus eight parts, each in Chinese with an English mirror. Against CoreCoder's actual code, it walks through how agents like Claude Code work under the hood. One hard rule I set myself: every line count and every snippet is re-read and re-checked from the repo, never written from memory. The first six get you reading, the seventh gets you forking, and the eighth is about extending it without touching the loop; read them in any order.
|
|
146
153
|
|
|
147
154
|
- **[Intro · Read Claude Code through CoreCoder, then build your own](article/00-index_EN.md)**
|
|
148
155
|
- **[01 · An agent, at its core, is a `while` loop](article/01-the-loop_EN.md)** — the main loop in `agent.py`, interrupts, and the round limit
|
|
149
|
-
- **[02 · The tool system: letting the model act, safely](article/02-tools_EN.md)** — the
|
|
156
|
+
- **[02 · The tool system: letting the model act, safely](article/02-tools_EN.md)** — the eight tools in `tools/` and the bash safety gate
|
|
150
157
|
- **[03 · Plug in any LLM, and keep the bill honest](article/03-llm-and-cost_EN.md)** — `llm.py`'s provider wrapper, retries, and cost accounting
|
|
151
158
|
- **[04 · Surviving a long task on a finite window](article/04-context_EN.md)** — `context.py`'s three-tier compaction and orphaned tool messages
|
|
152
159
|
- **[05 · Parallel execution and sub-agents](article/05-parallel-and-subagents_EN.md)** — thread-pool concurrency and sub-agent isolation
|
|
153
160
|
- **[06 · Turning it into a real command-line tool](article/06-session-and-cli_EN.md)** — `session.py` and path-traversal defense
|
|
154
161
|
- **[07 · Fork CoreCoder into your own coding agent](article/07-build-your-own_EN.md)** — from fork to custom tools to swapping models
|
|
162
|
+
- **[08 · Three ways to extend without touching the loop: MCP, hooks, and plan mode](article/08-extensibility_EN.md)** — the v0.6.0 extensibility trio and the contract that makes them safe
|
|
155
163
|
|
|
156
164
|
## Fork it, build something better
|
|
157
165
|
|
|
158
166
|
Once you understand it, the natural next step is to fork. Getting started doesn't take much:
|
|
159
167
|
|
|
160
|
-
- **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (
|
|
161
|
-
- **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (
|
|
168
|
+
- **Swap in a model you actually use.** It's the two env vars from above; `llm.py` (332 lines) is the entry point for all provider adaptation.
|
|
169
|
+
- **Add a tool of your own.** Write a new file against the tool base class in `tools/base.py` (32 lines): run tests, fetch a page, call an LSP, whatever. The end of the second essay walks you through your first one by hand.
|
|
162
170
|
- **Rewrite the system prompt.** `prompt.py` is all of 41 lines; change one line and you'll watch the agent's temperament shift. It's the cheapest "change one thing, see a result" in the whole project.
|
|
163
171
|
- **Import it as a library.** The top level exports `Agent`, `LLM`, and `Config`, ready to embed in your own program:
|
|
164
172
|
|
|
@@ -193,13 +201,13 @@ Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
|
|
|
193
201
|
quit / exit exit (Ctrl+C cancels the current round)
|
|
194
202
|
```
|
|
195
203
|
|
|
196
|
-
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
|
|
204
|
+
Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out. Undo history persists the same way: checkpoints land in `~/.corecoder/checkpoints.json`, so `/undo` still reaches back after a restart.
|
|
197
205
|
|
|
198
206
|
## Permissions
|
|
199
207
|
|
|
200
208
|
Read-only tools (`read_file`, `glob`, `grep`, `todo_write`) run the moment the model asks. The mutating ones (`edit_file`, `write_file`, `bash`, and spawning a sub-agent) stop for consent first, and the REPL banner shows which mode you're in:
|
|
201
209
|
|
|
202
|
-
- In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool
|
|
210
|
+
- In the REPL you get one prompt per call: allow once, always allow this tool, or deny. "Always" is remembered per tool and persists in `~/.corecoder/permissions.json` across restarts; a sub-agent inherits the same layer, so consent follows the work wherever it happens.
|
|
203
211
|
- In one-shot mode (`-p`) there is nobody to ask, so a mutating call is refused on the spot and the refusal goes back to the model as an ordinary tool result: the loop never hangs on input that can't arrive. Pass `--yes` to approve everything up front (scripts, CI).
|
|
204
212
|
- The decision itself is pure logic in `permissions.py`, with the terminal only supplying the prompt callback. You can unit-test consent without a TTY, or reuse the layer in your own embedding.
|
|
205
213
|
|
|
@@ -220,6 +228,31 @@ Drop a `hooks.json` under `~/.corecoder` and your own shell commands run around
|
|
|
220
228
|
|
|
221
229
|
Each hook gets the call as JSON on stdin (`tool_name`, `tool_input`; post hooks also get `tool_response`). The matcher is an exact tool name; empty or `*` fires on every tool. A pre hook can veto the call with exit code 2, and its stderr travels back to the model as the reason so it can route around the block. Post hooks only observe and can never block. A hook that errors or runs past ten seconds is skipped with a warning: hooks assist the loop, they never get to kill it. The whole mechanism is `hooks.py`, and the REPL banner shows how many hooks loaded.
|
|
222
230
|
|
|
231
|
+
Two worth stealing (the commands lean on `jq`, the usual suspect):
|
|
232
|
+
|
|
233
|
+
```bash
|
|
234
|
+
# 1. lint gate: after every edit/write, run the project's fast linter on the
|
|
235
|
+
# touched file. The model sees the output and fixes its own mistakes in
|
|
236
|
+
# the same turn instead of waiting for CI.
|
|
237
|
+
{
|
|
238
|
+
"PostToolUse": [{
|
|
239
|
+
"matcher": "edit_file",
|
|
240
|
+
"command": "f=$(jq -r .tool_input.file_path); ruff check \"$f\" 2>&1 | head -20"
|
|
241
|
+
}]
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
# 2. write protect: refuse edits under paths you never want an agent to
|
|
245
|
+
# touch. Exit code 2 vetoes the call and the message reaches the model.
|
|
246
|
+
{
|
|
247
|
+
"PreToolUse": [{
|
|
248
|
+
"matcher": "edit_file",
|
|
249
|
+
"command": "case \"$(jq -r .tool_input.file_path)\" in .env*|*/secrets/*|*.pem) echo 'that path is off-limits' >&2; exit 2;; esac"
|
|
250
|
+
}]
|
|
251
|
+
}
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
Both are plain shell; nothing here is CoreCoder-specific syntax beyond the JSON shape and the exit-code-2 veto.
|
|
255
|
+
|
|
223
256
|
## MCP servers
|
|
224
257
|
|
|
225
258
|
Drop a `mcp.json` under `~/.corecoder` and tools from any MCP server join the agent over stdio, the same config shape as Claude Code's:
|
|
@@ -246,7 +279,7 @@ If working through CoreCoder was useful, here are a few other tools I've built a
|
|
|
246
279
|
|
|
247
280
|
## Contributing / License
|
|
248
281
|
|
|
249
|
-
Before you send anything, run `pytest tests/ -q` (
|
|
282
|
+
Before you send anything, run `pytest tests/ -q` (215 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
|
|
250
283
|
|
|
251
284
|
---
|
|
252
285
|
|