corecoder 0.4.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. {corecoder-0.4.0 → corecoder-0.4.1}/PKG-INFO +32 -20
  2. {corecoder-0.4.0 → corecoder-0.4.1}/README.md +30 -18
  3. {corecoder-0.4.0 → corecoder-0.4.1}/README_CN.md +30 -18
  4. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/__init__.py +3 -3
  5. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/agent.py +17 -5
  6. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/cli.py +19 -11
  7. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/context.py +9 -9
  8. corecoder-0.4.1/corecoder/demo.py +84 -0
  9. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/llm.py +26 -1
  10. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/__init__.py +5 -3
  11. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/agent.py +5 -2
  12. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/bash.py +6 -2
  13. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/edit.py +4 -2
  14. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/glob_tool.py +7 -2
  15. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/grep.py +19 -5
  16. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/read.py +5 -2
  17. corecoder-0.4.1/corecoder/tools/todo.py +79 -0
  18. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/write.py +5 -2
  19. {corecoder-0.4.0 → corecoder-0.4.1}/pyproject.toml +1 -1
  20. {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_core.py +50 -6
  21. corecoder-0.4.1/tests/test_demo.py +39 -0
  22. {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_litellm.py +1 -2
  23. {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_session.py +1 -1
  24. {corecoder-0.4.0 → corecoder-0.4.1}/tests/test_tools.py +106 -3
  25. {corecoder-0.4.0 → corecoder-0.4.1}/.github/workflows/ci.yml +0 -0
  26. {corecoder-0.4.0 → corecoder-0.4.1}/.github/workflows/publish.yml +0 -0
  27. {corecoder-0.4.0 → corecoder-0.4.1}/.gitignore +0 -0
  28. {corecoder-0.4.0 → corecoder-0.4.1}/LICENSE +0 -0
  29. {corecoder-0.4.0 → corecoder-0.4.1}/article/00-index.md +0 -0
  30. {corecoder-0.4.0 → corecoder-0.4.1}/article/00-index_EN.md +0 -0
  31. {corecoder-0.4.0 → corecoder-0.4.1}/article/01-the-loop.md +0 -0
  32. {corecoder-0.4.0 → corecoder-0.4.1}/article/01-the-loop_EN.md +0 -0
  33. {corecoder-0.4.0 → corecoder-0.4.1}/article/02-tools.md +0 -0
  34. {corecoder-0.4.0 → corecoder-0.4.1}/article/02-tools_EN.md +0 -0
  35. {corecoder-0.4.0 → corecoder-0.4.1}/article/03-llm-and-cost.md +0 -0
  36. {corecoder-0.4.0 → corecoder-0.4.1}/article/03-llm-and-cost_EN.md +0 -0
  37. {corecoder-0.4.0 → corecoder-0.4.1}/article/04-context.md +0 -0
  38. {corecoder-0.4.0 → corecoder-0.4.1}/article/04-context_EN.md +0 -0
  39. {corecoder-0.4.0 → corecoder-0.4.1}/article/05-parallel-and-subagents.md +0 -0
  40. {corecoder-0.4.0 → corecoder-0.4.1}/article/05-parallel-and-subagents_EN.md +0 -0
  41. {corecoder-0.4.0 → corecoder-0.4.1}/article/06-session-and-cli.md +0 -0
  42. {corecoder-0.4.0 → corecoder-0.4.1}/article/06-session-and-cli_EN.md +0 -0
  43. {corecoder-0.4.0 → corecoder-0.4.1}/article/07-build-your-own.md +0 -0
  44. {corecoder-0.4.0 → corecoder-0.4.1}/article/07-build-your-own_EN.md +0 -0
  45. {corecoder-0.4.0 → corecoder-0.4.1}/assets/demo.png +0 -0
  46. {corecoder-0.4.0 → corecoder-0.4.1}/assets/demo_en.png +0 -0
  47. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/__main__.py +0 -0
  48. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/config.py +0 -0
  49. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/prompt.py +0 -0
  50. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/session.py +0 -0
  51. {corecoder-0.4.0 → corecoder-0.4.1}/corecoder/tools/base.py +0 -0
  52. {corecoder-0.4.0 → corecoder-0.4.1}/tests/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: corecoder
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: Minimal AI coding agent (~1,000 lines of Python) inspired by Claude Code. Works with any LLM. (formerly NanoCoder)
5
5
  Project-URL: Homepage, https://github.com/he-yufeng/CoreCoder
6
6
  Project-URL: Repository, https://github.com/he-yufeng/CoreCoder
@@ -52,17 +52,28 @@ Description-Content-Type: text/markdown
52
52
 
53
53
  </div>
54
54
 
55
- - **Readable end to end.** Read the whole engine in an afternoon: 1,081 lines of pure Python, with no magic hidden anywhere you can't follow it.
56
- - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram. It just isn't meant to be your daily driver.
55
+ - **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
56
+ - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
57
57
  - **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
58
58
 
59
+ ## How it compares
60
+
61
+ | | CoreCoder | Claude Code | aider | nanoGPT |
62
+ |---|---|---|---|---|
63
+ | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
64
+ | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
65
+ | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
66
+ | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
67
+
68
+ The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
69
+
59
70
  ## What this is
60
71
 
61
72
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
62
73
 
63
74
  The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
64
75
 
65
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 86 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
76
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
66
77
 
67
78
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
68
79
 
@@ -93,6 +104,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
93
104
  |---|---|
94
105
  | OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
95
106
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
107
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
96
108
  | Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
97
109
 
98
110
  Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
@@ -108,7 +120,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
108
120
 
109
121
  ```
110
122
  corecoder/
111
- ├── agent.py agent loop + parallel tool exec 150 lines ← start here
123
+ ├── agent.py agent loop + parallel tool exec 162 lines ← start here
112
124
  ├── llm.py streaming client + retry + cost 336 lines
113
125
  ├── context.py three-tier context compaction 210 lines
114
126
  ├── session.py save / resume + path-traversal guard 97 lines
@@ -122,11 +134,12 @@ corecoder/
122
134
  ├── glob_tool.py filename matching 47 lines
123
135
  ├── read.py file read 53 lines
124
136
  ├── write.py file write 38 lines
137
+ ├── todo.py agent-maintained task checklist 79 lines
125
138
  ├── agent.py sub-agent spawning 58 lines
126
139
  └── base.py tool base class 27 lines
127
140
  ```
128
141
 
129
- Seven tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, and `agent` (which spawns a sub-agent). Add the packaging files like `__init__` and `__main__` and the package is 18 files; strip the CLI shell and config and the engine itself is about 1,081 lines.
142
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
130
143
 
131
144
  ## A `while` loop is the whole agent
132
145
 
@@ -147,7 +160,7 @@ def chat(self, user_input):
147
160
  return "(hit the round limit)"
148
161
  ```
149
162
 
150
- That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess that shows up once the loop runs against the real world. `llm.py` is the biggest file in the project, not because calling a model is hard, but because in a streamed response a single tool call's arguments arrive in several fragments, one after another, and you have to stitch them back together in order. A provider will occasionally hand you half a JSON object, or fill the `usage` field with null; rate limits (429), timeouts, dropped connections and 5xx all need backoff-and-retry, while the other 4xx should just raise instead of being retried into the ground. Even an OpenAI extension like `stream_options` gets handled: some providers reject it outright with a 400, so the code strips it and resends exactly once, only on a 400, and never stacks that on top of the retry backoff. Lay all this grunt work out and it's the genuinely hard engineering part of taking an agent from demo to delivery. The unglamorous part turns out to be the loop itself; the thousand lines of fallback around it are where the real work lives.
163
+ That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
151
164
 
152
165
  Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
153
166
 
@@ -197,17 +210,6 @@ Going deeper, the directions are out in the open too. None of the following is i
197
210
 
198
211
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
199
212
 
200
- ## How it compares
201
-
202
- | | CoreCoder | Claude Code | aider | nanoGPT |
203
- |---|---|---|---|---|
204
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
205
- | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
206
- | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
207
- | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
208
-
209
- The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
210
-
211
213
  ## Commands
212
214
 
213
215
  Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
@@ -223,9 +225,19 @@ quit / exit exit (Ctrl+C cancels the current round)
223
225
 
224
226
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
225
227
 
228
+ ## Related Projects
229
+
230
+ If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
231
+
232
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
233
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
234
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
235
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
236
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
237
+
226
238
  ## Contributing / License
227
239
 
228
- Before you send anything, run `pytest tests/ -q` (86 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
240
+ Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
229
241
 
230
242
  ---
231
243
 
@@ -17,17 +17,28 @@
17
17
 
18
18
  </div>
19
19
 
20
- - **Readable end to end.** Read the whole engine in an afternoon: 1,081 lines of pure Python, with no magic hidden anywhere you can't follow it.
21
- - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram. It just isn't meant to be your daily driver.
20
+ - **Readable end to end.** Read the whole engine in an afternoon, with no magic hidden anywhere you can't follow it.
21
+ - **Hackable.** Set a breakpoint on any line, change it, rerun, all on your own machine. It genuinely works, which makes this a living reference rather than a diagram.
22
22
  - **The gaps are the point.** It deliberately keeps only the minimal core; what's missing isn't half-finished, it's where you branch off and make it your own.
23
23
 
24
+ ## How it compares
25
+
26
+ | | CoreCoder | Claude Code | aider | nanoGPT |
27
+ |---|---|---|---|---|
28
+ | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
29
+ | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
30
+ | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
31
+ | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
32
+
33
+ The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
34
+
24
35
  ## What this is
25
36
 
26
37
  I've always felt coding agents get talked about as if they were arcane. Strip a tool like Claude Code or Cursor all the way down and the core is a `while` loop wrapped around a large model, plus seven or eight tools that let it actually do things. The hard part was never the loop; it's everything the loop has to cope with once it meets the real world. CoreCoder is the minimal version that writes that core out honestly.
27
38
 
28
39
  The engine (loop, model interface, context, tools, sessions) is 1,081 lines once you drop blank lines and comments. Counting the outer CLI, config and packaging too, the whole package is 18 files: 1,714 physical lines, 1,385 net, every one short enough to read in a single sitting.
29
40
 
30
- And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 86 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
41
+ And it really runs: reads and writes files, executes shell, spawns sub-agents, compacts context in three tiers, and tells you the tokens and dollars a run burned whenever you ask. 103 tests, all green. But the point of it running isn't to become your daily driver. It runs so the walkthrough can't lie: a reference that shows how an agent works has to actually work.
31
42
 
32
43
  The code came out of a public teardown: open analyses have already exposed a lot of the load-bearing architecture inside production agents like Claude Code. I took the most essential layer and rewrote it honestly, in as little code as I could. So reading CoreCoder is roughly like reading a runnable, annotated take on how that kind of agent works, except it's only a minimal reimplementation, sitting right there on your machine for you to take apart and change.
33
44
 
@@ -58,6 +69,7 @@ Give it a model and a key and it goes. It speaks the OpenAI-compatible API by de
58
69
  |---|---|
59
70
  | OpenAI (default `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
60
71
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
72
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
61
73
  | Local Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
62
74
 
63
75
  Kimi, Qwen and the like are the same two variables; for providers that don't even offer an OpenAI-compatible endpoint, the optional LiteLLM backend (`pip install "corecoder[litellm]"`) routes to a hundred-plus of them. The third essay goes into this in detail. The key can be `export`ed directly or dropped into a `.env` at the project root, which is loaded on startup. Then:
@@ -73,7 +85,7 @@ Laid out flat, the whole project is this big. Skim it before you clone and you'l
73
85
 
74
86
  ```
75
87
  corecoder/
76
- ├── agent.py agent loop + parallel tool exec 150 lines ← start here
88
+ ├── agent.py agent loop + parallel tool exec 162 lines ← start here
77
89
  ├── llm.py streaming client + retry + cost 336 lines
78
90
  ├── context.py three-tier context compaction 210 lines
79
91
  ├── session.py save / resume + path-traversal guard 97 lines
@@ -87,11 +99,12 @@ corecoder/
87
99
  ├── glob_tool.py filename matching 47 lines
88
100
  ├── read.py file read 53 lines
89
101
  ├── write.py file write 38 lines
102
+ ├── todo.py agent-maintained task checklist 79 lines
90
103
  ├── agent.py sub-agent spawning 58 lines
91
104
  └── base.py tool base class 27 lines
92
105
  ```
93
106
 
94
- Seven tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, and `agent` (which spawns a sub-agent). Add the packaging files like `__init__` and `__main__` and the package is 18 files; strip the CLI shell and config and the engine itself is about 1,081 lines.
107
+ Eight tools: `bash`, `read_file`, `write_file`, `edit_file`, `glob`, `grep`, `todo_write` (a task checklist the agent maintains for itself), and `agent` (which spawns a sub-agent). Everything else is the CLI shell, config, and packaging wrapped around that engine core.
95
108
 
96
109
  ## A `while` loop is the whole agent
97
110
 
@@ -112,7 +125,7 @@ def chat(self, user_input):
112
125
  return "(hit the round limit)"
113
126
  ```
114
127
 
115
- That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess that shows up once the loop runs against the real world. `llm.py` is the biggest file in the project, not because calling a model is hard, but because in a streamed response a single tool call's arguments arrive in several fragments, one after another, and you have to stitch them back together in order. A provider will occasionally hand you half a JSON object, or fill the `usage` field with null; rate limits (429), timeouts, dropped connections and 5xx all need backoff-and-retry, while the other 4xx should just raise instead of being retried into the ground. Even an OpenAI extension like `stream_options` gets handled: some providers reject it outright with a 400, so the code strips it and resends exactly once, only on a 400, and never stacks that on top of the retry backoff. Lay all this grunt work out and it's the genuinely hard engineering part of taking an agent from demo to delivery. The unglamorous part turns out to be the loop itself; the thousand lines of fallback around it are where the real work lives.
128
+ That's the whole thing. The core skeleton is about twenty lines; counting parallel execution and the bookkeeping after a Ctrl+C interrupt, maybe forty. Almost everything else in CoreCoder's thousand-odd lines is there to clean up the mess the loop runs into once it meets the real world. `llm.py` ends up the biggest file in the project, not because calling a model is hard, but because a streamed response splinters each tool call's arguments into fragments you have to restitch in order, a provider will hand you half a JSON object or a null `usage` field, and 429s, timeouts, dropped connections and 5xx all need backoff-and-retry while the other 4xx should just raise. That unglamorous grunt work, not the loop, is where the real engineering of taking an agent from demo to delivery actually lives; the third essay follows it down to the line.
116
129
 
117
130
  Three decisions are worth a closer look, because they're the kind of call you can only make after you've understood how others did it, and they're judgments you can lift straight into your own fork.
118
131
 
@@ -162,17 +175,6 @@ Going deeper, the directions are out in the open too. None of the following is i
162
175
 
163
176
  The README only points; the seventh essay picks up the code details for each. Pick one and start; that's the whole reason the core is kept this small.
164
177
 
165
- ## How it compares
166
-
167
- | | CoreCoder | Claude Code | aider | nanoGPT |
168
- |---|---|---|---|---|
169
- | Lines of code | ~1,081 engine / 1,714 total | hundreds of thousands (closed) | tens of thousands of Python | ~600 (two files) |
170
- | Time to read it all | one afternoon | can't (closed) | a few days of slogging | one afternoon |
171
- | Breakpoint, change, rerun? | yes, every line | no | yes, but there's a lot | yes |
172
- | What it's for | understand one, then fork your own | production coding assistant | terminal pair-programming | minimal GPT for teaching |
173
-
174
- The nanoGPT column is there as a reference point: minimal, readable, but it teaches you to train a GPT. CoreCoder is after the same thing, only the subject is an agent that actually edits code. Sitting it next to Claude Code and aider isn't about competing for their users. CoreCoder is the foundation you stand on while you learn from them and get going; it isn't in the same race.
175
-
176
178
  ## Commands
177
179
 
178
180
  Inside the REPL, `/help` lists everything; these are the ones you'll reach for:
@@ -188,9 +190,19 @@ quit / exit exit (Ctrl+C cancels the current round)
188
190
 
189
191
  Session IDs are sanitized to safe characters before they become filenames, every archive lands under `~/.corecoder/sessions`, and a malicious session name can't traverse out.
190
192
 
193
+ ## Related Projects
194
+
195
+ If working through CoreCoder was useful, here are a few other tools I've built around agents and LLM systems:
196
+
197
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — dropped into an unfamiliar codebase? It gives you a guided wiki and a where-to-start reading path, a self-hostable DeepWiki alternative.
198
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — stop sifting job boards by hand: it ranks postings against your resume and runs mock interviews.
199
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — catch the risky clauses before you sign: it reads contracts and flags the dangerous bits.
200
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — want to contribute to open source? It finds issues worth your time and gauges whether your PR will get merged.
201
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — understand any codebase even if you don't code, built for non-programmers.
202
+
191
203
  ## Contributing / License
192
204
 
193
- Before you send anything, run `pytest tests/ -q` (86 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
205
+ Before you send anything, run `pytest tests/ -q` (103 tests), `ruff check`, and `compileall`, and make sure they're green. MIT licensed: fork it, learn from it, ship something better. A mention of this project is appreciated.
194
206
 
195
207
  ---
196
208
 
@@ -17,17 +17,28 @@
17
17
 
18
18
  </div>
19
19
 
20
- - **读得完。** 一个下午读完整个引擎,1081 行纯 Python,没有一处藏着你看不懂的魔法。
21
- - **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图;但它不是拿来当日常工具的。
20
+ - **读得完。** 一个下午读完整个引擎,没有一处藏着你看不懂的魔法。
21
+ - **改得动。** 每一行都能在你自己机器上下断点、改了再跑。它真能干活,所以这份参考是活的,不是示意图。
22
22
  - **留白即起点。** 刻意只留最小核心,没做的那些不是半成品,是留给你 fork 出更好东西的地方。
23
23
 
24
+ ## 和谁比
25
+
26
+ | | CoreCoder | Claude Code | aider | nanoGPT |
27
+ |---|---|---|---|---|
28
+ | 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
29
+ | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
30
+ | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
31
+ | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
32
+
33
+ nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
34
+
24
35
  ## 这是什么
25
36
 
26
37
  我一直觉得 coding agent 被讲得太玄了。把 Claude Code、Cursor 这类工具扒到底,核心是一个 while 循环套着一个大模型,外加七八个让它能真正动手的工具。难的从来不是这个循环,而是循环跑进真实世界以后要兜的那些底。CoreCoder 就是把这个核心老老实实写出来的最小版本。
27
38
 
28
39
  引擎部分(循环、模型接口、上下文、工具、会话)去掉空行和注释是 1081 行。连最外层的 CLI、配置、打包一起算,整个包 18 个文件、物理 1714 行、净 1385 行,每个文件都短到能一口气读完。
29
40
 
30
- 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,86 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
41
+ 它真能跑:读写文件、执行 shell、派子 agent、分三层压上下文,还能随时把这趟烧掉的 token 和美元数报给你,103 个测试是绿的。但能跑不是为了劝你拿去日用,而是为了让这份「注释」不撒谎:一个解释 agent 怎么运作的范例,自己得真能运作。
31
42
 
32
43
  代码来自一次公开拆解。公开的源码分析里,Claude Code 这类生产级 agent 暴露出不少关键架构,我挑出最核心的一层,用尽量少的代码诚实地复写了一遍。所以读 CoreCoder,约等于读一份基于公开源码分析的「可运行注释版」:讲的是这类 agent 的核心思路,而它本身只是最小复写,就摆在你机器上,随你拆、随你改。
33
44
 
@@ -58,6 +69,7 @@ pip install -e .
58
69
  |---|---|
59
70
  | OpenAI(默认 `gpt-5.5`) | `OPENAI_API_KEY=sk-...` |
60
71
  | DeepSeek | `OPENAI_API_KEY=sk-... OPENAI_BASE_URL=https://api.deepseek.com CORECODER_MODEL=deepseek-chat` |
72
+ | OmniRoute | `OPENAI_API_KEY=your-key OPENAI_BASE_URL=http://localhost:20128/v1 CORECODER_MODEL=auto` |
61
73
  | 本地 Ollama | `OPENAI_API_KEY=ollama OPENAI_BASE_URL=http://localhost:11434/v1 CORECODER_MODEL=qwen2.5-coder` |
62
74
 
63
75
  Kimi、Qwen 这些同样是改这两个变量;连 OpenAI 兼容接口都不给的 provider,装上可选的 LiteLLM 后端(`pip install "corecoder[litellm]"`)能路由一百多家。第三篇文章把这块讲得更细。key 可以直接 `export`,也可以在项目根目录扔个 `.env`,启动时自动加载。然后:
@@ -73,7 +85,7 @@ corecoder -p "给 parse_config() 加错误处理" # 一次性模式,干完
73
85
 
74
86
  ```
75
87
  corecoder/
76
- ├── agent.py agent 主循环 + 并行工具执行 150 行 ← 从这里开始读
88
+ ├── agent.py agent 主循环 + 并行工具执行 162 行 ← 从这里开始读
77
89
  ├── llm.py 流式客户端 + 重试 + 成本统计 336 行
78
90
  ├── context.py 三层上下文压缩 210 行
79
91
  ├── session.py 会话存盘 / 续聊 + 路径穿越防护 97 行
@@ -87,11 +99,12 @@ corecoder/
87
99
  ├── glob_tool.py 文件名匹配 47 行
88
100
  ├── read.py 文件读取 53 行
89
101
  ├── write.py 文件写入 38 行
102
+ ├── todo.py agent 自维护的任务清单 79 行
90
103
  ├── agent.py 子 agent 派生 58 行
91
104
  └── base.py 工具基类 27 行
92
105
  ```
93
106
 
94
- 七个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`agent`(派子 agent)。再加上 `__init__`、`__main__` 这些打包文件,整包 18 个文件;剥掉 CLI 外壳和配置,引擎本身约 1081 行。
107
+ 八个工具:`bash`、`read_file`、`write_file`、`edit_file`、`glob`、`grep`、`todo_write`(agent 自己维护的任务清单)、`agent`(派子 agent)。其余都是包在引擎核心外面的 CLI 外壳、配置和打包。
95
108
 
96
109
  ## 一个 while 循环就是 agent 的本体
97
110
 
@@ -112,7 +125,7 @@ def chat(self, user_input):
112
125
  return "(已达轮次上限)"
113
126
  ```
114
127
 
115
- 就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 是全项目最大的文件,不是因为调模型有多难,而是流式返回里,一个工具调用的参数会被切成好几段先后送到,你得按顺序拼回去;provider 偶尔吐半截 JSON,或者把 usage 字段填成 null;限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。连 `stream_options` 这种 OpenAI 扩展都照顾到了:有的 provider 直接回 400 拒收,代码就只在 400 时把它摘掉重发一次,绝不和退避重试叠上去。这些脏活摊开来,恰恰是一个 agent 从能演示走到能交付,工程上最费劲的那部分。祛魅祛的是那个循环,不是这一千行兜底。
128
+ 就这么点。这个循环的核心骨架就二十来行,把并行执行和被 Ctrl+C 打断后的回填都算上,也才四十多行。CoreCoder 一千多行里剩下的,几乎全在收拾它真跑起来之后冒出来的岔子。`llm.py` 最后成了全项目最大的文件,不是因为调模型有多难,而是流式返回里一个工具调用的参数会被切成好几段先后送到、得按顺序拼回去,provider 偶尔吐半截 JSON 或把 usage 填成 null,限流(429)、超时、连接中断和 5xx 都得退避重试,其余 4xx 该直接抛就别硬试。这些不起眼的脏活,而不是那个循环,才是一个 agent 从能演示走到能交付真正吃工程功夫的地方;第三篇文章顺着它拆到每一行。
116
129
 
117
130
  有三个决定值得单独看,因为它们是「先读懂别人怎么做」之后才做得出的取舍,也是你 fork 自己 agent 时可以直接抄走的判断。
118
131
 
@@ -162,17 +175,6 @@ print(Agent(llm=llm).chat("找出项目里所有 TODO 注释并列出来"))
162
175
 
163
176
  README 只给方向,每条的代码细节第七篇接着讲。挑一个动手,就是把它做得更好的开始。
164
177
 
165
- ## 和谁比
166
-
167
- | | CoreCoder | Claude Code | aider | nanoGPT |
168
- |---|---|---|---|---|
169
- | 代码量 | 引擎约 1081 行 / 整包 1714 行 | 几十万行(闭源) | 数万行 Python | 约 600 行(两个文件) |
170
- | 读完要多久 | 一个下午 | 读不了(闭源) | 得啃几天 | 一个下午 |
171
- | 能不能下断点改了再跑 | 能,每一行 | 不能 | 能,但量大 | 能 |
172
- | 定位 | 读懂并 fork 出你自己的 agent | 生产级编程助手 | 终端结对编程 | 教学用最小 GPT |
173
-
174
- nanoGPT 那一列是拿来对照的:它最小、可读,但教的是训一个 GPT。CoreCoder 想干的是同一件事,只是把对象换成一个能真正改代码的 agent。和 Claude Code、aider 摆在一起,不是要跟它们抢用户,CoreCoder 是借它们来学、来起步的那块地基,根本不在一个赛道。
175
-
176
178
  ## 命令
177
179
 
178
180
  进了 REPL,`/help` 列全部,常用的这几个:
@@ -188,9 +190,19 @@ quit / exit 退出(Ctrl+C 取消当前回合)
188
190
 
189
191
  会话 ID 会先清洗成安全字符再拿去当文件名,存档统统落在 `~/.corecoder/sessions` 里,恶意会话名穿越不出去。
190
192
 
193
+ ## 相关项目
194
+
195
+ 如果你读 CoreCoder 读得还顺,下面几个我做的 agent / LLM 系统方向的工具也许用得上:
196
+
197
+ - **[RepoWiki](https://github.com/he-yufeng/RepoWiki)** — 被丢进一个陌生代码库?它给你一份带「从哪读起」路径的 wiki,一个可自托管的 DeepWiki 替代。
198
+ - **[FindJobs-Agent](https://github.com/he-yufeng/FindJobs-Agent)** — 别再手动刷招聘网站:它按你的简历给岗位排序,还能跑模拟面试。
199
+ - **[ContractGuard](https://github.com/he-yufeng/ContractGuard)** — 签字前先把有风险的条款挑出来:它读合同、标出危险点。
200
+ - **[GitSense](https://github.com/he-yufeng/GitSense)** — 想给开源做贡献?它帮你找到值得做的 issue,还能估你的 PR 多大概率被合。
201
+ - **[CodeABC](https://github.com/he-yufeng/CodeABC)** — 不会写代码也能看懂一个项目,专给小白做的。
202
+
191
203
  ## 贡献 / License
192
204
 
193
- 动手之前先跑一遍 `pytest tests/ -q`(86 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
205
+ 动手之前先跑一遍 `pytest tests/ -q`(103 个测试)、`ruff check` 和 `compileall`,绿了再提。MIT License,欢迎 fork 拿去造更好的东西,能在 README 里留一句出处就更好。
194
206
 
195
207
  ---
196
208
 
@@ -1,10 +1,10 @@
1
1
  """CoreCoder - Minimal AI coding agent inspired by Claude Code's architecture."""
2
2
 
3
- __version__ = "0.4.0"
3
+ __version__ = "0.4.1"
4
4
 
5
5
  from corecoder.agent import Agent
6
- from corecoder.llm import LLM
7
6
  from corecoder.config import Config
7
+ from corecoder.llm import LLM
8
8
  from corecoder.tools import ALL_TOOLS
9
9
 
10
- __all__ = ["Agent", "LLM", "Config", "ALL_TOOLS", "__version__"]
10
+ __all__ = ["ALL_TOOLS", "LLM", "Agent", "Config", "__version__"]
@@ -11,12 +11,14 @@ which means it's done working and ready to report back.
11
11
 
12
12
  import concurrent.futures
13
13
  import inspect
14
+
15
+ from .context import ContextManager
14
16
  from .llm import LLM
17
+ from .prompt import system_prompt
15
18
  from .tools import ALL_TOOLS
16
- from .tools.base import Tool
17
19
  from .tools.agent import AgentTool
18
- from .prompt import system_prompt
19
- from .context import ContextManager
20
+ from .tools.base import Tool
21
+ from .tools.todo import TodoWriteTool
20
22
 
21
23
 
22
24
  class Agent:
@@ -40,8 +42,17 @@ class Agent:
40
42
  if isinstance(t, AgentTool):
41
43
  t._parent_agent = self
42
44
 
45
+ self._todo = next((t for t in self.tools if isinstance(t, TodoWriteTool)), None)
46
+
43
47
  def _full_messages(self) -> list[dict]:
44
- return [{"role": "system", "content": self._system}] + self.messages
48
+ system = self._system
49
+ # the task list is re-injected every round, so the model always sees the
50
+ # current state rather than a stale copy buried in old tool results
51
+ if self._todo is not None:
52
+ rendered = self._todo.render()
53
+ if rendered:
54
+ system += "\n\n# Current task list\n" + rendered
55
+ return [{"role": "system", "content": system}] + self.messages
45
56
 
46
57
  def _tool_schemas(self) -> list[dict]:
47
58
  return [t.schema() for t in self.tools]
@@ -109,9 +120,10 @@ class Agent:
109
120
  inspect.signature(tool.execute).bind(**tc.arguments)
110
121
  except TypeError as e:
111
122
  return f"Error: bad arguments for {tc.name}: {e}"
123
+ # a tool that blows up gets reported back as text, never kills the loop
112
124
  try:
113
125
  return tool.execute(**tc.arguments)
114
- except Exception as e:
126
+ except Exception as e: # noqa: BLE001
115
127
  return f"Error executing {tc.name}: {e}"
116
128
 
117
129
  def _exec_tools_parallel(self, tool_calls, on_tool=None) -> list[str]:
@@ -1,21 +1,21 @@
1
1
  """Interactive REPL - the user-facing terminal interface."""
2
2
 
3
- import sys
4
- import os
5
3
  import argparse
4
+ import os
5
+ import sys
6
6
 
7
- from rich.console import Console
8
- from rich.markdown import Markdown
9
- from rich.panel import Panel
10
7
  from prompt_toolkit import prompt as pt_prompt
11
8
  from prompt_toolkit.history import FileHistory
12
9
  from prompt_toolkit.key_binding import KeyBindings
10
+ from rich.console import Console
11
+ from rich.markdown import Markdown
12
+ from rich.panel import Panel
13
13
 
14
+ from . import __version__
14
15
  from .agent import Agent
15
- from .llm import LLM, LiteLLM
16
16
  from .config import Config
17
- from .session import save_session, load_session, list_sessions
18
- from . import __version__
17
+ from .llm import LLM, LiteLLM
18
+ from .session import list_sessions, load_session, save_session
19
19
 
20
20
  console = Console()
21
21
 
@@ -29,6 +29,7 @@ def _parse_args():
29
29
  p.add_argument("--base-url", help="API base URL (default: $OPENAI_BASE_URL)")
30
30
  p.add_argument("--api-key", help="API key (default: $OPENAI_API_KEY)")
31
31
  p.add_argument("-p", "--prompt", help="One-shot prompt (non-interactive mode)")
32
+ p.add_argument("--demo", action="store_true", help="Run the offline scripted demo (no API key needed)")
32
33
  p.add_argument("-r", "--resume", metavar="ID", help="Resume a saved session")
33
34
  p.add_argument("-v", "--version", action="version", version=f"%(prog)s {__version__}")
34
35
  return p.parse_args()
@@ -36,6 +37,11 @@ def _parse_args():
36
37
 
37
38
  def main():
38
39
  args = _parse_args()
40
+
41
+ if args.demo:
42
+ from .demo import run_demo
43
+ raise SystemExit(run_demo())
44
+
39
45
  config = Config.from_env()
40
46
 
41
47
  # CLI args override env vars
@@ -108,7 +114,8 @@ def _run_once(agent: Agent, prompt: str):
108
114
  except KeyboardInterrupt:
109
115
  console.print("\n[yellow]Interrupted.[/yellow]")
110
116
  sys.exit(130)
111
- except Exception as e:
117
+ except Exception as e: # noqa: BLE001
118
+ # one-shot mode: print whatever went wrong and exit non-zero
112
119
  console.print(f"\n[red]Error: {e}[/red]")
113
120
  sys.exit(1)
114
121
  print()
@@ -223,7 +230,7 @@ def _repl(agent: Agent, config: Config):
223
230
  # call the agent
224
231
  streamed: list[str] = []
225
232
 
226
- def on_token(tok):
233
+ def on_token(tok, streamed=streamed):
227
234
  streamed.append(tok)
228
235
  print(tok, end="", flush=True)
229
236
 
@@ -239,7 +246,8 @@ def _repl(agent: Agent, config: Config):
239
246
  console.print(Markdown(response))
240
247
  except KeyboardInterrupt:
241
248
  console.print("\n[yellow]Interrupted.[/yellow]")
242
- except Exception as e:
249
+ except Exception as e: # noqa: BLE001
250
+ # keep the REPL alive no matter what chat() throws
243
251
  console.print(f"\n[red]Error: {e}[/red]")
244
252
 
245
253
 
@@ -13,6 +13,7 @@ CoreCoder implements the same idea in 3 layers:
13
13
  """
14
14
 
15
15
  from __future__ import annotations
16
+
16
17
  from typing import TYPE_CHECKING
17
18
 
18
19
  if TYPE_CHECKING:
@@ -48,16 +49,14 @@ class ContextManager:
48
49
  compressed = False
49
50
 
50
51
  # Layer 1: snip verbose tool outputs
51
- if current > self._snip_at:
52
- if self._snip_tool_outputs(messages):
53
- compressed = True
54
- current = estimate_tokens(messages)
52
+ if current > self._snip_at and self._snip_tool_outputs(messages):
53
+ compressed = True
54
+ current = estimate_tokens(messages)
55
55
 
56
56
  # Layer 2: LLM-powered summarization of old turns
57
- if current > self._summarize_at and len(messages) > 10:
58
- if self._summarize_old(messages, llm, keep_recent=8):
59
- compressed = True
60
- current = estimate_tokens(messages)
57
+ if current > self._summarize_at and len(messages) > 10 and self._summarize_old(messages, llm, keep_recent=8):
58
+ compressed = True
59
+ current = estimate_tokens(messages)
61
60
 
62
61
  # Layer 3: hard collapse - last resort
63
62
  if current > self._collapse_at and len(messages) > 4:
@@ -169,7 +168,8 @@ class ContextManager:
169
168
  ],
170
169
  )
171
170
  return resp.content
172
- except Exception:
171
+ except Exception: # noqa: BLE001, S110
172
+ # summarization is best-effort; fall back to extraction below
173
173
  pass
174
174
 
175
175
  # fallback: extract key lines
@@ -0,0 +1,84 @@
1
+ """Offline scripted demo: watch the full agent loop with no API key.
2
+
3
+ Runs one scripted task through the real Agent loop with a ScriptedLLM, so the
4
+ thought -> tool call -> observation cycle renders exactly as it would against
5
+ a live model. Useful for demos, screenshots, and as a smoke test of the loop.
6
+ """
7
+
8
+ import tempfile
9
+ from pathlib import Path
10
+
11
+ from rich.console import Console
12
+ from rich.markdown import Markdown
13
+ from rich.panel import Panel
14
+
15
+ from .agent import Agent
16
+ from .llm import LLMResponse, ScriptedLLM, ToolCall
17
+
18
+ console = Console()
19
+
20
+ _TASK = (
21
+ "Write a Python function `fib(n)` in fib.py returning the nth Fibonacci "
22
+ "number, add a pytest file, then run the tests."
23
+ )
24
+
25
+
26
+ def _script(workdir: Path) -> list[LLMResponse]:
27
+ fib_py = (
28
+ "def fib(n):\n"
29
+ " if n < 2:\n"
30
+ " return n\n"
31
+ " a, b = 0, 1\n"
32
+ " for _ in range(2, n + 1):\n"
33
+ " a, b = b, a + b\n"
34
+ " return b\n"
35
+ )
36
+ test_py = (
37
+ "from fib import fib\n"
38
+ "\n"
39
+ "def test_fib():\n"
40
+ " assert fib(0) == 0\n"
41
+ " assert fib(1) == 1\n"
42
+ " assert fib(10) == 55\n"
43
+ )
44
+ return [
45
+ LLMResponse(
46
+ content="I'll write fib.py with an iterative implementation.",
47
+ tool_calls=[ToolCall(id="c1", name="write_file", arguments={"file_path": str(workdir / "fib.py"), "content": fib_py})],
48
+ ),
49
+ LLMResponse(
50
+ content="Now the test file covering the base cases and fib(10).",
51
+ tool_calls=[ToolCall(id="c2", name="write_file", arguments={"file_path": str(workdir / "test_fib.py"), "content": test_py})],
52
+ ),
53
+ LLMResponse(
54
+ content="Running the tests.",
55
+ tool_calls=[ToolCall(id="c3", name="bash", arguments={"command": f"cd {workdir} && python -m pytest test_fib.py -q"})],
56
+ ),
57
+ LLMResponse(
58
+ content="All three assertions pass. `fib` is iterative, and the test covers the base cases plus fib(10) == 55."
59
+ ),
60
+ ]
61
+
62
+
63
+ def _summarize(args: dict) -> str:
64
+ parts = []
65
+ for key, value in args.items():
66
+ text = str(value)
67
+ if len(text) > 60:
68
+ text = text[:57] + "..."
69
+ parts.append(f"{key}={text}")
70
+ return " ".join(parts)
71
+
72
+
73
+ def run_demo() -> int:
74
+ workdir = Path(tempfile.mkdtemp(prefix="corecoder-demo-"))
75
+ agent = Agent(llm=ScriptedLLM(_script(workdir)))
76
+
77
+ console.print(Panel.fit(f"[bold]{_TASK}[/]", title="corecoder demo (offline)"))
78
+ result = agent.chat(
79
+ _TASK,
80
+ on_tool=lambda name, args: console.print(f"[cyan]tool:[/] {name} {_summarize(args)}"),
81
+ )
82
+ console.print(Panel.fit(Markdown(result), title="final"))
83
+ console.print(f"[dim]workspace kept at {workdir}[/]")
84
+ return 0