qai-cli 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,13 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ venv/
5
+ .env
6
+ .pytest_cache/
7
+ .mypy_cache/
8
+ .ruff_cache/
9
+ dist/
10
+ build/
11
+ *.egg-info/
12
+ models/
13
+ *.gguf
qai_cli-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nodoubvt
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
qai_cli-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,304 @@
1
+ Metadata-Version: 2.5
2
+ Name: qai-cli
3
+ Version: 0.1.0
4
+ Summary: Offline terminal coding agent. Runs Qwen2.5-Coder 1.5B locally via llama.cpp - no API key, no cloud, no telemetry.
5
+ Project-URL: Homepage, https://github.com/Nodoubvt/local-coding-agent
6
+ Project-URL: Repository, https://github.com/Nodoubvt/local-coding-agent
7
+ Project-URL: Issues, https://github.com/Nodoubvt/local-coding-agent/issues
8
+ Project-URL: Changelog, https://github.com/Nodoubvt/local-coding-agent/blob/main/CHANGELOG.md
9
+ Author: Nodoubvt
10
+ License: MIT
11
+ License-File: LICENSE
12
+ Keywords: agent,cli,coding-agent,coding-assistant,gguf,llama-cpp,llama.cpp,llm,local-llm,offline,privacy,qwen
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Environment :: Console
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: License :: OSI Approved :: MIT License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Software Development :: Code Generators
24
+ Classifier: Topic :: Utilities
25
+ Classifier: Typing :: Typed
26
+ Requires-Python: >=3.10
27
+ Requires-Dist: huggingface-hub>=0.23
28
+ Requires-Dist: pathspec>=0.12
29
+ Requires-Dist: platformdirs>=4.2
30
+ Requires-Dist: prompt-toolkit>=3.0.43
31
+ Requires-Dist: rich>=13.7
32
+ Requires-Dist: typer>=0.12
33
+ Provides-Extra: dev
34
+ Requires-Dist: build>=1.2; extra == 'dev'
35
+ Requires-Dist: mypy>=1.11; extra == 'dev'
36
+ Requires-Dist: pytest>=8.2; extra == 'dev'
37
+ Requires-Dist: ruff>=0.5; extra == 'dev'
38
+ Requires-Dist: twine>=5.1; extra == 'dev'
39
+ Provides-Extra: local
40
+ Requires-Dist: llama-cpp-python>=0.3.2; extra == 'local'
41
+ Description-Content-Type: text/markdown
42
+
43
+ <div align="center">
44
+
45
+ ![CI](https://github.com/Nodoubvt/local-coding-agent/actions/workflows/ci.yml/badge.svg)
46
+ ![License: MIT](https://img.shields.io/badge/license-MIT-2563eb.svg)
47
+ ![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-3776ab.svg)
48
+ ![Model: Qwen2.5-Coder 1.5B](https://img.shields.io/badge/model-Qwen2.5--Coder--1.5B-f97316.svg)
49
+ ![Inference: local / offline](https://img.shields.io/badge/inference-100%25%20local-16a34a.svg)
50
+ ![No API key](https://img.shields.io/badge/API%20keys-none-16a34a.svg)
51
+ ![Stars](https://img.shields.io/github/stars/Nodoubvt/local-coding-agent?style=social)
52
+
53
+ </div>
54
+
55
+ # local-coding-agent
56
+
57
+ **An offline AI coding assistant that runs entirely on your own machine.**
58
+ `qai` is a terminal coding agent powered by **Qwen2.5-Coder 1.5B** running on
59
+ [llama.cpp](https://github.com/ggml-org/llama.cpp) — no cloud, no API keys, no
60
+ telemetry, no data leaving your laptop.
61
+
62
+ It is deliberately small: about **1,700 lines** of readable Python you can finish
63
+ in one sitting, built to be extended rather than depended on.
64
+
65
+ > **CLI command:** `qai` · **PyPI package:** `qai-cli` · **Repository:**
66
+ > `local-coding-agent` · **License:** MIT
67
+
68
+ ```
69
+ qai> add a docstring to every function in utils.py
70
+ tool read_file(path=utils.py)
71
+ ok read_file
72
+ tool edit_file(path=utils.py, old_string=..., new_string=...)
73
+ Approve edit utils.py
74
+ ▸ 12 lines changed
75
+ ```
76
+
77
+ ---
78
+
79
+ ## Why this exists
80
+
81
+ Most "AI coding agent" tools assume a 70B model, a GPU, and a subscription.
82
+ This one assumes nothing:
83
+
84
+ - **Runs a 1.5B model on a laptop CPU.** No CUDA, no 24 GB of VRAM, no
85
+ electricity bill. It is a *small* model, and the tooling around it is built to
86
+ make a small model useful.
87
+ - **~1.1 GB download.** Q4_K_M GGUF weights, cached after the first run.
88
+ - **1,700 lines, fully readable.** The whole agent — sandbox, tools, approval
89
+ gate, context optimizer — is short enough to read, audit and fork.
90
+ - **MIT, no strings.** The safety model is code you can inspect, not a policy
91
+ page.
92
+
93
+ ## Features
94
+
95
+ - **Local inference** — GGUF weights from Hugging Face, run through
96
+ `llama-cpp-python`. CPU-only works out of the box; GPU offload is on by default
97
+ when a backend exists.
98
+ - **Agentic file tools** — `read_file`, `edit_file`, `write_file`, `list_dir`,
99
+ `grep`, all sandboxed to the workspace root.
100
+ - **Approval gate** — every write shows a unified diff and waits for you. `--yes`
101
+ skips it, `--read-only` disables writes entirely.
102
+ - **32k context with a real optimizer** — tool results and file dumps are trimmed
103
+ head-and-tail, then older turns are auto-compacted into a summary before the
104
+ buffer fills. Tool calls are never orphaned from their results.
105
+ - **Context injection** — `@path` mentions, `--attach` globs, and an automatic
106
+ workspace manifest so a 1.5B model knows what files exist.
107
+ - **Two modes** — one-shot `qai "..."` for scripting, or a REPL with `@file`
108
+ completion, history and slash commands.
109
+
110
+ ## Requirements
111
+
112
+ - Python 3.10 or newer
113
+ - ~2 GB free disk for weights, ~2 GB RAM at inference
114
+ - A C++ toolchain, *or* a prebuilt `llama-cpp-python` wheel (below)
115
+
116
+ ## Install
117
+
118
+ ### Fastest — no install, no virtualenv
119
+
120
+ ```bash
121
+ uvx qai-cli # run it instantly
122
+ ```
123
+
124
+ ### Global install, still isolated
125
+
126
+ ```bash
127
+ uv tool install qai-cli # or: pipx install qai-cli
128
+ qai "what does main() do?"
129
+ ```
130
+
131
+ > **Local inference needs `llama-cpp-python`**, a compiled C++ extension. If you
132
+ > have no C++ toolchain, use the prebuilt CPU wheel:
133
+ >
134
+ > ```bash
135
+ > pip install llama-cpp-python --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
136
+ > ```
137
+ >
138
+ > Then run `uvx --with llama-cpp-python qai-cli` to try it in one shot.
139
+
140
+ ### From source
141
+
142
+ ```bash
143
+ git clone https://github.com/Nodoubvt/local-coding-agent
144
+ cd local-coding-agent
145
+
146
+ python -m venv .venv
147
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
148
+ pip install -e ".[local]" # [local] adds llama-cpp-python
149
+ ```
150
+
151
+ Inspect resolved settings, or fetch weights ahead of time:
152
+
153
+ ```bash
154
+ qai --print-config
155
+ python -c "from qai.config import load_settings; from qai.backend import ensure_model_file; print(ensure_model_file(load_settings()))"
156
+ ```
157
+
158
+ ## Usage
159
+
160
+ ```bash
161
+ qai # interactive session
162
+ qai "why does main() hang here?" # one-shot
163
+ qai "refactor this" -a src/api.py # attach a file
164
+ qai "find every TODO" --no-tools # pure chat, no filesystem access
165
+
166
+ qai --model qwen2.5-coder-1.5b-instruct-q8_0.gguf # higher-quality quant
167
+ qai --gpu-layers 0 --threads 8 # force CPU, pin thread count
168
+ qai --read-only # refuse every write
169
+ qai --compact-at 0.5 # summarize earlier, at 50% of budget
170
+ qai --no-compact # disable the compactor
171
+ ```
172
+
173
+ ### REPL commands
174
+
175
+ | command | effect |
176
+ | --- | --- |
177
+ | `/help` | show commands |
178
+ | `/new` | clear conversation history |
179
+ | `/context [path]` | re-send the manifest, or attach specific paths |
180
+ | `/stats` | context usage, budget, compaction count |
181
+ | `/tools` | list available tools |
182
+ | `/model` | show active model and context size |
183
+ | `/read-only on\|off` | toggle write access |
184
+ | `/exit` | quit (ctrl-d also works) |
185
+
186
+ ## Configuration
187
+
188
+ Settings merge in this order: **defaults → `config.json` → env vars → CLI flags**.
189
+
190
+ ```bash
191
+ python -c "from qai.config import Settings, save_settings; save_settings(Settings(n_ctx=32768))"
192
+ ```
193
+
194
+ | setting | default | meaning |
195
+ | --- | --- | --- |
196
+ | `n_ctx` | `32768` | context window |
197
+ | `max_tokens` | `4096` | reply ceiling, also reserved from the budget |
198
+ | `compact_threshold` | `0.70` | auto-compact above this context ratio |
199
+ | `max_tool_result_chars` | `8000` | per-result trim limit |
200
+ | `keep_recent_blocks` | `6` | turns kept verbatim through compaction |
201
+ | `auto_compact` | `true` | enable the compactor |
202
+ | `n_gpu_layers` | `-1` | `-1` offload all, `0` CPU only |
203
+ | `temperature` | `0.2` | sampling temperature |
204
+ | `auto_approve` | `false` | skip write approval |
205
+
206
+ Env equivalents: `QAI_N_CTX`, `QAI_TEMPERATURE`, `QAI_AUTO_COMPACT`,
207
+ `QAI_MODEL_DIR`, `QAI_CONFIG_DIR`, and the rest follow `QAI_` + the setting name.
208
+
209
+ ## How the context optimizer works
210
+
211
+ A 1.5B model with a 32k window still hits the wall fast, because a single
212
+ `read_file` of a large file can eat a quarter of it. Three layers keep a long
213
+ session degrading instead of failing:
214
+
215
+ 1. **Budget.** `budget = n_ctx - max_tokens - tool_schema_reserve` — 27,472
216
+ tokens at defaults. Everything is measured with a deliberately pessimistic
217
+ chars-per-token estimate, so the agent under-fills rather than overflows.
218
+ 2. **Trim.** Oversized tool results and `<context>` blocks are cut head-and-tail,
219
+ so the end of an error message survives. Assistant messages carrying
220
+ `tool_calls` are grouped with their results into a single atomic block, so
221
+ trimming can never leave an orphan that breaks the chat template.
222
+ 3. **Compact.** Past `compact_threshold`, older turns are summarized — by the
223
+ model when it is available, otherwise by a deterministic extractive fallback —
224
+ into a single `<summary>` block, with the last `keep_recent_blocks` turns kept
225
+ verbatim.
226
+
227
+ Compaction runs between tool rounds, never mid-flight, so the model always sees a
228
+ consistent transcript, and the compacted transcript is carried into the next turn
229
+ so a long REPL session stays bounded instead of being re-trimmed from scratch
230
+ every time. `/stats` shows live usage.
231
+
232
+ ## Safety
233
+
234
+ - All tool paths resolve through `Workspace`, which refuses anything escaping the
235
+ workspace root after symlink resolution.
236
+ - `.git`, `node_modules`, `.venv` and model weights are hidden from listing,
237
+ search and context injection by default; add your own with `--ignore`.
238
+ - Writes require explicit approval with a diff shown first.
239
+ - `--read-only` disables every mutating tool at the registry level, not just in
240
+ the prompt.
241
+
242
+ ## Development
243
+
244
+ ```bash
245
+ pip install -e ".[dev]"
246
+ pytest # 70 tests, no model or network required
247
+ ruff check .
248
+ mypy src
249
+ ```
250
+
251
+ Tests run against a scripted fake backend, so nothing downloads 1 GB. Coverage
252
+ includes sandbox escapes, the approval gate, tool-call pairing across trims, and
253
+ a simulated 32k session that reads a 20k-line file fifteen times.
254
+
255
+ ## Releasing
256
+
257
+ Push a `v*` tag. The workflow verifies the tag matches the `pyproject.toml`
258
+ version, builds the sdist and wheel, checks metadata, smoke-tests the wheel, and
259
+ publishes to PyPI via OIDC trusted publishing — no API token is stored in the
260
+ repository.
261
+
262
+ ```bash
263
+ # 1. bump version in pyproject.toml + CHANGELOG.md, commit
264
+ # 2. tag and push
265
+ git tag v0.1.0 && git push origin v0.1.0
266
+ ```
267
+
268
+ ## License
269
+
270
+ MIT — see [LICENSE](LICENSE).
271
+
272
+ ---
273
+
274
+ ## Beyond the terminal
275
+
276
+ <p align="center">
277
+ <img src="docs/images/vibe.png" alt="Agentic Studio in Vibe Mode: a multi-pane IDE with a VS-style code editor, a live tool-call log showing grep_files and execute_shell, an agent conversation panel, and L1-L4 context cache gauges along the bottom" width="100%">
278
+ </p>
279
+
280
+ <p align="center"><em>Vibe Mode — the agent snapshots the workspace, then runs
281
+ unattended with the sandbox still hard-caged. Note the live tool log and the
282
+ L1–L4 context gauges along the bottom edge.</em></p>
283
+
284
+ <p align="center">
285
+ <img src="docs/images/ss-ctx1.png" alt="The Context window in Agentic Studio: a model picker, a 125k token slider with 16k through 1M presets, and below it the Token Context Heap heatmap colouring each cluster as free space, pinned, reading, writing, bloated or defragmented" width="100%">
286
+ </p>
287
+
288
+ <p align="center"><em>The Context Engine — a per-model window from 16k to 1M,
289
+ with the Token Context Heap showing exactly which clusters are pinned, bloating
290
+ or still free.</em></p>
291
+
292
+ `qai` is the terminal. If you want the graphical version of the same idea —
293
+ **5 coordinated agents, an inspectable context engine, embedded PHP/MariaDB/WebGPU
294
+ emulators and one-click rollback** — that is
295
+ **[Devhead Agentic Studio](https://dev-head.com/products/astudio.html)**, a
296
+ separate commercial product built by the same author.
297
+
298
+ The two are deliberately different tools: `qai` is free, MIT, and 1,700 lines of
299
+ Python you can audit today; Agentic Studio is a Windows desktop IDE. The CLI
300
+ stays free and MIT-licensed either way — nothing here is a trial, and nothing
301
+ here phones home.
302
+
303
+ *(Screenshots above are from the author's own product page and remain the
304
+ property of their respective owner.)*
@@ -0,0 +1,262 @@
1
+ <div align="center">
2
+
3
+ ![CI](https://github.com/Nodoubvt/local-coding-agent/actions/workflows/ci.yml/badge.svg)
4
+ ![License: MIT](https://img.shields.io/badge/license-MIT-2563eb.svg)
5
+ ![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-3776ab.svg)
6
+ ![Model: Qwen2.5-Coder 1.5B](https://img.shields.io/badge/model-Qwen2.5--Coder--1.5B-f97316.svg)
7
+ ![Inference: local / offline](https://img.shields.io/badge/inference-100%25%20local-16a34a.svg)
8
+ ![No API key](https://img.shields.io/badge/API%20keys-none-16a34a.svg)
9
+ ![Stars](https://img.shields.io/github/stars/Nodoubvt/local-coding-agent?style=social)
10
+
11
+ </div>
12
+
13
+ # local-coding-agent
14
+
15
+ **An offline AI coding assistant that runs entirely on your own machine.**
16
+ `qai` is a terminal coding agent powered by **Qwen2.5-Coder 1.5B** running on
17
+ [llama.cpp](https://github.com/ggml-org/llama.cpp) — no cloud, no API keys, no
18
+ telemetry, no data leaving your laptop.
19
+
20
+ It is deliberately small: about **1,700 lines** of readable Python you can finish
21
+ in one sitting, built to be extended rather than depended on.
22
+
23
+ > **CLI command:** `qai` · **PyPI package:** `qai-cli` · **Repository:**
24
+ > `local-coding-agent` · **License:** MIT
25
+
26
+ ```
27
+ qai> add a docstring to every function in utils.py
28
+ tool read_file(path=utils.py)
29
+ ok read_file
30
+ tool edit_file(path=utils.py, old_string=..., new_string=...)
31
+ Approve edit utils.py
32
+ ▸ 12 lines changed
33
+ ```
34
+
35
+ ---
36
+
37
+ ## Why this exists
38
+
39
+ Most "AI coding agent" tools assume a 70B model, a GPU, and a subscription.
40
+ This one assumes nothing:
41
+
42
+ - **Runs a 1.5B model on a laptop CPU.** No CUDA, no 24 GB of VRAM, no
43
+ electricity bill. It is a *small* model, and the tooling around it is built to
44
+ make a small model useful.
45
+ - **~1.1 GB download.** Q4_K_M GGUF weights, cached after the first run.
46
+ - **1,700 lines, fully readable.** The whole agent — sandbox, tools, approval
47
+ gate, context optimizer — is short enough to read, audit and fork.
48
+ - **MIT, no strings.** The safety model is code you can inspect, not a policy
49
+ page.
50
+
51
+ ## Features
52
+
53
+ - **Local inference** — GGUF weights from Hugging Face, run through
54
+ `llama-cpp-python`. CPU-only works out of the box; GPU offload is on by default
55
+ when a backend exists.
56
+ - **Agentic file tools** — `read_file`, `edit_file`, `write_file`, `list_dir`,
57
+ `grep`, all sandboxed to the workspace root.
58
+ - **Approval gate** — every write shows a unified diff and waits for you. `--yes`
59
+ skips it, `--read-only` disables writes entirely.
60
+ - **32k context with a real optimizer** — tool results and file dumps are trimmed
61
+ head-and-tail, then older turns are auto-compacted into a summary before the
62
+ buffer fills. Tool calls are never orphaned from their results.
63
+ - **Context injection** — `@path` mentions, `--attach` globs, and an automatic
64
+ workspace manifest so a 1.5B model knows what files exist.
65
+ - **Two modes** — one-shot `qai "..."` for scripting, or a REPL with `@file`
66
+ completion, history and slash commands.
67
+
68
+ ## Requirements
69
+
70
+ - Python 3.10 or newer
71
+ - ~2 GB free disk for weights, ~2 GB RAM at inference
72
+ - A C++ toolchain, *or* a prebuilt `llama-cpp-python` wheel (below)
73
+
74
+ ## Install
75
+
76
+ ### Fastest — no install, no virtualenv
77
+
78
+ ```bash
79
+ uvx qai-cli # run it instantly
80
+ ```
81
+
82
+ ### Global install, still isolated
83
+
84
+ ```bash
85
+ uv tool install qai-cli # or: pipx install qai-cli
86
+ qai "what does main() do?"
87
+ ```
88
+
89
+ > **Local inference needs `llama-cpp-python`**, a compiled C++ extension. If you
90
+ > have no C++ toolchain, use the prebuilt CPU wheel:
91
+ >
92
+ > ```bash
93
+ > pip install llama-cpp-python --extra-index-url https://abetlen.github.io/llama-cpp-python/whl/cpu
94
+ > ```
95
+ >
96
+ > Then run `uvx --with llama-cpp-python qai-cli` to try it in one shot.
97
+
98
+ ### From source
99
+
100
+ ```bash
101
+ git clone https://github.com/Nodoubvt/local-coding-agent
102
+ cd local-coding-agent
103
+
104
+ python -m venv .venv
105
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
106
+ pip install -e ".[local]" # [local] adds llama-cpp-python
107
+ ```
108
+
109
+ Inspect resolved settings, or fetch weights ahead of time:
110
+
111
+ ```bash
112
+ qai --print-config
113
+ python -c "from qai.config import load_settings; from qai.backend import ensure_model_file; print(ensure_model_file(load_settings()))"
114
+ ```
115
+
116
+ ## Usage
117
+
118
+ ```bash
119
+ qai # interactive session
120
+ qai "why does main() hang here?" # one-shot
121
+ qai "refactor this" -a src/api.py # attach a file
122
+ qai "find every TODO" --no-tools # pure chat, no filesystem access
123
+
124
+ qai --model qwen2.5-coder-1.5b-instruct-q8_0.gguf # higher-quality quant
125
+ qai --gpu-layers 0 --threads 8 # force CPU, pin thread count
126
+ qai --read-only # refuse every write
127
+ qai --compact-at 0.5 # summarize earlier, at 50% of budget
128
+ qai --no-compact # disable the compactor
129
+ ```
130
+
131
+ ### REPL commands
132
+
133
+ | command | effect |
134
+ | --- | --- |
135
+ | `/help` | show commands |
136
+ | `/new` | clear conversation history |
137
+ | `/context [path]` | re-send the manifest, or attach specific paths |
138
+ | `/stats` | context usage, budget, compaction count |
139
+ | `/tools` | list available tools |
140
+ | `/model` | show active model and context size |
141
+ | `/read-only on\|off` | toggle write access |
142
+ | `/exit` | quit (ctrl-d also works) |
143
+
144
+ ## Configuration
145
+
146
+ Settings merge in this order: **defaults → `config.json` → env vars → CLI flags**.
147
+
148
+ ```bash
149
+ python -c "from qai.config import Settings, save_settings; save_settings(Settings(n_ctx=32768))"
150
+ ```
151
+
152
+ | setting | default | meaning |
153
+ | --- | --- | --- |
154
+ | `n_ctx` | `32768` | context window |
155
+ | `max_tokens` | `4096` | reply ceiling, also reserved from the budget |
156
+ | `compact_threshold` | `0.70` | auto-compact above this context ratio |
157
+ | `max_tool_result_chars` | `8000` | per-result trim limit |
158
+ | `keep_recent_blocks` | `6` | turns kept verbatim through compaction |
159
+ | `auto_compact` | `true` | enable the compactor |
160
+ | `n_gpu_layers` | `-1` | `-1` offload all, `0` CPU only |
161
+ | `temperature` | `0.2` | sampling temperature |
162
+ | `auto_approve` | `false` | skip write approval |
163
+
164
+ Env equivalents: `QAI_N_CTX`, `QAI_TEMPERATURE`, `QAI_AUTO_COMPACT`,
165
+ `QAI_MODEL_DIR`, `QAI_CONFIG_DIR`, and the rest follow `QAI_` + the setting name.
166
+
167
+ ## How the context optimizer works
168
+
169
+ A 1.5B model with a 32k window still hits the wall fast, because a single
170
+ `read_file` of a large file can eat a quarter of it. Three layers keep a long
171
+ session degrading instead of failing:
172
+
173
+ 1. **Budget.** `budget = n_ctx - max_tokens - tool_schema_reserve` — 27,472
174
+ tokens at defaults. Everything is measured with a deliberately pessimistic
175
+ chars-per-token estimate, so the agent under-fills rather than overflows.
176
+ 2. **Trim.** Oversized tool results and `<context>` blocks are cut head-and-tail,
177
+ so the end of an error message survives. Assistant messages carrying
178
+ `tool_calls` are grouped with their results into a single atomic block, so
179
+ trimming can never leave an orphan that breaks the chat template.
180
+ 3. **Compact.** Past `compact_threshold`, older turns are summarized — by the
181
+ model when it is available, otherwise by a deterministic extractive fallback —
182
+ into a single `<summary>` block, with the last `keep_recent_blocks` turns kept
183
+ verbatim.
184
+
185
+ Compaction runs between tool rounds, never mid-flight, so the model always sees a
186
+ consistent transcript, and the compacted transcript is carried into the next turn
187
+ so a long REPL session stays bounded instead of being re-trimmed from scratch
188
+ every time. `/stats` shows live usage.
189
+
190
+ ## Safety
191
+
192
+ - All tool paths resolve through `Workspace`, which refuses anything escaping the
193
+ workspace root after symlink resolution.
194
+ - `.git`, `node_modules`, `.venv` and model weights are hidden from listing,
195
+ search and context injection by default; add your own with `--ignore`.
196
+ - Writes require explicit approval with a diff shown first.
197
+ - `--read-only` disables every mutating tool at the registry level, not just in
198
+ the prompt.
199
+
200
+ ## Development
201
+
202
+ ```bash
203
+ pip install -e ".[dev]"
204
+ pytest # 70 tests, no model or network required
205
+ ruff check .
206
+ mypy src
207
+ ```
208
+
209
+ Tests run against a scripted fake backend, so nothing downloads 1 GB. Coverage
210
+ includes sandbox escapes, the approval gate, tool-call pairing across trims, and
211
+ a simulated 32k session that reads a 20k-line file fifteen times.
212
+
213
+ ## Releasing
214
+
215
+ Push a `v*` tag. The workflow verifies the tag matches the `pyproject.toml`
216
+ version, builds the sdist and wheel, checks metadata, smoke-tests the wheel, and
217
+ publishes to PyPI via OIDC trusted publishing — no API token is stored in the
218
+ repository.
219
+
220
+ ```bash
221
+ # 1. bump version in pyproject.toml + CHANGELOG.md, commit
222
+ # 2. tag and push
223
+ git tag v0.1.0 && git push origin v0.1.0
224
+ ```
225
+
226
+ ## License
227
+
228
+ MIT — see [LICENSE](LICENSE).
229
+
230
+ ---
231
+
232
+ ## Beyond the terminal
233
+
234
+ <p align="center">
235
+ <img src="docs/images/vibe.png" alt="Agentic Studio in Vibe Mode: a multi-pane IDE with a VS-style code editor, a live tool-call log showing grep_files and execute_shell, an agent conversation panel, and L1-L4 context cache gauges along the bottom" width="100%">
236
+ </p>
237
+
238
+ <p align="center"><em>Vibe Mode — the agent snapshots the workspace, then runs
239
+ unattended with the sandbox still hard-caged. Note the live tool log and the
240
+ L1–L4 context gauges along the bottom edge.</em></p>
241
+
242
+ <p align="center">
243
+ <img src="docs/images/ss-ctx1.png" alt="The Context window in Agentic Studio: a model picker, a 125k token slider with 16k through 1M presets, and below it the Token Context Heap heatmap colouring each cluster as free space, pinned, reading, writing, bloated or defragmented" width="100%">
244
+ </p>
245
+
246
+ <p align="center"><em>The Context Engine — a per-model window from 16k to 1M,
247
+ with the Token Context Heap showing exactly which clusters are pinned, bloating
248
+ or still free.</em></p>
249
+
250
+ `qai` is the terminal. If you want the graphical version of the same idea —
251
+ **5 coordinated agents, an inspectable context engine, embedded PHP/MariaDB/WebGPU
252
+ emulators and one-click rollback** — that is
253
+ **[Devhead Agentic Studio](https://dev-head.com/products/astudio.html)**, a
254
+ separate commercial product built by the same author.
255
+
256
+ The two are deliberately different tools: `qai` is free, MIT, and 1,700 lines of
257
+ Python you can audit today; Agentic Studio is a Windows desktop IDE. The CLI
258
+ stays free and MIT-licensed either way — nothing here is a trial, and nothing
259
+ here phones home.
260
+
261
+ *(Screenshots above are from the author's own product page and remain the
262
+ property of their respective owner.)*
@@ -0,0 +1,90 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "qai-cli"
7
+ version = "0.1.0"
8
+ description = "Offline terminal coding agent. Runs Qwen2.5-Coder 1.5B locally via llama.cpp - no API key, no cloud, no telemetry."
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ authors = [{ name = "Nodoubvt" }]
12
+ requires-python = ">=3.10"
13
+ keywords = [
14
+ "llm",
15
+ "cli",
16
+ "agent",
17
+ "coding-agent",
18
+ "llama.cpp",
19
+ "llama-cpp",
20
+ "gguf",
21
+ "qwen",
22
+ "local-llm",
23
+ "offline",
24
+ "privacy",
25
+ "coding-assistant",
26
+ ]
27
+ classifiers = [
28
+ "Development Status :: 3 - Alpha",
29
+ "Environment :: Console",
30
+ "Intended Audience :: Developers",
31
+ "License :: OSI Approved :: MIT License",
32
+ "Operating System :: OS Independent",
33
+ "Programming Language :: Python :: 3",
34
+ "Programming Language :: Python :: 3.10",
35
+ "Programming Language :: Python :: 3.11",
36
+ "Programming Language :: Python :: 3.12",
37
+ "Programming Language :: Python :: 3.13",
38
+ "Topic :: Software Development :: Code Generators",
39
+ "Topic :: Utilities",
40
+ "Typing :: Typed",
41
+ ]
42
+ dependencies = [
43
+ "typer>=0.12",
44
+ "rich>=13.7",
45
+ "prompt-toolkit>=3.0.43",
46
+ "platformdirs>=4.2",
47
+ "huggingface-hub>=0.23",
48
+ "pathspec>=0.12",
49
+ ]
50
+
51
+ [project.optional-dependencies]
52
+ local = ["llama-cpp-python>=0.3.2"]
53
+ dev = [
54
+ "pytest>=8.2",
55
+ "ruff>=0.5",
56
+ "mypy>=1.11",
57
+ "build>=1.2",
58
+ "twine>=5.1",
59
+ ]
60
+
61
+ [project.scripts]
62
+ qai = "qai.cli:run"
63
+
64
+ [project.urls]
65
+ Homepage = "https://github.com/Nodoubvt/local-coding-agent"
66
+ Repository = "https://github.com/Nodoubvt/local-coding-agent"
67
+ Issues = "https://github.com/Nodoubvt/local-coding-agent/issues"
68
+ Changelog = "https://github.com/Nodoubvt/local-coding-agent/blob/main/CHANGELOG.md"
69
+
70
+ [tool.hatch.build.targets.wheel]
71
+ packages = ["src/qai"]
72
+
73
+ [tool.hatch.build.targets.sdist]
74
+ include = ["src/qai", "tests", "README.md", "LICENSE", "pyproject.toml"]
75
+
76
+ [tool.pytest.ini_options]
77
+ testpaths = ["tests"]
78
+ addopts = "-q"
79
+
80
+ [tool.ruff]
81
+ line-length = 100
82
+ target-version = "py310"
83
+
84
+ [tool.ruff.lint]
85
+ select = ["E", "F", "I", "UP", "B", "SIM"]
86
+
87
+ [tool.mypy]
88
+ python_version = "3.10"
89
+ warn_unused_ignores = true
90
+ ignore_missing_imports = true