ai-code-engineer 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. ai_code_engineer-0.1.0/LICENSE +21 -0
  2. ai_code_engineer-0.1.0/PKG-INFO +7 -0
  3. ai_code_engineer-0.1.0/README.md +301 -0
  4. ai_code_engineer-0.1.0/pyproject.toml +23 -0
  5. ai_code_engineer-0.1.0/setup.cfg +4 -0
  6. ai_code_engineer-0.1.0/src/ai_code_engineer/__init__.py +2 -0
  7. ai_code_engineer-0.1.0/src/ai_code_engineer/catalog.py +143 -0
  8. ai_code_engineer-0.1.0/src/ai_code_engineer/chat.py +181 -0
  9. ai_code_engineer-0.1.0/src/ai_code_engineer/cli.py +384 -0
  10. ai_code_engineer-0.1.0/src/ai_code_engineer/config.py +405 -0
  11. ai_code_engineer-0.1.0/src/ai_code_engineer/engine.py +1282 -0
  12. ai_code_engineer-0.1.0/src/ai_code_engineer/errors.py +27 -0
  13. ai_code_engineer-0.1.0/src/ai_code_engineer/git_integration.py +443 -0
  14. ai_code_engineer-0.1.0/src/ai_code_engineer/gui.py +2646 -0
  15. ai_code_engineer-0.1.0/src/ai_code_engineer/host.py +81 -0
  16. ai_code_engineer-0.1.0/src/ai_code_engineer/ignore.py +269 -0
  17. ai_code_engineer-0.1.0/src/ai_code_engineer/intent.py +222 -0
  18. ai_code_engineer-0.1.0/src/ai_code_engineer/labels.py +871 -0
  19. ai_code_engineer-0.1.0/src/ai_code_engineer/memory.py +91 -0
  20. ai_code_engineer-0.1.0/src/ai_code_engineer/modes.py +156 -0
  21. ai_code_engineer-0.1.0/src/ai_code_engineer/overrides.py +540 -0
  22. ai_code_engineer-0.1.0/src/ai_code_engineer/planbook.py +192 -0
  23. ai_code_engineer-0.1.0/src/ai_code_engineer/providers.py +404 -0
  24. ai_code_engineer-0.1.0/src/ai_code_engineer/redaction.py +54 -0
  25. ai_code_engineer-0.1.0/src/ai_code_engineer/repair.py +564 -0
  26. ai_code_engineer-0.1.0/src/ai_code_engineer/report.py +352 -0
  27. ai_code_engineer-0.1.0/src/ai_code_engineer/runner.py +854 -0
  28. ai_code_engineer-0.1.0/src/ai_code_engineer/setup.py +386 -0
  29. ai_code_engineer-0.1.0/src/ai_code_engineer/symbols.py +1286 -0
  30. ai_code_engineer-0.1.0/src/ai_code_engineer/verification.py +218 -0
  31. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/__init__.py +1 -0
  32. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/__main__.py +45 -0
  33. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/contract.py +36 -0
  34. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/controller.py +3556 -0
  35. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/fake.py +1141 -0
  36. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/launch.py +108 -0
  37. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/server.py +349 -0
  38. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/app.css +780 -0
  39. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/app.js +2118 -0
  40. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/boot.js +19 -0
  41. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/index.html +89 -0
  42. ai_code_engineer-0.1.0/src/ai_code_engineer/webapp/static/tokens.css +173 -0
  43. ai_code_engineer-0.1.0/src/ai_code_engineer/workspace.py +385 -0
  44. ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/PKG-INFO +7 -0
  45. ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/SOURCES.txt +72 -0
  46. ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/dependency_links.txt +1 -0
  47. ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/entry_points.txt +2 -0
  48. ai_code_engineer-0.1.0/src/ai_code_engineer.egg-info/top_level.txt +1 -0
  49. ai_code_engineer-0.1.0/tests/test_agent.py +1949 -0
  50. ai_code_engineer-0.1.0/tests/test_chat.py +216 -0
  51. ai_code_engineer-0.1.0/tests/test_cli.py +595 -0
  52. ai_code_engineer-0.1.0/tests/test_connection.py +925 -0
  53. ai_code_engineer-0.1.0/tests/test_controller.py +3947 -0
  54. ai_code_engineer-0.1.0/tests/test_doubles.py +150 -0
  55. ai_code_engineer-0.1.0/tests/test_git.py +911 -0
  56. ai_code_engineer-0.1.0/tests/test_gui.py +1258 -0
  57. ai_code_engineer-0.1.0/tests/test_host.py +189 -0
  58. ai_code_engineer-0.1.0/tests/test_ignore.py +267 -0
  59. ai_code_engineer-0.1.0/tests/test_intent.py +188 -0
  60. ai_code_engineer-0.1.0/tests/test_labels.py +586 -0
  61. ai_code_engineer-0.1.0/tests/test_memory.py +95 -0
  62. ai_code_engineer-0.1.0/tests/test_modes.py +153 -0
  63. ai_code_engineer-0.1.0/tests/test_overrides.py +620 -0
  64. ai_code_engineer-0.1.0/tests/test_planbook.py +214 -0
  65. ai_code_engineer-0.1.0/tests/test_redaction.py +95 -0
  66. ai_code_engineer-0.1.0/tests/test_repair.py +721 -0
  67. ai_code_engineer-0.1.0/tests/test_report.py +264 -0
  68. ai_code_engineer-0.1.0/tests/test_runner.py +831 -0
  69. ai_code_engineer-0.1.0/tests/test_setup.py +443 -0
  70. ai_code_engineer-0.1.0/tests/test_symbols.py +1204 -0
  71. ai_code_engineer-0.1.0/tests/test_transport.py +217 -0
  72. ai_code_engineer-0.1.0/tests/test_verification.py +514 -0
  73. ai_code_engineer-0.1.0/tests/test_webapp.py +1670 -0
  74. ai_code_engineer-0.1.0/tests/test_workspace.py +451 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Mohamed Saad (mohamedsaad208)
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,7 @@
1
+ Metadata-Version: 2.4
2
+ Name: ai-code-engineer
3
+ Version: 0.1.0
4
+ Summary: A local-first, review-first developer agent for Ollama and OpenRouter
5
+ Requires-Python: >=3.11
6
+ License-File: LICENSE
7
+ Dynamic: license-file
@@ -0,0 +1,301 @@
1
+ <div align="center">
2
+
3
+ <img src="assets/banner.svg" alt="AI Code Engineer Banner" width="100%"/>
4
+
5
+ # AI Code Engineer
6
+ ### Give your coding workflow an autonomous, privacy-first software engineer.
7
+
8
+ AI Code Engineer is an open-source autonomous agent framework for real-world software engineering: AST repository indexing, cryptographic diff proposals, zero-trust workspace security, automated JUnit test loops, and multi-provider LLM support (Ollama local, OpenRouter, OpenAI, Groq, DeepSeek).
9
+
10
+ [![License](https://img.shields.io/badge/License-MIT-F59E0B?style=for-the-badge&logo=opensourceinitiative&logoColor=white)](LICENSE)
11
+ [![Python](https://img.shields.io/badge/Python-3.11%2B-3B82F6?style=for-the-badge&logo=python&logoColor=white)](https://www.python.org/)
12
+ [![Platform](https://img.shields.io/badge/Platform-Windows%20%7C%20Linux%20%7C%20macOS-10B981?style=for-the-badge&logo=linux&logoColor=white)]()
13
+ [![Developed with AI](https://img.shields.io/badge/Built%20With-AI%20%26%20Human%20Pairing-8B5CF6?style=for-the-badge&logo=openai&logoColor=white)]()
14
+ [![Tests](https://github.com/mohamedsaad208/AI-Agent/actions/workflows/ci.yml/badge.svg?style=for-the-badge&logo=python&logoColor=white)](https://github.com/mohamedsaad208/AI-Agent/actions/workflows/ci.yml)
15
+
16
+ [Quick Start](#-quick-start) •
17
+ [Why AI Code Engineer](#-why-ai-code-engineer) •
18
+ [Platform Guide](#-platform-guide) •
19
+ [Try These First](#-try-these-first) •
20
+ [Architecture](#-architecture) •
21
+ [Contributing](#-contributing)
22
+
23
+ <br/>
24
+
25
+ <img src="assets/demo-complex.gif" alt="AI Code Engineer Multi-File Refactor & Self-Healing Loop" width="100%"/>
26
+
27
+ </div>
28
+
29
+ ---
30
+
31
+ # 🚀 Why AI Code Engineer
32
+
33
+ | Feature | Why it matters |
34
+ | :--- | :--- |
35
+ | 🔒 **Zero-Trust Security** | Path-traversal guards, symlink blocking, and automatic secret redaction (AWS, GitHub tokens, Bearer keys, `.env`) prevent leakage to logs or LLMs. |
36
+ | 🛡️ **Reviewed diffs** | Every change is a proposal with a `SHA-256` hash, and applying it is a decision you make (or hand over per folder, explicitly). Rollback is one step, as long as nothing else edited those files afterwards. |
37
+ | 🌿 **Git Checkpoints & Restore** | Applying an approved proposal inside a git repo automatically creates a non-destructive checkpoint commit (`--no-verify`, skips hooks). If subsequent edits block local rollback, targeted single-file git restore offers a safe escalation path back to the pre-task commit. Never pushes, pulls, or rewrites history. |
38
+ | 🌐 **Native Bilingual & RTL** | First-class Arabic and English dual-engine. Dynamic Right-to-Left (RTL) support in the WebApp, automatic language detection (`is_arabic`), and fully localized system notices, diagnostic reports, and `--arabic` CLI flags. |
39
+ | ⚡ **100% Offline & Local** | Full first-class support for **Ollama** (`qwen2.5-coder`, `deepseek-coder`, `llama3`). Code stays on your hardware. |
40
+ | ☁️ **Multi-Provider Cloud** | Seamlessly switch between **OpenRouter**, **OpenAI**, **Groq**, **DeepSeek**, or custom OpenAI-compatible endpoints. |
41
+ | 🔌 **Endpoints are configuration** | No model server's address is written in the code — the provider table carries one default per row, and a URL resolves as *what you typed → your environment (`OLLAMA_HOST`, `GROQ_BASE_URL`, …) → the table's own row*. Point Ollama at another port without editing a file, and a profile that forgets its endpoint gets **its own** provider's address, never a local default. |
42
+ | 🗺️ **AST Symbol Indexing** | In-memory symbol extractor (Python, Java/Kotlin, TypeScript/JS, Go, Rust) provides classes, methods, and types without burning context window tokens. |
43
+ | 🧪 **Self-Healing Test Loop** | Auto-detects `pytest`, `unittest`, `Maven`, `Gradle`, `npm`, `cargo`, `go test`. Parses JUnit XML output and feeds failures back to the agent for autonomous repair (up to 3 rounds). |
44
+ | ⏱️ **Zero-Drop Task Queueing** | Messages typed while a task or build is in progress are safely enqueued without race conditions, running automatically in FIFO order when the active job finishes. |
45
+ | 📑 **Session Audit & Export** | Transcripts, diffs, and proof tallies are exportable to structured JSON or clean, readable Markdown reports (`agent export-session`) for documentation and audits. |
46
+ | 🖥️ **Desktop WebApp & CLI** | Beautiful local WebApp with real-time streaming, diff previews, task queuing, and an interactive terminal menu. |
47
+
48
+ ---
49
+
50
+ # 🧭 Three modes, and the limits that go with them
51
+
52
+ **Chat** answers in prose and reads no files until you hand it a folder. **Change** produces a
53
+ proposal you review, and only `Apply` writes. **Auto-Apply** is a switch you turn on *per folder*:
54
+ the agent then writes what it proposes without a click, runs that folder's own command afterwards,
55
+ and keeps the rollback. It still asks first if the proposal would empty or delete an existing file, or
56
+ if a previous task left that folder half-written. The window says so on the card and in the Activity
57
+ log when a write happened without a click — that is the one thing about this tool that is easiest to
58
+ forget and hardest to undo.
59
+
60
+ | What you should know before you rely on it | |
61
+ | :--- | :--- |
62
+ | **Run and Check syntax execute your project's own code** | They invoke `mvn`, `gradle`, `npm`, `pytest`, `cargo`, `go` in that folder with *your* permissions. A build script is code, and code from a repository you did not write gets run here. The command list is an allowlist and the environment is stripped of credentials — that is a **limit**, not a sandbox. The row below is the sandbox. |
63
+ | **The sandbox is a tick on the Run button** | `Run in Docker` runs the project's command against a **copy** of the folder, in a container with no network, no capabilities, a read-only root and an image pinned by `sha256` digest — so the build's reports are still read afterwards, and your tree is never mounted into it. Without Docker the run is refused rather than quietly done on the host. Verified here as an argv, not as a build: no container has run on the machine this was written on. |
64
+ | **Read-only is a fact about the folder, not about a window** | Setting it in either window, or sealing a folder with `agent read-only --repo PATH`, is written down where every surface reads it back: a terminal that never opened a window cannot `apply` or `rollback` into it, and one window saving its own preferences cannot unseal what the other was told. Lifting it is an explicit act, and the refusal names the line that does it. What it never does is stop reading, searching, mapping, a static check, or a command you asked for by name. |
65
+ | **The settings you change from inside the program are signed, not encrypted** | `Settings → Overrides` and `agent overrides --set TARGET KEY VALUE` write `.agent-overrides.json` in this tool's own folder — created at first run, `.gitignore`d, never written into your project. The program signs every row, so a row you typed into the JSON by hand (or one moved, changed, added or deleted there) is **refused and named** rather than quietly obeyed, and the run falls back to your profile. That is tamper evidence, not a password: your own account can rewrite the file, and the standard library has no cipher. It never redirects an address a profile or a field already states, never swaps the provider, and never holds a key — a credential-shaped value is refused on the way in. |
66
+ | **Git checkpoints back every applied proposal** | Inside a git repository, applying a proposal commits the approved files with `--no-verify` (skipping hooks) and names the session. If files are edited after review, local rollback is blocked to prevent clobbering your later edits, and the tool offers **Targeted Git Restore** to put only that task's files back to its pre-task commit. Strictly no network commands (`push`/`pull`) and no rewritten history. |
67
+ | **The repository map is context, not a compiler** | Symbols are parsed with `ast` for Python and bounded scanners for Java, Kotlin, Go, Rust and TypeScript. It tells the model what files declare; it does not type-check, resolve imports or prove the code works. Only running the project's command does that, and a run that never ran is reported as `unverified`, not as a pass. |
68
+ | **Small local models write small diffs** | The reference setup is a CPU-only `qwen2.5-coder` on Ollama. Larger models produce better proposals; none of them produce a diff you should apply without reading. |
69
+ | **A cloud endpoint means your code leaves the device** | Cloud rows are refused until you approve, the approval is asked per task, cleartext to a remote host is refused outright, and API keys live in memory only — never in a config file, a log line or an error message. |
70
+
71
+ Running `python agent.py doctor` prints what this machine can actually reach — the local model list,
72
+ whether Docker is installed, and which key variables are set (their values are never printed).
73
+
74
+ ---
75
+
76
+ # ⚡ Quick Start
77
+
78
+ ## 1. Clone & Verify
79
+ ```bash
80
+ git clone https://github.com/mohamedsaad208/AI-Agent.git
81
+ cd AI-Agent
82
+ ```
83
+
84
+ ```bash
85
+ # First run, in order: what this machine can run, who answers, which models exist, what the folder
86
+ # grants, the offline proof, the five promises, and what `Send` may become. Exit 1 if a row blocks.
87
+ python agent.py setup # add --repo PATH to check a folder, --arabic, or --yes
88
+
89
+ # Self-diagnostics: python version, Ollama reachability, Docker, key variables (never their values)
90
+ python agent.py doctor
91
+
92
+ # Deterministic offline demo — no model, no network, and nothing from your repository is executed
93
+ python agent.py demo
94
+
95
+ # The suite. It is stdlib-only and needs no install step: `tests/*` add `src/` to sys.path itself.
96
+ python -m unittest discover -s tests
97
+
98
+ # Windows: the same command, with the interpreter this project is counted against.
99
+ # 3.11 is named on purpose — newer interpreters tally subtests differently, so the total moves.
100
+ run-tests.cmd
101
+ ```
102
+
103
+ The suite is the gate CI runs (`.github/workflows/ci.yml`: 3.11 on Windows and Linux, `compileall`,
104
+ `node --check` on the two UI scripts, a wheel build checked for the files the window needs). There is
105
+ no pytest, no `pip install -e .` step and no network access in it.
106
+
107
+ To open the local web window without an engine behind it — the same UI, scripted data, useful for
108
+ reading the interface before trusting a folder to it:
109
+
110
+ ```bash
111
+ python -m ai_code_engineer.webapp --fake --no-browser --port 8765
112
+ ```
113
+
114
+ ---
115
+
116
+ # 💻 Platform Guide
117
+
118
+ | Platform | GUI WebApp Launcher | Interactive CLI Launcher | Direct Terminal Command |
119
+ | :--- | :--- | :--- | :--- |
120
+ | **🪟 Windows** | Double-click `Run-Agent.bat` | Double-click `Run-Agent-CLI.bat` | `python desktop.pyw` |
121
+ | **🐧 Linux** | `./run-agent.sh` | `./run-agent-cli.sh` | `python3 desktop.pyw` |
122
+ | **🍎 macOS** | `./run-agent.sh` | `./run-agent-cli.sh` | `python3 desktop.pyw` |
123
+
124
+ <details>
125
+ <summary><strong>🐧 Linux & 🍎 macOS First-Time Setup</strong></summary>
126
+
127
+ Make the shell launchers executable:
128
+ ```bash
129
+ chmod +x run-agent.sh run-agent-cli.sh
130
+ ```
131
+
132
+ Launch the GUI:
133
+ ```bash
134
+ ./run-agent.sh
135
+ ```
136
+
137
+ Launch the interactive CLI:
138
+ ```bash
139
+ ./run-agent-cli.sh
140
+ ```
141
+ </details>
142
+
143
+ <details>
144
+ <summary><strong>⚙️ Advanced Headless CLI Usage</strong></summary>
145
+
146
+ ```powershell
147
+ # Index repository symbols
148
+ python agent.py map --repo examples/demo_repo
149
+
150
+ # Declare a folder Read-only for every surface — window and terminal alike — and lift it again
151
+ python agent.py read-only --repo examples/demo_repo
152
+ python agent.py read-only --repo examples/demo_repo --off
153
+
154
+ # Plan and propose code changes with local Ollama
155
+ python agent.py plan "Fix add in calculator.py so it adds two numbers" --repo examples/demo_repo --config profiles/local.toml
156
+
157
+ # Review proposed diff
158
+ python agent.py review "<session_id>"
159
+
160
+ # Apply approved proposal (cryptographically verified)
161
+ python agent.py apply "<session_id>" --approve "<sha256_hash>"
162
+
163
+ # Execute automated test suite
164
+ python agent.py verify "<session_id>"
165
+
166
+ # Check status and outcome of any session
167
+ python agent.py status "<session_id>"
168
+
169
+ # Export session transcript and diff report to Markdown or JSON
170
+ python agent.py export-session "<session_id>" --format markdown --out session-report.md
171
+
172
+ # Rollback if needed
173
+ python agent.py rollback "<session_id>" --approve "<sha256_hash>"
174
+ ```
175
+ </details>
176
+
177
+ ---
178
+
179
+ # 🖥️ Interactive Desktop WebApp
180
+
181
+ <div align="center">
182
+ <img src="assets/demo-webapp.gif" alt="AI Code Engineer Desktop WebApp Interface" width="100%"/>
183
+ </div>
184
+
185
+ The application features a sleek, local WebApp interface served on `127.0.0.1` with:
186
+ - 📂 **Multi-Project Workspace:** Manage isolated branches, project memory notes, and saved tasks.
187
+ - ⚡ **Review-First Diff Inspector:** Side-by-side **Diff / Now / Was** inspector with one-click rollbacks.
188
+ - 🤖 **Universal Model Selector:** Switch on the fly between local models (Ollama/DeepSeek) and cloud APIs (Groq, OpenAI, OpenRouter). A thinking model's deliberation arrives as its own collapsible row — capped, redacted, and never folded into the JSON the loop acts on — and an answer or a proposal streams in as it is written instead of appearing all at once after a minute of dots.
189
+ - 🧪 **Evidence-Based Checks:** Native test suite runner with JUnit XML proofs and self-healing fix rounds.
190
+ - 🐳 **A sandbox you can tick:** the same Checks card runs the project's command inside a pinned image —
191
+ no network, no capabilities, the project as a copy — and says so in the run line and in the exported
192
+ report, so a green from a container never reads as a green from your machine.
193
+ - 🧭 **First-Run Card:** On a machine that has never granted a folder, the window opens with the same
194
+ audit `agent setup` prints — what it can run, what answers, the five promises, the three positions —
195
+ built from rows that ask nothing over the network until you press **Run the checks**.
196
+ - ✏️ **Overrides you can see:** Settings → Overrides lists every configuration row the program is
197
+ running on — whose profile, whose field, whose signed file — and refuses to pretend a row it cannot
198
+ verify is in force. Both windows and the terminal read the same one file.
199
+ - 🌐 **Full Bilingual Arabic & RTL Support:** Dynamic Right-to-Left (RTL) layout when interacting in Arabic, with comprehensive Arabic localization across system notices, error diagnostics, step cards, and review audits.
200
+
201
+ Both windows are the same product: the web window and the `--tk` fallback share the engine, the
202
+ sentences and the decisions, and each keeps its proposed files in a viewer of its own — a sheet over the
203
+ chat there, a second window you can move beside the conversation on the desktop.
204
+
205
+ ---
206
+
207
+ # 🎯 Try These First
208
+
209
+ - **Fix a Bug with Automated Verification:**
210
+ *"Read the failing tests in `tests/test_auth.py` and inspect `src/auth.py`. Fix the token expiration validation without breaking backwards compatibility, then run tests."*
211
+
212
+ - **Implement a Multi-Phase Plan:**
213
+ Attach a `plan.md` file using the **+ Plan** button in the WebApp:
214
+ *"Implement Phase 1 from the attached plan only. Create missing DTO classes and verify syntax."*
215
+
216
+ - **Refactor with AST Context:**
217
+ *"Inspect the repository structure and refactor `UserService` to extract email notifications into an independent `NotificationService` interface."*
218
+
219
+ - **Autonomous Self-Healing Loop:**
220
+ Click **Run & Fix** on the Checks card to allow the agent to run the test suite, read compiler errors, and rewrite code until all tests turn green.
221
+
222
+ ---
223
+
224
+ # 🏗️ Architecture
225
+
226
+ ```
227
+ +---------------------------------------+
228
+ | Desktop WebApp / Native UI |
229
+ +---------------------------------------+
230
+ |
231
+ v
232
+ +---------------------------------------------------------------------------------+
233
+ | Agent Orchestration Engine |
234
+ | - Task Planner & Queue Manager - Step-by-Step Ledger (Planbook) |
235
+ | - AST Symbol Indexer & Repo Map - Persistent Project Memory |
236
+ +---------------------------------------------------------------------------------+
237
+ | |
238
+ v v
239
+ +--------------------------+ +-------------------------------+
240
+ | Model Providers | | Workspace & Security |
241
+ | - Ollama (Local) | | - Path Traversal Guard |
242
+ | - OpenRouter (Cloud) | | - Zero-Trust Secret Redactor |
243
+ | - OpenAI / Groq / Custom| | - SHA-256 Hash-Locked Diffs |
244
+ +--------------------------+ +-------------------------------+
245
+ |
246
+ v
247
+ +-------------------------------+
248
+ | Verification & Checks |
249
+ | - Toolchain Detectors |
250
+ | - JUnit XML Evidence Parser |
251
+ | - Autonomous Repair Loop |
252
+ | - Optional Docker Sandbox |
253
+ +-------------------------------+
254
+ ```
255
+
256
+ ---
257
+
258
+ # 📁 Where everything lives
259
+
260
+ | path | what it is |
261
+ | :--- | :--- |
262
+ | `src/ai_code_engineer/` | **the product.** `engine.py` runs the loop, `config.py` is the provider table, `providers.py` and `catalog.py` speak to a model, `runner.py` runs *your* project's command — here, or inside the one container shape the tool knows how to seal — `labels.py` holds every sentence in both languages, `intent.py` holds the three write positions and every refusal they speak, `modes.py` keeps a folder's position where the other window and the terminal both read it, `host.py` is the seam the two windows share, `redaction.py` keeps credentials out of what gets stored. |
263
+ | `src/ai_code_engineer/webapp/` | the local web window: `server.py` (loopback-only, per-launch token, Host/Origin/CSP), `controller.py` (the state the UI reads), `static/`. |
264
+ | `src/ai_code_engineer/gui.py` | the Tk window. Same engine, same sentences, different screen. |
265
+ | `tests/` | **the gate.** 1534 offline tests, stdlib `unittest`, no network. `doubles.py` and `helpers.py` are the shared fixtures. |
266
+ | `agent.py` · `desktop.pyw` · `launcher.py` | entry points: CLI, the desktop window, the interactive menu. |
267
+ | `profiles/` | TOML model presets. They name the *variable* holding a key and never a key. |
268
+ | `docs/` | plans, implementation status, code reviews — **local working notes, gitignored.** They quote this machine's paths, ports and counts, so they are kept on the device that measured them instead of shipped as product files. The rules they describe live in the modules' own docstrings, and the test suite is the gate: nothing in `src/`, `tests/` or CI reads a `docs/` file. |
269
+ | `tools/` | development aids for this repository — contrast checks, the dogfood ledger, wheel inspection, demo generators. |
270
+ | `sandbox/` | probes that produced a number someone quoted, kept so the number can be re-measured. |
271
+ | `examples/` | folders the agent is pointed at to try it out. |
272
+ | `archive/` | see `archive/README.md` — including why two experiment folders were **not** moved into it. |
273
+ | `.agent-chats/`, `.agent-runs/`, `.agent-projects.json` | what the tool writes next to itself: conversations, run records, the granted-folder registry. Ignored by git, and the history in them is addressed by absolute path. |
274
+
275
+ ---
276
+
277
+ # 🤝 Contributing
278
+
279
+ We welcome contributions from the global open-source community!
280
+
281
+ 1. **Fork** the repository.
282
+ 2. **Create your feature branch:**
283
+ ```bash
284
+ git checkout -b feature/amazing-feature
285
+ ```
286
+ 3. **Ensure all tests pass:**
287
+ ```bash
288
+ python -m unittest discover -s tests -v
289
+ ```
290
+ 4. **Commit your changes:**
291
+ ```bash
292
+ git commit -m "feat: add amazing new feature"
293
+ ```
294
+ 5. **Push to your fork and submit a Pull Request (PR).**
295
+
296
+ > ⚠️ **Branch Protection Note:** Direct pushes to `main` are restricted. All contributions must go through Pull Requests and pass automated verification.
297
+
298
+ ---
299
+
300
+ # 📄 License
301
+ This project is open-source software licensed under the [MIT License](LICENSE).
@@ -0,0 +1,23 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "ai-code-engineer"
7
+ version = "0.1.0"
8
+ description = "A local-first, review-first developer agent for Ollama and OpenRouter"
9
+ requires-python = ">=3.11"
10
+ dependencies = []
11
+
12
+ [project.scripts]
13
+ agent = "ai_code_engineer.cli:main"
14
+
15
+ [tool.setuptools.packages.find]
16
+ where = ["src"]
17
+
18
+ # webapp/server.py resolves `STATIC = Path(__file__).resolve().parent / "static"` at runtime,
19
+ # so the UI has to travel inside the package. Without this the built wheel contains only .py
20
+ # files and an installed app starts with a blank window. `static/*` also covers any file later
21
+ # added to that folder, which is the only place under src/ holding non-.py resources today.
22
+ [tool.setuptools.package-data]
23
+ "ai_code_engineer.webapp" = ["static/*"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,2 @@
1
+ """AI Code Engineer: bounded planning, review, and isolated verification."""
2
+ __version__ = "0.1.0"
@@ -0,0 +1,143 @@
1
+ """Read-only model discovery. Catalog requests do not submit source code or generate tokens.
2
+
3
+ Discovery and generation must read the *same* endpoint. They used to disagree — the Ollama list was
4
+ a literal loopback URL while generation used ``settings.endpoint`` — so a relocated service listed
5
+ zero models and the window called that "no models found" instead of "wrong address".
6
+ """
7
+ from decimal import Decimal, InvalidOperation
8
+ import os
9
+
10
+ from .config import Kind, OLLAMA, OPENROUTER, check_endpoint
11
+ from .errors import ProviderError
12
+ from .providers import request_json
13
+
14
+ LIVE = "live"
15
+ BUILT_IN = "built-in"
16
+
17
+
18
+ def ollama_models(endpoint: str = "") -> list[dict]:
19
+ base = check_endpoint(OLLAMA, endpoint)
20
+ data = request_json(base + "/api/tags", timeout=10)
21
+ entries = data.get("models")
22
+ if not isinstance(entries, list):
23
+ raise ProviderError("Ollama returned an invalid model list.")
24
+ found = {}
25
+ for item in entries:
26
+ if not isinstance(item, dict):
27
+ continue
28
+ name = item.get("name") or item.get("model")
29
+ if not isinstance(name, str) or not name:
30
+ continue
31
+ cloud = bool("cloud" in name.casefold() or item.get("remote_host") or item.get("remote_model"))
32
+ size = item.get("size")
33
+ size_label = f"{size / 1_000_000_000:.2f} GB" if isinstance(size, (int, float)) and size > 0 else ""
34
+ found[name] = {"id": name, "name": name, "cloud": cloud,
35
+ "description": ("Ollama cloud model — internet and Ollama account access required. "
36
+ "Billing depends on your Ollama plan." if cloud else "Runs locally on your device.")
37
+ + (" | " + size_label if size_label else "")}
38
+ return sorted(found.values(), key=lambda item: item["id"].casefold())
39
+
40
+
41
+ def price(value) -> Decimal | None:
42
+ try:
43
+ result = Decimal(str(value))
44
+ return result if result.is_finite() and result >= 0 else None
45
+ except (InvalidOperation, ValueError, TypeError):
46
+ return None
47
+
48
+
49
+ def openrouter_models(api_key: str | None = None, endpoint: str = "") -> list[dict]:
50
+ base = check_endpoint(OPENROUTER, endpoint)
51
+ data = request_json(base + "/models", key=api_key or os.environ.get("OPENROUTER_API_KEY"),
52
+ timeout=20, max_bytes=16_000_000)
53
+ entries = data.get("data")
54
+ if not isinstance(entries, list):
55
+ raise ProviderError("OpenRouter returned an invalid model catalog.")
56
+ found = {}
57
+ for item in entries:
58
+ if not isinstance(item, dict):
59
+ continue
60
+ name = item.get("id")
61
+ if not isinstance(name, str) or not name:
62
+ continue
63
+ architecture = item.get("architecture") or {}
64
+ if "text" not in architecture.get("output_modalities", []):
65
+ continue
66
+ pricing = item.get("pricing") or {}
67
+ prompt, completion = price(pricing.get("prompt")), price(pricing.get("completion"))
68
+ request_cost = price(pricing.get("request", "0"))
69
+ free = (name == "openrouter/free" or name.endswith(":free")) and all(
70
+ amount == 0 for amount in (prompt, completion, request_cost))
71
+ # A free suffix with missing/contradictory pricing is not advertised as free.
72
+ if (name.endswith(":free") or name == "openrouter/free") and not free:
73
+ continue
74
+ label = lambda cost: "unknown" if cost is None else f"${cost * 1_000_000:,.4f}"
75
+ pricing_text = "Free inference; provider limits apply." if free else (
76
+ f"Input {label(prompt)} / 1M tokens | Output {label(completion)} / 1M tokens")
77
+ if request_cost:
78
+ pricing_text += f" | ${request_cost} / request"
79
+ context = item.get("context_length")
80
+ context_text = f" | Context: {context:,}" if isinstance(context, int) else ""
81
+ params = item.get("supported_parameters") or []
82
+ json_support = "response_format" in params or "structured_outputs" in params
83
+ found[name] = {"id": name, "name": item.get("name", name), "free": free, "cloud": True,
84
+ "description": pricing_text + context_text +
85
+ (" | JSON output listed" if json_support else " | JSON support not listed")}
86
+ return sorted(found.values(), key=lambda item: item["id"].casefold())
87
+
88
+
89
+ def openai_models(base: str, api_key: str | None = None, *, cloud: bool = True) -> list[dict]:
90
+ """``GET {base}/models`` — the OpenAI-shaped list every compatible server answers.
91
+
92
+ The payload is a list of identifiers and nothing else that is useful here: pricing is not in
93
+ it, so the description says what the request can prove (where it runs) and nothing more.
94
+ """
95
+ data = request_json(base + "/models", key=api_key, timeout=20, max_bytes=8_000_000)
96
+ entries = data.get("data")
97
+ if not isinstance(entries, list):
98
+ entries = data.get("models")
99
+ if not isinstance(entries, list):
100
+ raise ProviderError("This provider returned an invalid model list.")
101
+ found = {}
102
+ for item in entries:
103
+ name = item.get("id") or item.get("name") if isinstance(item, dict) else item
104
+ if not isinstance(name, str) or not name.strip():
105
+ continue
106
+ found[name] = {"id": name, "name": name, "cloud": cloud,
107
+ "free": False,
108
+ "description": "Listed by this provider's /models endpoint. Pricing is not "
109
+ "reported there — check the service."}
110
+ return sorted(found.values(), key=lambda item: item["id"].casefold())
111
+
112
+
113
+ def built_in(kind: Kind) -> list[dict]:
114
+ """The names shipped with the row, used when the live request fails.
115
+
116
+ They are a starting point, not a claim that the service still lists them: the sentence that
117
+ presents them says so, because a stale id costs one refused request while a false promise of
118
+ "available" costs a task.
119
+ """
120
+ return [{"id": name, "name": name, "cloud": kind.cloud, "free": False,
121
+ "description": "Built-in name for this provider — not confirmed by a live request."}
122
+ for name in kind.verified]
123
+
124
+
125
+ def models_for(kind: Kind, endpoint: str = "",
126
+ api_key: str | None = None) -> tuple[list[dict], str]:
127
+ """Discover what a provider row has, and say *where* the answer came from.
128
+
129
+ Returns ``(entries, source)`` with ``source`` one of ``LIVE`` or ``BUILT_IN``. Only rows that
130
+ carry verified names fall back; a local server with nothing on it correctly reports zero models
131
+ rather than a list of guesses.
132
+ """
133
+ base = check_endpoint(kind, endpoint)
134
+ try:
135
+ if kind.shape == "ollama":
136
+ return ollama_models(base), LIVE
137
+ if kind.key == OPENROUTER.key:
138
+ return openrouter_models(api_key), LIVE
139
+ return openai_models(base, api_key, cloud=kind.cloud), LIVE
140
+ except ProviderError:
141
+ if kind.verified:
142
+ return built_in(kind), BUILT_IN
143
+ raise