ai-code-engineer 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. ai_code_engineer/__init__.py +2 -0
  2. ai_code_engineer/catalog.py +143 -0
  3. ai_code_engineer/chat.py +181 -0
  4. ai_code_engineer/cli.py +384 -0
  5. ai_code_engineer/config.py +405 -0
  6. ai_code_engineer/engine.py +1282 -0
  7. ai_code_engineer/errors.py +27 -0
  8. ai_code_engineer/git_integration.py +443 -0
  9. ai_code_engineer/gui.py +2646 -0
  10. ai_code_engineer/host.py +81 -0
  11. ai_code_engineer/ignore.py +269 -0
  12. ai_code_engineer/intent.py +222 -0
  13. ai_code_engineer/labels.py +871 -0
  14. ai_code_engineer/memory.py +91 -0
  15. ai_code_engineer/modes.py +156 -0
  16. ai_code_engineer/overrides.py +540 -0
  17. ai_code_engineer/planbook.py +192 -0
  18. ai_code_engineer/providers.py +404 -0
  19. ai_code_engineer/redaction.py +54 -0
  20. ai_code_engineer/repair.py +564 -0
  21. ai_code_engineer/report.py +352 -0
  22. ai_code_engineer/runner.py +854 -0
  23. ai_code_engineer/setup.py +386 -0
  24. ai_code_engineer/symbols.py +1286 -0
  25. ai_code_engineer/verification.py +218 -0
  26. ai_code_engineer/webapp/__init__.py +1 -0
  27. ai_code_engineer/webapp/__main__.py +45 -0
  28. ai_code_engineer/webapp/contract.py +36 -0
  29. ai_code_engineer/webapp/controller.py +3556 -0
  30. ai_code_engineer/webapp/fake.py +1141 -0
  31. ai_code_engineer/webapp/launch.py +108 -0
  32. ai_code_engineer/webapp/server.py +349 -0
  33. ai_code_engineer/webapp/static/app.css +780 -0
  34. ai_code_engineer/webapp/static/app.js +2118 -0
  35. ai_code_engineer/webapp/static/boot.js +19 -0
  36. ai_code_engineer/webapp/static/index.html +89 -0
  37. ai_code_engineer/webapp/static/tokens.css +173 -0
  38. ai_code_engineer/workspace.py +385 -0
  39. ai_code_engineer-0.1.0.dist-info/METADATA +7 -0
  40. ai_code_engineer-0.1.0.dist-info/RECORD +44 -0
  41. ai_code_engineer-0.1.0.dist-info/WHEEL +5 -0
  42. ai_code_engineer-0.1.0.dist-info/entry_points.txt +2 -0
  43. ai_code_engineer-0.1.0.dist-info/licenses/LICENSE +21 -0
  44. ai_code_engineer-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1141 @@
1
+ """A scripted controller so the interface is reviewable before the real one exists.
2
+
3
+ Nothing here reads or writes a project. It answers the same five methods the eventual
4
+ engine-backed controller will, with canned data that mirrors a real task, so design
5
+ decisions can be made against a running window instead of a screenshot.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import difflib
10
+ import secrets
11
+ import threading
12
+ import time
13
+ from datetime import datetime, timedelta, timezone
14
+ from pathlib import Path
15
+
16
+ from .. import config, git_integration, intent, labels, modes, overrides, repair, runner, setup
17
+ from ..errors import PolicyError
18
+ from .controller import MODES, PROJECT_ICONS as ICONS
19
+
20
+ BEFORE = """package com.demo.users;
21
+
22
+ import org.springframework.stereotype.Service;
23
+ import java.util.UUID;
24
+
25
+ @Service
26
+ public class UserService {
27
+
28
+ private final UserRepository repo;
29
+
30
+ public UserService(UserRepository repo) {
31
+ this.repo = repo;
32
+ }
33
+
34
+ public User create(String email) {
35
+ return repo.save(new User(UUID.randomUUID(), email));
36
+ }
37
+ }
38
+ """
39
+
40
+ AFTER = """package com.demo.users;
41
+
42
+ import java.util.Optional;
43
+ import org.springframework.stereotype.Service;
44
+ import java.util.UUID;
45
+
46
+ @Service
47
+ public class UserService {
48
+
49
+ private final UserRepository repo;
50
+
51
+ public UserService(UserRepository repo) {
52
+ this.repo = repo;
53
+ }
54
+
55
+ public User register(RegisterRequest req) {
56
+ if (repo.existsByEmail(req.email())) {
57
+ throw new DuplicateEmailException(req.email());
58
+ }
59
+ return repo.save(User.from(req));
60
+ }
61
+
62
+ public User create(String email) {
63
+ return repo.save(new User(UUID.randomUUID(), email));
64
+ }
65
+ }
66
+ """
67
+
68
+ CTRL_BEFORE = """@PostMapping("/register")
69
+ public ResponseEntity<UserDto> register(@RequestBody RegisterRequest req) {
70
+ return ResponseEntity.ok(dto(userService.create(req.email())));
71
+ }
72
+ """
73
+ CTRL_AFTER = """@PostMapping("/register")
74
+ public ResponseEntity<UserDto> register(@RequestBody RegisterRequest req) {
75
+ return ResponseEntity.ok(dto(userService.register(req)));
76
+ }
77
+
78
+ @ExceptionHandler(DuplicateEmailException.class)
79
+ public ResponseEntity<Void> duplicate(DuplicateEmailException exc) {
80
+ return ResponseEntity.status(HttpStatus.CONFLICT).build();
81
+ }
82
+ """
83
+ EXC = """package com.demo.users;
84
+
85
+ public class DuplicateEmailException extends RuntimeException {
86
+ public DuplicateEmailException(String email) {
87
+ super("An account already exists for " + email);
88
+ }
89
+ }
90
+ """
91
+
92
+ FILES = [
93
+ {"path": "src/main/java/com/demo/users/UserService.java", "before": BEFORE, "after": AFTER},
94
+ {"path": "src/main/java/com/demo/users/RegisterController.java", "before": CTRL_BEFORE, "after": CTRL_AFTER},
95
+ {"path": "src/main/java/com/demo/users/DuplicateEmailException.java", "before": None, "after": EXC},
96
+ ]
97
+
98
+ # The reasoning row, scripted. A thinking model answers twice and only one answer is the action, so the
99
+ # preview needs the other half visible to be designed against: capped, redacted, and opened as its own
100
+ # section rather than folded into the envelope.
101
+ REASONING_SAMPLE = ("The register flow needs the email check before the token mint, so the typed error "
102
+ "has to live outside the service or the controller cannot map it to 409.\n"
103
+ "Adding a guard inside save() would leave the duplicate row half-written, so the "
104
+ "check belongs at the top of register().\n"
105
+ "The password is never in this file — the store hashes it, so nothing to redact here.")
106
+
107
+ # The module graph, scripted. `show_graph` answers with this so the sheet, the column layout, the edge
108
+ # widths and the caption's caveat clauses can all be reviewed in the preview window with no engine and no
109
+ # folder attached. The back edge from `common-lib` to `auth` is deliberate: a shared library that imports
110
+ # a service is the cycle this project's own dogfood target really has, and a graph that only ever draws a
111
+ # clean tree never shows what it does when the layout cannot be exact.
112
+ GRAPH_NODES = [
113
+ {"name": "common-lib", "files": 9, "column": 0},
114
+ {"name": "auth", "files": 6, "column": 1},
115
+ {"name": "users-service", "files": 12, "column": 2},
116
+ {"name": "api-gateway", "files": 5, "column": 3},
117
+ ]
118
+ GRAPH_EDGES = [
119
+ {"from": "users-service", "to": "common-lib", "count": 6},
120
+ {"from": "auth", "to": "common-lib", "count": 4},
121
+ {"from": "api-gateway", "to": "users-service", "count": 3},
122
+ {"from": "users-service", "to": "auth", "count": 2},
123
+ {"from": "common-lib", "to": "auth", "count": 1},
124
+ ]
125
+ GRAPH_COLUMNS = 4
126
+ GRAPH_HIDDEN = 2
127
+
128
+ # The state names and their tones come from labels.py rather than a copy of them here, because a
129
+ # second table drifts: this one had no BLOCKED row and no DISCOVERING tone, so the two previews of
130
+ # a blocked task disagreed with the real window.
131
+ STATES = labels.STATES
132
+ TONE = labels.TONE
133
+
134
+
135
+ def _ago(minutes: int) -> str:
136
+ return (datetime.now(timezone.utc) - timedelta(minutes=minutes)).isoformat(timespec="seconds")
137
+
138
+
139
+ def _diff(before: str, after: str, name: str) -> list[str]:
140
+ return list(difflib.unified_diff(before.splitlines(), after.splitlines(),
141
+ fromfile="a/" + name, tofile="b/" + name, lineterm=""))
142
+
143
+
144
+ class FakeController:
145
+ """In-memory only. Restarts clean, which is exactly what a design review wants."""
146
+
147
+ def __init__(self) -> None:
148
+ self.prefs = {"style": "claude", "theme": "light", "collapsed": False}
149
+ self.state = "WAITING_APPROVAL"
150
+ self.view = "task"
151
+ self.busy = False
152
+ self.cancellable = False
153
+ self.chained = True
154
+ # The container switch is scripted as if this machine had Docker: on a machine without it the
155
+ # preview would show one frozen sentence, and the four answers are the thing to review here.
156
+ # The real window's `available` comes from `runner.sandbox_available()`.
157
+ self.sandbox_on = False
158
+ self.sandbox_image = ""
159
+ self.composer = "chat"
160
+ # The scripted twin of the controller's `declared` block: the folder's own position, kept by
161
+ # whichever surface wrote it. The preview needs the field or the lock on the badge cannot be
162
+ # reviewed at all, and a surface nobody has seen is exactly what a design review should catch.
163
+ self.declared = {"mode": "", "by": "", "at": ""}
164
+ # The real window keys the composer placeholder and the auto-write note off this switch, so
165
+ # the preview has to carry a live one or neither can be reviewed here.
166
+ self.auto_apply = False
167
+ # The chip is reviewable only if the scripted window can move it, so the branch is state
168
+ # here rather than a literal in snapshot(): see the `git_branch` action below.
169
+ self.branch = "main"
170
+ self.git_base = ""
171
+ # The restore pill is drawn from here. The real window only offers a restore after a rollback
172
+ # refuses because something edited the files afterwards, which a scripted window has no disk
173
+ # to reproduce — so `rollback` here raises the offer on purpose, to keep the surface reviewable.
174
+ self.git_restore_offer = None
175
+ self.timeout = 300
176
+ self.recipe = "Maven test"
177
+ # Three modules so the monorepo picker can be reviewed here: the scripted window answers to
178
+ # what the real one would list, and choosing a module changes the commands offered.
179
+ self.targets = [{"path": ".", "label": "demo2 (whole project)",
180
+ "recipes": ["Maven test", "Maven compile"]},
181
+ {"path": "backend", "label": "backend", "recipes": ["Maven test"]},
182
+ {"path": "frontend", "label": "frontend", "recipes": ["npm test",
183
+ "Node test runner"]}]
184
+ self.target = "."
185
+ # The model picker is a list you search, so the preview needs a catalog and a filter that
186
+ # behaves like the real view — the settings used to hand back one constant.
187
+ self.mode = "Ollama"
188
+ # The same table the real window builds, so the preview is reviewing the provider list and
189
+ # not a shorter copy of it.
190
+ self.modes = list(MODES)
191
+ self.catalogs = {label: (OLLAMA_MODELS if kind.shape == "ollama" else
192
+ FREE_MODELS if label == config.free_mode(kind) else
193
+ PAID_MODELS if label == config.paid_mode(kind) else SERVED_MODELS)
194
+ for label, kind in config.mode_rows()}
195
+ self.endpoints = {kind.key: config.default_endpoint(kind) for kind in config.KINDS
196
+ if config.default_endpoint(kind)}
197
+ self.profile = ""
198
+ self.catalog_source = {label: "live" for label in MODES}
199
+ self.model = "qwen2.5-coder:1.5b"
200
+ self.selections = {self.mode: self.model}
201
+ self.model_filter = ""
202
+ self.draft = ""
203
+ self.consent = False
204
+ self.key = ""
205
+ self.memory = "Java 17, Spring Boot 3.2. Do not add dependencies. " \
206
+ "Keep controllers thin; put rules in the service layer."
207
+ self.memory_info = "412 chars saved · demo2-8f2a.md · sent with every task here"
208
+ self.step = 2
209
+ self.runs = 0
210
+ self.fix_round = 0
211
+ self.auto_fix = False
212
+ self.tab = "diff"
213
+ self.pending: str | None = None
214
+ # The activity strip draws from this, so the preview needs a live-looking line of its own.
215
+ self.status_line = "Turn 3/12: asking qwen2.5-coder:1.5b..."
216
+ # The rows the Overrides section shows, held in memory: a preview that wrote to this machine's
217
+ # real signed file would be a preview that changed how the real window runs.
218
+ self.override_rows = [{"target": "ollama", "key": "timeout_seconds", "value": 600,
219
+ "state": "in force", "by": "cli", "at": "2026-09-30 09:40", "why": ""},
220
+ {"target": "*", "key": "context_chars", "value": 40000,
221
+ "state": "in force", "by": "web", "at": "2026-09-30 09:12", "why": ""},
222
+ {"target": "*", "key": "endpoint", "value": "",
223
+ "state": "refused", "by": "", "at": "",
224
+ "why": "per-provider"}]
225
+ # Two scripted rows so the strip can be laid out and reviewed: a plain wait, and one the
226
+ # user has sent to a chat of its own.
227
+ self.queue = [
228
+ {"id": "q1", "text": "And while you are in there, add a test for the duplicate case.",
229
+ "at": "14:04:11", "detached": False},
230
+ {"id": "q2", "text": "Why does the controller return 409 instead of 400?",
231
+ "at": "14:04:32", "detached": True},
232
+ ]
233
+ self.queue_held = False
234
+ self.log = [{"ts": "14:02:11", "kind": "plan", "text": "Reading attached plan: plan.md"}]
235
+ self.log_dropped = 0
236
+ # The one step row the preview has open. The real window fetches it from the session's records;
237
+ # here it is scripted, so the chevron, the sections and the file rows can be laid out.
238
+ self.step_detail: dict | None = None
239
+ self.messages = [
240
+ {"role": "assistant", "author": "AI Code Engineer", "time": "14:02",
241
+ "text": "Project **demo2** is attached, so this becomes a reviewable change request. "
242
+ "Nothing is written to disk until you approve the proposal."},
243
+ {"role": "user", "author": "You", "time": "14:02",
244
+ "text": "Implement step 2: register the user and reject duplicate emails. Keep the existing controller shape."},
245
+ # The step rows the engine now announces. Four of them, because the question this window
246
+ # exists to answer is what the thread looks like while a task works. Each carries the same
247
+ # `step` handle the real controller attaches, or the chevrons cannot be reviewed here at all.
248
+ {"role": "tool", "author": "Steps", "text": "\U0001f4c1 Scanning project files...",
249
+ "step": {"id": "st-1", "action": "list_files", "fields": {"count": 42},
250
+ "detail": labels.step_has_detail("list_files", {"count": 42})}},
251
+ # No digest on this one, so it draws without a chevron: the preview has to show what a row
252
+ # with nothing behind it looks like, or that state gets designed blind.
253
+ {"role": "tool", "author": "Steps",
254
+ "text": "\U0001f4d6 Reading file: src/main/java/com/demo/users/UserRepository.java",
255
+ "step": {"id": "st-2", "action": "read_file",
256
+ "fields": {"path": "src/main/java/com/demo/users/UserRepository.java"},
257
+ "detail": labels.step_has_detail("read_file", {"path": "x"})}},
258
+ {"role": "tool", "author": "Steps", "text": "\U0001f50d Searching code: existsByEmail",
259
+ "step": {"id": "st-3", "action": "search_code",
260
+ "fields": {"query": "existsByEmail", "count": 2},
261
+ "detail": labels.step_has_detail("search_code", {"count": 2})}},
262
+ {"role": "tool", "author": "Steps",
263
+ # Built through `labels` rather than typed out, because the preview is what a reviewer reads
264
+ # before approving a row's shape: a scripted sentence the engine could not produce reviews a
265
+ # fiction. The count is the record's own length so the two cannot disagree.
266
+ "text": labels.step_line(False, "model_reasoning", count=len(REASONING_SAMPLE),
267
+ detail=REASONING_SAMPLE),
268
+ "step": {"id": "st-r", "action": "model_reasoning",
269
+ "fields": {"count": len(REASONING_SAMPLE), "detail": REASONING_SAMPLE},
270
+ "detail": labels.step_has_detail("model_reasoning", {"detail": "x"})}},
271
+ {"role": "tool", "author": "Steps",
272
+ "text": "\u270d\ufe0f Proposed changes for 3 file(s): UserService.java, RegisterController.java, "
273
+ "DuplicateEmailException.java",
274
+ "step": {"id": "st-4", "action": "propose",
275
+ "fields": {"count": 3, "names": ["UserService.java", "RegisterController.java",
276
+ "DuplicateEmailException.java"]},
277
+ "detail": labels.step_has_detail("propose", {"names": ["UserService.java"]})}},
278
+ {"role": "tool", "author": "Steps",
279
+ "text": "\u2699\ufe0f Ran mvn -B test — failed · exit 1 · 41.2s · 14 tests",
280
+ "step": {"id": "st-5", "action": "executed", "fields": {"command": "mvn -B test"},
281
+ "detail": labels.step_has_detail("executed", {"command": "mvn -B test"})}},
282
+ {"role": "assistant", "author": "AI Code Engineer", "time": "14:03",
283
+ "text": "I read `UserRepository.java` and `RegisterRequest.java` first, then proposed 3 files. "
284
+ "Duplicate emails now raise a typed error the controller maps to **409**.\n\n"
285
+ "```java\n// duplicate guard added before persisting\n"
286
+ "public User register(RegisterRequest req) {\n"
287
+ " if (repo.existsByEmail(req.email())) {\n"
288
+ " throw new DuplicateEmailException(req.email());\n"
289
+ " }\n return repo.save(User.from(req));\n}\n```\n\n"
290
+ "- `RegisterController.java` keeps the existing method shape.\n"
291
+ "- A new exception type means no string matching in the controller."},
292
+ {"role": "tool", "author": "Changes",
293
+ "text": "3 files · UserService.java, RegisterController.java, DuplicateEmailException.java"},
294
+ ]
295
+ self._replies: dict[str, threading.Event] = {}
296
+ self._answers: dict[str, dict] = {}
297
+ self._emit = None
298
+ # The first-run card, built from the real `setup` rows so the preview cannot drift from the
299
+ # sentences that actually ship. Only the provider and model lines are scripted: they are the
300
+ # two a machine without Ollama running would otherwise leave blank on a design surface.
301
+ self.setup_open = True
302
+ self.setup_rows = self._script_setup()
303
+ self.setup_demo: dict | None = None
304
+
305
+ def _script_setup(self) -> list[dict]:
306
+ rows = setup.audit(repo="", provider="ollama",
307
+ endpoint=config.default_endpoint(config.OLLAMA),
308
+ model=self.model, probe=False)
309
+ out = []
310
+ for row in rows:
311
+ if row["id"] == "provider":
312
+ row = {"id": "provider", "status": "ok", "advice": "",
313
+ "text": "Ollama answers at " + config.default_endpoint(config.OLLAMA) + "."}
314
+ elif row["id"] == "model":
315
+ row = {"id": "model", "status": "ok", "advice": "",
316
+ "text": "11 models (10 of them local). Using " + self.model + "."}
317
+ elif row["id"] == "demo":
318
+ row = dict(row, text="The offline proof has not been run yet.")
319
+ out.append(row)
320
+ return out
321
+
322
+ # ----------------------------- contract -----------------------------
323
+ @property
324
+ def reading_only(self) -> bool:
325
+ """The same question the real controller asks every write and every run gate."""
326
+ return intent.read_only(self.composer)
327
+
328
+ def snapshot(self) -> dict:
329
+ return {
330
+ "prefs": self.prefs, "busy": self.busy, "cancellable": self.cancellable, "pending": self.pending,
331
+ "status": self.status_line,
332
+ "current": "s-1", "header": {"title": "Spring Boot authentication",
333
+ "subtitle": "demo2 · step %d of 5 · %s" % (self.step, self.model)},
334
+ "project": {"name": "demo2", "path": "D:\\AI\\AI-Agent\\examples\\demo2"},
335
+ # The branch in front and the composer under it are what the header, the mode badge
336
+ # and the drop targets read; a preview without them shows none of that furniture.
337
+ "branch": {"kind": "project", "key": "demo2", "id": "s-1", "bound": False,
338
+ "projectName": "demo2"},
339
+ "composer": self.composer,
340
+ "declared": {"sealed": self.declared["mode"] == intent.READ,
341
+ "mode": self.declared["mode"], "by": intent.source(self.declared["by"]),
342
+ "at": self.declared["at"],
343
+ "note": (intent.followed(self.declared["by"], self.declared["at"])
344
+ if self.declared["mode"] == intent.READ
345
+ and not intent.read_only(self.composer) else "")},
346
+ "icons": list(ICONS),
347
+ "git": dict({"repo": True, "branch": self.branch, "detached": False,
348
+ "head": "9f3c21a", "dirty": 2},
349
+ **({"base": self.git_base} if self.git_base else {}),
350
+ **({"restore": self.git_restore_offer}
351
+ if self.git_restore_offer else {})),
352
+ "banner": ({"count": len(FILES),
353
+ "text": labels.applied_note(arabic=False, count=len(FILES))}
354
+ if self.auto_apply and self.state in labels.MUTABLE_STATES
355
+ else {"count": 0, "text": ""}),
356
+ "plan": {"name": "plan.md", "step": self.step, "total": 5, "verified": self.step - 1,
357
+ "note": "Send works on step %d" % self.step,
358
+ "steps": [{"id": 1, "title": "Project foundation", "status": "verified", "current": False},
359
+ {"id": 2, "title": "Register a user", "status": "in_progress", "current": self.step == 2},
360
+ {"id": 3, "title": "Login and issue a JWT", "status": "pending", "current": self.step == 3},
361
+ {"id": 4, "title": "Protect the routes", "status": "pending", "current": self.step == 4},
362
+ {"id": 5, "title": "Refresh-token rotation", "status": "pending", "current": self.step == 5}]},
363
+ "provider": {"mode": self.mode, "modes": self.modes, "model": self.model,
364
+ "models": self.visible_models()},
365
+ "connection": self.connection_info(),
366
+ "overrides": self.overrides_info(),
367
+ "recipes": self.target_recipes(), "recipe": self.recipe,
368
+ "targets": [{"path": row["path"], "label": row["label"]} for row in self.targets],
369
+ "target": self.target, "targetLabel": self.target_label(),
370
+ "canRun": self.state in {"APPLIED_UNVERIFIED", "CHECKS_PASSED", "VERIFICATION_FAILED", "VERIFICATION_BLOCKED"},
371
+ "runInfo": "No command has run yet." if not self.runs else
372
+ f"{self.runs} run(s). Last: Maven test — passed (exit 0, 41.2s)",
373
+ "runWarning": labels.run_warning(arabic=False),
374
+ "sandbox": {"on": bool(self.sandbox_on), "image": self.sandbox_image, "available": True,
375
+ "note": labels.note(runner.sandbox_state(self.sandbox_on,
376
+ self.sandbox_image, True))},
377
+ # The loop's budget, from the same constant the real controller reads it from: a field the
378
+ # preview never sends is a field the window is never drawn with.
379
+ "fixRounds": {"of": repair.MAX_FIX_ROUNDS, "spent": self.fix_round},
380
+ "memory": {"info": "412 chars saved"},
381
+ "settings": {"project": "D:\\AI\\AI-Agent\\examples\\demo2", "plan": "plan.md", "chained": self.chained, "auto_apply": self.auto_apply,
382
+ "timeout": self.timeout, "model_info": self._model_info(),
383
+ "memory": self.memory,
384
+ "memory_info": self.memory_info,
385
+ "consent": self.consent},
386
+ "draft": self.draft,
387
+ "queue": {"items": self.queue, "held": self.queue_held, "elsewhere": 1,
388
+ **labels.queue_notes(False, 1, False)},
389
+ "artifact": labels.artifact_card(self.state, arabic=False, count=len(FILES),
390
+ project="demo2",
391
+ summary="Duplicate emails now raise a typed error the "
392
+ "controller maps to 409; the existing create() "
393
+ "behaviour is untouched.",
394
+ written=self.state in labels.MUTABLE_STATES,
395
+ has_project=True),
396
+ # The preview is where a card like this gets reviewed, so it has to carry the same shape the
397
+ # real controller sends — including the counts the header prints.
398
+ "setup": {"show": self.setup_open, "rows": self.setup_rows,
399
+ "counts": setup.counts(self.setup_rows),
400
+ "tally": setup.tally(setup.counts(self.setup_rows)),
401
+ "demo": self.setup_demo, "busy": False},
402
+ "review": self._review(), "messages": self.messages, "log": self.log,
403
+ "log_dropped": self.log_dropped, "log_note": "", "step_detail": self.step_detail,
404
+ "projects": [{"key": "demo2", "name": "demo2", "initials": "d2",
405
+ "path": "D:\\AI\\AI-Agent\\examples\\demo2",
406
+ "chats": [{"id": "s-1", "title": "Spring Boot authentication", "state": self.state,
407
+ "updated": _ago(120), "busy": self.busy},
408
+ {"id": "s-2", "title": "Add refresh-token rotation", "state": "APPLIED_UNVERIFIED",
409
+ "updated": _ago(300), "busy": False},
410
+ {"id": "s-3", "title": "Fix password hashing", "state": "CHECKS_PASSED",
411
+ "updated": _ago(1440), "busy": False}]},
412
+ {"key": "demo_repo", "name": "demo_repo", "initials": "dr",
413
+ "path": "D:\\AI\\AI-Agent\\examples\\demo_repo",
414
+ "chats": [{"id": "s-4", "title": "Fix add in calculator.py", "state": "CHECKS_PASSED",
415
+ "updated": _ago(4300), "busy": False},
416
+ {"id": "s-5", "title": "Extract a UserService", "state": "ROLLED_BACK",
417
+ "updated": _ago(5760), "busy": False}]}],
418
+ "chats": [{"id": "c-1", "title": "Explain Maven surefire reports", "updated": _ago(360)},
419
+ {"id": "c-2", "title": "Best way to gate a plan step?", "updated": _ago(2880)}],
420
+ }
421
+
422
+ def _review(self) -> dict:
423
+ reading = self.reading_only
424
+ files = []
425
+ for change in FILES:
426
+ lines = _diff(change["before"] or "", change["after"], change["path"])
427
+ add = sum(1 for l in lines if l.startswith("+") and not l.startswith("+++"))
428
+ dele = sum(1 for l in lines if l.startswith("-") and not l.startswith("---"))
429
+ files.append({"path": Path(change["path"]).name, "kind": "A" if change["before"] is None else "M",
430
+ "add": add, "del": dele})
431
+ chosen = FILES[min(self._file, len(FILES) - 1)]
432
+ name = Path(chosen["path"]).name
433
+ return {
434
+ "state": STATES.get(self.state, self.state), "tone": TONE.get(self.state, ""),
435
+ "title": "Implement step 2: register the user and reject duplicate emails",
436
+ "detail": f"{len(FILES)} files · attached plan plan.md · proposal 8f2a…c41b",
437
+ "canApply": self.state == "WAITING_APPROVAL" and not reading,
438
+ "canMutate": self.state in labels.MUTABLE_STATES,
439
+ # Roll back answers to a wider set than the other two, because an interrupted apply is
440
+ # exactly when the escape has to be on screen. Without this field the preview window —
441
+ # the one the design is reviewed in — shows a button that can never light up.
442
+ "canRollback": (self.state in labels.MUTABLE_STATES | labels.INTERRUPTED_STATES) and not reading,
443
+ "files": files, "selected": self._file, "tab": self.tab,
444
+ "view": {"diff": _diff(chosen["before"] or "", chosen["after"], name),
445
+ "before": (chosen["before"] or "").splitlines(),
446
+ "after": chosen["after"].splitlines(),
447
+ "checks": ["Proposed checks (not execution results):",
448
+ "• mvn -B test passes", "• duplicate email returns 409",
449
+ "• existing create() behaviour unchanged",
450
+ "", "Latest check: not run yet." if not self.runs else
451
+ f"Latest check: passed · {self.runs} run(s) recorded"]},
452
+ }
453
+
454
+ _file = 0
455
+
456
+ def visible_models(self, mode: str | None = None) -> list[dict]:
457
+ """The catalog as filtered — the same read-only view `controller.visible_models` computes.
458
+
459
+ The preview has to narrow the way the real window does, or the filter gets reviewed here as
460
+ a box that lists everything and shipped as one that loses models.
461
+ """
462
+ query = self.model_filter.strip().casefold()
463
+ entries = self.catalogs.get(mode if mode is not None else self.mode, [])
464
+ if not query:
465
+ return list(entries)
466
+ return [entry for entry in entries
467
+ if query in " ".join([entry.get("id", ""), entry.get("name", ""),
468
+ entry.get("description", "")]).casefold()]
469
+
470
+ def _model_info(self) -> str:
471
+ entry = next((item for item in self.catalogs.get(self.mode, [])
472
+ if item["id"] == self.model), None)
473
+ if entry:
474
+ return entry["name"] + " — " + entry["description"]
475
+ loaded = len(self.catalogs.get(self.mode, []))
476
+ shown = len(self.visible_models())
477
+ if shown != loaded:
478
+ return (f'{shown} of {loaded} models match "{self.model_filter.strip()}". '
479
+ "Clear the filter to see the rest.")
480
+ return f"{loaded} models available. Select one from the list."
481
+
482
+ def overrides_info(self) -> dict:
483
+ """The preview's own rows, kept in memory.
484
+
485
+ Reviewing the Overrides section against a scripted list is what makes the design checkable
486
+ without touching this machine's real signing key or its real settings.
487
+ """
488
+ return {"rows": overrides.spoken(self.override_rows, arabic=False),
489
+ "keys": overrides.fields(),
490
+ "targets": [overrides.EVERY] + [kind.key for kind in config.KINDS],
491
+ "path": "(preview) " + overrides.FILE, "kind": self.kind().key,
492
+ "note": overrides.scope(arabic=False)}
493
+
494
+ def kind(self):
495
+ return config.MODE_KIND.get(self.mode, config.DEFAULT_KIND)
496
+
497
+ def set_override(self, payload: dict) -> None:
498
+ """Sign nothing here: the preview stores the same shape the real window writes to disk."""
499
+ target = str(payload.get("target", "")).strip().casefold() or overrides.EVERY
500
+ key = str(payload.get("key", "")).strip().casefold()
501
+ raw = payload.get("value", "")
502
+ field = next((item for item in overrides.fields() if item["key"] == key), None)
503
+ if field is None or (target == overrides.EVERY and key in overrides.PER_PROVIDER):
504
+ self.status_line = overrides.bad_value(key, "unknown" if field is None else
505
+ "per-provider", arabic=False)
506
+ return
507
+ if field["number"]:
508
+ try:
509
+ value = int(str(raw).strip())
510
+ except ValueError:
511
+ self.status_line = overrides.bad_value(key, "not a number", arabic=False)
512
+ return
513
+ low, high = config.LIMITS[key]
514
+ if not low <= value <= high:
515
+ self.status_line = f"{key} must be between {low} and {high}."
516
+ return
517
+ else:
518
+ value = str(raw).strip()
519
+ self.override_rows = [row for row in self.override_rows
520
+ if not (row["target"] == target and row["key"] == key)]
521
+ self.override_rows.insert(0, {"target": target, "key": key, "value": value, "state": "in force",
522
+ "by": "web", "at": "just now", "why": ""})
523
+ self.status_line = overrides.written({"key": key, "value": value, "target": target},
524
+ arabic=False)
525
+
526
+ def unset_override(self, payload: dict) -> None:
527
+ target = str(payload.get("target", "")).strip().casefold() or overrides.EVERY
528
+ key = str(payload.get("key", "")).strip().casefold()
529
+ before = len(self.override_rows)
530
+ self.override_rows = [row for row in self.override_rows
531
+ if not (row["target"] == target and row["key"] == key)]
532
+ self.status_line = (overrides.removed(key, arabic=False) if len(self.override_rows) < before
533
+ else overrides.absent(key, arabic=False))
534
+
535
+ def connection_info(self) -> dict:
536
+ """The same block the real controller sends, so the Connection tab is reviewable here.
537
+
538
+ The scripted window never contacts a provider: an endpoint change re-reads the row's own
539
+ rules and nothing else, which is what makes it safe to click in a preview.
540
+ """
541
+ kind = config.MODE_KIND.get(self.mode, config.DEFAULT_KIND)
542
+ endpoint = self.endpoints.get(kind.key, "") or config.default_endpoint(kind)
543
+ return {"kind": kind.key, "label": kind.label, "endpoint": endpoint,
544
+ "default_endpoint": config.default_endpoint(kind), "cloud": kind.cloud, "shape": kind.shape,
545
+ "needs_key": kind.needs_key, "key_env": kind.key_env,
546
+ "consent": config.needs_consent(kind, endpoint),
547
+ "paid": self.mode == config.paid_mode(kind),
548
+ "profile": self.profile, "profiles": config.profile_names(),
549
+ "source": self.catalog_source.get(self.mode, ""),
550
+ "key_present": bool(self.key.strip())}
551
+
552
+ def set_profile(self, label: str) -> None:
553
+ """Move the preview onto a profile's row — the same three fields the real window sets."""
554
+ self.profile = label
555
+ if not label:
556
+ return
557
+ try:
558
+ settings = config.load_profile(label)
559
+ except Exception: # noqa: BLE001 - a preview reports, never raises
560
+ return
561
+ self.mode = config.mode_for(settings.provider, settings.model) or self.mode
562
+ if settings.endpoint:
563
+ self.endpoints[config.MODE_KIND[self.mode].key] = settings.endpoint
564
+ self.model = settings.model
565
+
566
+ def project_info(self, key: str) -> dict:
567
+ """One drawer payload per scripted project, with the same keys the real one sends.
568
+
569
+ The drawer reads every field it draws, so an answer that is missing one shows up as a
570
+ blank instead of an error and the drift goes unnoticed.
571
+ """
572
+ if key not in PROJECT_INFO:
573
+ raise PolicyError("Unknown project.")
574
+ return dict(PROJECT_INFO[key])
575
+
576
+ def list_dir(self, path: str, want_files=None) -> dict:
577
+ root = Path(path or Path.home())
578
+ if not root.is_dir():
579
+ root = Path.home()
580
+ dirs = sorted((p for p in root.iterdir() if _visible_dir(p)), key=lambda p: p.name.casefold())
581
+ files = sorted((p for p in root.iterdir() if p.is_file() and p.suffix.lower() in {".md", ".txt"}),
582
+ key=lambda p: p.name.casefold()) if want_files else []
583
+ return {"path": str(root), "parent": str(root.parent) if root.parent != root else None,
584
+ "dirs": [{"name": p.name, "path": str(p)} for p in dirs[:300]],
585
+ "files": [{"name": p.name, "path": str(p)} for p in files[:300]]}
586
+
587
+ def set_reply(self, request_id: str, reply: dict) -> None:
588
+ self._answers[request_id] = reply
589
+ waiter = self._replies.pop(request_id, None)
590
+ if waiter:
591
+ waiter.set()
592
+
593
+ # ----------------------------- actions -----------------------------
594
+ def action(self, type: str, payload: dict, emit) -> dict | None:
595
+ self._emit = emit
596
+ if type == "set_override":
597
+ self.set_override(payload)
598
+ return None
599
+ if type == "unset_override":
600
+ self.unset_override(payload)
601
+ return None
602
+ if type == "send":
603
+ return self._send(payload.get("text", ""), emit)
604
+ if type == "apply":
605
+ if self.reading_only:
606
+ return self._refuse(intent.no_write("Apply"))
607
+ return self._confirm_then("apply", emit)
608
+ if type == "run":
609
+ return self._run(payload.get("fix"), emit)
610
+ if type == "rollback":
611
+ if self.reading_only:
612
+ return self._refuse(intent.no_write("Roll back"))
613
+ self.state = "ROLLED_BACK"
614
+ self._note(emit, "rolled_back", "Task changes rolled back.")
615
+ self.git_restore_offer = {"commit": "9f3c21a", "paths": len(FILES)}
616
+ elif type == "git_restore":
617
+ if self.reading_only:
618
+ # The preview has to refuse the escalation exactly like the real window does, or the
619
+ # design gets reviewed against a button the shipped thing will not press.
620
+ return self._refuse(intent.no_write("Restoring files from git"))
621
+ offer = self.git_restore_offer or {"commit": "9f3c21a", "paths": len(FILES)}
622
+ self.git_restore_offer = None
623
+ text = labels.restore_done(arabic=False, commit=offer["commit"],
624
+ restored=[item["path"] for item in FILES], skipped=[])
625
+ self.messages.append({"role": "tool", "author": "Git", "text": text, "time": _clock()})
626
+ emit({"kind": "message", "message": self.messages[-1]})
627
+ self._note(emit, "git_restore", text)
628
+ elif type == "git_branch":
629
+ if self.reading_only:
630
+ # A switch rewrites the tracked files, so a read-only preview refuses it for the same
631
+ # reason the controller does.
632
+ return self._refuse(intent.no_write("Switching branches"))
633
+ # The scripted branch comes from the same generator the real window uses, so reviewing
634
+ # the chip here shows the name a user would actually get.
635
+ back = str(payload.get("back", "")).strip()
636
+ if back:
637
+ self.branch, self.git_base = back, ""
638
+ text = labels.branch_switched(arabic=False, branch=back)
639
+ else:
640
+ self.git_base = self.branch
641
+ self.branch = git_integration.task_branch_name("Review the preview script",
642
+ "fakesession01")
643
+ text = labels.branch_started(arabic=False, branch=self.branch, back=self.git_base)
644
+ self.messages.append({"role": "tool", "author": "Git", "text": text, "time": _clock()})
645
+ emit({"kind": "message", "message": self.messages[-1]})
646
+ self._note(emit, "git_branch", text)
647
+ elif type == "verify":
648
+ self.state = "VERIFICATION_BLOCKED"
649
+ self._note(emit, "verify", "Syntax checks finished. Project tests have not run.")
650
+ elif type == "stop":
651
+ self.busy = self.cancellable = False
652
+ self.pending = None
653
+ self.state = "CANCELLED"
654
+ self._note(emit, "cancelled", "Task cancelled. No project files were changed.")
655
+ elif type == "set_style":
656
+ self.prefs["style"] = payload.get("style", "claude")
657
+ elif type == "set_theme":
658
+ self.prefs["theme"] = payload.get("theme", "light")
659
+ elif type == "pick_project":
660
+ self._ask("folder", {"title": "Choose a project folder", "hint": "Nothing is read until you approve a proposal."}, emit)
661
+ elif type == "select_file":
662
+ self._file = int(payload.get("index", 0))
663
+ elif type == "select_tab":
664
+ self.tab = str(payload.get("tab", "diff"))
665
+ elif type == "queue_add":
666
+ text = str(payload.get("text", "")).strip()
667
+ if text:
668
+ self.queue.append({"id": "q%d" % (len(self.queue) + 3), "text": text,
669
+ "at": "now", "detached": False})
670
+ self.queue_held = False
671
+ elif type == "queue_edit":
672
+ for item in self.queue:
673
+ if item["id"] == payload.get("id"):
674
+ item["text"] = str(payload.get("text", "")).strip()
675
+ self.queue_held = False
676
+ elif type == "queue_drop":
677
+ self.queue = [item for item in self.queue if item["id"] != payload.get("id")]
678
+ elif type == "queue_now":
679
+ picked = [item for item in self.queue if item["id"] == payload.get("id")]
680
+ if picked:
681
+ self.queue.remove(picked[0])
682
+ self.queue.insert(0, picked[0])
683
+ self.queue_held = False
684
+ elif type == "queue_chat":
685
+ for item in self.queue:
686
+ if item["id"] == payload.get("id"):
687
+ item["detached"] = True
688
+ self.queue_held = False
689
+ elif type == "queue_resume":
690
+ self.queue_held = False
691
+ elif type == "set_chained":
692
+ self.chained = bool(payload.get("value"))
693
+ elif type == "sandbox":
694
+ # Both halves in one action, exactly as the real window sends them: the sentence under the
695
+ # box is the thing being reviewed, and it changes on the tick and on every keystroke.
696
+ if "on" in payload:
697
+ self.sandbox_on = bool(payload.get("on"))
698
+ if "image" in payload:
699
+ self.sandbox_image = str(payload.get("image") or "")
700
+ elif type == "set_auto_apply":
701
+ # The pill next to Send and the composer placeholder both key off this, so a preview
702
+ # that ignored the click could not be used to review either of them.
703
+ if self.reading_only and not bool(payload.get("value")):
704
+ pass # turning a switch that is already off asks nothing of anyone
705
+ elif self.reading_only:
706
+ self._refuse(intent.no_auto_apply())
707
+ else:
708
+ self.auto_apply = bool(payload.get("value"))
709
+ elif type == "set_composer":
710
+ self.composer = intent.normalise(payload.get("value"))
711
+ if self.composer in (intent.READ, intent.CHANGE):
712
+ # Read and Change are the two positions that say what happens to the folder's files,
713
+ # so either is written down — the same rule the real window keeps in `modes`.
714
+ self.declared = {"mode": self.composer, "by": modes.WEB, "at": _clock()}
715
+ # The switch belongs to Change mode, exactly as it does in the real window: a folder
716
+ # moved to Read-only mid-session has to stop showing "writes itself" on the next pill.
717
+ self.auto_apply = self.auto_apply and self.composer == "change"
718
+ elif type == "set_timeout":
719
+ self.timeout = int(payload.get("value") or 300)
720
+ elif type == "set_recipe":
721
+ self.recipe = str(payload.get("value") or self.recipe)
722
+ elif type == "set_target":
723
+ self.set_target(str(payload.get("value") or ""))
724
+ elif type == "set_filter":
725
+ self.model_filter = str(payload.get("value", ""))
726
+ elif type == "set_model":
727
+ self.model = str(payload.get("value", ""))
728
+ self.selections[self.mode] = self.model
729
+ elif type == "set_mode":
730
+ value = str(payload.get("value", ""))
731
+ if value in self.modes:
732
+ self.selections[self.mode] = self.model
733
+ self.mode, self.model_filter, self.consent = value, "", False
734
+ self.model = self.selections.get(value, "")
735
+ elif type == "set_endpoint":
736
+ # Checked by the same rule the real window uses, so a rejected URL is refused here too
737
+ # and the tab can be reviewed against a refusal rather than only against a success.
738
+ kind = config.MODE_KIND.get(self.mode, config.DEFAULT_KIND)
739
+ text = str(payload.get("value", "")).strip() or config.default_endpoint(kind)
740
+ try:
741
+ self.endpoints[kind.key] = config.check_endpoint(kind, text)
742
+ except PolicyError as exc:
743
+ emit({"kind": "toast", "text": str(exc)})
744
+ else:
745
+ emit({"kind": "toast", "text": "%s now points at %s" % (kind.label, self.endpoints[kind.key])})
746
+ elif type == "set_profile":
747
+ self.set_profile(str(payload.get("value", "")))
748
+ elif type == "refresh_models":
749
+ emit({"kind": "toast",
750
+ "text": "%d models listed for %s in this preview — the real window asks the "
751
+ "provider." % (len(self.visible_models()), self.mode)})
752
+ elif type == "set_draft":
753
+ self.draft = str(payload.get("text", ""))
754
+ elif type == "set_key":
755
+ self.key = str(payload.get("value", ""))
756
+ elif type == "set_consent":
757
+ self.consent = bool(payload.get("value"))
758
+ elif type == "set_collapsed":
759
+ self.prefs["collapsed"] = bool(payload.get("value"))
760
+ elif type == "save_memory":
761
+ self.memory = str(payload.get("text", ""))
762
+ self.memory_info = ("%d chars saved · demo2-8f2a.md · sent with every task here"
763
+ % len(self.memory)) if self.memory else "Notes cleared for this folder."
764
+ elif type == "apply_block":
765
+ # The block button writes nothing here either: it opens a proposal, which is the state
766
+ # the review cards are drawn for.
767
+ if self.reading_only:
768
+ return self._refuse(intent.no_proposal())
769
+ self.state = "WAITING_APPROVAL"
770
+ emit({"kind": "toast", "text": "Proposing %s — review the diff, then Apply"
771
+ % str(payload.get("path", "that file"))})
772
+ elif type == "setup_check":
773
+ self.setup_open = True
774
+ emit({"kind": "toast", "text": "Checks done — 4 ok, 1 to watch, 0 blocking. "
775
+ "The provider rows are scripted here."})
776
+ elif type == "setup_demo":
777
+ # Scripted, and it says so: the preview never writes anywhere, so the proof it shows is the
778
+ # sentence the real run prints rather than a run of its own.
779
+ self.setup_demo = {"proposal_apply_rollback": "passed", "llm_used": False,
780
+ "static_checks": {"status": "unverified"},
781
+ "note": "Scripted in this preview: the real proof runs in a temporary "
782
+ "folder."}
783
+ self.setup_rows = [setup.demo_row(self.setup_demo, arabic=False) if item["id"] == "demo" else item
784
+ for item in self.setup_rows]
785
+ self.setup_open = True
786
+ emit({"kind": "state", "data": self.snapshot()})
787
+ elif type == "setup_hide":
788
+ # "Don't show this again" is a decision the preview has to be able to demonstrate, not only
789
+ # the real window: the note under the button promises it stops appearing at launch.
790
+ self.setup_open = False
791
+ elif type == "step_detail":
792
+ self.open_step(str(payload.get("id", "")))
793
+ elif type == "show_graph":
794
+ # The one scripted verb that answers with data rather than with an event: the sheet is drawn
795
+ # from the reply, so the preview has to hand back the same shape the real window does.
796
+ return self.open_graph()
797
+ elif type in PREVIEW_ONLY:
798
+ emit({"kind": "toast", "text": PREVIEW_ONLY[type]})
799
+ else:
800
+ # An action nobody scripted used to fall off the end silently, which is how the model
801
+ # filter sat dead in this window for a week: say it out loud instead.
802
+ emit({"kind": "toast", "text": "Not scripted in this preview: " + str(type)[:40]})
803
+ return None
804
+
805
+ def _send(self, text: str, emit) -> None:
806
+ if not text:
807
+ return None
808
+ self.messages.append({"role": "user", "author": "You", "time": _clock(), "text": text})
809
+ emit({"kind": "message", "message": self.messages[-1]})
810
+ if self.reading_only:
811
+ return self._analyse(emit)
812
+ self.busy = self.cancellable = True
813
+ self.pending = "connecting to the model…"
814
+ emit({"kind": "busy", "value": True, "cancellable": True})
815
+
816
+ def work():
817
+ for line, note in [("plan", "Reading attached plan: plan.md"),
818
+ ("read", "Read 6 files · 18.4k characters"),
819
+ ("propose", "Model produced a proposal envelope")]:
820
+ time.sleep(0.7)
821
+ self._note(emit, line, note)
822
+ # The step sentences arrive in the thread as the task works, and the same line is
823
+ # what the strip shows — that pairing is the feature being reviewed here. Each row
824
+ # carries a step handle so its chevron can be opened mid-run, not only after it.
825
+ action = "read_file" if line == "read" else "list_files"
826
+ fields = ({"path": "src/main/java/com/demo/users/UserRepository.java",
827
+ "digest": "8f2a5c31"} if action == "read_file" else {"count": 42})
828
+ text = labels.step_line(False, action, **fields)
829
+ self.status_line = text
830
+ self.messages.append({"role": "tool", "author": "Steps", "text": text,
831
+ "step": {"id": "live-%d" % len(self.messages),
832
+ "action": action, "fields": fields,
833
+ "detail": labels.step_has_detail(action, fields)}})
834
+ emit({"kind": "message", "message": self.messages[-1]})
835
+ emit({"kind": "status", "text": text})
836
+ self.pending = "proposing changes…"
837
+ emit({"kind": "status", "text": "Connecting to the model and preparing changes…"})
838
+ # The envelope arriving in pieces, into Activity: what a reader watches during a long turn
839
+ # on a CPU model, and the same sink the real window streams a proposal turn to.
840
+ for fragment in ('{"action": "propose", "summary": "Guard the duplicate email",',
841
+ '"changes": [{"path": "UserService.java"},',
842
+ '{"path": "RegisterController.java"}]}'):
843
+ time.sleep(0.25)
844
+ emit({"kind": "log_chunk", "ts": _clock(), "text": fragment})
845
+ time.sleep(0.4)
846
+
847
+ self.pending = None
848
+ self.busy = self.cancellable = False
849
+ self.state = "WAITING_APPROVAL"
850
+ reply = {"role": "assistant", "author": "AI Code Engineer", "time": _clock(),
851
+ "text": "Here is the smallest change that satisfies step 2. Three files, and the "
852
+ "existing `create()` path is untouched.\n\n"
853
+ "- A typed `DuplicateEmailException` keeps the controller free of string matching.\n"
854
+ "- The repository gains `existsByEmail`, so the guard is one query."}
855
+ self.messages.append(reply)
856
+ emit({"kind": "message", "message": reply})
857
+ emit({"kind": "toast", "text": "Proposal ready — review the changes, then apply them if you want."})
858
+ emit({"kind": "state", "data": self.snapshot()})
859
+
860
+ threading.Thread(target=work, name="ui-fake-job", daemon=True).start()
861
+
862
+ def _refuse(self, text: str) -> None:
863
+ """A gate the preview cannot show is a gate nobody reviewed — so refusals land on screen."""
864
+ self.status_line = text
865
+ self._emit({"kind": "toast", "text": text})
866
+
867
+ def _analyse(self, emit) -> None:
868
+ """Read-only's scripted answer: the folder was read, the refusal is its own row, and no
869
+ proposal state is entered. Everything the mode promises has to be visible here."""
870
+ self._note(emit, "read", "Read 6 files · 18.4k characters")
871
+ self.messages.append({"role": "tool", "author": "Tool", "time": _clock(),
872
+ "text": intent.no_proposal()})
873
+ emit({"kind": "message", "message": self.messages[-1]})
874
+ reply = {"role": "assistant", "author": "AI Code Engineer", "time": _clock(),
875
+ "text": "The duplicate-email guard lives in `UserService.create()`, and it compares "
876
+ "strings in two places that disagree about the field name.\n"
877
+ "That is the whole of what I found; the fix would touch `UserService.java` "
878
+ "and its test."}
879
+ # Streamed the way the real window streams: whole lines, one at a time, into the bubble the
880
+ # client draws from whatever has arrived. This is the only place an answer landing piece by
881
+ # piece can be reviewed at the pace a person reads at. No `busy` claim — Read-only starts no
882
+ # job, and the preview has always had to show that.
883
+ def work():
884
+ for line in reply["text"].split("\n"):
885
+ time.sleep(0.35)
886
+ emit({"kind": "token", "ts": _clock(), "text": line})
887
+ self.messages.append(reply)
888
+ emit({"kind": "message", "message": reply})
889
+ emit({"kind": "state", "data": self.snapshot()})
890
+
891
+ threading.Thread(target=work, name="ui-fake-answer", daemon=True).start()
892
+
893
+ def _confirm_then(self, type: str, emit) -> None:
894
+ answer = self._ask("confirm", {"title": "Apply changes",
895
+ "message": "Write 3 file(s) to D:\\AI\\AI-Agent\\examples\\demo2?\n"
896
+ "You can roll back afterwards as long as the files are not edited later.",
897
+ "warning": "Removes most of an existing file:\n• UserService.java keeps 12 of 26 lines",
898
+ "confirm": "Apply"}, emit)
899
+ if answer.get("ok"):
900
+ self.state = "APPLIED_UNVERIFIED"
901
+ self._note(emit, "apply", "Applied 3 file(s).")
902
+ emit({"kind": "toast", "text": "Changes applied. You can run the project command now."})
903
+
904
+ def target_row(self, path: str) -> dict:
905
+ return next((row for row in self.targets if row["path"] == path), self.targets[0])
906
+
907
+ def target_label(self) -> str:
908
+ return self.target_row(self.target)["label"]
909
+
910
+ def target_recipes(self) -> list:
911
+ return self.target_row(self.target)["recipes"]
912
+
913
+ def set_target(self, label: str) -> None:
914
+ """The same rule the real window follows: pick by label, and the commands change with it."""
915
+ chosen = next((row for row in self.targets if row["label"] == label), None)
916
+ if chosen and chosen["path"] != self.target:
917
+ self.target = chosen["path"]
918
+ if self.recipe not in chosen["recipes"]:
919
+ self.recipe = chosen["recipes"][0]
920
+
921
+ def _run(self, fix, emit) -> None:
922
+ if self.reading_only:
923
+ # The one action in this mode that runs anything, so it asks per command. The refusal has
924
+ # to be reviewable here too: a preview that only shows the yes path hides the whole rule.
925
+ answer = self._ask("confirm", {"title": "Run this command?",
926
+ "message": intent.run_ask("mvn -B test", project="demo2"),
927
+ "confirm": "Run it"}, emit)
928
+ if not answer.get("ok"):
929
+ return self._refuse(intent.run_declined())
930
+ # A round's output is a proposal, and this mode builds none: the scripted chip never moves.
931
+ fix = False
932
+ self.busy = True
933
+ # The scripted command always passes, so a loop has to be driven by hand: "Run & fix" spends
934
+ # the budget the same way the real window does — reset, then one round per further run — so
935
+ # the chip can be reviewed at any value the real one can reach.
936
+ if fix:
937
+ self.fix_round, self.auto_fix = 0, True
938
+ elif self.auto_fix:
939
+ self.fix_round = min(self.fix_round + 1, repair.MAX_FIX_ROUNDS)
940
+ emit({"kind": "busy", "value": True, "cancellable": False})
941
+
942
+ def work():
943
+ emit({"kind": "status", "text": "Running Maven test in demo2…"})
944
+ time.sleep(1.1)
945
+ self._note(emit, "run", "mvn -B test → exit 0 · 41.2s · 14 tests, 0 failures")
946
+ self.runs += 1
947
+ self.state = "CHECKS_PASSED"
948
+ if self.step == 2:
949
+ self.step = 3
950
+ emit({"kind": "toast", "text": "Plan step 2/5 verified. Starting step 3: Login and issue a JWT"})
951
+ self.busy = False
952
+ emit({"kind": "busy", "value": False, "cancellable": False})
953
+ emit({"kind": "state", "data": self.snapshot()})
954
+
955
+ threading.Thread(target=work, name="ui-fake-run", daemon=True).start()
956
+
957
+ def _note(self, emit, kind: str, text: str) -> None:
958
+ row = {"ts": _clock(), "kind": kind, "text": text}
959
+ self.log.append(row)
960
+ emit({"kind": "log", **row})
961
+
962
+ def open_graph(self) -> dict:
963
+ """The preview's copy of `controller.open_graph` — same shape, scripted contents.
964
+
965
+ The caption is written by `labels` here exactly as the real window writes it, so the sentence
966
+ under a reviewed graph is the sentence a real project gets.
967
+ """
968
+ return {"nodes": list(GRAPH_NODES), "edges": list(GRAPH_EDGES),
969
+ "columns": GRAPH_COLUMNS, "cyclic": True, "hidden": GRAPH_HIDDEN,
970
+ "caption": labels.graph_caption(
971
+ False, nodes=len(GRAPH_NODES), edges=len(GRAPH_EDGES), cyclic=True,
972
+ hidden=GRAPH_HIDDEN)}
973
+
974
+ def open_step(self, step_id: str) -> None:
975
+ """The preview's copy of `controller.open_step` — same block shape, scripted contents.
976
+
977
+ Every action the real window can open gets a block here, because this is the window the design
978
+ is reviewed in: a row that cannot be opened in the preview is a row nobody has ever seen open.
979
+ """
980
+ self.step_detail = None
981
+ if not step_id:
982
+ # The client's close is a fetch for the empty id, and the real window answers it by showing
983
+ # nothing. Answering it with "this step is not in the task" would review a different window.
984
+ return
985
+ row = next(((m.get("step") or {}) for m in self.messages
986
+ if (m.get("step") or {}).get("id") == step_id), None)
987
+ if not row:
988
+ # The same shape the real window answers with when the page is behind the task.
989
+ self.step_detail = {"id": step_id, "sections": [], "files": [],
990
+ "note": labels.step_missing_line(arabic=False)}
991
+ return
992
+ action = str(row.get("action", ""))
993
+ fields = row.get("fields") or {}
994
+ block = {"id": step_id, "sections": [], "files": [], "note": ""}
995
+ if action == "executed":
996
+ block["sections"] = [
997
+ [labels.detail_section(False, "command"), [str(fields.get("command", ""))]],
998
+ [labels.detail_section(False, "result"), ["failed · exit 1 · 41.2s · 14 tests"]],
999
+ [labels.detail_section(False, "problems"),
1000
+ ["[ERROR] Tests run: 14, Failures: 2, Errors: 0",
1001
+ "[ERROR] UserRegistrationTest.rejectsDuplicateEmail:48 expected 409 but was 400"]],
1002
+ [labels.detail_section(False, "output"),
1003
+ ["[INFO] BUILD FAILURE", "[INFO] Total time: 41.182 s",
1004
+ "[INFO] Finished at: 2026-09-28T14:06:31+03:00"]]]
1005
+ elif action in {"propose", "applied"}:
1006
+ block["files"] = [str(name) for name in fields.get("names") or []]
1007
+ elif action == "search_code":
1008
+ block["sections"] = [[labels.detail_section(False, "search"),
1009
+ [str(fields.get("query", "")),
1010
+ "%d match(es)" % int(fields.get("count") or 0)]]]
1011
+ elif action == "list_files":
1012
+ block["sections"] = [[labels.detail_section(False, "files"),
1013
+ ["%d file(s) in the folder" % int(fields.get("count") or 0)]]]
1014
+ elif action == "model_reasoning":
1015
+ block["sections"] = [[labels.detail_section(False, "reasoning"),
1016
+ str(fields.get("detail") or "").splitlines()]]
1017
+ elif action == "read_file":
1018
+ block["sections"] = [[labels.detail_section(False, "read"),
1019
+ [str(fields.get("path", "")), "sha256 8f2a5c31"]]]
1020
+ self.step_detail = block
1021
+
1022
+ def _ask(self, kind: str, payload: dict, emit) -> dict:
1023
+ """Push a modal request to the browser and block this thread until it answers."""
1024
+ request_id = secrets.token_hex(8)
1025
+ waiter = threading.Event()
1026
+ self._replies[request_id] = waiter
1027
+ emit({"kind": kind, "id": request_id, **payload})
1028
+ if kind == "folder":
1029
+ return {}
1030
+ waiter.wait(timeout=1800)
1031
+ return self._answers.pop(request_id, {}) or {}
1032
+
1033
+
1034
+ """Model catalogs the preview lists, shaped like the real ones (``id``/``name``/``description``),
1035
+ so the filter the user types into has something to match against."""
1036
+ OLLAMA_MODELS = [
1037
+ {"id": "qwen2.5-coder:1.5b", "name": "qwen2.5-coder:1.5b",
1038
+ "description": "Runs locally on your device. | 0.99 GB"},
1039
+ {"id": "qwen2.5-coder:3b", "name": "qwen2.5-coder:3b",
1040
+ "description": "Runs locally on your device. | 1.99 GB"},
1041
+ {"id": "qwen3:4b", "name": "qwen3:4b",
1042
+ "description": "Runs locally on your device. | 2.50 GB"},
1043
+ {"id": "codegemma:2b", "name": "codegemma:2b",
1044
+ "description": "Runs locally on your device. | 1.61 GB"},
1045
+ {"id": "codellama:13b", "name": "codellama:13b",
1046
+ "description": "Runs locally on your device. | 7.37 GB"},
1047
+ {"id": "granite-code:3b", "name": "granite-code:3b",
1048
+ "description": "Runs locally on your device. | 1.96 GB"},
1049
+ {"id": "llama3.1:70b-cloud", "name": "llama3.1:70b-cloud",
1050
+ "description": "Ollama cloud model — internet and Ollama account access required. | 43.0 GB"},
1051
+ ]
1052
+ FREE_MODELS = [
1053
+ {"id": "deepseek/deepseek-coder:free", "name": "DeepSeek Coder",
1054
+ "description": "16k context · free variant"},
1055
+ {"id": "qwen/qwen-2.5-coder-32b-instruct:free", "name": "Qwen2.5 Coder 32B",
1056
+ "description": "128k context · free variant"},
1057
+ {"id": "meta-llama/llama-3.3-70b-instruct:free", "name": "Llama 3.3 70B",
1058
+ "description": "131k context · free variant"},
1059
+ ]
1060
+ PAID_MODELS = [
1061
+ {"id": "anthropic/claude-3.5-sonnet", "name": "Claude 3.5 Sonnet",
1062
+ "description": "200k context · $3/M input"},
1063
+ {"id": "openai/gpt-4o", "name": "GPT-4o", "description": "128k context · $2.50/M input"},
1064
+ ]
1065
+ # What an OpenAI-shaped server lists: identifiers and nothing else, which is exactly what a real
1066
+ # `/models` response carries. Pricing is not in that payload, so the preview does not invent any.
1067
+ SERVED_MODELS = [
1068
+ {"id": "qwen2.5-coder-3b-instruct", "name": "qwen2.5-coder-3b-instruct",
1069
+ "description": "Listed by this provider's /models endpoint. Pricing is not reported there — "
1070
+ "check the service."},
1071
+ {"id": "deepseek-coder", "name": "deepseek-coder",
1072
+ "description": "Listed by this provider's /models endpoint. Pricing is not reported there — "
1073
+ "check the service."},
1074
+ ]
1075
+ MODELS = OLLAMA_MODELS
1076
+
1077
+ """Gestures this window answers with a sentence rather than a state change.
1078
+
1079
+ They are the folder and plan pickers, the sidebar's navigation and the sample — each interesting
1080
+ part is a dialog or a disk read the scripted thread has no equivalent of. A silent no-op is the
1081
+ failure this list exists to prevent: something that looked broken in review would pass review.
1082
+ """
1083
+ PREVIEW_ONLY = {
1084
+ "new_project": "Preview only: the real window creates a new empty folder and grants it.",
1085
+ "pick_plan": "Preview only: the real window browses that project for a plan file.",
1086
+ "clear_plan": "Preview only: this scripted thread keeps its attached plan.",
1087
+ "new_chat": "Preview only: a new chat starts with an empty thread.",
1088
+ "new_chat_in": "Preview only: the real window opens a fresh chat on that project.",
1089
+ "bind_chat": "Preview only: moving a chat onto a project is a sidebar drag.",
1090
+ "open": "Preview only: opening a saved task reads its session from disk.",
1091
+ "reveal": "Preview only: this opens Explorer on the granted folder.",
1092
+ "example": "Preview only: the sample fills the composer with a real task.",
1093
+ "set_icon": "Preview only: the icon is saved with that project's preferences.",
1094
+ }
1095
+
1096
+ """One drawer payload per scripted project. Every key the real controller sends is here,
1097
+ because a missing field in the preview renders as a blank rather than an error."""
1098
+ PROJECT_INFO = {
1099
+ "demo2": {
1100
+ "key": "demo2", "name": "demo2", "path": "D:\\AI\\AI-Agent\\examples\\demo2",
1101
+ "exists": True, "icon": "☕", "notes": "Spring Boot 3, Java 17. Tests are JUnit 5 and run "
1102
+ "through Maven; the surefire reports are the proof a run passed.",
1103
+ "notes_limit": 4000, "notes_file": "demo2-8f2a.md",
1104
+ "notes_dir": "D:\\AI\\AI-Agent\\.agent-memory",
1105
+ "context": {"system": 2180, "context": 6120, "turns": 4890, "kept": 3, "used": 13190,
1106
+ "budget": 24000, "remaining": 10810, "est_tokens": 3298, "map": 6120,
1107
+ "notes": 128, "files": 41, "bound": True},
1108
+ "toolchain": {
1109
+ "detected": [{"name": "maven-test", "label": "Maven test", "command": "mvn -B test"},
1110
+ {"name": "maven-compile", "label": "Maven compile",
1111
+ "command": "mvn -B -DskipTests compile"}],
1112
+ "selected": "maven-test", "timeout": 1500, "proof": "surefire XML",
1113
+ "request_timeout": 300},
1114
+ },
1115
+ "demo_repo": {
1116
+ "key": "demo_repo", "name": "demo_repo", "path": "D:\\AI\\AI-Agent\\examples\\demo_repo",
1117
+ "exists": True, "icon": "🐍", "notes": "",
1118
+ "notes_limit": 4000, "notes_file": "demo_repo-1c4d.md",
1119
+ "notes_dir": "D:\\AI\\AI-Agent\\.agent-memory",
1120
+ "context": {"system": 2180, "context": 1740, "turns": 0, "kept": 0, "used": 3920,
1121
+ "budget": 24000, "remaining": 20080, "est_tokens": 980, "map": 1740,
1122
+ "notes": 0, "files": 12, "bound": False},
1123
+ "toolchain": {
1124
+ "detected": [{"name": "python-unittest", "label": "Python unittest",
1125
+ "command": "python -m unittest discover -s tests -v"}],
1126
+ "selected": "python-unittest", "timeout": 600,
1127
+ "proof": "the command's own summary line", "request_timeout": 300},
1128
+ },
1129
+ }
1130
+
1131
+
1132
+ def _visible_dir(path: Path) -> bool:
1133
+ if not path.is_dir():
1134
+ return False
1135
+ name = path.name
1136
+ return not (name.startswith(".") or name.casefold() in
1137
+ {"node_modules", "__pycache__", "venv", ".venv", "library", "windows", "program files"})
1138
+
1139
+
1140
+ def _clock() -> str:
1141
+ return datetime.now().strftime("%H:%M")