locdex 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
locdex/agent.py ADDED
@@ -0,0 +1,955 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ import shlex
6
+ from dataclasses import dataclass
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ from .agent_tools import TOOL_DESCRIPTIONS, ToolError, execute_tool
11
+ from .config import LocalModelConfig, load_local_model_config
12
+ from .local_runtime import get_runtime
13
+ from .planner import record_usage
14
+
15
+ MUTATING_GIT_TOOLS = {"git_add", "git_commit", "git_pull", "git_push"}
16
+ WORKSPACE_MUTATING_TOOLS = {"write_file", "replace_in_file", "delete_path"}
17
+ SUBSTANTIVE_INSPECTION_TOOLS = {"read_file", "search_code"}
18
+
19
+ _CHANGE_KEYWORDS = (
20
+ "fix",
21
+ "correct",
22
+ "change",
23
+ "edit",
24
+ "modify",
25
+ "update",
26
+ "implement",
27
+ "add",
28
+ "remove",
29
+ "delete",
30
+ "create",
31
+ "write",
32
+ "refactor",
33
+ "rename",
34
+ )
35
+
36
+ _VALIDATION_KEYWORDS = (
37
+ "test",
38
+ "tests",
39
+ "pytest",
40
+ "unittest",
41
+ "check",
42
+ "checks",
43
+ "lint",
44
+ "verify",
45
+ "validation",
46
+ )
47
+
48
+ _TEST_COMMAND_MARKERS = {
49
+ "pytest",
50
+ "unittest",
51
+ "test",
52
+ "tests",
53
+ "jest",
54
+ "vitest",
55
+ "mocha",
56
+ "cargo",
57
+ "go",
58
+ }
59
+
60
+ # For very small repositories, deterministic context gathering is cheaper and
61
+ # more reliable than asking a tiny local model to discover the project itself.
62
+ SMALL_REPO_MAX_FILES = 8
63
+ SMALL_REPO_MAX_TOTAL_BYTES = 64 * 1024
64
+
65
+ _DIRECT_EDIT_PATTERN = re.compile(
66
+ r"(?:^|[,.;!?]\s+|\band\s+)"
67
+ r"(?:(?:now|just|please)\b[\s,]*)*"
68
+ r"(?:(?:can|could|would|will)\s+you\s+|"
69
+ r"i\s+(?:need|want)\s+you\s+to\s+)?"
70
+ r"(?:(?:now|just|please)\b[\s,]*)*"
71
+ r"(?:fix|correct|change|edit|modify|update|implement|add|remove|delete|"
72
+ r"create|write|refactor|rename)\b",
73
+ re.IGNORECASE,
74
+ )
75
+ _DELETE_PATH_PATTERN = re.compile(
76
+ r"\b(?:delete|remove)\s+(?:the\s+)?"
77
+ r"(?:(?:file|folder|directory|path)\b|[\w./\\-]+\.[A-Za-z0-9]{1,12}\b)",
78
+ re.IGNORECASE,
79
+ )
80
+
81
+
82
+ @dataclass(frozen=True)
83
+ class TaskPermission:
84
+ can_edit: bool
85
+ can_delete_path: bool
86
+
87
+ @property
88
+ def label(self) -> str:
89
+ if self.can_edit:
90
+ return "edit"
91
+ return "read-only"
92
+
93
+
94
+ def _git_tool_allowed(task: str, tool_name: str) -> bool:
95
+ text = task.lower()
96
+ if "ship it" in text or text.strip() == "ship":
97
+ return tool_name in {"git_add", "git_commit", "git_push"}
98
+ keywords = {
99
+ "git_add": ("git add", "stage ", "stage these", "stage the", "add these changes", "add changes"),
100
+ "git_commit": ("commit",),
101
+ "git_pull": ("git pull", "pull latest", "pull from", "sync from remote", "sync with remote"),
102
+ "git_push": ("git push", "push", "publish branch"),
103
+ }
104
+ return any(phrase in text for phrase in keywords.get(tool_name, ()))
105
+
106
+
107
+ def _task_requires_workspace_change(task: str) -> bool:
108
+ return _DIRECT_EDIT_PATTERN.search(task.strip()) is not None
109
+
110
+
111
+ def _task_authorizes_delete_path(task: str) -> bool:
112
+ return _DELETE_PATH_PATTERN.search(task) is not None
113
+
114
+
115
+ def _task_permission(task: str) -> TaskPermission:
116
+ can_edit = _task_requires_workspace_change(task)
117
+ return TaskPermission(
118
+ can_edit=can_edit,
119
+ can_delete_path=can_edit and _task_authorizes_delete_path(task),
120
+ )
121
+
122
+
123
+ def _print_permission(permission: TaskPermission) -> None:
124
+ if permission.can_edit:
125
+ print("[Permission] Task authorizes scoped workspace edits.")
126
+ else:
127
+ print("[Permission] Read-only task: workspace mutations are blocked.")
128
+
129
+
130
+ _MUTATION_CLAIM_PATTERN = re.compile(
131
+ r"\b(?:changed|fixed|updated|edited|modified|created|wrote|written|"
132
+ r"removed|deleted|renamed|refactored|implemented|added)\b",
133
+ re.IGNORECASE,
134
+ )
135
+
136
+ _NO_MUTATION_PHRASES = (
137
+ "no changes were made",
138
+ "no change was made",
139
+ "nothing was changed",
140
+ "did not change",
141
+ "didn't change",
142
+ "has not been changed",
143
+ "have not been changed",
144
+ "without changing",
145
+ "left unchanged",
146
+ )
147
+
148
+
149
+ def _summary_claims_workspace_change(summary: str) -> bool:
150
+ lowered = summary.lower()
151
+ if any(phrase in lowered for phrase in _NO_MUTATION_PHRASES):
152
+ return False
153
+ return _MUTATION_CLAIM_PATTERN.search(summary) is not None
154
+
155
+
156
+ def _task_requests_validation(task: str) -> bool:
157
+ text = task.lower()
158
+ return any(keyword in text for keyword in _VALIDATION_KEYWORDS)
159
+
160
+
161
+ def _mutation_result_changed(result: Any) -> bool:
162
+ if not isinstance(result, dict) or result.get("ok") is not True:
163
+ return False
164
+ # Real Locdex mutation tools now report `changed`. Keep compatibility with
165
+ # older/fake test tools that only report ok=True.
166
+ if "changed" in result:
167
+ return result.get("changed") is True
168
+ return True
169
+
170
+
171
+ def _successful_workspace_mutation(tool_calls: list[dict[str, Any]]) -> bool:
172
+ for call in tool_calls:
173
+ if call.get("tool") not in WORKSPACE_MUTATING_TOOLS:
174
+ continue
175
+ if _mutation_result_changed(call.get("result")):
176
+ return True
177
+ return False
178
+
179
+
180
+ def _has_substantive_inspection(tool_calls: list[dict[str, Any]]) -> bool:
181
+ for call in tool_calls:
182
+ if call.get("tool") not in SUBSTANTIVE_INSPECTION_TOOLS:
183
+ continue
184
+ result = call.get("result")
185
+ if isinstance(result, dict) and "error" not in result:
186
+ return True
187
+ return False
188
+
189
+
190
+ def _looks_like_validation_call(call: dict[str, Any]) -> bool:
191
+ tool = str(call.get("tool", ""))
192
+ if tool == "run_tests":
193
+ return True
194
+ if tool != "run_command":
195
+ return False
196
+
197
+ args = call.get("args")
198
+ if not isinstance(args, dict):
199
+ return False
200
+ argv = args.get("argv")
201
+ if not isinstance(argv, list):
202
+ return False
203
+
204
+ normalized = {str(part).lower() for part in argv}
205
+ return bool(normalized & _TEST_COMMAND_MARKERS)
206
+
207
+
208
+ def _validation_state(tool_calls: list[dict[str, Any]]) -> tuple[bool, bool, bool]:
209
+ """Return (attempted, passed, unavailable)."""
210
+ attempted = False
211
+ passed = False
212
+ unavailable = False
213
+
214
+ for call in tool_calls:
215
+ if not _looks_like_validation_call(call):
216
+ continue
217
+
218
+ attempted = True
219
+ result = call.get("result")
220
+ if not isinstance(result, dict):
221
+ continue
222
+
223
+ if result.get("ok") is True:
224
+ passed = True
225
+ continue
226
+
227
+ if result.get("returncode") is None and "No supported test runner" in str(result.get("output", "")):
228
+ unavailable = True
229
+
230
+ return attempted, passed, unavailable
231
+
232
+
233
+ def _last_successful_mutation_index(tool_calls: list[dict[str, Any]]) -> int | None:
234
+ last_index: int | None = None
235
+ for index, call in enumerate(tool_calls):
236
+ if call.get("tool") not in WORKSPACE_MUTATING_TOOLS:
237
+ continue
238
+ if _mutation_result_changed(call.get("result")):
239
+ last_index = index
240
+ return last_index
241
+
242
+
243
+ def _permission_error(permission: TaskPermission, tool_name: str) -> str | None:
244
+ if tool_name in {"write_file", "replace_in_file"} and not permission.can_edit:
245
+ return (
246
+ f"{tool_name} blocked: the current task is read-only. "
247
+ "Ask the user to explicitly request a fix/change/edit before mutating files."
248
+ )
249
+ if tool_name == "delete_path" and not permission.can_delete_path:
250
+ return (
251
+ "delete_path blocked: deleting a file/folder requires an explicit deletion "
252
+ "instruction in the current user request."
253
+ )
254
+ return None
255
+
256
+
257
+ def _edit_tool_response_schema(permission: TaskPermission, task: str) -> dict[str, Any]:
258
+ text = task.lower()
259
+ creating_file = bool(re.search(r"\bcreate\b|\bnew\s+file\b", text))
260
+ # Recovery mode favors the least-destructive tool. Existing-file fixes,
261
+ # changes and refactors should use exact replacement rather than rewriting
262
+ # the whole file. Whole-file write remains available for explicit creation.
263
+ tools = ["write_file"] if creating_file else ["replace_in_file"]
264
+ if permission.can_delete_path:
265
+ tools.append("delete_path")
266
+ return {
267
+ "type": "object",
268
+ "properties": {
269
+ "action": {"type": "string", "enum": ["tool"]},
270
+ "tool": {"type": "string", "enum": tools},
271
+ "args": {"type": "object"},
272
+ "summary": {"type": "string"},
273
+ "confidence": {"type": "number", "minimum": 0, "maximum": 1},
274
+ },
275
+ "required": ["action", "tool", "args"],
276
+ }
277
+
278
+
279
+ def _progress_message(tool_name: str, args: dict[str, Any]) -> str:
280
+ if tool_name == "list_files":
281
+ return "[Agent] Inspecting workspace files..."
282
+ if tool_name == "read_file":
283
+ return f"[Agent] Reading {args.get('path', 'file')}..."
284
+ if tool_name == "search_code":
285
+ query = str(args.get("query", ""))
286
+ return f"[Agent] Searching code for {query!r}..."
287
+ if tool_name == "write_file":
288
+ return f"[Agent] Writing {args.get('path', 'file')}..."
289
+ if tool_name == "replace_in_file":
290
+ return f"[Agent] Editing {args.get('path', 'file')}..."
291
+ if tool_name == "delete_path":
292
+ return f"[Agent] Deleting {args.get('path', 'path')}..."
293
+ if tool_name == "run_tests":
294
+ return "[Agent] Running tests..."
295
+ if tool_name == "run_command":
296
+ argv = args.get("argv")
297
+ if isinstance(argv, list):
298
+ command = shlex.join(str(part) for part in argv)
299
+ return f"[Agent] Running: {command}"
300
+ return "[Agent] Running command..."
301
+ if tool_name == "git_status":
302
+ return "[Agent] Checking Git status..."
303
+ if tool_name == "git_diff":
304
+ return "[Agent] Inspecting Git diff..."
305
+ if tool_name == "git_add":
306
+ return "[Agent] Staging requested changes..."
307
+ if tool_name == "git_commit":
308
+ return "[Agent] Creating requested commit..."
309
+ if tool_name == "git_pull":
310
+ return "[Agent] Pulling requested changes..."
311
+ if tool_name == "git_push":
312
+ return "[Agent] Pushing requested commits..."
313
+ return f"[Agent] Running tool: {tool_name}..."
314
+
315
+
316
+ def _print_tool_result(tool_name: str, result: dict[str, Any]) -> None:
317
+ if "error" in result:
318
+ print(f"[Agent] ✗ {tool_name} failed: {result['error']}")
319
+ return
320
+
321
+ if tool_name in {"write_file", "replace_in_file", "delete_path"} and result.get("ok") is True:
322
+ path = result.get("path")
323
+ if result.get("changed") is False:
324
+ print(f"[Agent] ✗ No workspace change occurred in {path or 'target'}.")
325
+ else:
326
+ print(f"[Agent] ✓ Updated {path or 'workspace'}.")
327
+ return
328
+
329
+ if tool_name in {"run_tests", "run_command"}:
330
+ if result.get("ok") is True:
331
+ print("[Agent] ✓ Command passed.")
332
+ else:
333
+ code = result.get("returncode")
334
+ suffix = f" (exit {code})" if code is not None else ""
335
+ print(f"[Agent] ✗ Command failed{suffix}; result added to context.")
336
+ return
337
+
338
+ if result.get("ok") is True:
339
+ print(f"[Agent] ✓ {tool_name} completed.")
340
+
341
+
342
+ def _run_tool(
343
+ repo_path: str,
344
+ task: str,
345
+ permission: TaskPermission,
346
+ tool_name: str,
347
+ tool_args: dict[str, Any],
348
+ ) -> dict[str, Any]:
349
+ print(_progress_message(tool_name, tool_args))
350
+ try:
351
+ permission_error = _permission_error(permission, tool_name)
352
+ if permission_error is not None:
353
+ raise ToolError(permission_error)
354
+ if tool_name in MUTATING_GIT_TOOLS and not _git_tool_allowed(task, tool_name):
355
+ raise ToolError(
356
+ f"{tool_name} requires an explicit Git instruction from the user in the current request."
357
+ )
358
+ result = execute_tool(repo_path, tool_name, tool_args)
359
+ except ToolError as exc:
360
+ result = {"error": str(exc)}
361
+ except Exception as exc: # noqa: BLE001
362
+ result = {"error": f"Tool execution failed: {exc}"}
363
+
364
+ _print_tool_result(tool_name, result)
365
+ return result
366
+
367
+
368
+ def _append_tool_result(
369
+ messages: list[dict[str, str]],
370
+ tool_name: str,
371
+ result: dict[str, Any],
372
+ ) -> None:
373
+ messages.append(
374
+ {
375
+ "role": "user",
376
+ "content": f"TOOL RESULT for {tool_name}:\n{json.dumps(result, ensure_ascii=False)[:24000]}",
377
+ }
378
+ )
379
+
380
+
381
+ def _append_mutation_readback(
382
+ repo_path: str,
383
+ messages: list[dict[str, str]],
384
+ tool_name: str,
385
+ result: dict[str, Any],
386
+ ) -> None:
387
+ if tool_name not in {"write_file", "replace_in_file"} or not _mutation_result_changed(result):
388
+ return
389
+ path = result.get("path")
390
+ if not isinstance(path, str) or not path:
391
+ return
392
+
393
+ print(f"[Agent] Verifying written content in {path}...")
394
+ try:
395
+ readback = execute_tool(
396
+ repo_path,
397
+ "read_file",
398
+ {"path": path, "start_line": 1, "end_line": 400},
399
+ )
400
+ except Exception as exc: # noqa: BLE001
401
+ readback = {"error": f"Post-write readback failed: {exc}"}
402
+
403
+ messages.append(
404
+ {
405
+ "role": "user",
406
+ "content": (
407
+ f"POST-MUTATION READBACK for {path}:\n"
408
+ f"{json.dumps(readback, ensure_ascii=False)[:24000]}"
409
+ ),
410
+ }
411
+ )
412
+
413
+
414
+ def _append_failed_edit_readback(
415
+ repo_path: str,
416
+ messages: list[dict[str, str]],
417
+ tool_name: str,
418
+ tool_args: dict[str, Any],
419
+ ) -> None:
420
+ if tool_name not in {"write_file", "replace_in_file"}:
421
+ return
422
+ path = tool_args.get("path")
423
+ if not isinstance(path, str) or not path:
424
+ return
425
+
426
+ print(f"[Agent] Refreshing {path} after failed edit attempt...")
427
+ try:
428
+ readback = execute_tool(
429
+ repo_path,
430
+ "read_file",
431
+ {"path": path, "start_line": 1, "end_line": 400},
432
+ )
433
+ except Exception as exc: # noqa: BLE001
434
+ readback = {"error": f"Recovery readback failed: {exc}"}
435
+
436
+ messages.append(
437
+ {
438
+ "role": "user",
439
+ "content": (
440
+ f"CURRENT FILE STATE after failed edit attempt for {path}:\n"
441
+ f"{json.dumps(readback, ensure_ascii=False)[:24000]}"
442
+ ),
443
+ }
444
+ )
445
+
446
+
447
+ def _small_repo_paths(repo_path: str, files: list[str]) -> list[str]:
448
+ if not files or len(files) > SMALL_REPO_MAX_FILES:
449
+ return []
450
+
451
+ root = Path(repo_path).resolve()
452
+ total = 0
453
+ accepted: list[str] = []
454
+
455
+ for relative in files:
456
+ candidate = (root / relative).resolve()
457
+ try:
458
+ candidate.relative_to(root)
459
+ except ValueError:
460
+ return []
461
+ if not candidate.is_file():
462
+ continue
463
+ try:
464
+ size = candidate.stat().st_size
465
+ except OSError:
466
+ return []
467
+ total += size
468
+ if total > SMALL_REPO_MAX_TOTAL_BYTES:
469
+ return []
470
+ accepted.append(relative)
471
+
472
+ return accepted
473
+
474
+
475
+ def _bootstrap_context(
476
+ task: str,
477
+ repo_path: str,
478
+ permission: TaskPermission,
479
+ messages: list[dict[str, str]],
480
+ bootstrap_calls: list[dict[str, Any]],
481
+ requests_validation: bool,
482
+ ) -> bool:
483
+ """Gather deterministic evidence before spending model generations.
484
+
485
+ Returns True when the whole small repository was pre-read.
486
+ """
487
+ print("[Agent] Preparing workspace evidence...")
488
+
489
+ list_args: dict[str, Any] = {"path": ".", "limit": 200}
490
+ list_result = _run_tool(repo_path, task, permission, "list_files", list_args)
491
+ bootstrap_calls.append({"tool": "list_files", "args": list_args, "result": list_result})
492
+ _append_tool_result(messages, "list_files", list_result)
493
+
494
+ files = list_result.get("files") if isinstance(list_result, dict) else None
495
+ small_paths = _small_repo_paths(repo_path, files if isinstance(files, list) else [])
496
+
497
+ if small_paths:
498
+ print(f"[Agent] Small workspace detected ({len(small_paths)} files); pre-reading project...")
499
+ for path in small_paths:
500
+ read_args: dict[str, Any] = {"path": path, "start_line": 1, "end_line": 400}
501
+ read_result = _run_tool(repo_path, task, permission, "read_file", read_args)
502
+ bootstrap_calls.append({"tool": "read_file", "args": read_args, "result": read_result})
503
+ _append_tool_result(messages, "read_file", read_result)
504
+
505
+ if requests_validation:
506
+ test_args: dict[str, Any] = {}
507
+ test_result = _run_tool(repo_path, task, permission, "run_tests", test_args)
508
+ bootstrap_calls.append({"tool": "run_tests", "args": test_args, "result": test_result})
509
+ _append_tool_result(messages, "run_tests", test_result)
510
+
511
+ return bool(small_paths)
512
+
513
+
514
+ def _append_premature_final_feedback(
515
+ messages: list[dict[str, str]],
516
+ decision: dict[str, Any],
517
+ reason: str,
518
+ ) -> None:
519
+ print(f"[Agent] Completion deferred: {reason}")
520
+ messages.append({"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)})
521
+ messages.append(
522
+ {
523
+ "role": "user",
524
+ "content": (
525
+ "You attempted to finish, but Locdex cannot mark this task complete yet. "
526
+ f"{reason} Continue using the available tools. Do not claim work was performed "
527
+ "unless the corresponding tool result proves it."
528
+ ),
529
+ }
530
+ )
531
+
532
+
533
+ def _append_escalation_feedback(
534
+ messages: list[dict[str, str]],
535
+ decision: dict[str, Any],
536
+ small_repo: bool,
537
+ ) -> None:
538
+ if small_repo:
539
+ reason = (
540
+ "This is a small workspace whose files and test evidence are already in context. "
541
+ "Attempt the minimal concrete fix locally before escalating."
542
+ )
543
+ else:
544
+ reason = (
545
+ "Do not escalate yet. Read or search relevant source code first, then attempt the "
546
+ "smallest plausible local fix."
547
+ )
548
+
549
+ print(f"[Agent] Escalation deferred: {reason}")
550
+ messages.append({"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)})
551
+ messages.append({"role": "user", "content": reason})
552
+
553
+
554
+ AGENT_RESPONSE_SCHEMA: dict[str, Any] = {
555
+ "type": "object",
556
+ "properties": {
557
+ "action": {"type": "string", "enum": ["tool", "final", "escalate"]},
558
+ "tool": {"type": "string"},
559
+ "args": {"type": "object"},
560
+ "summary": {"type": "string"},
561
+ "confidence": {"type": "number", "minimum": 0, "maximum": 1},
562
+ "reason": {"type": "string"},
563
+ },
564
+ "required": ["action"],
565
+ }
566
+
567
+
568
+ def _system_prompt(extra_context: str, model_key: str) -> str:
569
+ tool_text = "\n".join(f"- {name}: {desc}" for name, desc in TOOL_DESCRIPTIONS.items())
570
+ smoke_hint = ""
571
+ if model_key == "smoke":
572
+ smoke_hint = (
573
+ "\nSMOKE MODEL MODE: Prefer direct, minimal tool actions over broad planning. "
574
+ "If the workspace is tiny and its files/test output are supplied, solve the concrete "
575
+ "local problem rather than describing the repository as complex.\n"
576
+ )
577
+
578
+ return f"""You are Locdex, an autonomous coding agent working directly in the user's current workspace.
579
+
580
+ You should behave like a modern terminal coding agent: inspect the repository, edit files in place, run commands/tests, inspect diffs, and continue until the user's task is complete. File edits made with tools are REAL and immediate. Do not merely describe code changes when you can make them.
581
+ {smoke_hint}
582
+ Available tools:
583
+ {tool_text}
584
+
585
+ Operating rules:
586
+ 1. Inspect relevant files before editing. Never claim you inspected something you did not read/search.
587
+ 2. For code tasks, use write_file or replace_in_file to make the requested changes directly in the workspace.
588
+ 3. After editing, run the most relevant tests/checks you can reasonably detect. Use git_diff/status to inspect your work when useful.
589
+ 4. Workspace mutation is permission-scoped. write_file/replace_in_file are allowed only when the current user request explicitly asks for a change. delete_path additionally requires an explicit file/folder deletion instruction.
590
+ 5. Do NOT call git_add, git_commit, git_pull, or git_push unless the user's current request explicitly asks for that Git operation.
591
+ 6. Never access paths outside the workspace. Do not try to read .git internals, virtualenvs, caches, credentials, or secrets.
592
+ 7. Prefer argv-style run_command calls. Do not invoke privilege escalation or machine power/admin commands.
593
+ 8. Keep edits scoped to the request. Do not rewrite unrelated files.
594
+ 9. Never escalate before using the available workspace evidence and attempting a concrete local solution.
595
+ 10. If the task truly exceeds the local model after inspection and an attempted solution, return action='escalate' with a concise reason.
596
+ 11. When finished, return action='final' with a concise summary of what you changed, tests/checks run, and any relevant caveat.
597
+ 12. A final answer is accepted only when tool evidence supports it. For edit/fix tasks, a successful workspace mutation must have occurred. If the user explicitly requested tests/checks, Locdex must observe the requested validation before accepting completion.
598
+
599
+ WORKSPACE CONTEXT:
600
+ {extra_context}
601
+ """
602
+
603
+
604
+ def run_agent(
605
+ task: str,
606
+ repo_path: str = ".",
607
+ context: dict | None = None,
608
+ config: LocalModelConfig | None = None,
609
+ ) -> dict:
610
+ config = config or load_local_model_config()
611
+ runtime = get_runtime(config)
612
+ extra_context = (context or {}).get("system_prompt", "")
613
+
614
+ messages: list[dict[str, str]] = [
615
+ {"role": "system", "content": _system_prompt(extra_context, getattr(config, "model_key", ""))},
616
+ {"role": "user", "content": task},
617
+ ]
618
+ tool_calls: list[dict[str, Any]] = []
619
+ bootstrap_calls: list[dict[str, Any]] = []
620
+
621
+ permission = _task_permission(task)
622
+ _print_permission(permission)
623
+ requires_change = permission.can_edit
624
+ requests_validation = _task_requests_validation(task)
625
+ small_repo = _bootstrap_context(
626
+ task,
627
+ repo_path,
628
+ permission,
629
+ messages,
630
+ bootstrap_calls,
631
+ requests_validation,
632
+ )
633
+
634
+ # One early escalation is treated as a weak-model planning failure rather than
635
+ # proof that the task truly requires cloud fallback.
636
+ escalation_deferrals_remaining = 1 if requires_change else 0
637
+ premature_final_count = 0
638
+ force_edit_next = False
639
+ forced_edit_attempts = 0
640
+ unsupported_summary_count = 0
641
+ post_mutation_validation_calls: list[dict[str, Any]] = []
642
+ validation_repair_attempts = 0
643
+ max_validation_repair_attempts = 2
644
+ edit_tool_retry_attempts = 0
645
+ max_edit_tool_retry_attempts = 2
646
+ smoke_mode = getattr(config, "model_key", "") == "smoke"
647
+
648
+ for step in range(1, config.max_agent_steps + 1):
649
+ forced_edit_mode = force_edit_next
650
+ if forced_edit_mode:
651
+ forced_edit_attempts += 1
652
+ print(f"[Agent] Step {step}/{config.max_agent_steps}: protocol recovery requires an edit tool...")
653
+ messages.append(
654
+ {
655
+ "role": "user",
656
+ "content": (
657
+ "PROTOCOL RECOVERY: previous answers claimed completion without making the "
658
+ "authorized workspace change. Your next response MUST use the edit tool "
659
+ "permitted by the provided schema. For an existing file, make the smallest "
660
+ "exact replacement that actually changes the requested text. Do not return "
661
+ "final or escalate in this response."
662
+ ),
663
+ }
664
+ )
665
+ schema = _edit_tool_response_schema(permission, task)
666
+ force_edit_next = False
667
+ else:
668
+ print(f"[Agent] Step {step}/{config.max_agent_steps}: planning next action...")
669
+ schema = AGENT_RESPONSE_SCHEMA
670
+
671
+ decision = runtime.json_completion(messages, schema)
672
+ record_usage("local")
673
+ action = decision.get("action")
674
+
675
+ if forced_edit_mode and action != "tool":
676
+ print("[Agent] Protocol recovery failed: local model still did not request an edit tool.")
677
+ return {
678
+ "status": "incomplete",
679
+ "confidence": 0.0,
680
+ "summary": (
681
+ "Local model could not follow the required edit-tool protocol after repeated "
682
+ "unsupported completion attempts."
683
+ ),
684
+ "steps": step,
685
+ "tool_calls": tool_calls,
686
+ }
687
+
688
+ if action == "final":
689
+ mutated = _successful_workspace_mutation(tool_calls)
690
+
691
+ if requires_change and mutated:
692
+ validation_attempted, validation_passed, validation_unavailable = _validation_state(
693
+ post_mutation_validation_calls
694
+ )
695
+ else:
696
+ validation_attempted, validation_passed, validation_unavailable = _validation_state(
697
+ bootstrap_calls + tool_calls
698
+ )
699
+
700
+ if requires_change and not mutated:
701
+ premature_final_count += 1
702
+ if premature_final_count == 1:
703
+ _append_premature_final_feedback(
704
+ messages,
705
+ decision,
706
+ "the request requires a workspace change, but no successful write/edit/delete tool has run.",
707
+ )
708
+ continue
709
+
710
+ if premature_final_count == 2:
711
+ print(
712
+ "[Agent] Repeated unsupported completion; the next generation will be "
713
+ "constrained to an authorized edit tool."
714
+ )
715
+ messages.append({"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)})
716
+ force_edit_next = True
717
+ continue
718
+
719
+ print("[Agent] Local model repeatedly claimed completion without performing the authorized edit.")
720
+ return {
721
+ "status": "incomplete",
722
+ "confidence": 0.0,
723
+ "summary": (
724
+ "Local model repeatedly claimed completion without performing the "
725
+ "authorized workspace edit."
726
+ ),
727
+ "steps": step,
728
+ "tool_calls": tool_calls,
729
+ }
730
+
731
+ if requests_validation and not validation_attempted:
732
+ _append_premature_final_feedback(
733
+ messages,
734
+ decision,
735
+ "the user explicitly requested tests/checks, but no validation command has been attempted.",
736
+ )
737
+ continue
738
+
739
+ if requests_validation and validation_attempted and not (validation_passed or validation_unavailable):
740
+ if requires_change and validation_repair_attempts < max_validation_repair_attempts:
741
+ validation_repair_attempts += 1
742
+ print(
743
+ "[Agent] Validation failed after the edit; the next generation will be "
744
+ "constrained to a repair edit."
745
+ )
746
+ messages.append(
747
+ {"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)}
748
+ )
749
+ messages.append(
750
+ {
751
+ "role": "user",
752
+ "content": (
753
+ "The latest validation failed after the workspace edit. Use the "
754
+ "failure output already in context and make a concrete repair. "
755
+ "Your next response must use the permitted edit tool."
756
+ ),
757
+ }
758
+ )
759
+ force_edit_next = True
760
+ continue
761
+
762
+ return {
763
+ "status": "incomplete",
764
+ "confidence": 0.0,
765
+ "summary": (
766
+ "Validation still failed after the allowed local repair attempts."
767
+ ),
768
+ "steps": step,
769
+ "tool_calls": tool_calls,
770
+ }
771
+
772
+ summary = str(decision.get("summary", "Task completed."))
773
+ if not mutated and _summary_claims_workspace_change(summary):
774
+ unsupported_summary_count += 1
775
+ if unsupported_summary_count == 1:
776
+ print(
777
+ "[Agent] Final claim rejected: the model described a workspace change "
778
+ "without mutation evidence."
779
+ )
780
+ messages.append(
781
+ {"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)}
782
+ )
783
+ messages.append(
784
+ {
785
+ "role": "user",
786
+ "content": (
787
+ "EVIDENCE CORRECTION: no workspace mutation occurred in this task. "
788
+ "Answer using only the observed file/tool evidence. Do not say that "
789
+ "anything was changed, fixed, updated, created, removed, or edited."
790
+ ),
791
+ }
792
+ )
793
+ continue
794
+
795
+ print("[Agent] Local model repeated an unsupported workspace-change claim.")
796
+ return {
797
+ "status": "incomplete",
798
+ "confidence": 0.0,
799
+ "summary": (
800
+ "Local model repeatedly claimed a workspace change without mutation evidence."
801
+ ),
802
+ "steps": step,
803
+ "tool_calls": tool_calls,
804
+ }
805
+
806
+ print("[Agent] ✓ Task evidence satisfied.")
807
+ return {
808
+ "status": "completed",
809
+ "confidence": float(decision.get("confidence", 0.85)),
810
+ "summary": summary,
811
+ "steps": step,
812
+ "tool_calls": tool_calls,
813
+ }
814
+
815
+ if action == "escalate":
816
+ should_defer = escalation_deferrals_remaining > 0
817
+ if not _has_substantive_inspection(bootstrap_calls + tool_calls) and requires_change:
818
+ should_defer = True
819
+
820
+ if should_defer:
821
+ escalation_deferrals_remaining = max(0, escalation_deferrals_remaining - 1)
822
+ _append_escalation_feedback(messages, decision, small_repo)
823
+ continue
824
+
825
+ reason = str(decision.get("reason") or decision.get("summary") or "Local agent requested escalation.")
826
+ print(f"[Agent] Local escalation requested: {reason}")
827
+ return {
828
+ "status": "escalate",
829
+ "confidence": float(decision.get("confidence", 0.0)),
830
+ "summary": reason,
831
+ "steps": step,
832
+ "tool_calls": tool_calls,
833
+ }
834
+
835
+ if action != "tool":
836
+ print("[Agent] Model returned an invalid protocol action; retrying...")
837
+ messages.append(
838
+ {"role": "user", "content": "Protocol error: choose action 'tool', 'final', or 'escalate'."}
839
+ )
840
+ continue
841
+
842
+ tool_name = str(decision.get("tool", ""))
843
+ tool_args = decision.get("args") if isinstance(decision.get("args"), dict) else {}
844
+ result = _run_tool(repo_path, task, permission, tool_name, tool_args)
845
+
846
+ tool_calls.append({"tool": tool_name, "args": tool_args, "result": result})
847
+ messages.append({"role": "assistant", "content": json.dumps(decision, ensure_ascii=False)})
848
+ _append_tool_result(messages, tool_name, result)
849
+ _append_mutation_readback(repo_path, messages, tool_name, result)
850
+
851
+ if _looks_like_validation_call(tool_calls[-1]):
852
+ post_mutation_validation_calls.append(tool_calls[-1])
853
+
854
+ if tool_name in WORKSPACE_MUTATING_TOOLS and _mutation_result_changed(result):
855
+ # A real mutation succeeded, so tool-level recovery starts fresh.
856
+ edit_tool_retry_attempts = 0
857
+ # Any validation from before this mutation is stale.
858
+ post_mutation_validation_calls = []
859
+
860
+ if smoke_mode and requests_validation:
861
+ print("[Agent] Running requested validation after the edit...")
862
+ validation_args: dict[str, Any] = {}
863
+ validation_result = _run_tool(
864
+ repo_path,
865
+ task,
866
+ permission,
867
+ "run_tests",
868
+ validation_args,
869
+ )
870
+ validation_call = {
871
+ "tool": "run_tests",
872
+ "args": validation_args,
873
+ "result": validation_result,
874
+ }
875
+ post_mutation_validation_calls.append(validation_call)
876
+ _append_tool_result(messages, "run_tests", validation_result)
877
+
878
+ attempted, passed, unavailable = _validation_state(post_mutation_validation_calls)
879
+ if attempted and not (passed or unavailable):
880
+ if validation_repair_attempts < max_validation_repair_attempts:
881
+ validation_repair_attempts += 1
882
+ print(
883
+ "[Agent] Validation still failing; the next generation will be "
884
+ "constrained to a repair edit."
885
+ )
886
+ messages.append(
887
+ {
888
+ "role": "user",
889
+ "content": (
890
+ "POST-EDIT VALIDATION FAILED. Use the failure output above to "
891
+ "repair the implementation. Your next response must use the "
892
+ "permitted edit tool and make a real workspace change."
893
+ ),
894
+ }
895
+ )
896
+ force_edit_next = True
897
+ continue
898
+
899
+ return {
900
+ "status": "incomplete",
901
+ "confidence": 0.0,
902
+ "summary": (
903
+ "Validation still failed after the allowed local repair attempts."
904
+ ),
905
+ "steps": step,
906
+ "tool_calls": tool_calls,
907
+ }
908
+
909
+ if ( tool_name in WORKSPACE_MUTATING_TOOLS
910
+ and not _mutation_result_changed(result)
911
+ and requires_change
912
+ ):
913
+ _append_failed_edit_readback(repo_path, messages, tool_name, tool_args)
914
+ edit_tool_retry_attempts += 1
915
+
916
+ if edit_tool_retry_attempts <= max_edit_tool_retry_attempts:
917
+ print(
918
+ "[Agent] Edit tool failed or made no change; retrying constrained edit "
919
+ f"({edit_tool_retry_attempts}/{max_edit_tool_retry_attempts})..."
920
+ )
921
+ messages.append(
922
+ {
923
+ "role": "user",
924
+ "content": (
925
+ "The previous edit tool call failed or made no file-content change. "
926
+ "Use the tool error and refreshed CURRENT FILE STATE above. Make a "
927
+ "different concrete edit: the replacement old/new text must differ, "
928
+ "the old text must match the current file exactly, and the new text "
929
+ "must address the latest validation failure. Your next response must "
930
+ "use the permitted edit tool."
931
+ ),
932
+ }
933
+ )
934
+ force_edit_next = True
935
+ continue
936
+
937
+ return {
938
+ "status": "incomplete",
939
+ "confidence": 0.0,
940
+ "summary": (
941
+ "The local model could not produce a verified workspace change after "
942
+ "the allowed edit-tool recovery attempts."
943
+ ),
944
+ "steps": step,
945
+ "tool_calls": tool_calls,
946
+ }
947
+
948
+ print(f"[Agent] Step budget exhausted after {config.max_agent_steps} local generations.")
949
+ return {
950
+ "status": "incomplete",
951
+ "confidence": 0.0,
952
+ "summary": "Agent step limit reached before it could finish.",
953
+ "steps": config.max_agent_steps,
954
+ "tool_calls": tool_calls,
955
+ }