localcode 0.4.0__py3-none-macosx_13_0_arm64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (173) hide show
  1. localcode/__init__.py +11 -0
  2. localcode/__main__.py +5 -0
  3. localcode/_subproc_env.py +54 -0
  4. localcode/agent/__init__.py +177 -0
  5. localcode/agent/app_tasks.py +157 -0
  6. localcode/agent/constants.py +275 -0
  7. localcode/agent/context.py +1152 -0
  8. localcode/agent/goal.py +244 -0
  9. localcode/agent/helpers.py +1036 -0
  10. localcode/agent/hollow_module.py +230 -0
  11. localcode/agent/hooks.py +414 -0
  12. localcode/agent/loop.py +2855 -0
  13. localcode/agent/plateau.py +297 -0
  14. localcode/agent/project_check_gate.py +91 -0
  15. localcode/agent/prompt_context.py +264 -0
  16. localcode/agent/prompts.py +368 -0
  17. localcode/agent/reasoning_loop.py +79 -0
  18. localcode/agent/recovery.py +606 -0
  19. localcode/agent/recovery_controller.py +25 -0
  20. localcode/agent/round_policy.py +23 -0
  21. localcode/agent/sections.py +241 -0
  22. localcode/agent/state_machine.py +100 -0
  23. localcode/agent/streaming.py +324 -0
  24. localcode/agent/tool_execution.py +368 -0
  25. localcode/agent/tool_orchestration.py +54 -0
  26. localcode/agent/turn_finalization.py +135 -0
  27. localcode/app.py +1595 -0
  28. localcode/approvals.py +17 -0
  29. localcode/auto_compact.py +176 -0
  30. localcode/autonomy.py +103 -0
  31. localcode/bin/__init__.py +0 -0
  32. localcode/bin/llama-server +0 -0
  33. localcode/bin/localcode-ui +4 -0
  34. localcode/bootstrap.py +1191 -0
  35. localcode/cache.py +194 -0
  36. localcode/checkpoint.py +375 -0
  37. localcode/compact.py +73 -0
  38. localcode/compaction.py +341 -0
  39. localcode/composer.py +72 -0
  40. localcode/config.py +521 -0
  41. localcode/context.py +161 -0
  42. localcode/display.py +58 -0
  43. localcode/embeddings.py +480 -0
  44. localcode/entrypoint.py +536 -0
  45. localcode/errors.py +350 -0
  46. localcode/events.py +409 -0
  47. localcode/evidence.py +102 -0
  48. localcode/execution_policy.py +48 -0
  49. localcode/features.py +308 -0
  50. localcode/formatting.py +91 -0
  51. localcode/headless_json.py +265 -0
  52. localcode/health.py +288 -0
  53. localcode/hf_quants.py +203 -0
  54. localcode/history.py +594 -0
  55. localcode/indexer.py +144 -0
  56. localcode/injection_defense.py +204 -0
  57. localcode/launcher.py +373 -0
  58. localcode/logging_utils.py +28 -0
  59. localcode/lsp.py +153 -0
  60. localcode/mcp/__init__.py +88 -0
  61. localcode/mcp/_bridge.py +61 -0
  62. localcode/mcp/_config.py +33 -0
  63. localcode/mcp/_transports.py +128 -0
  64. localcode/mcp/client.py +395 -0
  65. localcode/memory_guard.py +291 -0
  66. localcode/model_config.py +288 -0
  67. localcode/model_delete.py +444 -0
  68. localcode/model_families.py +279 -0
  69. localcode/models.py +236 -0
  70. localcode/models_catalog.py +901 -0
  71. localcode/notebook.py +123 -0
  72. localcode/output.py +403 -0
  73. localcode/patching.py +65 -0
  74. localcode/paths.py +303 -0
  75. localcode/performance.py +623 -0
  76. localcode/permissions.py +79 -0
  77. localcode/permissions_v2.py +310 -0
  78. localcode/plans.py +198 -0
  79. localcode/process_registry.py +271 -0
  80. localcode/protocol/__init__.py +56 -0
  81. localcode/protocol/events.py +116 -0
  82. localcode/protocol/jsonl.py +153 -0
  83. localcode/protocol/outcomes.py +231 -0
  84. localcode/reasoning_capabilities.py +59 -0
  85. localcode/recommendations.py +87 -0
  86. localcode/recovery.py +134 -0
  87. localcode/redaction.py +164 -0
  88. localcode/runtime.py +2331 -0
  89. localcode/server_manager.py +1069 -0
  90. localcode/session.py +296 -0
  91. localcode/shell.py +83 -0
  92. localcode/skills/debug.md +21 -0
  93. localcode/skills/edit-verified.md +30 -0
  94. localcode/skills/explain.md +19 -0
  95. localcode/skills/finish-verified.md +11 -0
  96. localcode/skills/git-commit-safely.md +30 -0
  97. localcode/skills/locate.md +21 -0
  98. localcode/skills/plan-task.md +51 -0
  99. localcode/skills/review.md +22 -0
  100. localcode/skills/run-tests.md +24 -0
  101. localcode/skills.py +728 -0
  102. localcode/snapshots.py +195 -0
  103. localcode/sounds.py +50 -0
  104. localcode/telemetry.py +273 -0
  105. localcode/theme.py +137 -0
  106. localcode/thermal.py +160 -0
  107. localcode/thinking.py +114 -0
  108. localcode/tool_router.py +246 -0
  109. localcode/toolkit.py +1241 -0
  110. localcode/tools/__init__.py +400 -0
  111. localcode/tools/agent.py +259 -0
  112. localcode/tools/append_file.py +55 -0
  113. localcode/tools/background_process.py +115 -0
  114. localcode/tools/base.py +168 -0
  115. localcode/tools/bash.py +1160 -0
  116. localcode/tools/code_navigation.py +97 -0
  117. localcode/tools/edit_diff.py +80 -0
  118. localcode/tools/edit_file.py +477 -0
  119. localcode/tools/facts.py +149 -0
  120. localcode/tools/glob_tool.py +60 -0
  121. localcode/tools/grep.py +65 -0
  122. localcode/tools/inspect_symbol.py +266 -0
  123. localcode/tools/launch_app.py +93 -0
  124. localcode/tools/list_files.py +42 -0
  125. localcode/tools/multi_edit.py +167 -0
  126. localcode/tools/plan_mode.py +86 -0
  127. localcode/tools/project_check.py +685 -0
  128. localcode/tools/read_file.py +261 -0
  129. localcode/tools/read_state.py +223 -0
  130. localcode/tools/skill_tool.py +41 -0
  131. localcode/tools/syntax_check.py +216 -0
  132. localcode/tools/todo_write.py +200 -0
  133. localcode/tools/tool_call_repair.py +157 -0
  134. localcode/tools/web_fetch.py +182 -0
  135. localcode/tools/web_search.py +45 -0
  136. localcode/tools/write_file.py +281 -0
  137. localcode/tui/__init__.py +1 -0
  138. localcode/tui/app.py +485 -0
  139. localcode/tui/bridge.py +78 -0
  140. localcode/tui/clipboard_image.py +153 -0
  141. localcode/tui/paste_collapse.py +95 -0
  142. localcode/tui/screens/__init__.py +0 -0
  143. localcode/tui/screens/chat.py +5052 -0
  144. localcode/tui/screens/confirm.py +123 -0
  145. localcode/tui/screens/mode_picker.py +79 -0
  146. localcode/tui/screens/model_picker.py +962 -0
  147. localcode/tui/screens/setup.py +891 -0
  148. localcode/tui/styles/__init__.py +0 -0
  149. localcode/tui/styles/app.tcss +197 -0
  150. localcode/tui/widgets/__init__.py +0 -0
  151. localcode/tui/widgets/approval.py +2 -0
  152. localcode/tui/widgets/chat_log.py +2063 -0
  153. localcode/tui/widgets/messages/__init__.py +21 -0
  154. localcode/tui/widgets/messages/diff.py +287 -0
  155. localcode/tui/widgets/voice_visualizer.py +122 -0
  156. localcode/turn_diff.py +180 -0
  157. localcode/ui/FORK_COMMIT +1 -0
  158. localcode/ui/__init__.py +58 -0
  159. localcode/ui/launch.py +204 -0
  160. localcode/ui/picker_cli.py +165 -0
  161. localcode/ui/plugin/localcode.ts +626 -0
  162. localcode/ui/ports.py +42 -0
  163. localcode/ui/server_cmd.py +89 -0
  164. localcode/ui/supervisor.py +736 -0
  165. localcode/undo.py +121 -0
  166. localcode/verification.py +87 -0
  167. localcode/voice.py +948 -0
  168. localcode-0.4.0.dist-info/METADATA +160 -0
  169. localcode-0.4.0.dist-info/RECORD +173 -0
  170. localcode-0.4.0.dist-info/WHEEL +5 -0
  171. localcode-0.4.0.dist-info/entry_points.txt +3 -0
  172. localcode-0.4.0.dist-info/licenses/LICENSE +201 -0
  173. localcode-0.4.0.dist-info/top_level.txt +1 -0
localcode/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ __all__ = ["__version__"]
2
+
3
+ # Read the version from installed package metadata so it can never drift from
4
+ # pyproject.toml (it was hardcoded and went ~12 releases stale). Falls back to a
5
+ # hardcoded value only when running from a source tree that isn't installed.
6
+ from importlib.metadata import PackageNotFoundError, version as _pkg_version
7
+
8
+ try:
9
+ __version__ = _pkg_version("localcode")
10
+ except PackageNotFoundError: # not installed (e.g. raw source checkout)
11
+ __version__ = "0.4.0"
localcode/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ from .entrypoint import main
2
+
3
+
4
+ if __name__ == "__main__":
5
+ main()
@@ -0,0 +1,54 @@
1
+ """Single source of truth for the env dict passed to subprocesses.
2
+
3
+ Background: macOS's libsystem prints
4
+ "MallocStackLogging: can't turn off malloc stack logging
5
+ because it was not enabled."
6
+ to stderr at libsystem-init of every spawned child whenever certain
7
+ env vars are set in the child's environment. The vars are commonly
8
+ set by Xcode CLI tools, IDE shell integrations, and Conda env-activate
9
+ hooks. The warning fires before any user code runs, so Python can't
10
+ suppress it from inside the child — the only fix is to strip the vars
11
+ from the env dict we hand to subprocess.Popen / subprocess.run.
12
+
13
+ the console-script entrypoint pops them from os.environ on entry, but every site that builds
14
+ an explicit env dict must also strip them or risk re-introducing the
15
+ var. Real failure 2026-04-26: terminal flooded with ~60 of these
16
+ warnings because spawn sites filtered MallocStackLogging* but missed
17
+ MallocNanoZone, which produces the same warning class.
18
+
19
+ This module is the SINGLE place that knows the full ban list. Every
20
+ subprocess spawn that needs to construct an env dict should use
21
+ clean_env() rather than rolling its own filter.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import os
26
+ from typing import Mapping
27
+
28
+
29
+ # Env vars whose presence triggers the libsystem malloc-stack-logging
30
+ # warning at child-process startup on macOS. Sources:
31
+ # • MallocStackLogging[NoCompact]: enabled by Xcode Instruments and
32
+ # a few legacy debug profiles; setting it to "0" doesn't disable —
33
+ # the var must be UNSET.
34
+ # • MallocNanoZone: typically "0" set by Xcode CLI tools / IDE
35
+ # terminal integrations to opt-out of the nano malloc zone.
36
+ # Triggers the same class of warning at libsystem init.
37
+ _MALLOC_NOISE_VARS = {
38
+ "MallocStackLogging",
39
+ "MallocStackLoggingNoCompact",
40
+ "MallocNanoZone",
41
+ }
42
+
43
+
44
+ def clean_env(base: Mapping[str, str] | None = None) -> dict[str, str]:
45
+ """Return a copy of `base` (default: os.environ) with malloc-noise
46
+ vars stripped. Safe to pass to subprocess.Popen / subprocess.run as
47
+ `env=`.
48
+
49
+ Always returns a NEW dict — never mutates the input. Callers that
50
+ need to add or override keys (`env["GGML_BACKEND_PATH"] = ""`,
51
+ etc.) can do so on the returned dict without affecting os.environ.
52
+ """
53
+ src = os.environ if base is None else base
54
+ return {k: v for k, v in src.items() if k not in _MALLOC_NOISE_VARS}
@@ -0,0 +1,177 @@
1
+ """localcode.agent — public re-export surface.
2
+
3
+ This package contains the agent turn engine, split across several
4
+ focused modules (T0.1 refactor). Everything listed below is
5
+ re-exported here so external callers can continue to import from
6
+ `localcode.agent` without caring about the internal split.
7
+
8
+ loop.py — `run_agent_loop`, the main entry point
9
+ prompts.py — SYSTEM_PROMPT, REASONING_RULES,
10
+ _load_project_instructions
11
+ constants.py — policy knobs, safety caps, tables
12
+ context.py — message aging / redaction / compaction pipeline
13
+ recovery.py — stall detection + auto-nudge
14
+ helpers.py — tool-dispatch + display helpers
15
+
16
+ Background: before T0.1, agent.py was a 1792-line monolith. The split
17
+ broke it into focused modules (each ≤ ~720 LoC), with __init__.py
18
+ reduced to this re-export surface — well under the 400-LoC cap the
19
+ plan sets for god modules (see dev/eval/OPTIMIZATION_PLAN.md § T0).
20
+
21
+ Unused-import warnings in this file are expected — every imported
22
+ name is intentionally re-exported. The `# noqa: F401` comments
23
+ document this for linters that honour them; Pylance doesn't, so its
24
+ "not accessed" warnings on this module are false positives.
25
+ """
26
+ from __future__ import annotations
27
+
28
+
29
+ # Public contract for `localcode.agent`. Anything not in this list is
30
+ # internal and may be renamed / moved / deleted without warning. Names
31
+ # starting with an underscore are included for back-compat with
32
+ # tests/test_context_pipeline_e2e.py which reaches into the context
33
+ # pipeline by name — once that test is migrated to a public helper,
34
+ # those entries can come out of `__all__` and move to underscore-only
35
+ # "internal import at your own risk."
36
+ __all__ = [
37
+ # Loop entry
38
+ "run_agent_loop",
39
+ # Prompts
40
+ "SYSTEM_PROMPT",
41
+ "REASONING_RULES",
42
+ # Constants (policy knobs external code may want to read)
43
+ "MAX_ROUNDS",
44
+ "MAX_OUTPUT_TOKENS",
45
+ "MAX_THINKING_SECONDS",
46
+ "MAX_THINKING_CHARS",
47
+ "RESULT_LIMITS",
48
+ "MAX_AGGREGATE_PER_TURN",
49
+ "DESTRUCTIVE_PATTERNS",
50
+ "COMPACT_KEEP_RECENT_TOOL_RESULTS",
51
+ "COMPACT_MIN_CONTENT_CHARS",
52
+ "REDACT_KEEP_RECENT_WRITES",
53
+ "REDACT_MIN_CONTENT_CHARS",
54
+ "READ_UNCHANGED_STUB_PREFIX",
55
+ # Recovery
56
+ "StallMode",
57
+ "detect_stall",
58
+ "nudge_for",
59
+ "MAX_EMPTY_ROUND_RETRIES",
60
+ # Context pipeline — underscore-prefixed, test-only back-compat
61
+ "_prepare_model_messages",
62
+ "_redact_old_write_args",
63
+ "_redact_duplicate_reads",
64
+ "_compact_old_tool_results",
65
+ ]
66
+
67
+
68
+ # ── Constants ────────────────────────────────────────────────────────
69
+ # Policy knobs / safety caps / table data moved to agent/constants.py
70
+ # during the T0.1 split. Re-exported here so external callers (tests,
71
+ # eval, app.py) that do `from localcode.agent import MAX_THINKING_SECONDS`
72
+ # keep working unchanged.
73
+
74
+ from .constants import ( # noqa: F401 — re-exports for back-compat
75
+ MAX_ROUNDS,
76
+ MAX_OUTPUT_TOKENS,
77
+ MAX_THINKING_SECONDS,
78
+ MAX_THINKING_CHARS,
79
+ RESULT_LIMITS,
80
+ MAX_AGGREGATE_PER_TURN,
81
+ DESTRUCTIVE_PATTERNS,
82
+ COMPACT_KEEP_RECENT_TOOL_RESULTS,
83
+ COMPACT_MIN_CONTENT_CHARS,
84
+ REDACT_KEEP_RECENT_WRITES,
85
+ REDACT_MIN_CONTENT_CHARS,
86
+ PROJECT_FILES as _PROJECT_FILES,
87
+ READ_UNCHANGED_STUB_PREFIX,
88
+ )
89
+
90
+ # ── Context-management pipeline ─────────────────────────────────────
91
+ # Moved to agent/context.py during T0.1-c. Re-exported here so
92
+ # tests/test_context_pipeline_e2e.py + other callers that do
93
+ # `from localcode.agent import _prepare_model_messages`
94
+ # keep working unchanged.
95
+
96
+ from .context import ( # noqa: F401
97
+ _truncate_result,
98
+ _compact_old_tool_results,
99
+ _redact_old_write_args,
100
+ _redact_duplicate_reads,
101
+ _msg_bytes,
102
+ _prepare_model_messages,
103
+ _estimate_tokens,
104
+ _compact_messages,
105
+ _summarize_args,
106
+ )
107
+
108
+
109
+ # ── Stall detection + auto-nudge recovery ──────────────────────────
110
+ # Moved to agent/recovery.py during T0.1-d. The loop calls
111
+ # `detect_stall(...)` after each round and, if the round stalled,
112
+ # appends `nudge_for(mode)` as a synthetic user message before
113
+ # looping. MAX_EMPTY_ROUND_RETRIES bounds how many consecutive
114
+ # stalls we tolerate per turn.
115
+
116
+ from .recovery import ( # noqa: F401
117
+ StallMode,
118
+ detect_stall,
119
+ nudge_for,
120
+ MAX_EMPTY_ROUND_RETRIES,
121
+ )
122
+
123
+
124
+ # ── Loop-adjacent helpers ──────────────────────────────────────────
125
+ # Moved to agent/helpers.py during T0.1-e. Re-exported here so any
126
+ # internal caller that still imports via `from localcode.agent import
127
+ # _execute_tool` (or other private helpers) keeps working unchanged.
128
+ # These are not part of the public API — they're named here only to
129
+ # preserve the back-compat surface during the refactor.
130
+
131
+ from .helpers import ( # noqa: F401
132
+ _execute_tool,
133
+ _first_token,
134
+ _needs_confirmation,
135
+ _render_markdown,
136
+ _brief_result,
137
+ _grounded_file_summary,
138
+ _tool_stage_label,
139
+ )
140
+
141
+ # ── Prompt templates + project-instructions loader ────────────────────────
142
+ # Moved to agent/prompts.py during the T0.1-b split. Re-exported here so
143
+ # external callers (dev/eval/prompt_variants.py, tests/promptfoo, app.py,
144
+ # tests/test_context_pipeline_e2e.py) that do
145
+ # `from localcode.agent import SYSTEM_PROMPT`
146
+ # keep working unchanged. See agent/prompts.py for the commented
147
+ # MINIMAL-CORE variant preserved there for visual diffing.
148
+
149
+ from .prompts import ( # noqa: F401 — re-exports for back-compat
150
+ SYSTEM_PROMPT,
151
+ REASONING_RULES,
152
+ _load_project_instructions,
153
+ )
154
+
155
+
156
+ # ── Tool registry ────────────────────────────────────────────────────────
157
+ #
158
+ # Every tool lives in its own file under src/localcode/tools/. That package
159
+ # assembles the registry; we just pull in the schemas (for the LLM call)
160
+ # and the dispatcher (for _execute_tool). Plan-mode gating still lives
161
+ # here because it's cross-tool policy, not tool-specific logic.
162
+
163
+
164
+
165
+ # ── Result Management ────────────────────────────────────────────────────
166
+
167
+ # ── Context Management ───────────────────────────────────────────────────
168
+
169
+ # ── Display Helpers ──────────────────────────────────────────────────────
170
+
171
+ # ── Main agent loop ─────────────────────────────────────────────────
172
+ # Moved to agent/loop.py during T0.1-f. Re-exported here so every
173
+ # existing caller that does `from localcode.agent import run_agent_loop`
174
+ # keeps working unchanged. This is the public entry point into the
175
+ # agent turn engine.
176
+
177
+ from .loop import run_agent_loop # noqa: F401
@@ -0,0 +1,157 @@
1
+ """App/build/run task helpers for the agent loop."""
2
+ from __future__ import annotations
3
+
4
+ from pathlib import Path
5
+ import re
6
+
7
+
8
+ PARTIAL_HANDOFF_RE = re.compile(
9
+ r"(?:\bnext steps\b|\bimplemented features so far\b|\bi have started building\b|"
10
+ r"\bready to proceed\b|\bready to continue\b|\barchitecture overview\b|"
11
+ r"\bwould you like me to\b|\bI'?m ready to proceed\b)",
12
+ re.IGNORECASE,
13
+ )
14
+ BLOCKING_QUESTION_RE = re.compile(r"^(?:[^?]{0,320}\?)$", re.DOTALL)
15
+ PORT_RE = re.compile(r"(?:--port\s+|-p\s+|localhost:|127\.0\.0\.1:)(\d{2,5})")
16
+ __all__ = [
17
+ "looks_like_partial_handoff",
18
+ "is_focused_blocking_question",
19
+ "extract_port",
20
+ "has_runtime_verification_signal",
21
+ "app_source_line_stats",
22
+ "has_launch_signal",
23
+ "ground_run_or_launch_text",
24
+ "format_run_or_launch_summary",
25
+ ]
26
+
27
+
28
+ def looks_like_partial_handoff(content: str) -> bool:
29
+ return bool(PARTIAL_HANDOFF_RE.search(content or ""))
30
+
31
+
32
+ def is_focused_blocking_question(content: str) -> bool:
33
+ text = (content or "").strip()
34
+ if not text:
35
+ return False
36
+ if text.count("?") != 1:
37
+ return False
38
+ if not BLOCKING_QUESTION_RE.match(text):
39
+ return False
40
+ lower = text.lower()
41
+ if re.match(r"^(?:hi|hello|hey|yo)[!.\s,]*(?:how can i help|what can i do)", lower):
42
+ return False
43
+ return not any(
44
+ bad in lower for bad in (
45
+ "next steps",
46
+ "ready to proceed",
47
+ "implemented features so far",
48
+ "architecture overview",
49
+ )
50
+ )
51
+
52
+
53
+ def extract_port(text: str) -> int:
54
+ for raw in PORT_RE.findall(text or ""):
55
+ try:
56
+ port = int(raw)
57
+ except Exception:
58
+ continue
59
+ if 1 <= port <= 65535:
60
+ return port
61
+ return 0
62
+
63
+
64
+ def has_runtime_verification_signal(bash_history: list[tuple[str, str]]) -> bool:
65
+ for cmd, result in bash_history:
66
+ cmd_l = (cmd or "").lower()
67
+ result_l = (result or "").lower()
68
+ if result_l.startswith("error:") or result_l.startswith("rejected:"):
69
+ continue
70
+ if "curl " in cmd_l or "http://localhost:" in cmd_l or "http://127.0.0.1:" in cmd_l:
71
+ if any(
72
+ bad in result_l for bad in (
73
+ "address already in use",
74
+ "error while attempting to bind",
75
+ "failed to start",
76
+ "connection refused",
77
+ "not found",
78
+ )
79
+ ):
80
+ continue
81
+ return True
82
+ if "open http://localhost:" in cmd_l or "open http://127.0.0.1:" in cmd_l:
83
+ return True
84
+ return False
85
+
86
+
87
+ def app_source_line_stats(repo_root: Path | str, changed_files: list[str]) -> tuple[int, int]:
88
+ repo = Path(repo_root)
89
+ source_exts = {".py", ".js", ".jsx", ".ts", ".tsx", ".html", ".css"}
90
+ source_count = 0
91
+ total_lines = 0
92
+ for rel in changed_files:
93
+ try:
94
+ path = repo / rel
95
+ if path.suffix.lower() not in source_exts or not path.is_file():
96
+ continue
97
+ source_count += 1
98
+ total_lines += len(path.read_text(errors="replace").splitlines())
99
+ except Exception:
100
+ continue
101
+ return source_count, total_lines
102
+
103
+
104
+ def has_launch_signal(bash_history: list[tuple[str, str]]) -> bool:
105
+ for cmd, result in bash_history:
106
+ cmd_l = (cmd or "").lower()
107
+ result_l = (result or "").lower()
108
+ if result_l.startswith("error:") or result_l.startswith("rejected:"):
109
+ continue
110
+ if any(
111
+ token in cmd_l for token in (
112
+ "npm run dev",
113
+ "vite",
114
+ "uvicorn",
115
+ "flask run",
116
+ "python -m http.server",
117
+ "streamlit run",
118
+ )
119
+ ):
120
+ if any(
121
+ bad in result_l for bad in (
122
+ "exit code 1",
123
+ "address already in use",
124
+ "failed to start",
125
+ "connection refused",
126
+ "error:",
127
+ )
128
+ ):
129
+ continue
130
+ return True
131
+ return False
132
+
133
+
134
+ def ground_run_or_launch_text(text: str, port: int) -> str:
135
+ if port <= 0 or not text:
136
+ return text
137
+ grounded = text
138
+ grounded = grounded.replace("http://localhost:[FRONTEND_PORT]", f"http://localhost:{port}")
139
+ grounded = grounded.replace("http://127.0.0.1:[FRONTEND_PORT]", f"http://127.0.0.1:{port}")
140
+ grounded = grounded.replace("http://localhost:[PORT]", f"http://localhost:{port}")
141
+ grounded = grounded.replace("http://127.0.0.1:[PORT]", f"http://127.0.0.1:{port}")
142
+ grounded = grounded.replace("localhost:[FRONTEND_PORT]", f"localhost:{port}")
143
+ grounded = grounded.replace("127.0.0.1:[FRONTEND_PORT]", f"127.0.0.1:{port}")
144
+ grounded = grounded.replace("localhost:[PORT]", f"localhost:{port}")
145
+ grounded = grounded.replace("127.0.0.1:[PORT]", f"127.0.0.1:{port}")
146
+ return grounded
147
+
148
+
149
+ def format_run_or_launch_summary(port: int, verified: bool) -> str:
150
+ if port > 0:
151
+ if verified:
152
+ return f"The app is now running and verified.\n\nOpen it at http://localhost:{port}."
153
+ return f"The app is now running.\n\nOpen it at http://localhost:{port}."
154
+ if verified:
155
+ return "The app is now running and verified."
156
+ return "The app is now running."
157
+
@@ -0,0 +1,275 @@
1
+ """Agent loop constants, kept separate from the loop logic.
2
+
3
+ Pulled out of agent/__init__.py during the T0.1 split. Every name here
4
+ was previously a module-level constant in the old agent.py — nothing
5
+ has changed semantically. Keeping them in their own module lets us:
6
+
7
+ 1. Reason about policy knobs without scrolling past 1,800 lines of
8
+ loop logic.
9
+ 2. Re-export them from `localcode.agent` so external callers
10
+ (tests, eval, app.py) don't notice the split.
11
+ 3. Import them from sibling modules (context.py, recovery.py,
12
+ loop.py) without creating an import cycle through the package
13
+ `__init__`.
14
+ """
15
+ from __future__ import annotations
16
+
17
+
18
+ __all__ = [
19
+ "MAX_ROUNDS",
20
+ "MAX_OUTPUT_TOKENS",
21
+ "MAX_THINKING_SECONDS",
22
+ "MAX_THINKING_CHARS",
23
+ "RESULT_LIMITS",
24
+ "MAX_AGGREGATE_PER_TURN",
25
+ "DESTRUCTIVE_PATTERNS",
26
+ "COMPACT_KEEP_RECENT_TOOL_RESULTS",
27
+ "COMPACT_MIN_CONTENT_CHARS",
28
+ "REDACT_KEEP_RECENT_WRITES",
29
+ "REDACT_MIN_CONTENT_CHARS",
30
+ "PROJECT_FILES",
31
+ "READ_UNCHANGED_STUB_PREFIX",
32
+ "CHURN_FILE_WRITE_LIMIT",
33
+ "CHURN_COMMAND_FAIL_LIMIT",
34
+ "CHURN_READONLY_STREAK_LIMIT",
35
+ "CHURN_PLANNING_STREAK_LIMIT",
36
+ "CROSS_ROUND_REPEAT_LIMIT",
37
+ ]
38
+
39
+
40
+ # ── Turn-level caps ─────────────────────────────────────────────────
41
+
42
+ MAX_ROUNDS = 0
43
+ """Upper bound on round-trips the agent loop takes per turn. Each
44
+ round = one model call + any tool dispatches that follow.
45
+
46
+ `0` means NO HARD CAP — matches agent (`maxTurns` opt-in, default
47
+ unlimited) and terminal coding tools (no per-turn limit). Loop termination is
48
+ delegated to the targeted safety nets that catch REAL failure
49
+ patterns rather than counting rounds:
50
+
51
+ - 3-in-a-row identical-call breaker (`recent_tool_sigs`)
52
+ - same-tool > 10 in a turn (`tool_name_counts`)
53
+ - file-edit > 3 same path (`file_edit_counts`)
54
+ - investigation-spin (≥10 read-only) (`_readonly_streak`)
55
+ - looks-fine streak (≥3 rounds) (`_looks_fine_streak`)
56
+ - MAX_THINKING_SECONDS / CHARS (per-round thinking cap)
57
+ - empty-round nudge (no content/tools)
58
+ - `cancel_requested` + Ctrl+C (user-initiated)
59
+
60
+ Bumped through 20 → 50 → 0 on 2026-04-26 after observing legitimate
61
+ "redesign this section" investigations need 10+ rounds, and that
62
+ every pathological loop in our telemetry trips one of the targeted
63
+ guards inside 15 rounds. MAX_ROUNDS as a hard cap was emergency
64
+ insurance with no observed claims. Set to a positive integer to
65
+ re-enable a hard ceiling (eval / batch / unattended modes may want
66
+ this; interactive sessions don't)."""
67
+
68
+ MAX_OUTPUT_TOKENS = -1
69
+ """Per-round generation cap. -1 = unlimited; the model stops at its
70
+ natural EOS. This stays -1 deliberately: capping mid-stream truncates
71
+ valid tool-call JSON into unparseable garbage, which then triggers a
72
+ useless recovery round. A prior 5.8-minute round burn was caused by an
73
+ 8192 cap chopping a long-but-valid write_file call. So a small per-round
74
+ token cap is the WRONG lever.
75
+
76
+ The runaway backstop is NOT ctx-size. An earlier version of this note
77
+ claimed `--ctx-size 32768` bounded prompt+generation so "the model can't
78
+ run forever" — that is FALSE for long-context models. LocalCode launches
79
+ Qwen (262K trained) with `--ctx-size 131072`; a 6K-token prompt then
80
+ leaves ~125K tokens of generation headroom ≈ ~28 minutes of nonstop
81
+ decode. A real 40-minute / no-output hang (thinking on, no answer emitted)
82
+ was traced to exactly this.
83
+
84
+ The primary control is llama.cpp's per-request `thinking_budget_tokens`: it
85
+ forces the template's end-thinking sequence and lets the SAME generation move
86
+ on to a tool call. Feature.THINKING_CAPS + MAX_THINKING_SECONDS / CHARS are
87
+ compatibility backstops for templates whose thinking tags the server cannot
88
+ identify. Leaving MAX_OUTPUT_TOKENS at -1 keeps valid long tool calls intact
89
+ without leaving the reasoning channel unbounded."""
90
+
91
+
92
+ # ── Thinking-phase safety caps ──────────────────────────────────────
93
+ #
94
+ # Bound how much reasoning Python will accept if the server-side token budget
95
+ # cannot recognize the model's thinking delimiters. Either cap trips a
96
+ # `_thinking_abort`. Exact periodic loops retry once with thinking disabled;
97
+ # other runaways surface a clear user message.
98
+ #
99
+ # These are a RUNAWAY guard, not a reasoning budget. The earlier tight
100
+ # values (90 s / 4000 chars ≈ 1000 tokens) aborted legitimate long
101
+ # reasoning and were disabled 2026-04-27 (see features.THINKING_CAPS).
102
+ # They are re-enabled now with far more generous bounds after a real
103
+ # 29-minute / 94.8k-token runaway on Qwen Q8 with thinking on: the
104
+ # cap must never touch normal reasoning, only a pathological loop.
105
+ #
106
+ # * 600 s (10 min): a genuine slow-reason path on a heavy Q8 model
107
+ # (~50-75 tok/s) fits comfortably; a 29-minute loop does not.
108
+ # * 80000 chars (~20k reasoning tokens): well above any legitimate
109
+ # planning trace; the observed runaway was ~380k chars.
110
+ #
111
+ # With reasoning now streamed live in the TUI, the user also SEES a
112
+ # runaway and can `esc` — the cap is the backstop for unattended /
113
+ # headless runs where no one is watching.
114
+
115
+ # NOTE: the CHAR cap is the real, speed-independent runaway guard (~20k
116
+ # reasoning tokens); the structural loop detector catches periodic loops in
117
+ # ~1s. The TIME cap is only a final backstop, and it must scale to the SLOWEST
118
+ # model we ship: at ~15 tok/s (dense Qwen 3.8 27B Q8 / Muse Glimmer, not the
119
+ # ~50-75 tok/s MoEs this was first tuned for), 80k chars ≈ 20k tokens ≈ 22 min
120
+ # of legitimate reasoning. A 600s cap tripped that at ~5.7k tokens — aborting
121
+ # NORMAL slow reasoning, the exact thing this cap must never do. 1800s (30 min)
122
+ # keeps the char cap + loop detector as the real guards. Either way a trip now
123
+ # recovers via a no-think retry (see loop.py), so the turn is never thrown away.
124
+ MAX_THINKING_SECONDS = 1800
125
+ MAX_THINKING_CHARS = 80000
126
+
127
+
128
+ # ── Tool-result size policy ─────────────────────────────────────────
129
+
130
+ RESULT_LIMITS: dict[str, int] = {
131
+ "grep": 20_000,
132
+ "bash": 30_000,
133
+ "read_file": 50_000,
134
+ "web_search": 10_000,
135
+ "default": 50_000,
136
+ }
137
+ """Per-tool truncation budgets (chars). Tools whose payload exceeds
138
+ their budget get truncated with a clear "[truncated N chars]" tail
139
+ so the judge / user can tell content was dropped. Tune per tool
140
+ because grep output and bash stdout have different density."""
141
+
142
+ MAX_AGGREGATE_PER_TURN = 100_000
143
+ """Sum of tool-result chars across a single turn. If a turn ingests
144
+ more than this across its tool calls, later tool calls get their
145
+ results aggressively truncated or stubbed — prevents a single turn
146
+ from blowing past the context window via 10× big grep results."""
147
+
148
+
149
+ # ── Destructive-command detection ───────────────────────────────────
150
+
151
+ DESTRUCTIVE_PATTERNS: list[str] = [
152
+ "rm -rf", "rm -r", "rmdir", "git push", "git reset --hard",
153
+ "sudo ", "pip install", "npm install", "brew install",
154
+ "docker rm", "kubectl delete", "DROP TABLE", "DELETE FROM",
155
+ "python ", "python3 ", "node ", "npm run", "npm start",
156
+ ]
157
+ """Prefixes that trigger the approval flow in the bash tool. Matching
158
+ is substring-wise so `bash -c 'rm -rf foo'` still fires. NOT a
159
+ security boundary — a determined user can bypass via e.g. `\\rm`
160
+ or `eval` — but catches the common footgun cases."""
161
+
162
+
163
+ # ── Context-management policy ───────────────────────────────────────
164
+
165
+ COMPACT_KEEP_RECENT_TOOL_RESULTS = 4
166
+ """Number of most-recent tool_result messages to preserve verbatim
167
+ before aging starts. Older tool_result payloads get replaced with a
168
+ brief summary stub."""
169
+
170
+ COMPACT_MIN_CONTENT_CHARS = 400
171
+ """Only tool_result messages longer than this get considered for
172
+ aging. Short results (exit codes, one-line stdout) stay inline
173
+ because they're cheap and often load-bearing."""
174
+
175
+ REDACT_KEEP_RECENT_WRITES = 1
176
+ """Number of most-recent write_file/edit_file tool_call args to
177
+ preserve verbatim. Older ones get their `content`/`new_string` args
178
+ redacted with a stub telling the model to read_file the path if it
179
+ needs the content. Prevents the ~10× context bloat that would
180
+ otherwise accumulate from repeated full-file writes."""
181
+
182
+ REDACT_MIN_CONTENT_CHARS = 400
183
+ """Threshold for redaction — small writes (short configs, tiny
184
+ shims) stay inline because the savings aren't worth the indirection."""
185
+
186
+
187
+ # ── Project-instruction files (load order matters) ──────────────────
188
+
189
+ PROJECT_FILES: list[str] = ["LOCALCODE.md", "localcode.md", ".localcode.md"]
190
+ """Filenames the agent looks for at repo root for project-specific
191
+ instructions. First match wins. Kept in this list so the agent
192
+ behaviour is documented in one place and other callers (eval /
193
+ tests / setup UI) can list them without duplicating the literal."""
194
+
195
+
196
+ # ── Duplicate-read stub text ────────────────────────────────────────
197
+
198
+ READ_UNCHANGED_STUB_PREFIX = (
199
+ "[FILE UNCHANGED — a later read_file call for this same path is in "
200
+ "history below with the current content. Scroll forward, or re-call "
201
+ "read_file if needed.]"
202
+ )
203
+ """Replacement text for older duplicate read_file results, so the
204
+ model sees "the content exists further down" rather than the raw
205
+ bytes twice. `_redact_duplicate_reads` inserts this prefix."""
206
+
207
+
208
+ # ── Semantic-churn thresholds ───────────────────────────────────────
209
+ #
210
+ # These catch the "thrashing without converging" pattern that the
211
+ # byte-identical-call breakers miss: the model keeps WRITING a file
212
+ # (different content each time) or keeps RE-RUNNING a failing command,
213
+ # never reading the actual error and making a targeted fix. Distinct
214
+ # from the exact-repeat guards (`recent_tool_sigs`, `success_counts`)
215
+ # which only trip on identical args. Tuned conservatively so normal
216
+ # multi-step flows (a couple of edits to one file, running a build
217
+ # command twice while iterating) do NOT trip — only sustained churn.
218
+
219
+ CHURN_FILE_WRITE_LIMIT = 3
220
+ """How many times the SAME path may be written/edited in one turn
221
+ before we nudge "stop rewriting it, read the error and make a
222
+ targeted fix." Counts every write/edit/append to the path regardless
223
+ of whether content differs — semantic churn, not byte-identical
224
+ repeats. 3 chosen because a legit flow is typically write-once then
225
+ one corrective edit (2); a 3rd full rewrite of the same file in a
226
+ turn is the churn signal (the real incident rewrote package.json ~5×)."""
227
+
228
+ CHURN_COMMAND_FAIL_LIMIT = 3
229
+ """How many times the SAME command (keyed by its first token, e.g.
230
+ `npm`) may FAIL in one turn before we nudge "read its error output
231
+ and fix the root cause before re-running." 3 (not 2) so running a
232
+ build/install command, fixing, and one more failure while iterating
233
+ is tolerated; the 3rd failure of the same command family means the
234
+ model is re-running without absorbing the error."""
235
+
236
+ CHURN_READONLY_STREAK_LIMIT = 6
237
+ """Consecutive rounds of PURE read-only investigation (read_file /
238
+ grep / glob / list_files / web_*) with no mutating or server action
239
+ before we nudge "take a concrete action now." Tighter than the prior
240
+ generic streak of 10: a 10-round read-only run is already 5+ minutes
241
+ of the user staring at a frozen screen. 6 still allows reading the
242
+ handful of files needed to understand a redesign before committing,
243
+ but interrupts a genuine investigation spin sooner."""
244
+
245
+ CHURN_PLANNING_STREAK_LIMIT = 4
246
+ """Consecutive rounds of PLANNING-WITHOUT-PROGRESS before we nudge
247
+ "you've planned enough; take a concrete action now."
248
+
249
+ A round counts toward this streak when ALL of:
250
+ • it changed NO new file this turn (changed_files count didn't grow),
251
+ • it ran NO build/test/verify command, and
252
+ • it produced thinking or narration content (i.e. the model was
253
+ reasoning/planning, not idle).
254
+
255
+ This catches the model that re-derives the SAME plan across many rounds
256
+ — lots of thinking, no concrete action — which the read-only-spin
257
+ signal misses because that streak resets on any round with zero tool
258
+ calls (a pure think-then-read-then-think alternation never accumulates
259
+ a pure read-only streak). Set to 4 (one higher than the read-only
260
+ limit's effective reach) so legitimate "read two files, think, read a
261
+ third, then edit" flows do NOT trip: as soon as a round changes a file
262
+ or runs a build, the streak resets to 0."""
263
+
264
+ CROSS_ROUND_REPEAT_LIMIT = 4
265
+ """How many times the SAME (tool, canonical-args) call may run ACROSS the
266
+ turn before we nudge "you already have this result — stop repeating it."
267
+
268
+ The in-round breaker catches identical calls within ONE round; this catches
269
+ the cross-ROUND spin the logs show — read_file on the same path 53x over many
270
+ rounds, or a pkill->curl->read loop where each command succeeds (so the
271
+ command-FAILURE breaker never trips). Crucially this only NUDGES — it never
272
+ withholds the tool result (the 2026-04-29 read-dedup STUB starved legitimate
273
+ debug re-reads and hung a turn 17 min; we do not repeat that). A write/edit to
274
+ a path resets that path's read counts, so a legitimate read-after-edit isn't
275
+ counted. 4 tolerates a couple of genuine re-looks before flagging a true loop."""