localcode 0.4.0a1__py3-none-macosx_13_0_arm64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- localcode/__init__.py +11 -0
- localcode/__main__.py +5 -0
- localcode/_subproc_env.py +54 -0
- localcode/agent/__init__.py +177 -0
- localcode/agent/app_tasks.py +157 -0
- localcode/agent/constants.py +275 -0
- localcode/agent/context.py +1152 -0
- localcode/agent/goal.py +244 -0
- localcode/agent/helpers.py +1036 -0
- localcode/agent/hollow_module.py +230 -0
- localcode/agent/hooks.py +414 -0
- localcode/agent/loop.py +2855 -0
- localcode/agent/plateau.py +297 -0
- localcode/agent/project_check_gate.py +91 -0
- localcode/agent/prompt_context.py +264 -0
- localcode/agent/prompts.py +368 -0
- localcode/agent/reasoning_loop.py +79 -0
- localcode/agent/recovery.py +606 -0
- localcode/agent/recovery_controller.py +25 -0
- localcode/agent/round_policy.py +23 -0
- localcode/agent/sections.py +241 -0
- localcode/agent/state_machine.py +100 -0
- localcode/agent/streaming.py +324 -0
- localcode/agent/tool_execution.py +368 -0
- localcode/agent/tool_orchestration.py +54 -0
- localcode/agent/turn_finalization.py +135 -0
- localcode/app.py +1595 -0
- localcode/approvals.py +17 -0
- localcode/auto_compact.py +176 -0
- localcode/autonomy.py +103 -0
- localcode/bin/__init__.py +0 -0
- localcode/bin/llama-server +0 -0
- localcode/bin/localcode-ui +4 -0
- localcode/bootstrap.py +1191 -0
- localcode/cache.py +194 -0
- localcode/checkpoint.py +375 -0
- localcode/compact.py +73 -0
- localcode/compaction.py +341 -0
- localcode/composer.py +72 -0
- localcode/config.py +518 -0
- localcode/context.py +161 -0
- localcode/display.py +58 -0
- localcode/embeddings.py +480 -0
- localcode/entrypoint.py +536 -0
- localcode/errors.py +350 -0
- localcode/events.py +409 -0
- localcode/evidence.py +102 -0
- localcode/execution_policy.py +48 -0
- localcode/features.py +308 -0
- localcode/formatting.py +91 -0
- localcode/headless_json.py +265 -0
- localcode/health.py +288 -0
- localcode/hf_quants.py +203 -0
- localcode/history.py +594 -0
- localcode/indexer.py +144 -0
- localcode/injection_defense.py +204 -0
- localcode/launcher.py +373 -0
- localcode/logging_utils.py +28 -0
- localcode/lsp.py +153 -0
- localcode/mcp/__init__.py +88 -0
- localcode/mcp/_bridge.py +61 -0
- localcode/mcp/_config.py +33 -0
- localcode/mcp/_transports.py +128 -0
- localcode/mcp/client.py +395 -0
- localcode/memory_guard.py +291 -0
- localcode/model_config.py +288 -0
- localcode/model_delete.py +444 -0
- localcode/model_families.py +279 -0
- localcode/models.py +236 -0
- localcode/models_catalog.py +901 -0
- localcode/notebook.py +123 -0
- localcode/output.py +403 -0
- localcode/patching.py +65 -0
- localcode/paths.py +303 -0
- localcode/performance.py +623 -0
- localcode/permissions.py +79 -0
- localcode/permissions_v2.py +310 -0
- localcode/plans.py +198 -0
- localcode/process_registry.py +271 -0
- localcode/protocol/__init__.py +56 -0
- localcode/protocol/events.py +116 -0
- localcode/protocol/jsonl.py +153 -0
- localcode/protocol/outcomes.py +231 -0
- localcode/reasoning_capabilities.py +59 -0
- localcode/recommendations.py +87 -0
- localcode/recovery.py +134 -0
- localcode/redaction.py +164 -0
- localcode/runtime.py +2331 -0
- localcode/server_manager.py +1069 -0
- localcode/session.py +296 -0
- localcode/shell.py +83 -0
- localcode/skills/debug.md +21 -0
- localcode/skills/edit-verified.md +30 -0
- localcode/skills/explain.md +19 -0
- localcode/skills/finish-verified.md +11 -0
- localcode/skills/git-commit-safely.md +30 -0
- localcode/skills/locate.md +21 -0
- localcode/skills/plan-task.md +51 -0
- localcode/skills/review.md +22 -0
- localcode/skills/run-tests.md +24 -0
- localcode/skills.py +728 -0
- localcode/snapshots.py +195 -0
- localcode/sounds.py +50 -0
- localcode/telemetry.py +273 -0
- localcode/theme.py +137 -0
- localcode/thermal.py +160 -0
- localcode/thinking.py +114 -0
- localcode/tool_router.py +246 -0
- localcode/toolkit.py +1241 -0
- localcode/tools/__init__.py +400 -0
- localcode/tools/agent.py +259 -0
- localcode/tools/append_file.py +55 -0
- localcode/tools/background_process.py +115 -0
- localcode/tools/base.py +168 -0
- localcode/tools/bash.py +1160 -0
- localcode/tools/code_navigation.py +97 -0
- localcode/tools/edit_diff.py +80 -0
- localcode/tools/edit_file.py +477 -0
- localcode/tools/facts.py +149 -0
- localcode/tools/glob_tool.py +60 -0
- localcode/tools/grep.py +65 -0
- localcode/tools/inspect_symbol.py +266 -0
- localcode/tools/launch_app.py +93 -0
- localcode/tools/list_files.py +42 -0
- localcode/tools/multi_edit.py +167 -0
- localcode/tools/plan_mode.py +86 -0
- localcode/tools/project_check.py +685 -0
- localcode/tools/read_file.py +261 -0
- localcode/tools/read_state.py +223 -0
- localcode/tools/skill_tool.py +41 -0
- localcode/tools/syntax_check.py +216 -0
- localcode/tools/todo_write.py +200 -0
- localcode/tools/tool_call_repair.py +157 -0
- localcode/tools/web_fetch.py +182 -0
- localcode/tools/web_search.py +45 -0
- localcode/tools/write_file.py +281 -0
- localcode/tui/__init__.py +1 -0
- localcode/tui/app.py +485 -0
- localcode/tui/bridge.py +78 -0
- localcode/tui/clipboard_image.py +153 -0
- localcode/tui/paste_collapse.py +95 -0
- localcode/tui/screens/__init__.py +0 -0
- localcode/tui/screens/chat.py +5052 -0
- localcode/tui/screens/confirm.py +123 -0
- localcode/tui/screens/mode_picker.py +79 -0
- localcode/tui/screens/model_picker.py +962 -0
- localcode/tui/screens/setup.py +891 -0
- localcode/tui/styles/__init__.py +0 -0
- localcode/tui/styles/app.tcss +197 -0
- localcode/tui/widgets/__init__.py +0 -0
- localcode/tui/widgets/approval.py +2 -0
- localcode/tui/widgets/chat_log.py +2063 -0
- localcode/tui/widgets/messages/__init__.py +21 -0
- localcode/tui/widgets/messages/diff.py +287 -0
- localcode/tui/widgets/voice_visualizer.py +122 -0
- localcode/turn_diff.py +180 -0
- localcode/ui/FORK_COMMIT +1 -0
- localcode/ui/__init__.py +58 -0
- localcode/ui/launch.py +201 -0
- localcode/ui/picker_cli.py +165 -0
- localcode/ui/plugin/localcode.ts +626 -0
- localcode/ui/ports.py +42 -0
- localcode/ui/server_cmd.py +89 -0
- localcode/ui/supervisor.py +736 -0
- localcode/undo.py +121 -0
- localcode/verification.py +87 -0
- localcode/voice.py +948 -0
- localcode-0.4.0a1.dist-info/METADATA +160 -0
- localcode-0.4.0a1.dist-info/RECORD +173 -0
- localcode-0.4.0a1.dist-info/WHEEL +5 -0
- localcode-0.4.0a1.dist-info/entry_points.txt +3 -0
- localcode-0.4.0a1.dist-info/licenses/LICENSE +201 -0
- localcode-0.4.0a1.dist-info/top_level.txt +1 -0
localcode/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
__all__ = ["__version__"]
|
|
2
|
+
|
|
3
|
+
# Read the version from installed package metadata so it can never drift from
|
|
4
|
+
# pyproject.toml (it was hardcoded and went ~12 releases stale). Falls back to a
|
|
5
|
+
# hardcoded value only when running from a source tree that isn't installed.
|
|
6
|
+
from importlib.metadata import PackageNotFoundError, version as _pkg_version
|
|
7
|
+
|
|
8
|
+
try:
|
|
9
|
+
__version__ = _pkg_version("localcode")
|
|
10
|
+
except PackageNotFoundError: # not installed (e.g. raw source checkout)
|
|
11
|
+
__version__ = "0.4.0a1"
|
localcode/__main__.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Single source of truth for the env dict passed to subprocesses.
|
|
2
|
+
|
|
3
|
+
Background: macOS's libsystem prints
|
|
4
|
+
"MallocStackLogging: can't turn off malloc stack logging
|
|
5
|
+
because it was not enabled."
|
|
6
|
+
to stderr at libsystem-init of every spawned child whenever certain
|
|
7
|
+
env vars are set in the child's environment. The vars are commonly
|
|
8
|
+
set by Xcode CLI tools, IDE shell integrations, and Conda env-activate
|
|
9
|
+
hooks. The warning fires before any user code runs, so Python can't
|
|
10
|
+
suppress it from inside the child — the only fix is to strip the vars
|
|
11
|
+
from the env dict we hand to subprocess.Popen / subprocess.run.
|
|
12
|
+
|
|
13
|
+
the console-script entrypoint pops them from os.environ on entry, but every site that builds
|
|
14
|
+
an explicit env dict must also strip them or risk re-introducing the
|
|
15
|
+
var. Real failure 2026-04-26: terminal flooded with ~60 of these
|
|
16
|
+
warnings because spawn sites filtered MallocStackLogging* but missed
|
|
17
|
+
MallocNanoZone, which produces the same warning class.
|
|
18
|
+
|
|
19
|
+
This module is the SINGLE place that knows the full ban list. Every
|
|
20
|
+
subprocess spawn that needs to construct an env dict should use
|
|
21
|
+
clean_env() rather than rolling its own filter.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import os
|
|
26
|
+
from typing import Mapping
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Env vars whose presence triggers the libsystem malloc-stack-logging
|
|
30
|
+
# warning at child-process startup on macOS. Sources:
|
|
31
|
+
# • MallocStackLogging[NoCompact]: enabled by Xcode Instruments and
|
|
32
|
+
# a few legacy debug profiles; setting it to "0" doesn't disable —
|
|
33
|
+
# the var must be UNSET.
|
|
34
|
+
# • MallocNanoZone: typically "0" set by Xcode CLI tools / IDE
|
|
35
|
+
# terminal integrations to opt-out of the nano malloc zone.
|
|
36
|
+
# Triggers the same class of warning at libsystem init.
|
|
37
|
+
_MALLOC_NOISE_VARS = {
|
|
38
|
+
"MallocStackLogging",
|
|
39
|
+
"MallocStackLoggingNoCompact",
|
|
40
|
+
"MallocNanoZone",
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def clean_env(base: Mapping[str, str] | None = None) -> dict[str, str]:
|
|
45
|
+
"""Return a copy of `base` (default: os.environ) with malloc-noise
|
|
46
|
+
vars stripped. Safe to pass to subprocess.Popen / subprocess.run as
|
|
47
|
+
`env=`.
|
|
48
|
+
|
|
49
|
+
Always returns a NEW dict — never mutates the input. Callers that
|
|
50
|
+
need to add or override keys (`env["GGML_BACKEND_PATH"] = ""`,
|
|
51
|
+
etc.) can do so on the returned dict without affecting os.environ.
|
|
52
|
+
"""
|
|
53
|
+
src = os.environ if base is None else base
|
|
54
|
+
return {k: v for k, v in src.items() if k not in _MALLOC_NOISE_VARS}
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""localcode.agent — public re-export surface.
|
|
2
|
+
|
|
3
|
+
This package contains the agent turn engine, split across several
|
|
4
|
+
focused modules (T0.1 refactor). Everything listed below is
|
|
5
|
+
re-exported here so external callers can continue to import from
|
|
6
|
+
`localcode.agent` without caring about the internal split.
|
|
7
|
+
|
|
8
|
+
loop.py — `run_agent_loop`, the main entry point
|
|
9
|
+
prompts.py — SYSTEM_PROMPT, REASONING_RULES,
|
|
10
|
+
_load_project_instructions
|
|
11
|
+
constants.py — policy knobs, safety caps, tables
|
|
12
|
+
context.py — message aging / redaction / compaction pipeline
|
|
13
|
+
recovery.py — stall detection + auto-nudge
|
|
14
|
+
helpers.py — tool-dispatch + display helpers
|
|
15
|
+
|
|
16
|
+
Background: before T0.1, agent.py was a 1792-line monolith. The split
|
|
17
|
+
broke it into focused modules (each ≤ ~720 LoC), with __init__.py
|
|
18
|
+
reduced to this re-export surface — well under the 400-LoC cap the
|
|
19
|
+
plan sets for god modules (see dev/eval/OPTIMIZATION_PLAN.md § T0).
|
|
20
|
+
|
|
21
|
+
Unused-import warnings in this file are expected — every imported
|
|
22
|
+
name is intentionally re-exported. The `# noqa: F401` comments
|
|
23
|
+
document this for linters that honour them; Pylance doesn't, so its
|
|
24
|
+
"not accessed" warnings on this module are false positives.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
# Public contract for `localcode.agent`. Anything not in this list is
|
|
30
|
+
# internal and may be renamed / moved / deleted without warning. Names
|
|
31
|
+
# starting with an underscore are included for back-compat with
|
|
32
|
+
# tests/test_context_pipeline_e2e.py which reaches into the context
|
|
33
|
+
# pipeline by name — once that test is migrated to a public helper,
|
|
34
|
+
# those entries can come out of `__all__` and move to underscore-only
|
|
35
|
+
# "internal import at your own risk."
|
|
36
|
+
__all__ = [
|
|
37
|
+
# Loop entry
|
|
38
|
+
"run_agent_loop",
|
|
39
|
+
# Prompts
|
|
40
|
+
"SYSTEM_PROMPT",
|
|
41
|
+
"REASONING_RULES",
|
|
42
|
+
# Constants (policy knobs external code may want to read)
|
|
43
|
+
"MAX_ROUNDS",
|
|
44
|
+
"MAX_OUTPUT_TOKENS",
|
|
45
|
+
"MAX_THINKING_SECONDS",
|
|
46
|
+
"MAX_THINKING_CHARS",
|
|
47
|
+
"RESULT_LIMITS",
|
|
48
|
+
"MAX_AGGREGATE_PER_TURN",
|
|
49
|
+
"DESTRUCTIVE_PATTERNS",
|
|
50
|
+
"COMPACT_KEEP_RECENT_TOOL_RESULTS",
|
|
51
|
+
"COMPACT_MIN_CONTENT_CHARS",
|
|
52
|
+
"REDACT_KEEP_RECENT_WRITES",
|
|
53
|
+
"REDACT_MIN_CONTENT_CHARS",
|
|
54
|
+
"READ_UNCHANGED_STUB_PREFIX",
|
|
55
|
+
# Recovery
|
|
56
|
+
"StallMode",
|
|
57
|
+
"detect_stall",
|
|
58
|
+
"nudge_for",
|
|
59
|
+
"MAX_EMPTY_ROUND_RETRIES",
|
|
60
|
+
# Context pipeline — underscore-prefixed, test-only back-compat
|
|
61
|
+
"_prepare_model_messages",
|
|
62
|
+
"_redact_old_write_args",
|
|
63
|
+
"_redact_duplicate_reads",
|
|
64
|
+
"_compact_old_tool_results",
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# ── Constants ────────────────────────────────────────────────────────
|
|
69
|
+
# Policy knobs / safety caps / table data moved to agent/constants.py
|
|
70
|
+
# during the T0.1 split. Re-exported here so external callers (tests,
|
|
71
|
+
# eval, app.py) that do `from localcode.agent import MAX_THINKING_SECONDS`
|
|
72
|
+
# keep working unchanged.
|
|
73
|
+
|
|
74
|
+
from .constants import ( # noqa: F401 — re-exports for back-compat
|
|
75
|
+
MAX_ROUNDS,
|
|
76
|
+
MAX_OUTPUT_TOKENS,
|
|
77
|
+
MAX_THINKING_SECONDS,
|
|
78
|
+
MAX_THINKING_CHARS,
|
|
79
|
+
RESULT_LIMITS,
|
|
80
|
+
MAX_AGGREGATE_PER_TURN,
|
|
81
|
+
DESTRUCTIVE_PATTERNS,
|
|
82
|
+
COMPACT_KEEP_RECENT_TOOL_RESULTS,
|
|
83
|
+
COMPACT_MIN_CONTENT_CHARS,
|
|
84
|
+
REDACT_KEEP_RECENT_WRITES,
|
|
85
|
+
REDACT_MIN_CONTENT_CHARS,
|
|
86
|
+
PROJECT_FILES as _PROJECT_FILES,
|
|
87
|
+
READ_UNCHANGED_STUB_PREFIX,
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
# ── Context-management pipeline ─────────────────────────────────────
|
|
91
|
+
# Moved to agent/context.py during T0.1-c. Re-exported here so
|
|
92
|
+
# tests/test_context_pipeline_e2e.py + other callers that do
|
|
93
|
+
# `from localcode.agent import _prepare_model_messages`
|
|
94
|
+
# keep working unchanged.
|
|
95
|
+
|
|
96
|
+
from .context import ( # noqa: F401
|
|
97
|
+
_truncate_result,
|
|
98
|
+
_compact_old_tool_results,
|
|
99
|
+
_redact_old_write_args,
|
|
100
|
+
_redact_duplicate_reads,
|
|
101
|
+
_msg_bytes,
|
|
102
|
+
_prepare_model_messages,
|
|
103
|
+
_estimate_tokens,
|
|
104
|
+
_compact_messages,
|
|
105
|
+
_summarize_args,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
# ── Stall detection + auto-nudge recovery ──────────────────────────
|
|
110
|
+
# Moved to agent/recovery.py during T0.1-d. The loop calls
|
|
111
|
+
# `detect_stall(...)` after each round and, if the round stalled,
|
|
112
|
+
# appends `nudge_for(mode)` as a synthetic user message before
|
|
113
|
+
# looping. MAX_EMPTY_ROUND_RETRIES bounds how many consecutive
|
|
114
|
+
# stalls we tolerate per turn.
|
|
115
|
+
|
|
116
|
+
from .recovery import ( # noqa: F401
|
|
117
|
+
StallMode,
|
|
118
|
+
detect_stall,
|
|
119
|
+
nudge_for,
|
|
120
|
+
MAX_EMPTY_ROUND_RETRIES,
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ── Loop-adjacent helpers ──────────────────────────────────────────
|
|
125
|
+
# Moved to agent/helpers.py during T0.1-e. Re-exported here so any
|
|
126
|
+
# internal caller that still imports via `from localcode.agent import
|
|
127
|
+
# _execute_tool` (or other private helpers) keeps working unchanged.
|
|
128
|
+
# These are not part of the public API — they're named here only to
|
|
129
|
+
# preserve the back-compat surface during the refactor.
|
|
130
|
+
|
|
131
|
+
from .helpers import ( # noqa: F401
|
|
132
|
+
_execute_tool,
|
|
133
|
+
_first_token,
|
|
134
|
+
_needs_confirmation,
|
|
135
|
+
_render_markdown,
|
|
136
|
+
_brief_result,
|
|
137
|
+
_grounded_file_summary,
|
|
138
|
+
_tool_stage_label,
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
# ── Prompt templates + project-instructions loader ────────────────────────
|
|
142
|
+
# Moved to agent/prompts.py during the T0.1-b split. Re-exported here so
|
|
143
|
+
# external callers (dev/eval/prompt_variants.py, tests/promptfoo, app.py,
|
|
144
|
+
# tests/test_context_pipeline_e2e.py) that do
|
|
145
|
+
# `from localcode.agent import SYSTEM_PROMPT`
|
|
146
|
+
# keep working unchanged. See agent/prompts.py for the commented
|
|
147
|
+
# MINIMAL-CORE variant preserved there for visual diffing.
|
|
148
|
+
|
|
149
|
+
from .prompts import ( # noqa: F401 — re-exports for back-compat
|
|
150
|
+
SYSTEM_PROMPT,
|
|
151
|
+
REASONING_RULES,
|
|
152
|
+
_load_project_instructions,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
# ── Tool registry ────────────────────────────────────────────────────────
|
|
157
|
+
#
|
|
158
|
+
# Every tool lives in its own file under src/localcode/tools/. That package
|
|
159
|
+
# assembles the registry; we just pull in the schemas (for the LLM call)
|
|
160
|
+
# and the dispatcher (for _execute_tool). Plan-mode gating still lives
|
|
161
|
+
# here because it's cross-tool policy, not tool-specific logic.
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# ── Result Management ────────────────────────────────────────────────────
|
|
166
|
+
|
|
167
|
+
# ── Context Management ───────────────────────────────────────────────────
|
|
168
|
+
|
|
169
|
+
# ── Display Helpers ──────────────────────────────────────────────────────
|
|
170
|
+
|
|
171
|
+
# ── Main agent loop ─────────────────────────────────────────────────
|
|
172
|
+
# Moved to agent/loop.py during T0.1-f. Re-exported here so every
|
|
173
|
+
# existing caller that does `from localcode.agent import run_agent_loop`
|
|
174
|
+
# keeps working unchanged. This is the public entry point into the
|
|
175
|
+
# agent turn engine.
|
|
176
|
+
|
|
177
|
+
from .loop import run_agent_loop # noqa: F401
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""App/build/run task helpers for the agent loop."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
PARTIAL_HANDOFF_RE = re.compile(
|
|
9
|
+
r"(?:\bnext steps\b|\bimplemented features so far\b|\bi have started building\b|"
|
|
10
|
+
r"\bready to proceed\b|\bready to continue\b|\barchitecture overview\b|"
|
|
11
|
+
r"\bwould you like me to\b|\bI'?m ready to proceed\b)",
|
|
12
|
+
re.IGNORECASE,
|
|
13
|
+
)
|
|
14
|
+
BLOCKING_QUESTION_RE = re.compile(r"^(?:[^?]{0,320}\?)$", re.DOTALL)
|
|
15
|
+
PORT_RE = re.compile(r"(?:--port\s+|-p\s+|localhost:|127\.0\.0\.1:)(\d{2,5})")
|
|
16
|
+
__all__ = [
|
|
17
|
+
"looks_like_partial_handoff",
|
|
18
|
+
"is_focused_blocking_question",
|
|
19
|
+
"extract_port",
|
|
20
|
+
"has_runtime_verification_signal",
|
|
21
|
+
"app_source_line_stats",
|
|
22
|
+
"has_launch_signal",
|
|
23
|
+
"ground_run_or_launch_text",
|
|
24
|
+
"format_run_or_launch_summary",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def looks_like_partial_handoff(content: str) -> bool:
|
|
29
|
+
return bool(PARTIAL_HANDOFF_RE.search(content or ""))
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def is_focused_blocking_question(content: str) -> bool:
|
|
33
|
+
text = (content or "").strip()
|
|
34
|
+
if not text:
|
|
35
|
+
return False
|
|
36
|
+
if text.count("?") != 1:
|
|
37
|
+
return False
|
|
38
|
+
if not BLOCKING_QUESTION_RE.match(text):
|
|
39
|
+
return False
|
|
40
|
+
lower = text.lower()
|
|
41
|
+
if re.match(r"^(?:hi|hello|hey|yo)[!.\s,]*(?:how can i help|what can i do)", lower):
|
|
42
|
+
return False
|
|
43
|
+
return not any(
|
|
44
|
+
bad in lower for bad in (
|
|
45
|
+
"next steps",
|
|
46
|
+
"ready to proceed",
|
|
47
|
+
"implemented features so far",
|
|
48
|
+
"architecture overview",
|
|
49
|
+
)
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def extract_port(text: str) -> int:
|
|
54
|
+
for raw in PORT_RE.findall(text or ""):
|
|
55
|
+
try:
|
|
56
|
+
port = int(raw)
|
|
57
|
+
except Exception:
|
|
58
|
+
continue
|
|
59
|
+
if 1 <= port <= 65535:
|
|
60
|
+
return port
|
|
61
|
+
return 0
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def has_runtime_verification_signal(bash_history: list[tuple[str, str]]) -> bool:
|
|
65
|
+
for cmd, result in bash_history:
|
|
66
|
+
cmd_l = (cmd or "").lower()
|
|
67
|
+
result_l = (result or "").lower()
|
|
68
|
+
if result_l.startswith("error:") or result_l.startswith("rejected:"):
|
|
69
|
+
continue
|
|
70
|
+
if "curl " in cmd_l or "http://localhost:" in cmd_l or "http://127.0.0.1:" in cmd_l:
|
|
71
|
+
if any(
|
|
72
|
+
bad in result_l for bad in (
|
|
73
|
+
"address already in use",
|
|
74
|
+
"error while attempting to bind",
|
|
75
|
+
"failed to start",
|
|
76
|
+
"connection refused",
|
|
77
|
+
"not found",
|
|
78
|
+
)
|
|
79
|
+
):
|
|
80
|
+
continue
|
|
81
|
+
return True
|
|
82
|
+
if "open http://localhost:" in cmd_l or "open http://127.0.0.1:" in cmd_l:
|
|
83
|
+
return True
|
|
84
|
+
return False
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def app_source_line_stats(repo_root: Path | str, changed_files: list[str]) -> tuple[int, int]:
|
|
88
|
+
repo = Path(repo_root)
|
|
89
|
+
source_exts = {".py", ".js", ".jsx", ".ts", ".tsx", ".html", ".css"}
|
|
90
|
+
source_count = 0
|
|
91
|
+
total_lines = 0
|
|
92
|
+
for rel in changed_files:
|
|
93
|
+
try:
|
|
94
|
+
path = repo / rel
|
|
95
|
+
if path.suffix.lower() not in source_exts or not path.is_file():
|
|
96
|
+
continue
|
|
97
|
+
source_count += 1
|
|
98
|
+
total_lines += len(path.read_text(errors="replace").splitlines())
|
|
99
|
+
except Exception:
|
|
100
|
+
continue
|
|
101
|
+
return source_count, total_lines
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def has_launch_signal(bash_history: list[tuple[str, str]]) -> bool:
|
|
105
|
+
for cmd, result in bash_history:
|
|
106
|
+
cmd_l = (cmd or "").lower()
|
|
107
|
+
result_l = (result or "").lower()
|
|
108
|
+
if result_l.startswith("error:") or result_l.startswith("rejected:"):
|
|
109
|
+
continue
|
|
110
|
+
if any(
|
|
111
|
+
token in cmd_l for token in (
|
|
112
|
+
"npm run dev",
|
|
113
|
+
"vite",
|
|
114
|
+
"uvicorn",
|
|
115
|
+
"flask run",
|
|
116
|
+
"python -m http.server",
|
|
117
|
+
"streamlit run",
|
|
118
|
+
)
|
|
119
|
+
):
|
|
120
|
+
if any(
|
|
121
|
+
bad in result_l for bad in (
|
|
122
|
+
"exit code 1",
|
|
123
|
+
"address already in use",
|
|
124
|
+
"failed to start",
|
|
125
|
+
"connection refused",
|
|
126
|
+
"error:",
|
|
127
|
+
)
|
|
128
|
+
):
|
|
129
|
+
continue
|
|
130
|
+
return True
|
|
131
|
+
return False
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def ground_run_or_launch_text(text: str, port: int) -> str:
|
|
135
|
+
if port <= 0 or not text:
|
|
136
|
+
return text
|
|
137
|
+
grounded = text
|
|
138
|
+
grounded = grounded.replace("http://localhost:[FRONTEND_PORT]", f"http://localhost:{port}")
|
|
139
|
+
grounded = grounded.replace("http://127.0.0.1:[FRONTEND_PORT]", f"http://127.0.0.1:{port}")
|
|
140
|
+
grounded = grounded.replace("http://localhost:[PORT]", f"http://localhost:{port}")
|
|
141
|
+
grounded = grounded.replace("http://127.0.0.1:[PORT]", f"http://127.0.0.1:{port}")
|
|
142
|
+
grounded = grounded.replace("localhost:[FRONTEND_PORT]", f"localhost:{port}")
|
|
143
|
+
grounded = grounded.replace("127.0.0.1:[FRONTEND_PORT]", f"127.0.0.1:{port}")
|
|
144
|
+
grounded = grounded.replace("localhost:[PORT]", f"localhost:{port}")
|
|
145
|
+
grounded = grounded.replace("127.0.0.1:[PORT]", f"127.0.0.1:{port}")
|
|
146
|
+
return grounded
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def format_run_or_launch_summary(port: int, verified: bool) -> str:
|
|
150
|
+
if port > 0:
|
|
151
|
+
if verified:
|
|
152
|
+
return f"The app is now running and verified.\n\nOpen it at http://localhost:{port}."
|
|
153
|
+
return f"The app is now running.\n\nOpen it at http://localhost:{port}."
|
|
154
|
+
if verified:
|
|
155
|
+
return "The app is now running and verified."
|
|
156
|
+
return "The app is now running."
|
|
157
|
+
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""Agent loop constants, kept separate from the loop logic.
|
|
2
|
+
|
|
3
|
+
Pulled out of agent/__init__.py during the T0.1 split. Every name here
|
|
4
|
+
was previously a module-level constant in the old agent.py — nothing
|
|
5
|
+
has changed semantically. Keeping them in their own module lets us:
|
|
6
|
+
|
|
7
|
+
1. Reason about policy knobs without scrolling past 1,800 lines of
|
|
8
|
+
loop logic.
|
|
9
|
+
2. Re-export them from `localcode.agent` so external callers
|
|
10
|
+
(tests, eval, app.py) don't notice the split.
|
|
11
|
+
3. Import them from sibling modules (context.py, recovery.py,
|
|
12
|
+
loop.py) without creating an import cycle through the package
|
|
13
|
+
`__init__`.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"MAX_ROUNDS",
|
|
20
|
+
"MAX_OUTPUT_TOKENS",
|
|
21
|
+
"MAX_THINKING_SECONDS",
|
|
22
|
+
"MAX_THINKING_CHARS",
|
|
23
|
+
"RESULT_LIMITS",
|
|
24
|
+
"MAX_AGGREGATE_PER_TURN",
|
|
25
|
+
"DESTRUCTIVE_PATTERNS",
|
|
26
|
+
"COMPACT_KEEP_RECENT_TOOL_RESULTS",
|
|
27
|
+
"COMPACT_MIN_CONTENT_CHARS",
|
|
28
|
+
"REDACT_KEEP_RECENT_WRITES",
|
|
29
|
+
"REDACT_MIN_CONTENT_CHARS",
|
|
30
|
+
"PROJECT_FILES",
|
|
31
|
+
"READ_UNCHANGED_STUB_PREFIX",
|
|
32
|
+
"CHURN_FILE_WRITE_LIMIT",
|
|
33
|
+
"CHURN_COMMAND_FAIL_LIMIT",
|
|
34
|
+
"CHURN_READONLY_STREAK_LIMIT",
|
|
35
|
+
"CHURN_PLANNING_STREAK_LIMIT",
|
|
36
|
+
"CROSS_ROUND_REPEAT_LIMIT",
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
# ── Turn-level caps ─────────────────────────────────────────────────
|
|
41
|
+
|
|
42
|
+
MAX_ROUNDS = 0
|
|
43
|
+
"""Upper bound on round-trips the agent loop takes per turn. Each
|
|
44
|
+
round = one model call + any tool dispatches that follow.
|
|
45
|
+
|
|
46
|
+
`0` means NO HARD CAP — matches agent (`maxTurns` opt-in, default
|
|
47
|
+
unlimited) and terminal coding tools (no per-turn limit). Loop termination is
|
|
48
|
+
delegated to the targeted safety nets that catch REAL failure
|
|
49
|
+
patterns rather than counting rounds:
|
|
50
|
+
|
|
51
|
+
- 3-in-a-row identical-call breaker (`recent_tool_sigs`)
|
|
52
|
+
- same-tool > 10 in a turn (`tool_name_counts`)
|
|
53
|
+
- file-edit > 3 same path (`file_edit_counts`)
|
|
54
|
+
- investigation-spin (≥10 read-only) (`_readonly_streak`)
|
|
55
|
+
- looks-fine streak (≥3 rounds) (`_looks_fine_streak`)
|
|
56
|
+
- MAX_THINKING_SECONDS / CHARS (per-round thinking cap)
|
|
57
|
+
- empty-round nudge (no content/tools)
|
|
58
|
+
- `cancel_requested` + Ctrl+C (user-initiated)
|
|
59
|
+
|
|
60
|
+
Bumped through 20 → 50 → 0 on 2026-04-26 after observing legitimate
|
|
61
|
+
"redesign this section" investigations need 10+ rounds, and that
|
|
62
|
+
every pathological loop in our telemetry trips one of the targeted
|
|
63
|
+
guards inside 15 rounds. MAX_ROUNDS as a hard cap was emergency
|
|
64
|
+
insurance with no observed claims. Set to a positive integer to
|
|
65
|
+
re-enable a hard ceiling (eval / batch / unattended modes may want
|
|
66
|
+
this; interactive sessions don't)."""
|
|
67
|
+
|
|
68
|
+
MAX_OUTPUT_TOKENS = -1
|
|
69
|
+
"""Per-round generation cap. -1 = unlimited; the model stops at its
|
|
70
|
+
natural EOS. This stays -1 deliberately: capping mid-stream truncates
|
|
71
|
+
valid tool-call JSON into unparseable garbage, which then triggers a
|
|
72
|
+
useless recovery round. A prior 5.8-minute round burn was caused by an
|
|
73
|
+
8192 cap chopping a long-but-valid write_file call. So a small per-round
|
|
74
|
+
token cap is the WRONG lever.
|
|
75
|
+
|
|
76
|
+
The runaway backstop is NOT ctx-size. An earlier version of this note
|
|
77
|
+
claimed `--ctx-size 32768` bounded prompt+generation so "the model can't
|
|
78
|
+
run forever" — that is FALSE for long-context models. LocalCode launches
|
|
79
|
+
Qwen (262K trained) with `--ctx-size 131072`; a 6K-token prompt then
|
|
80
|
+
leaves ~125K tokens of generation headroom ≈ ~28 minutes of nonstop
|
|
81
|
+
decode. A real 40-minute / no-output hang (thinking on, no answer emitted)
|
|
82
|
+
was traced to exactly this.
|
|
83
|
+
|
|
84
|
+
The primary control is llama.cpp's per-request `thinking_budget_tokens`: it
|
|
85
|
+
forces the template's end-thinking sequence and lets the SAME generation move
|
|
86
|
+
on to a tool call. Feature.THINKING_CAPS + MAX_THINKING_SECONDS / CHARS are
|
|
87
|
+
compatibility backstops for templates whose thinking tags the server cannot
|
|
88
|
+
identify. Leaving MAX_OUTPUT_TOKENS at -1 keeps valid long tool calls intact
|
|
89
|
+
without leaving the reasoning channel unbounded."""
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# ── Thinking-phase safety caps ──────────────────────────────────────
|
|
93
|
+
#
|
|
94
|
+
# Bound how much reasoning Python will accept if the server-side token budget
|
|
95
|
+
# cannot recognize the model's thinking delimiters. Either cap trips a
|
|
96
|
+
# `_thinking_abort`. Exact periodic loops retry once with thinking disabled;
|
|
97
|
+
# other runaways surface a clear user message.
|
|
98
|
+
#
|
|
99
|
+
# These are a RUNAWAY guard, not a reasoning budget. The earlier tight
|
|
100
|
+
# values (90 s / 4000 chars ≈ 1000 tokens) aborted legitimate long
|
|
101
|
+
# reasoning and were disabled 2026-04-27 (see features.THINKING_CAPS).
|
|
102
|
+
# They are re-enabled now with far more generous bounds after a real
|
|
103
|
+
# 29-minute / 94.8k-token runaway on Qwen Q8 with thinking on: the
|
|
104
|
+
# cap must never touch normal reasoning, only a pathological loop.
|
|
105
|
+
#
|
|
106
|
+
# * 600 s (10 min): a genuine slow-reason path on a heavy Q8 model
|
|
107
|
+
# (~50-75 tok/s) fits comfortably; a 29-minute loop does not.
|
|
108
|
+
# * 80000 chars (~20k reasoning tokens): well above any legitimate
|
|
109
|
+
# planning trace; the observed runaway was ~380k chars.
|
|
110
|
+
#
|
|
111
|
+
# With reasoning now streamed live in the TUI, the user also SEES a
|
|
112
|
+
# runaway and can `esc` — the cap is the backstop for unattended /
|
|
113
|
+
# headless runs where no one is watching.
|
|
114
|
+
|
|
115
|
+
# NOTE: the CHAR cap is the real, speed-independent runaway guard (~20k
|
|
116
|
+
# reasoning tokens); the structural loop detector catches periodic loops in
|
|
117
|
+
# ~1s. The TIME cap is only a final backstop, and it must scale to the SLOWEST
|
|
118
|
+
# model we ship: at ~15 tok/s (dense Qwen 3.8 27B Q8 / Muse Glimmer, not the
|
|
119
|
+
# ~50-75 tok/s MoEs this was first tuned for), 80k chars ≈ 20k tokens ≈ 22 min
|
|
120
|
+
# of legitimate reasoning. A 600s cap tripped that at ~5.7k tokens — aborting
|
|
121
|
+
# NORMAL slow reasoning, the exact thing this cap must never do. 1800s (30 min)
|
|
122
|
+
# keeps the char cap + loop detector as the real guards. Either way a trip now
|
|
123
|
+
# recovers via a no-think retry (see loop.py), so the turn is never thrown away.
|
|
124
|
+
MAX_THINKING_SECONDS = 1800
|
|
125
|
+
MAX_THINKING_CHARS = 80000
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# ── Tool-result size policy ─────────────────────────────────────────
|
|
129
|
+
|
|
130
|
+
RESULT_LIMITS: dict[str, int] = {
|
|
131
|
+
"grep": 20_000,
|
|
132
|
+
"bash": 30_000,
|
|
133
|
+
"read_file": 50_000,
|
|
134
|
+
"web_search": 10_000,
|
|
135
|
+
"default": 50_000,
|
|
136
|
+
}
|
|
137
|
+
"""Per-tool truncation budgets (chars). Tools whose payload exceeds
|
|
138
|
+
their budget get truncated with a clear "[truncated N chars]" tail
|
|
139
|
+
so the judge / user can tell content was dropped. Tune per tool
|
|
140
|
+
because grep output and bash stdout have different density."""
|
|
141
|
+
|
|
142
|
+
MAX_AGGREGATE_PER_TURN = 100_000
|
|
143
|
+
"""Sum of tool-result chars across a single turn. If a turn ingests
|
|
144
|
+
more than this across its tool calls, later tool calls get their
|
|
145
|
+
results aggressively truncated or stubbed — prevents a single turn
|
|
146
|
+
from blowing past the context window via 10× big grep results."""
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
# ── Destructive-command detection ───────────────────────────────────
|
|
150
|
+
|
|
151
|
+
DESTRUCTIVE_PATTERNS: list[str] = [
|
|
152
|
+
"rm -rf", "rm -r", "rmdir", "git push", "git reset --hard",
|
|
153
|
+
"sudo ", "pip install", "npm install", "brew install",
|
|
154
|
+
"docker rm", "kubectl delete", "DROP TABLE", "DELETE FROM",
|
|
155
|
+
"python ", "python3 ", "node ", "npm run", "npm start",
|
|
156
|
+
]
|
|
157
|
+
"""Prefixes that trigger the approval flow in the bash tool. Matching
|
|
158
|
+
is substring-wise so `bash -c 'rm -rf foo'` still fires. NOT a
|
|
159
|
+
security boundary — a determined user can bypass via e.g. `\\rm`
|
|
160
|
+
or `eval` — but catches the common footgun cases."""
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# ── Context-management policy ───────────────────────────────────────
|
|
164
|
+
|
|
165
|
+
COMPACT_KEEP_RECENT_TOOL_RESULTS = 4
|
|
166
|
+
"""Number of most-recent tool_result messages to preserve verbatim
|
|
167
|
+
before aging starts. Older tool_result payloads get replaced with a
|
|
168
|
+
brief summary stub."""
|
|
169
|
+
|
|
170
|
+
COMPACT_MIN_CONTENT_CHARS = 400
|
|
171
|
+
"""Only tool_result messages longer than this get considered for
|
|
172
|
+
aging. Short results (exit codes, one-line stdout) stay inline
|
|
173
|
+
because they're cheap and often load-bearing."""
|
|
174
|
+
|
|
175
|
+
REDACT_KEEP_RECENT_WRITES = 1
|
|
176
|
+
"""Number of most-recent write_file/edit_file tool_call args to
|
|
177
|
+
preserve verbatim. Older ones get their `content`/`new_string` args
|
|
178
|
+
redacted with a stub telling the model to read_file the path if it
|
|
179
|
+
needs the content. Prevents the ~10× context bloat that would
|
|
180
|
+
otherwise accumulate from repeated full-file writes."""
|
|
181
|
+
|
|
182
|
+
REDACT_MIN_CONTENT_CHARS = 400
|
|
183
|
+
"""Threshold for redaction — small writes (short configs, tiny
|
|
184
|
+
shims) stay inline because the savings aren't worth the indirection."""
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# ── Project-instruction files (load order matters) ──────────────────
|
|
188
|
+
|
|
189
|
+
PROJECT_FILES: list[str] = ["LOCALCODE.md", "localcode.md", ".localcode.md"]
|
|
190
|
+
"""Filenames the agent looks for at repo root for project-specific
|
|
191
|
+
instructions. First match wins. Kept in this list so the agent
|
|
192
|
+
behaviour is documented in one place and other callers (eval /
|
|
193
|
+
tests / setup UI) can list them without duplicating the literal."""
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# ── Duplicate-read stub text ────────────────────────────────────────
|
|
197
|
+
|
|
198
|
+
READ_UNCHANGED_STUB_PREFIX = (
|
|
199
|
+
"[FILE UNCHANGED — a later read_file call for this same path is in "
|
|
200
|
+
"history below with the current content. Scroll forward, or re-call "
|
|
201
|
+
"read_file if needed.]"
|
|
202
|
+
)
|
|
203
|
+
"""Replacement text for older duplicate read_file results, so the
|
|
204
|
+
model sees "the content exists further down" rather than the raw
|
|
205
|
+
bytes twice. `_redact_duplicate_reads` inserts this prefix."""
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
# ── Semantic-churn thresholds ───────────────────────────────────────
|
|
209
|
+
#
|
|
210
|
+
# These catch the "thrashing without converging" pattern that the
|
|
211
|
+
# byte-identical-call breakers miss: the model keeps WRITING a file
|
|
212
|
+
# (different content each time) or keeps RE-RUNNING a failing command,
|
|
213
|
+
# never reading the actual error and making a targeted fix. Distinct
|
|
214
|
+
# from the exact-repeat guards (`recent_tool_sigs`, `success_counts`)
|
|
215
|
+
# which only trip on identical args. Tuned conservatively so normal
|
|
216
|
+
# multi-step flows (a couple of edits to one file, running a build
|
|
217
|
+
# command twice while iterating) do NOT trip — only sustained churn.
|
|
218
|
+
|
|
219
|
+
CHURN_FILE_WRITE_LIMIT = 3
|
|
220
|
+
"""How many times the SAME path may be written/edited in one turn
|
|
221
|
+
before we nudge "stop rewriting it, read the error and make a
|
|
222
|
+
targeted fix." Counts every write/edit/append to the path regardless
|
|
223
|
+
of whether content differs — semantic churn, not byte-identical
|
|
224
|
+
repeats. 3 chosen because a legit flow is typically write-once then
|
|
225
|
+
one corrective edit (2); a 3rd full rewrite of the same file in a
|
|
226
|
+
turn is the churn signal (the real incident rewrote package.json ~5×)."""
|
|
227
|
+
|
|
228
|
+
CHURN_COMMAND_FAIL_LIMIT = 3
|
|
229
|
+
"""How many times the SAME command (keyed by its first token, e.g.
|
|
230
|
+
`npm`) may FAIL in one turn before we nudge "read its error output
|
|
231
|
+
and fix the root cause before re-running." 3 (not 2) so running a
|
|
232
|
+
build/install command, fixing, and one more failure while iterating
|
|
233
|
+
is tolerated; the 3rd failure of the same command family means the
|
|
234
|
+
model is re-running without absorbing the error."""
|
|
235
|
+
|
|
236
|
+
CHURN_READONLY_STREAK_LIMIT = 6
|
|
237
|
+
"""Consecutive rounds of PURE read-only investigation (read_file /
|
|
238
|
+
grep / glob / list_files / web_*) with no mutating or server action
|
|
239
|
+
before we nudge "take a concrete action now." Tighter than the prior
|
|
240
|
+
generic streak of 10: a 10-round read-only run is already 5+ minutes
|
|
241
|
+
of the user staring at a frozen screen. 6 still allows reading the
|
|
242
|
+
handful of files needed to understand a redesign before committing,
|
|
243
|
+
but interrupts a genuine investigation spin sooner."""
|
|
244
|
+
|
|
245
|
+
CHURN_PLANNING_STREAK_LIMIT = 4
|
|
246
|
+
"""Consecutive rounds of PLANNING-WITHOUT-PROGRESS before we nudge
|
|
247
|
+
"you've planned enough; take a concrete action now."
|
|
248
|
+
|
|
249
|
+
A round counts toward this streak when ALL of:
|
|
250
|
+
• it changed NO new file this turn (changed_files count didn't grow),
|
|
251
|
+
• it ran NO build/test/verify command, and
|
|
252
|
+
• it produced thinking or narration content (i.e. the model was
|
|
253
|
+
reasoning/planning, not idle).
|
|
254
|
+
|
|
255
|
+
This catches the model that re-derives the SAME plan across many rounds
|
|
256
|
+
— lots of thinking, no concrete action — which the read-only-spin
|
|
257
|
+
signal misses because that streak resets on any round with zero tool
|
|
258
|
+
calls (a pure think-then-read-then-think alternation never accumulates
|
|
259
|
+
a pure read-only streak). Set to 4 (one higher than the read-only
|
|
260
|
+
limit's effective reach) so legitimate "read two files, think, read a
|
|
261
|
+
third, then edit" flows do NOT trip: as soon as a round changes a file
|
|
262
|
+
or runs a build, the streak resets to 0."""
|
|
263
|
+
|
|
264
|
+
CROSS_ROUND_REPEAT_LIMIT = 4
|
|
265
|
+
"""How many times the SAME (tool, canonical-args) call may run ACROSS the
|
|
266
|
+
turn before we nudge "you already have this result — stop repeating it."
|
|
267
|
+
|
|
268
|
+
The in-round breaker catches identical calls within ONE round; this catches
|
|
269
|
+
the cross-ROUND spin the logs show — read_file on the same path 53x over many
|
|
270
|
+
rounds, or a pkill->curl->read loop where each command succeeds (so the
|
|
271
|
+
command-FAILURE breaker never trips). Crucially this only NUDGES — it never
|
|
272
|
+
withholds the tool result (the 2026-04-29 read-dedup STUB starved legitimate
|
|
273
|
+
debug re-reads and hung a turn 17 min; we do not repeat that). A write/edit to
|
|
274
|
+
a path resets that path's read counts, so a legitimate read-after-edit isn't
|
|
275
|
+
counted. 4 tolerates a couple of genuine re-looks before flagging a true loop."""
|