code-context-control 2.54.0__py3-none-any.whl → 2.55.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
cli/c3.py CHANGED
@@ -92,7 +92,7 @@ console = Console() if HAS_RICH else None
92
92
  # Config
93
93
  CONFIG_DIR = ".c3"
94
94
  CONFIG_FILE = ".c3/config.json"
95
- __version__ = "2.54.0"
95
+ __version__ = "2.55.0"
96
96
 
97
97
 
98
98
  def _command_deps() -> CommandDeps:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: code-context-control
3
- Version: 2.54.0
3
+ Version: 2.55.0
4
4
  Summary: Local code-intelligence layer for AI coding tools (Claude Code, Codex, Copilot, Cursor, Antigravity). Retrieve less, read less, edit safer — and version the configs that shape your agent (CLAUDE.md, skills, hooks, MCP).
5
5
  Author-email: Dimitri Tselenchuk <dtselenc@gmail.com>
6
6
  License-Expression: Apache-2.0
@@ -1,6 +1,6 @@
1
1
  cli/__init__.py,sha256=ec66drCZGNMRU4V6ov0zVhYZph1us12Vn8OvG_LJyRY,22
2
2
  cli/_hook_utils.py,sha256=tJ4ZmXD5Y-JjFWb0uoR7ycYklYFUMnnjF2G7BO-wyBI,12600
3
- cli/c3.py,sha256=jTV4sTDaPkooaJtlwaQOAKYXY4Q6C5RfyioKX35vfTM,309220
3
+ cli/c3.py,sha256=vH0SuIP9w6imPbFVi8y-veKrAKfNYrYC0KYiA8ydb54,309220
4
4
  cli/docs.html,sha256=bNymSz7LatnWHjxSXL92QSUtGvB3jBzNHbOV-bRpAOo,142507
5
5
  cli/edits.html,sha256=UjAhoCmBmQ89cklGvJqzC6eyNP2tc8H6T-e01DVkLvE,43418
6
6
  cli/hook_artifact.py,sha256=Se1CNBfoBFyvJQlRmYdNtdRXIkQp9zaF0O6ndRMo7ts,2198
@@ -95,7 +95,7 @@ cli/ui/components/sessions.js,sha256=FIKtil76B8tCkAmcFV7hlj6GQ_DCJK2jCzvEmdK7NBE
95
95
  cli/ui/components/settings.js,sha256=ATbAjBlVIwCNpxq7s191b49a_INQV38iwmySqtJLYwY,79066
96
96
  cli/ui/components/sidebar.js,sha256=cAY_jwYB-o1X_wWn__VXlG4IegVObuE3NmVsuFWqxtg,7417
97
97
  cli/ui/components/tasks.js,sha256=vyKQ3uwoppMwvdEaHlhWXW4oWcAisx4NveqzMhsYqHo,38438
98
- code_context_control-2.54.0.dist-info/licenses/LICENSE,sha256=l8Kh5QCNWNvR6kIt8L0BUZvc2LAFiHv2c-FnsGnUZf4,11301
98
+ code_context_control-2.55.0.dist-info/licenses/LICENSE,sha256=l8Kh5QCNWNvR6kIt8L0BUZvc2LAFiHv2c-FnsGnUZf4,11301
99
99
  core/__init__.py,sha256=TSDCEcM4V7gcZVM3w2ykJaqEUch4Dkon-rivV17T73s,2501
100
100
  core/config.py,sha256=NUU1DuespWRotlJsv59CNnLiWTR4Bc7ysmD1UFA76RE,16429
101
101
  core/ide.py,sha256=V6VVMVsFdmmcsMyxikjQp7z9xa42CWiHKS-ya-MAcG4,6172
@@ -110,7 +110,7 @@ oracle/services/__init__.py,sha256=Nb4POd1_YIwLVYsGfr-DiK-iKTelkU0fh9m7wjeLQHA,2
110
110
  oracle/services/activity_reporter.py,sha256=ZAxNFyUy7QRP28anQLRbKEwm5SGG97KIQYFQDLjYqlM,11729
111
111
  oracle/services/api_auth.py,sha256=1PW3pG--1DJb_F6qMhP3gBTYHxxPE2bsHmVmIhC81Y8,3566
112
112
  oracle/services/c3_bridge.py,sha256=RqSw1tOoUDXA21fKDHmZ7s7qrIj3jiDYq8j8OvbxshA,17919
113
- oracle/services/chat_engine.py,sha256=y_1kQL4jvUw86R1lYWcvDGVyPk-YhG2Uj6Yj26bBrrM,66119
113
+ oracle/services/chat_engine.py,sha256=zvjyFsXGwqRHsvggBwQ6L0hcOovQe0AKyOy2ZdOU0No,55974
114
114
  oracle/services/chat_store.py,sha256=mizKwDyFGESKh63V-XgyNd3jQVrN2JCQc-7u8z3_1F4,6252
115
115
  oracle/services/cross_memory.py,sha256=F8JeYkEFSwQL3iGS9KV1onaMSFj3opr6f3nCxwvznV0,5484
116
116
  oracle/services/federated_graph.py,sha256=I9FWedcht6-FU2hoIlMHlXyjpwrspKueIN-jEYek6vg,19947
@@ -152,7 +152,7 @@ services/benchmark_dashboard.py,sha256=iR-DnqnoKbqHMJ4d-ZkIvJBYfzwTa7r-jzO6j2BYD
152
152
  services/bitbucket_client.py,sha256=JByovvtVZ6F4NcU611KVuTgKlT1rIsX2A2VlBDfvRV8,17509
153
153
  services/bitbucket_credentials.py,sha256=2qLA9pQMol4y95y4DJMNBsBBPUsJQCKbLFo2iiCnfvI,7364
154
154
  services/circuit_breaker.py,sha256=ES6PpjL40gfDhtK88GeS6GFzsz5EOEEkZsB-DzX_tbw,3179
155
- services/claude_md.py,sha256=vI2at-7DEsnRXYcSaE88uNTxCSxzhy8M77AM20VlYRc,42643
155
+ services/claude_md.py,sha256=-OwAv8cQRY8_i6tSsK5X-9rayZSJN_rrZMMjqJf3oEM,42728
156
156
  services/compressor.py,sha256=AK31TqHFpkOD_Lqbp9aEOkOx7zENUwpsQHQ9B2imEHo,33081
157
157
  services/context_snapshot.py,sha256=2f_bxY3xX4ZC653ncYno8YSeSTFSavwBjrorytHA0Ns,16879
158
158
  services/conversation_store.py,sha256=-E2H6f9pMSOHiBjsq7tFUjrFAwWYExf7LJkibi5MRMA,35650
@@ -189,9 +189,9 @@ services/retention.py,sha256=I2_RV233kWBBXox5rc_w-1h1aPua93o9huuPf-pJVuE,18629
189
189
  services/retrieval_broker.py,sha256=9X67VZ_6AkbAzopHuuMFKmP4CGZLnW576kjSKMenBnw,5261
190
190
  services/router.py,sha256=Cz10nx2fKTbaGn14mSBePWIDrw5rdcs_1JFYXeik084,15626
191
191
  services/runtime.py,sha256=b79gN0dFlzBO4rPaUMWier8FZFamMZlwIM3aywLmbB4,12840
192
- services/scanner.py,sha256=5rMwxAO8Qchz3-GtpZTNwCQDiJIyhAMSXdwNPmTaq-U,4701
192
+ services/scanner.py,sha256=1wnsH6WAoarf4L0MskRb8CBfjaF0pPw6ZYDX_4y2q3A,5985
193
193
  services/session_benchmark.py,sha256=GX3H8OwKC0X9Kk5U-vfY6Y7qq9Qaz2HwadNA7mjO8RY,105047
194
- services/session_manager.py,sha256=qUo7ool6dDWXfVgFpokIRcWqkWh10OzMcp7aZTdmTHY,51632
194
+ services/session_manager.py,sha256=cVsL45nfkAZ5P-elwLMBoesQ9QGwZsCl4ND3tQJXMMI,51838
195
195
  services/session_preloader.py,sha256=DsTAXMKVtrX9yu1sEFojYDi9-jkSAj1Ylt9JTy57Dow,9883
196
196
  services/subprojects.py,sha256=KNA-qGVPsd7Tg92Pt_kLWMLyER2YHmdQWO44sOelTIg,25123
197
197
  services/task_store.py,sha256=IHNeLipdyWtbWYu2-3ItIDeSDI9WoMKj6YwPPc1TgW0,46487
@@ -226,8 +226,8 @@ tui/screens/search_view.py,sha256=MMHjVdlk3HZSuDBSvq8IGrqv_Mh5Us6YqXQ80bcWSMk,19
226
226
  tui/screens/session_view.py,sha256=eZ1eDwHTvPOck1wCCviixtOaCxIkBT_95ytNNNriGNA,5991
227
227
  tui/screens/stats.py,sha256=p81PjzdaIv7hllb8f45-rlVe4lJZwSdIMqu7e86_u5s,6223
228
228
  tui/screens/ui_view.py,sha256=1QJCgLh2YfgWIpvzRG1KOGXYEaOYX6ojN61Azjf2oX0,2125
229
- code_context_control-2.54.0.dist-info/METADATA,sha256=jtQEMTmI5Z0UW8uxs6ImKZ8KKA9Fic3IrIgALMBz8zM,24759
230
- code_context_control-2.54.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
231
- code_context_control-2.54.0.dist-info/entry_points.txt,sha256=7kX_WUsDCF2hbXzvbNyscyaBb9AeA-DJY5v_5hN0DlU,93
232
- code_context_control-2.54.0.dist-info/top_level.txt,sha256=wRt41zBybVF3qAiNXHz9BURbkKvUvfhmWWtKMhaw6eE,29
233
- code_context_control-2.54.0.dist-info/RECORD,,
229
+ code_context_control-2.55.0.dist-info/METADATA,sha256=t426vjB6vV4mHLBpU55F_4GZtuKXvIRZ-yLYqTsVGc8,24759
230
+ code_context_control-2.55.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
231
+ code_context_control-2.55.0.dist-info/entry_points.txt,sha256=7kX_WUsDCF2hbXzvbNyscyaBb9AeA-DJY5v_5hN0DlU,93
232
+ code_context_control-2.55.0.dist-info/top_level.txt,sha256=wRt41zBybVF3qAiNXHz9BURbkKvUvfhmWWtKMhaw6eE,29
233
+ code_context_control-2.55.0.dist-info/RECORD,,
@@ -3,7 +3,6 @@
3
3
  import concurrent.futures
4
4
  import json
5
5
  import queue
6
- import re
7
6
  import threading
8
7
  import time
9
8
  import urllib.error
@@ -31,102 +30,6 @@ from oracle.services.memory_writer import MemoryWriter
31
30
  from oracle.services.project_scanner import ProjectScanner
32
31
  from services.ollama_bridge import OllamaBridge
33
32
 
34
- # ── Tool definitions (embedded in system prompt) ──────────
35
-
36
- _TOOL_DEFS = """
37
- Available tools — call ONE at a time by outputting exactly:
38
- <tool_call>{"name": "tool_name", "args": {…}}</tool_call>
39
-
40
- After you see the result, continue your response using the data.
41
- Do NOT call a tool if you can answer from context or prior results.
42
-
43
- Tools:
44
- 1. list_projects()
45
- Returns all registered C3 projects with fact counts and paths.
46
-
47
- 2. query_memory(project_path, query?, category?, limit=10)
48
- Search or list memory facts from a specific project.
49
- - project_path (required): full path to the project
50
- - query (optional): keyword filter on fact text
51
- - category (optional): filter by category
52
- - limit: max results (default 10)
53
-
54
- 3. search_facts(query, limit=20)
55
- Search facts across ALL projects. Returns matches with project source.
56
-
57
- 4. project_health(project_path)
58
- Run a health check on a project's memory. Returns status, issues, stats.
59
-
60
- 5. analyze_project(project_path)
61
- Deep LLM-powered analysis of a project's memory patterns and themes.
62
-
63
- 6. cross_insights(project_path?)
64
- Get cross-project insights. If project_path given, filter to that project.
65
-
66
- 7. suggest_action(project_path, action, fact_ids, reason)
67
- Create a pending suggestion (merge_facts, archive_facts, or add_fact).
68
- - action: "merge_facts" | "archive_facts" | "add_fact"
69
- - fact_ids: list of fact IDs involved
70
- - reason: explanation string
71
-
72
- 8. read_graph(project_path)
73
- Get memory graph statistics for a project (nodes, edges, types).
74
-
75
- --- C3 Code Intelligence Tools (require project_path) ---
76
-
77
- 9. c3_search(project_path, query, action="code", top_k=3, max_tokens=1200)
78
- Code intelligence search within a project.
79
- action: code|exact|files|semantic|transcript
80
-
81
- 10. c3_read(project_path, file_path, symbols=null, lines=null)
82
- Read file contents with optional symbol or line-range extraction.
83
-
84
- 11. c3_edits(project_path, action="history", file="", limit=50, since="", tag="")
85
- Query the edit ledger: history, versions, stats.
86
-
87
- 12. c3_edits_cross(action="history", tag="", limit=20, scope="")
88
- Query edit ledgers across ALL projects. No project_path needed.
89
- scope: ""=all, "top"=top-level only, or a project name/path = it + its sub-projects.
90
-
91
- 13. c3_memory_query(project_path, action="query", query="", category="", top_k=10)
92
- Query project memory (read-only: recall, query, list, score, graph, trends).
93
-
94
- 14. c3_compress(project_path, file_path, mode="map")
95
- Token-efficient file summary. mode: map|dense_map|smart|diff|bug_scan
96
-
97
- 15. c3_validate(project_path, file_path)
98
- Syntax validation on a file.
99
-
100
- 16. c3_status(project_path, view="health", detailed=false)
101
- Project health/budget/sessions overview. view: budget|health|sessions
102
-
103
- 17. c3_search_cross(query, action="code", top_k=3, scope="")
104
- Search code across ALL projects. No project_path needed.
105
- scope: ""=all, "top"=top-level only, or a project name/path = it + its sub-projects.
106
-
107
- 18. delegate_task(agent_id, task, project_path="")
108
- Delegate a specific sub-task to a specialized agent.
109
- - agent_id: The ID of the active agent to use.
110
- - task: A detailed prompt explaining what the agent needs to do.
111
- - project_path: required when the agent's backend is codex/gemini/claude/auto
112
- (CLI backends run inside a registered project's workspace, read-only).
113
-
114
- 19. activity_report(date?, since?, until?, project_path?, narrate=false)
115
- Cross-project daily activity digest: sessions, tool calls, edits, git
116
- mutations, and token/cost across ALL projects (or one, via project_path).
117
- Defaults to today (UTC). narrate=true adds a prose summary.
118
-
119
- 20. c3_project(action="list", project="", ...)
120
- Cross-project ops by registered project NAME or path. Read-only actions:
121
- list|info|subprojects|search|read|compress|status|memory|impact|edits|validate.
122
- Extra args mirror the underlying tool (query, file_path, mode, view,
123
- mem_action, edits_action, top_k, limit, target).
124
-
125
- 21. c3_artifacts(project_path, action="list", artifact="", version=0, against=0)
126
- Agent-config artifact tracking (read-only): list|history|show|diff|status.
127
- Version history for CLAUDE.md/settings/MCP configs/skills in a project.
128
- """
129
-
130
33
  _SYSTEM_BASE = """You are Oracle, an AI assistant specializing in cross-project code intelligence and memory analysis.
131
34
  You have access to memory facts, project health data, cross-project insights,
132
35
  AND full C3 code intelligence (code search, file reading, edit history, validation)
@@ -148,23 +51,23 @@ _DEPTH_INSTRUCTIONS = {
148
51
  "deep": "\nProvide thorough, detailed analysis with examples, data, and recommendations. Use multiple tool calls to gather comprehensive data when needed.\n",
149
52
  }
150
53
 
54
+ # Tools are declared via Ollama's tools= API, so the prompt carries only
55
+ # behavioral guidance — no syntax, no catalog.
151
56
  _SYSTEM_RULES = """
152
57
  Important rules:
153
- - You can call multiple tools at once by outputting multiple `<tool_call>...</tool_call>` blocks.
154
- - Call tools in parallel when tasks are independent.
155
- - If the user's question can be answered from conversation context, do NOT call a tool.
58
+ - Use the provided tools when the user asks about specific data you don't have in context; call independent tools in parallel in one turn.
59
+ - Do NOT call a tool if the question can be answered from conversation context or prior results.
156
60
  - After receiving tool results, synthesize them into a clear, helpful answer.
61
+ - Use list_projects first to discover project paths before calling project-specific tools.
157
62
  - Format your answers with markdown for readability.
158
63
  """
159
64
 
160
- # Native tool-calling mode: tools are declared via Ollama's tools= API, so the
161
- # prompt carries only behavioral guidance no <tool_call> syntax, no catalog.
162
- _SYSTEM_RULES_NATIVE = """
65
+ # Appended when the model rejected native tool calling: the round is rerun
66
+ # without tools, and the prompt must not promise capabilities it lacks.
67
+ _NO_TOOLS_RULES = """
163
68
  Important rules:
164
- - Use the provided tools when the user asks about specific data you don't have in context; call independent tools in parallel in one turn.
165
- - Do NOT call a tool if the question can be answered from conversation context or prior results.
166
- - After receiving tool results, synthesize them into a clear, helpful answer.
167
- - Use list_projects first to discover project paths before calling project-specific tools.
69
+ - No tools are available in this session. Answer from conversation context only.
70
+ - If the question requires live project data you do not have, say so plainly.
168
71
  - Format your answers with markdown for readability.
169
72
  """
170
73
 
@@ -181,31 +84,17 @@ COMMANDS = {
181
84
  "team": {"args": "", "desc": "Show active agents and their specializations"},
182
85
  }
183
86
 
184
- _TOOL_CALL_RE = re.compile(r"<tool_call>\s*(\{.*?\})\s*</tool_call>", re.DOTALL)
185
87
  _MAX_TOOL_ROUNDS = 8
186
88
  # Rounds = LLM calls. One tool use needs 2 rounds (call + response synthesis).
187
89
  _DEPTH_MAX_ROUNDS = {"brief": 2, "normal": 4, "deep": 8}
188
90
  _MAX_HISTORY_MESSAGES = 40
189
91
  _MAX_TOOL_RESULT_CHARS = 3000
190
92
  _VISIBLE_RETRY_PROMPT = (
191
- "Your previous response contained only hidden reasoning and no user-visible "
192
- "assistant content. Now provide the visible response. If you need a tool, "
193
- "output exactly one <tool_call>{...}</tool_call> block in assistant content; "
194
- "otherwise answer the user directly."
195
- )
196
- _VISIBLE_RETRY_PROMPT_NATIVE = (
197
93
  "Your previous response contained only hidden reasoning and no user-visible "
198
94
  "assistant content. Now provide the visible response. If you need a tool, "
199
95
  "use a native tool call; otherwise answer the user directly."
200
96
  )
201
97
 
202
- _TC_OPEN = "<tool_call>"
203
- _TC_CLOSE = "</tool_call>"
204
- # If the pre-strip visible answer in round 0 is at least this many chars, we
205
- # treat any trailing <tool_call> as speculative and do NOT regenerate. Prevents
206
- # the "correct answer then wrong answer on next round" failure mode.
207
- _TRUST_ANSWER_MIN_CHARS = 120
208
-
209
98
 
210
99
  _NATIVE_TOOLS_CACHE: list[dict] | None = None
211
100
 
@@ -231,76 +120,36 @@ def _native_tool_defs() -> list[dict]:
231
120
  return _NATIVE_TOOLS_CACHE
232
121
 
233
122
 
234
- def _classify_no_tools(exc: Exception) -> tuple[bool, bool]:
235
- """Classify a streaming error raised while native tools were attached.
123
+ def _classify_stream_400(exc: Exception, tools_attached: bool,
124
+ think_enabled: bool) -> tuple[bool, bool, bool]:
125
+ """Classify a streaming HTTP 400 into retriable capability downgrades.
236
126
 
237
- Returns ``(fallback_to_legacy, negative_cache_model)``. Any HTTP 400
238
- triggers fallback (retrying without tools is one cheap request); the
239
- model's capability is negative-cached only when the server names tools as
240
- the problem, so an unrelated 400 doesn't permanently demote the model.
127
+ Returns ``(drop_tools, drop_think, cache_no_tools)``. Ollama names the
128
+ offending capability in the error body ("does not support thinking" /
129
+ "... tools"). A 400 naming neither is still treated as a tools rejection
130
+ when tools were attached (retrying without them is one cheap request),
131
+ but is never negative-cached — so an unrelated 400 doesn't permanently
132
+ demote the model.
241
133
  """
242
134
  if not isinstance(exc, urllib.error.HTTPError) or exc.code != 400:
243
- return False, False
135
+ return False, False, False
244
136
  try:
245
- body = exc.read().decode("utf-8", "replace")
137
+ body = (exc.read().decode("utf-8", "replace") or "").lower()
246
138
  except Exception:
247
139
  body = ""
248
- return True, "tool" in body.lower()
249
-
140
+ if think_enabled and "think" in body:
141
+ return False, True, False
142
+ if tools_attached:
143
+ return True, False, "tool" in body
144
+ return False, False, False
250
145
 
251
- class _ToolCallStripper:
252
- """Streaming-friendly stripper for <tool_call>...</tool_call> blocks.
253
-
254
- Buffers chunks so partial open/close tags that straddle chunk boundaries
255
- are never leaked to the UI. feed() returns only visible (stripped) text;
256
- flush() emits any trailing buffer that is not inside a tool_call.
257
- """
258
146
 
259
- def __init__(self) -> None:
260
- self._buf = ""
261
- self._in_call = False
262
-
263
- def feed(self, chunk: str) -> str:
264
- self._buf += chunk
265
- out: list[str] = []
266
- while True:
267
- if self._in_call:
268
- end = self._buf.find(_TC_CLOSE)
269
- if end == -1:
270
- return "".join(out)
271
- self._buf = self._buf[end + len(_TC_CLOSE):]
272
- self._in_call = False
273
- continue
274
- start = self._buf.find(_TC_OPEN)
275
- if start == -1:
276
- hold = 0
277
- for i in range(1, len(_TC_OPEN)):
278
- if self._buf.endswith(_TC_OPEN[:i]):
279
- hold = i
280
- if hold:
281
- out.append(self._buf[:-hold])
282
- self._buf = self._buf[-hold:]
283
- else:
284
- out.append(self._buf)
285
- self._buf = ""
286
- return "".join(out)
287
- out.append(self._buf[:start])
288
- self._buf = self._buf[start + len(_TC_OPEN):]
289
- self._in_call = True
290
-
291
- def flush(self) -> str:
292
- if self._in_call:
293
- return ""
294
- tail = self._buf
295
- self._buf = ""
296
- return tail
297
-
298
-
299
- def _build_system_prompt(state: dict, native: bool = False) -> str:
147
+ def _build_system_prompt(state: dict, tools_enabled: bool = True) -> str:
300
148
  """Build system prompt dynamically based on conversation state.
301
149
 
302
- ``native=True`` omits the <tool_call> catalog/syntax (tools are declared
303
- via Ollama's native API instead), keeping only behavioral guidance.
150
+ Tools are declared via Ollama's native API; the prompt carries only
151
+ behavioral guidance. ``tools_enabled=False`` (model rejected native tool
152
+ calling) swaps in rules that promise no tool capabilities.
304
153
  """
305
154
  parts = [_SYSTEM_BASE]
306
155
 
@@ -326,11 +175,7 @@ def _build_system_prompt(state: dict, native: bool = False) -> str:
326
175
  "or 'my project', they mean one of the focused projects.\n"
327
176
  )
328
177
 
329
- if native:
330
- parts.append(_SYSTEM_RULES_NATIVE)
331
- else:
332
- parts.append(_TOOL_DEFS)
333
- parts.append(_SYSTEM_RULES)
178
+ parts.append(_SYSTEM_RULES if tools_enabled else _NO_TOOLS_RULES)
334
179
  return "".join(parts)
335
180
 
336
181
 
@@ -364,21 +209,19 @@ class ChatEngine:
364
209
  # ── Shared streaming drain ────────────────────────────
365
210
 
366
211
  def _drain_stream(self, messages: list[dict], model: str,
367
- stripper: "_ToolCallStripper | None" = None,
368
212
  tools: list[dict] | None = None, think: bool | None = True):
369
213
  """Stream one LLM round, yielding SSE event dicts as they arrive.
370
214
 
371
215
  The single streamer behind the main chat loop, the visible-response
372
216
  retry, and the delegate sub-agent loop. Returns (via StopIteration
373
- value — use ``yield from``) a summary dict: ``text`` (raw, including
374
- any <tool_call> markup in legacy mode), ``thinking``, ``tool_calls``
375
- (native protocol), ``stats``, ``chunks``, ``visible_chars``.
376
-
377
- In legacy mode pass a ``_ToolCallStripper`` so <tool_call> blocks are
378
- withheld from the emitted text events; native mode emits text as-is
379
- (the content channel is clean) and collects structured tool calls
380
- without emitting SSE for them — the executor emits the canonical
381
- ``tool_call`` event with its tool_id.
217
+ value — use ``yield from``) a summary dict: ``text``, ``thinking``,
218
+ ``tool_calls`` (native protocol), ``stats``, ``chunks``,
219
+ ``visible_chars``.
220
+
221
+ Text is emitted as-is (the content channel is clean under native
222
+ tool calling); structured tool calls are collected without emitting
223
+ SSE for them the executor emits the canonical ``tool_call`` event
224
+ with its tool_id.
382
225
  """
383
226
  full_text = ""
384
227
  thinking_text = ""
@@ -402,15 +245,9 @@ class ChatEngine:
402
245
  tool_calls.append(chunk)
403
246
  else:
404
247
  full_text += chunk
405
- visible = stripper.feed(chunk) if stripper is not None else chunk
406
- if visible:
407
- visible_chars += len(visible)
408
- yield {"type": "text", "content": visible}
409
- if stripper is not None:
410
- tail = stripper.flush()
411
- if tail:
412
- visible_chars += len(tail)
413
- yield {"type": "text", "content": tail}
248
+ if chunk:
249
+ visible_chars += len(chunk)
250
+ yield {"type": "text", "content": chunk}
414
251
  return {"text": full_text, "thinking": thinking_text,
415
252
  "tool_calls": tool_calls, "stats": stats, "chunks": chunks,
416
253
  "visible_chars": visible_chars}
@@ -454,15 +291,18 @@ class ChatEngine:
454
291
  # Save user message
455
292
  self.store.append_message(conv_id, {"role": "user", "content": user_message})
456
293
 
457
- # Native tool-calling mode: capability probe returns True/False, or
458
- # None (unknown) attempt native and fall back on live rejection.
294
+ # Capability probe returns True/False, or None (unknown) → attempt
295
+ # native tools and rerun without them on live rejection. Thinking is
296
+ # likewise attempted and dropped on a live "does not support" 400.
459
297
  probe = self.bridge.supports_tools(use_model) if self.bridge else False
460
- use_native = probe is not False
461
- tools = _native_tool_defs() if use_native else None
298
+ tools_enabled = probe is not False
299
+ tools = _native_tool_defs() if tools_enabled else None
300
+ think_enabled = True
462
301
 
463
302
  # Build messages for LLM
464
303
  history = self.store.get_conversation(conv_id)
465
- llm_messages = self._build_llm_messages(history, state, native=use_native)
304
+ llm_messages = self._build_llm_messages(history, state,
305
+ tools_enabled=tools_enabled)
466
306
  context_msgs = len(llm_messages) - 1 # exclude system prompt
467
307
  yield {"type": "status", "message": "Context ready", "detail": f"{context_msgs} messages in context"}
468
308
 
@@ -476,38 +316,43 @@ class ChatEngine:
476
316
  yield {"type": "status", "message": f"Streaming from {use_model}", "detail": round_label or "Generating response"}
477
317
 
478
318
  stream_start = time.time()
479
- try:
480
- result = yield from self._drain_stream(
481
- llm_messages, use_model,
482
- stripper=None if use_native else _ToolCallStripper(),
483
- tools=tools,
484
- )
485
- except Exception as e:
486
- fallback_ok, cache_neg = (
487
- _classify_no_tools(e) if (use_native and tools) else (False, False)
488
- )
489
- if not fallback_ok:
490
- yield {"type": "error", "message": f"LLM error: {e}"}
491
- break
492
- # Mid-turn fallback: the model rejected native tools —
493
- # demote to the legacy text protocol and rerun this round.
494
- if cache_neg and self.bridge:
495
- self.bridge.set_tools_support(use_model, False)
496
- use_native = False
497
- tools = None
498
- llm_messages[0] = {
499
- "role": "system",
500
- "content": _build_system_prompt(state, native=False),
501
- }
502
- yield {"type": "status", "message": "Falling back to text tool protocol",
503
- "detail": f"{use_model} rejected native tool calling"}
319
+ # Up to two capability downgrades (thinking, then tools) —
320
+ # each dropped at most once, so this cannot loop.
321
+ result = None
322
+ for _attempt in range(3):
504
323
  try:
505
324
  result = yield from self._drain_stream(
506
- llm_messages, use_model, stripper=_ToolCallStripper(),
325
+ llm_messages, use_model, tools=tools,
326
+ think=think_enabled,
507
327
  )
508
- except Exception as e2:
509
- yield {"type": "error", "message": f"LLM error: {e2}"}
510
328
  break
329
+ except Exception as e:
330
+ drop_tools, drop_think, cache_neg = _classify_stream_400(
331
+ e, tools_attached=bool(tools),
332
+ think_enabled=think_enabled)
333
+ if drop_think:
334
+ think_enabled = False
335
+ yield {"type": "status",
336
+ "message": "Continuing without thinking",
337
+ "detail": f"{use_model} does not support thinking"}
338
+ continue
339
+ if drop_tools:
340
+ if cache_neg and self.bridge:
341
+ self.bridge.set_tools_support(use_model, False)
342
+ tools_enabled = False
343
+ tools = None
344
+ llm_messages[0] = {
345
+ "role": "system",
346
+ "content": _build_system_prompt(state, tools_enabled=False),
347
+ }
348
+ yield {"type": "status",
349
+ "message": "Continuing without tools",
350
+ "detail": f"{use_model} rejected native tool calling"}
351
+ continue
352
+ yield {"type": "error", "message": f"LLM error: {e}"}
353
+ break
354
+ if result is None:
355
+ break
511
356
 
512
357
  full_text = result["text"]
513
358
  thinking_text = result["thinking"]
@@ -524,10 +369,8 @@ class ChatEngine:
524
369
  "message": "Retrying visible response",
525
370
  "detail": "Model returned thinking without assistant content",
526
371
  }
527
- retry_prompt = (
528
- _VISIBLE_RETRY_PROMPT_NATIVE if use_native else _VISIBLE_RETRY_PROMPT
529
- )
530
- retry_messages = llm_messages + [{"role": "user", "content": retry_prompt}]
372
+ retry_messages = llm_messages + [
373
+ {"role": "user", "content": _VISIBLE_RETRY_PROMPT}]
531
374
  try:
532
375
  retry = yield from self._drain_stream(
533
376
  retry_messages, use_model, tools=tools, think=False,
@@ -556,61 +399,30 @@ class ChatEngine:
556
399
  total_tokens += chunk_count
557
400
  yield {"type": "status", "message": "Response received", "detail": f"{chunk_count} chunks in {stream_ms}ms"}
558
401
 
559
- # Determine tool calls for this round.
560
- if use_native:
561
- # Structured calls from the native protocol; content is clean.
562
- valid_calls = [
563
- {"name": c.get("name", "unknown"), "args": c.get("arguments") or {}}
564
- for c in native_calls if c.get("name")
565
- ]
566
- visible_text = full_text.strip()
567
- final_text = visible_text
568
- else:
569
- tool_matches = list(_TOOL_CALL_RE.finditer(full_text))
570
- visible_text = _TOOL_CALL_RE.sub("", full_text).strip()
571
- final_text = visible_text or full_text
572
- valid_calls = []
573
- for match in tool_matches:
574
- try:
575
- valid_calls.append(json.loads(match.group(1)))
576
- except json.JSONDecodeError:
577
- continue
402
+ # Structured calls from the native protocol; content is clean.
403
+ valid_calls = [
404
+ {"name": c.get("name", "unknown"), "args": c.get("arguments") or {}}
405
+ for c in native_calls if c.get("name")
406
+ ]
407
+ final_text = full_text.strip()
578
408
 
579
409
  if not valid_calls:
580
410
  # No tool call — final answer
581
411
  round_messages.append({"role": "assistant", "content": final_text})
582
412
  break
583
413
 
584
- # Legacy-only heuristic: if the model already produced a
585
- # substantive answer alongside speculative <tool_call> blocks,
586
- # trust the answer and stop — regenerating with tool results
587
- # usually corrupts it. Native mode needs no such guard: models
588
- # emitting structured calls expect their results.
589
- if not use_native and len(visible_text) >= _TRUST_ANSWER_MIN_CHARS:
590
- round_messages.append({"role": "assistant", "content": visible_text})
591
- yield {
592
- "type": "status",
593
- "message": "Answer finalized",
594
- "detail": "Speculative tool calls skipped — answer already provided",
595
- }
596
- break
597
-
598
- # Record the assistant response before tool results (stripped so
599
- # the next LLM round sees clean context without <tool_call> noise).
414
+ # Record the assistant response before tool results, and echo
415
+ # the structured calls back so the model can pair the
416
+ # role:tool results that follow.
600
417
  round_messages.append({"role": "assistant", "content": final_text})
601
- if use_native:
602
- # Echo the structured calls back so the model can pair the
603
- # role:tool results that follow.
604
- llm_messages.append({
605
- "role": "assistant",
606
- "content": full_text,
607
- "tool_calls": [
608
- {"function": {"name": c["name"], "arguments": c.get("args") or {}}}
609
- for c in valid_calls
610
- ],
611
- })
612
- else:
613
- llm_messages.append({"role": "assistant", "content": final_text})
418
+ llm_messages.append({
419
+ "role": "assistant",
420
+ "content": full_text,
421
+ "tool_calls": [
422
+ {"function": {"name": c["name"], "arguments": c.get("args") or {}}}
423
+ for c in valid_calls
424
+ ],
425
+ })
614
426
 
615
427
  # Execute all tools in parallel
616
428
  tool_calls_count += len(valid_calls)
@@ -702,18 +514,11 @@ class ChatEngine:
702
514
  "tool_name": tool_name, "tool_id": tool_id,
703
515
  })
704
516
 
705
- # Extend next LLM context: native protocol feeds a
706
- # role:tool message; legacy re-injects as user text.
707
- if use_native:
708
- llm_messages.append({
709
- "role": "tool", "tool_name": tool_name,
710
- "content": truncated,
711
- })
712
- else:
713
- llm_messages.append({
714
- "role": "user",
715
- "content": f"<tool_result name=\"{tool_name}\">\n{truncated}\n</tool_result>",
716
- })
517
+ # Extend next LLM context with a role:tool message.
518
+ llm_messages.append({
519
+ "role": "tool", "tool_name": tool_name,
520
+ "content": truncated,
521
+ })
717
522
 
718
523
  # Final drain after all tools complete
719
524
  while True:
@@ -723,21 +528,6 @@ class ChatEngine:
723
528
  break
724
529
  yield ev
725
530
 
726
- if not use_native:
727
- # Legacy models tend to restart their answer after inlined
728
- # tool results; the nudge counters that. Native mode must
729
- # NOT inject a user message between tool_calls and their
730
- # role:tool results — the protocol handles continuation.
731
- llm_messages.append({
732
- "role": "user",
733
- "content": (
734
- "The tool results above contain the data you requested. "
735
- "Now write your final answer to the user based on those results. "
736
- "Do NOT output any <tool_call> blocks. Do NOT restart or repeat "
737
- "your previous message — the user has already seen it. "
738
- "Write only the remaining, finalized answer."
739
- ),
740
- })
741
531
  yield {"type": "status", "message": "Finalizing answer", "detail": f"After {len(valid_calls)} parallel results"}
742
532
  else:
743
533
  # Exhausted rounds without a final answer — synthesize one
@@ -806,14 +596,16 @@ class ChatEngine:
806
596
  # ── LLM message building ─────────────────────────────
807
597
 
808
598
  def _build_llm_messages(self, history: list[dict], state: dict | None = None,
809
- native: bool = False) -> list[dict]:
599
+ tools_enabled: bool = True) -> list[dict]:
810
600
  """Convert stored history to Ollama chat messages with sliding window.
811
601
 
812
602
  Historical tool results are re-injected as ``<tool_result>`` user
813
- messages in BOTH modes: safe for all models, and avoids orphaned
814
- ``role:tool`` messages (no paired tool_calls) on conversation reload.
603
+ messages deliberately NOT ``role:tool``: on conversation reload
604
+ there is no paired ``tool_calls`` assistant message, and orphaned
605
+ role:tool entries confuse several models. This is history
606
+ serialization, not the retired legacy tool-call protocol.
815
607
  """
816
- system_prompt = _build_system_prompt(state or {}, native=native)
608
+ system_prompt = _build_system_prompt(state or {}, tools_enabled=tools_enabled)
817
609
  messages = [{"role": "system", "content": system_prompt}]
818
610
 
819
611
  # Take last N messages, skip tool_call/tool_result (they were inlined)
@@ -1000,7 +792,15 @@ class ChatEngine:
1000
792
  return {"ok": True, "command": "help", "message": "\n".join(lines)}
1001
793
 
1002
794
  def _cmd_tools(self) -> dict:
1003
- return {"ok": True, "command": "tools", "message": _TOOL_DEFS.strip()}
795
+ # Listed from TOOL_SPECS (single source of truth) — the legacy
796
+ # prompt catalog this used to echo was removed with the <tool_call>
797
+ # text protocol.
798
+ from oracle.services.tool_registry import TOOL_SPECS
799
+ lines = ["Available tools:"]
800
+ for spec in TOOL_SPECS:
801
+ desc = (spec.get("description") or "").strip().split("\n")[0]
802
+ lines.append(f"- **{spec['name']}** — {desc}")
803
+ return {"ok": True, "command": "tools", "message": "\n".join(lines)}
1004
804
 
1005
805
  def _cmd_team(self) -> dict:
1006
806
  cfg = load_config()
@@ -1247,17 +1047,18 @@ class ChatEngine:
1247
1047
  project_path, _emit)
1248
1048
 
1249
1049
  model = agent.get("model") or self.bridge.model
1250
- # Sub-agents pick their tool protocol from their OWN model.
1050
+ # Sub-agents probe their OWN model for native tool support.
1251
1051
  probe = self.bridge.supports_tools(model) if self.bridge else False
1252
- use_native = probe is not False
1253
- tools = _native_tool_defs() if use_native else None
1052
+ tools_enabled = probe is not False
1053
+ tools = _native_tool_defs() if tools_enabled else None
1054
+ think_enabled = True
1254
1055
 
1255
- def _sys_prompt(native: bool) -> str:
1256
- rules = _SYSTEM_RULES_NATIVE if native else f"{_TOOL_DEFS}\n{_SYSTEM_RULES}"
1056
+ def _sys_prompt(enabled: bool) -> str:
1057
+ rules = _SYSTEM_RULES if enabled else _NO_TOOLS_RULES
1257
1058
  return f"{agent.get('system_prompt', '')}\n\n{rules}"
1258
1059
 
1259
1060
  llm_messages = [
1260
- {"role": "system", "content": _sys_prompt(use_native)},
1061
+ {"role": "system", "content": _sys_prompt(tools_enabled)},
1261
1062
  {"role": "user", "content": task}
1262
1063
  ]
1263
1064
 
@@ -1272,7 +1073,8 @@ class ChatEngine:
1272
1073
  while rounds < max_rounds:
1273
1074
  rounds += 1
1274
1075
  _emit("agent_round", agent_id=agent_id, round=rounds)
1275
- gen = self._drain_stream(llm_messages, model, tools=tools)
1076
+ gen = self._drain_stream(llm_messages, model, tools=tools,
1077
+ think=think_enabled)
1276
1078
  result = None
1277
1079
  try:
1278
1080
  while True:
@@ -1287,18 +1089,22 @@ class ChatEngine:
1287
1089
  _emit("agent_thinking", content=ev["content"])
1288
1090
  # stats chunks are ignored for sub-agents
1289
1091
  except Exception as e:
1290
- if use_native and tools:
1291
- fallback_ok, cache_neg = _classify_no_tools(e)
1292
- if fallback_ok:
1293
- # Model rejected native tools — demote to the legacy
1294
- # text protocol and rerun this round.
1295
- if cache_neg and self.bridge:
1296
- self.bridge.set_tools_support(model, False)
1297
- use_native = False
1298
- tools = None
1299
- llm_messages[0] = {"role": "system", "content": _sys_prompt(False)}
1300
- rounds -= 1
1301
- continue
1092
+ drop_tools, drop_think, cache_neg = _classify_stream_400(
1093
+ e, tools_attached=bool(tools), think_enabled=think_enabled)
1094
+ if drop_think:
1095
+ think_enabled = False
1096
+ rounds -= 1
1097
+ continue
1098
+ if drop_tools:
1099
+ # Model rejected native tools — rerun this round without
1100
+ # them (the agent answers from the prompt).
1101
+ if cache_neg and self.bridge:
1102
+ self.bridge.set_tools_support(model, False)
1103
+ tools_enabled = False
1104
+ tools = None
1105
+ llm_messages[0] = {"role": "system", "content": _sys_prompt(False)}
1106
+ rounds -= 1
1107
+ continue
1302
1108
  _emit("agent_done", agent_id=agent_id, rounds=rounds, error=str(e))
1303
1109
  return {"error": f"Agent '{agent_id}' encountered LLM error: {e}"}
1304
1110
 
@@ -1309,33 +1115,15 @@ class ChatEngine:
1309
1115
  _emit("agent_done", agent_id=agent_id, rounds=rounds, error="empty_response")
1310
1116
  return {"error": f"Agent '{agent_id}' returned empty response."}
1311
1117
 
1312
- if use_native:
1313
- if not native_calls:
1314
- total_result_chars = len(full_text)
1315
- dur_ms = (time.perf_counter_ns() - agent_start_ns) // 1_000_000
1316
- _emit("agent_done", agent_id=agent_id, rounds=rounds,
1317
- result_chars=total_result_chars, duration_ms=dur_ms)
1318
- return {"agent": agent_id, "result": full_text}
1319
- # Sub-agents run one tool per round; take the first call.
1320
- first = native_calls[0]
1321
- call = {"name": first.get("name", "unknown"), "args": first.get("arguments") or {}}
1322
- else:
1323
- tool_match = _TOOL_CALL_RE.search(full_text)
1324
- if not tool_match:
1325
- total_result_chars = len(full_text)
1326
- dur_ms = (time.perf_counter_ns() - agent_start_ns) // 1_000_000
1327
- _emit("agent_done", agent_id=agent_id, rounds=rounds,
1328
- result_chars=total_result_chars, duration_ms=dur_ms)
1329
- return {"agent": agent_id, "result": full_text}
1330
-
1331
- try:
1332
- call = json.loads(tool_match.group(1))
1333
- except json.JSONDecodeError:
1334
- total_result_chars = len(full_text)
1335
- dur_ms = (time.perf_counter_ns() - agent_start_ns) // 1_000_000
1336
- _emit("agent_done", agent_id=agent_id, rounds=rounds,
1337
- result_chars=total_result_chars, duration_ms=dur_ms)
1338
- return {"agent": agent_id, "result": full_text}
1118
+ if not native_calls:
1119
+ total_result_chars = len(full_text)
1120
+ dur_ms = (time.perf_counter_ns() - agent_start_ns) // 1_000_000
1121
+ _emit("agent_done", agent_id=agent_id, rounds=rounds,
1122
+ result_chars=total_result_chars, duration_ms=dur_ms)
1123
+ return {"agent": agent_id, "result": full_text}
1124
+ # Sub-agents run one tool per round; take the first call.
1125
+ first = native_calls[0]
1126
+ call = {"name": first.get("name", "unknown"), "args": first.get("arguments") or {}}
1339
1127
 
1340
1128
  tool_name = call.get("name", "unknown")
1341
1129
  tool_args = call.get("args", {})
@@ -1358,18 +1146,11 @@ class ChatEngine:
1358
1146
  if len(result_str) > _MAX_TOOL_RESULT_CHARS:
1359
1147
  truncated += "... (truncated)"
1360
1148
 
1361
- if use_native:
1362
- llm_messages.append({
1363
- "role": "assistant", "content": full_text,
1364
- "tool_calls": [{"function": {"name": tool_name, "arguments": tool_args}}],
1365
- })
1366
- llm_messages.append({"role": "tool", "tool_name": tool_name, "content": truncated})
1367
- else:
1368
- llm_messages.append({"role": "assistant", "content": full_text})
1369
- llm_messages.append({
1370
- "role": "user",
1371
- "content": f"<tool_result name=\"{tool_name}\">\n{truncated}\n</tool_result>\nContinue your response using this data."
1372
- })
1149
+ llm_messages.append({
1150
+ "role": "assistant", "content": full_text,
1151
+ "tool_calls": [{"function": {"name": tool_name, "arguments": tool_args}}],
1152
+ })
1153
+ llm_messages.append({"role": "tool", "tool_name": tool_name, "content": truncated})
1373
1154
 
1374
1155
  dur_ms = (time.perf_counter_ns() - agent_start_ns) // 1_000_000
1375
1156
  _emit("agent_done", agent_id=agent_id, rounds=rounds,
services/claude_md.py CHANGED
@@ -819,12 +819,13 @@ class ClaudeMdManager:
819
819
  if dirname:
820
820
  mentioned_dirs.add(dirname)
821
821
 
822
- # Scan actual top-level dirs
823
- skip = {'node_modules', '.git', '__pycache__', '.c3', 'venv',
824
- 'env', '.venv', 'dist', 'build', '.next', '.cache', '.claude'}
822
+ # Scan actual top-level dirs with the same pruning rules as the
823
+ # generator, so gitignored dirs are never reported as "new".
824
+ from services.scanner import make_dir_pruner
825
+ pruned = make_dir_pruner(self.project_path, extra_skip=('.claude',))
825
826
  actual_dirs = set()
826
827
  for item in self.project_path.iterdir():
827
- if item.is_dir() and item.name not in skip and not item.name.startswith('.'):
828
+ if item.is_dir() and not pruned(item.name) and not item.name.startswith('.'):
828
829
  actual_dirs.add(item.name)
829
830
 
830
831
  # Compare (use base names only)
services/scanner.py CHANGED
@@ -11,6 +11,7 @@ os.walk with in-place dirnames pruning never descends into skipped
11
11
  directories, yields files in deterministic order, exits as soon as the
12
12
  caller stops consuming, and reports progress so long scans are visible.
13
13
  """
14
+ import fnmatch
14
15
  import os
15
16
  from pathlib import Path
16
17
  from typing import Callable, Iterator, Optional, Set, Tuple
@@ -29,30 +30,60 @@ SKIP_DIRS = {
29
30
  }
30
31
 
31
32
 
32
- def gitignore_dir_names(root) -> set:
33
- """Literal directory names from the root .gitignore (best-effort).
33
+ def gitignore_dir_patterns(root) -> tuple:
34
+ """(literal_names, glob_patterns) for directory pruning from .gitignore.
34
35
 
35
- Only unambiguous entries are used - a bare name or ``/name/`` with no
36
- wildcard, negation, or nested separator - so pruning can never be
37
- broader than the ignore file itself. Data/log directories are exactly
36
+ Single-segment entries only - a bare name, ``/name/``, or a simple glob
37
+ like ``*.egg-info/`` - never negations or nested paths, so pruning can
38
+ only be narrower than the ignore file itself. Both sets are matched
39
+ against directory basenames. Data/log/build directories are exactly
38
40
  what makes large-project scans hang, and they are almost always
39
- plain-name entries.
41
+ single-segment entries.
40
42
  """
41
- names = set()
43
+ names: set = set()
44
+ patterns: list = []
42
45
  try:
43
46
  text = (Path(root) / '.gitignore').read_text(errors='replace')
44
47
  except OSError:
45
- return names
48
+ return names, patterns
46
49
  for line in text.splitlines():
47
50
  line = line.strip()
48
51
  if not line or line.startswith('#') or line.startswith('!'):
49
52
  continue
50
- if any(ch in line for ch in '*?['):
51
- continue
52
53
  cleaned = line.strip('/')
53
- if cleaned and '/' not in cleaned and '\\' not in cleaned:
54
+ if not cleaned or '/' in cleaned or '\\' in cleaned:
55
+ continue
56
+ if any(ch in cleaned for ch in '*?['):
57
+ patterns.append(cleaned)
58
+ else:
54
59
  names.add(cleaned)
55
- return names
60
+ return names, patterns
61
+
62
+
63
+ def gitignore_dir_names(root) -> set:
64
+ """Literal directory names from the root .gitignore (best-effort)."""
65
+ return gitignore_dir_patterns(root)[0]
66
+
67
+
68
+ def make_dir_pruner(root, extra_skip=(), respect_gitignore: bool = True):
69
+ """Predicate ``dirname -> bool`` (True = prune this directory).
70
+
71
+ Combines SKIP_DIRS, ``extra_skip``, and the root .gitignore (literal
72
+ names plus single-segment globs like ``*.egg-info/``). Shared by the
73
+ index scanners and the project-tree doc generator so every surface
74
+ agrees on what a distributable project looks like.
75
+ """
76
+ skip = set(SKIP_DIRS) | set(extra_skip)
77
+ patterns: list = []
78
+ if respect_gitignore:
79
+ names, patterns = gitignore_dir_patterns(root)
80
+ skip |= names
81
+
82
+ def pruned(dirname: str) -> bool:
83
+ return (dirname in skip
84
+ or any(fnmatch.fnmatch(dirname, p) for p in patterns))
85
+
86
+ return pruned
56
87
 
57
88
 
58
89
  def iter_files(
@@ -81,8 +112,10 @@ def iter_files(
81
112
  """
82
113
  root = Path(root)
83
114
  skip = set(SKIP_DIRS if skip_dirs is None else skip_dirs)
115
+ glob_skips: list = []
84
116
  if respect_gitignore:
85
- skip |= gitignore_dir_names(root)
117
+ names, glob_skips = gitignore_dir_patterns(root)
118
+ skip |= names
86
119
 
87
120
  yielded = 0
88
121
  seen = 0
@@ -97,6 +130,8 @@ def iter_files(
97
130
  for d in sorted(dirnames):
98
131
  if d in skip:
99
132
  continue
133
+ if glob_skips and any(fnmatch.fnmatch(d, p) for p in glob_skips):
134
+ continue
100
135
  if exclude_parts is not None and exclude_parts(rel_parts + (d,)):
101
136
  continue
102
137
  kept.append(d)
@@ -711,13 +711,18 @@ class SessionManager:
711
711
  return True
712
712
 
713
713
  def _scan_project_structure(self) -> str:
714
- """Scan project and generate compressed structure."""
715
- skip = {'node_modules', '.git', '__pycache__', '.c3', 'venv',
716
- 'env', '.venv', 'dist', 'build', '.next'}
714
+ """Scan project and generate compressed structure.
715
+
716
+ Prunes via the shared scanner rules (SKIP_DIRS + root .gitignore,
717
+ including simple globs like *.egg-info/), so the tree shipped in
718
+ instruction docs shows only what a clean distribution contains.
719
+ """
720
+ from services.scanner import make_dir_pruner
721
+ pruned = make_dir_pruner(self.project_path)
717
722
 
718
723
  structure = ["```"]
719
724
  for root, dirs, files in os.walk(self.project_path):
720
- dirs[:] = sorted(d for d in dirs if d not in skip)
725
+ dirs[:] = sorted(d for d in dirs if not pruned(d))
721
726
  level = len(Path(root).relative_to(self.project_path).parts)
722
727
 
723
728
  indent = " " * level