nimcode 0.1.21__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. {nimcode-0.1.21 → nimcode-0.3.0}/PKG-INFO +2 -3
  2. {nimcode-0.1.21 → nimcode-0.3.0}/setup.py +2 -3
  3. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/agent.py +25 -70
  4. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/cli.py +45 -48
  5. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/config.py +2 -2
  6. nimcode-0.3.0/src/nimcode/memory.py +69 -0
  7. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/nim_client.py +51 -66
  8. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/permissions.py +8 -34
  9. nimcode-0.3.0/src/nimcode/rag.py +121 -0
  10. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/repl.py +31 -6
  11. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/tools.py +53 -17
  12. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/PKG-INFO +2 -3
  13. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/SOURCES.txt +1 -9
  14. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/requires.txt +1 -2
  15. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_agent.py +6 -7
  16. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_cli.py +13 -25
  17. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_memory.py +4 -4
  18. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_permissions.py +29 -31
  19. nimcode-0.3.0/tests/test_plugin_manager.py +53 -0
  20. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_repl.py +9 -6
  21. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_tools.py +13 -3
  22. nimcode-0.3.0/tests/test_updater.py +41 -0
  23. nimcode-0.1.21/src/nimcode/analytics.py +0 -68
  24. nimcode-0.1.21/src/nimcode/memory.py +0 -140
  25. nimcode-0.1.21/src/nimcode/repo_map.py +0 -67
  26. nimcode-0.1.21/src/nimcode/secret_scanner.py +0 -48
  27. nimcode-0.1.21/src/nimcode/stdio_server.py +0 -163
  28. nimcode-0.1.21/src/nimcode/swarm.py +0 -42
  29. nimcode-0.1.21/tests/test_analytics.py +0 -22
  30. nimcode-0.1.21/tests/test_nimcoderules.py +0 -19
  31. nimcode-0.1.21/tests/test_plugin_manager.py +0 -35
  32. nimcode-0.1.21/tests/test_repo_map.py +0 -22
  33. nimcode-0.1.21/tests/test_secret_scanner.py +0 -31
  34. nimcode-0.1.21/tests/test_updater.py +0 -39
  35. {nimcode-0.1.21 → nimcode-0.3.0}/README.md +0 -0
  36. {nimcode-0.1.21 → nimcode-0.3.0}/setup.cfg +0 -0
  37. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/__init__.py +0 -0
  38. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/lenient_parser.py +0 -0
  39. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/mcp_client.py +0 -0
  40. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/plugin_manager.py +0 -0
  41. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/updater.py +0 -0
  42. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode/watcher.py +0 -0
  43. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/dependency_links.txt +0 -0
  44. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/entry_points.txt +0 -0
  45. {nimcode-0.1.21 → nimcode-0.3.0}/src/nimcode.egg-info/top_level.txt +0 -0
  46. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_config.py +0 -0
  47. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_lenient_parser.py +0 -0
  48. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_mcp_client.py +0 -0
  49. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_nim_client.py +0 -0
  50. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_repl_extra.py +0 -0
  51. {nimcode-0.1.21 → nimcode-0.3.0}/tests/test_repl_trust.py +0 -0
@@ -1,14 +1,13 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: nimcode
3
- Version: 0.1.21
3
+ Version: 0.3.0
4
4
  Summary: A standalone, robust coding agent for NVIDIA NIM models.
5
5
  Author: Autonomous Agent
6
6
  Requires-Dist: httpx>=0.27.0
7
7
  Requires-Dist: rich>=13.7.0
8
8
  Requires-Dist: prompt_toolkit>=3.0.0
9
9
  Requires-Dist: mcp>=1.2.0
10
- Requires-Dist: keyring>=24.0.0
11
- Requires-Dist: watchdog>=4.0.0
10
+ Requires-Dist: watchdog>=3.0.0
12
11
  Dynamic: author
13
12
  Dynamic: requires-dist
14
13
  Dynamic: summary
@@ -2,7 +2,7 @@ from setuptools import setup, find_packages
2
2
 
3
3
  setup(
4
4
  name="nimcode",
5
- version="0.1.21",
5
+ version="0.3.0",
6
6
  packages=find_packages(where="src"),
7
7
  package_dir={"": "src"},
8
8
  install_requires=[
@@ -10,8 +10,7 @@ setup(
10
10
  "rich>=13.7.0",
11
11
  "prompt_toolkit>=3.0.0",
12
12
  "mcp>=1.2.0",
13
- "keyring>=24.0.0",
14
- "watchdog>=4.0.0"
13
+ "watchdog>=3.0.0"
15
14
  ],
16
15
  entry_points={
17
16
  "console_scripts": [
@@ -8,7 +8,6 @@ from .permissions import PermissionEngine, PermissionMode
8
8
  from .config import load_settings, save_global_setting
9
9
  from .mcp_client import MCPManager
10
10
  from .memory import MemoryManager
11
- from .analytics import AnalyticsEngine
12
11
 
13
12
  logger = logging.getLogger(__name__)
14
13
 
@@ -28,7 +27,7 @@ Available Tools:
28
27
  - Bash: {"tool": "Bash", "args": {"command": "string"}}
29
28
  - Read: {"tool": "Read", "args": {"file_path": "string"}}
30
29
  - Write: {"tool": "Write", "args": {"file_path": "string", "content": "string"}}
31
- - Replace: {"tool": "Replace", "args": {"file_path": "string", "replacements": [{"old_string": "exact old", "new_string": "exact new"}]}}
30
+ - Edit: {"tool": "Edit", "args": {"file_path": "string", "old_string": "string", "new_string": "string"}}
32
31
  - Glob: {"tool": "Glob", "args": {"pattern": "string"}}
33
32
  - Grep: {"tool": "Grep", "args": {"query": "string", "directory": "string"}}
34
33
 
@@ -49,42 +48,19 @@ When you have completely fulfilled the user's request and have no more tools to
49
48
  """
50
49
 
51
50
  class Agent:
52
- def __init__(self, api_key: str, model: str = None, max_turns: int = 30, permission_mode: PermissionMode = PermissionMode.DEFAULT, max_tokens: int = 100000):
51
+ def __init__(self, api_key: str, model: str = None, max_turns: int = 30, permission_mode: PermissionMode = PermissionMode.DEFAULT, max_tokens: int = 4000):
53
52
  # Load global settings
54
53
  self.settings = load_settings()
55
54
  self.model = model or self.settings.get("model", "meta/llama-3.1-70b-instruct")
56
- self.is_local = self.settings.get("is_local", False)
57
- self.base_url = self.settings.get("base_url", None)
58
- self.client = NimClient(api_key=api_key, base_url=self.base_url, model=self.model, is_local=self.is_local)
55
+ api_base_url = self.settings.get("api_base_url", "https://integrate.api.nvidia.com/v1")
56
+ self.client = NimClient(api_key=api_key, base_url=api_base_url, model=self.model)
59
57
 
60
58
  # Initialize MCP Manager
61
59
  self.mcp = MCPManager(self.settings)
62
- self.analytics = AnalyticsEngine()
63
- self.memory = MemoryManager(model_name=model, fallback_max_tokens=max_tokens)
64
60
 
65
61
  # Base system prompt
66
62
  final_prompt = SYSTEM_PROMPT + self.mcp.get_system_prompt_additions()
67
63
 
68
- # Inject Repo Map
69
- try:
70
- from .repo_map import RepoMapper
71
- mapper = RepoMapper(os.getcwd())
72
- repo_map = mapper.generate_map()
73
- final_prompt += f"\n\n--- REPOSITORY MAP ---\n{repo_map}\n----------------------\n"
74
- except Exception as e:
75
- logger.error(f"Failed to generate repo map: {e}")
76
-
77
- # Load .nimcoderules if present
78
- rules_path = os.path.join(os.getcwd(), ".nimcoderules")
79
- if os.path.exists(rules_path):
80
- try:
81
- with open(rules_path, "r", encoding="utf-8") as f:
82
- rules = f.read()
83
- final_prompt += f"\n\nPROJECT-SPECIFIC RULES (.nimcoderules):\n{rules}\n"
84
- logger.info("Loaded .nimcoderules")
85
- except Exception as e:
86
- logger.error(f"Failed to load .nimcoderules: {e}")
87
-
88
64
  # Load skills if present
89
65
  skills_dir = os.path.join(os.getcwd(), ".nimcode", "skills")
90
66
  if os.path.exists(skills_dir) and os.path.isdir(skills_dir):
@@ -106,18 +82,13 @@ class Agent:
106
82
  final_prompt += f"\n\nGIT CONTEXT:\nBranch: {branch}\nUncommitted changes:\n{status if status else 'None'}"
107
83
  except Exception as e:
108
84
  logger.error(f"Failed to load git context: {e}")
109
-
110
- # Repo Map
111
- repo_map = self._generate_repo_map(os.getcwd())
112
- if repo_map:
113
- final_prompt += f"\n\nREPOSITORY MAP:\n{repo_map}"
114
85
 
115
86
  self.messages: List[Dict[str, Any]] = [
116
87
  {"role": "system", "content": final_prompt}
117
88
  ]
118
89
  self.max_turns = max_turns
119
90
  self.permission_engine = PermissionEngine(mode=permission_mode)
120
- self.memory = MemoryManager(model_name=model, fallback_max_tokens=max_tokens)
91
+ self.memory = MemoryManager(max_tokens=max_tokens)
121
92
 
122
93
  def save_history(self):
123
94
  """Saves current conversation to NIMCODE.md"""
@@ -138,28 +109,6 @@ class Agent:
138
109
  except Exception as e:
139
110
  logger.error(f"Failed to load history: {e}")
140
111
 
141
- def _generate_repo_map(self, cwd: str) -> str:
142
- """Generates a fast, lightweight file tree map."""
143
- ignore_dirs = {'.git', 'node_modules', '__pycache__', '.venv', 'venv', 'env', '.nimcode'}
144
- tree = []
145
- for root, dirs, files in os.walk(cwd):
146
- dirs[:] = [d for d in dirs if d not in ignore_dirs and not d.startswith('.')]
147
- level = root.replace(cwd, '').count(os.sep)
148
- indent = ' ' * 4 * level
149
- basename = os.path.basename(root)
150
- if basename:
151
- tree.append(f"{indent}{basename}/")
152
- subindent = ' ' * 4 * (level + 1)
153
- for f in files:
154
- if not f.endswith('.pyc') and not f.startswith('.'):
155
- tree.append(f"{subindent}{f}")
156
-
157
- # Limit to 500 lines to save context
158
- if len(tree) > 500:
159
- tree = tree[:500] + ["... (truncated for context limit)"]
160
-
161
- return "\n".join(tree)
162
-
163
112
  async def _stream_response(self) -> str:
164
113
  from rich.live import Live
165
114
  from rich.markdown import Markdown
@@ -222,16 +171,23 @@ class Agent:
222
171
  live.update(Markdown(response_text + "\n\n*[yellow]Stream interrupted by user.[/yellow]*", code_theme=code_theme))
223
172
  c.print("\n[yellow]Generation interrupted.[/yellow]")
224
173
 
225
- # Analytics Token Tracker Update
226
- est_prompt_tokens = self.client.count_tokens_approx(self.messages)
227
- est_completion_tokens = len(response_text) // 4
174
+ # Approximate Token Tracker Update
175
+ est_tokens = len(response_text) // 4
176
+ if not hasattr(self, "session_tokens"):
177
+ self.session_tokens = 0
178
+ self.session_tokens += est_tokens
228
179
 
229
- self.analytics.log_usage(self.model, est_prompt_tokens, est_completion_tokens)
180
+ # Approximate cost based on 70B typical rates ($3/1M tokens)
181
+ cost = (self.session_tokens / 1000000) * 3.0
182
+ c.print(f"[dim]Output est. tokens: {est_tokens} | Session Cost: ~${cost:.4f}[/dim]")
230
183
 
231
- # Display today's cost
232
- stats = self.analytics.get_summary()
233
- today_cost = stats["today"]["cost_usd"]
234
- c.print(f"[dim]Output est. tokens: {est_completion_tokens} | Today's Cost: ~${today_cost:.4f}[/dim]")
184
+ # Context usage warning
185
+ import json
186
+ total_context_chars = sum(len(str(m.get("content", ""))) for m in self.messages)
187
+ total_est_tokens = total_context_chars // 4
188
+ max_context = 128000 # Assume standard Llama-3.1 128k context for now
189
+ if total_est_tokens > max_context * 0.8:
190
+ c.print("[bold yellow]⚠️ Context window is over 80% full. Consider running /compact or /clear.[/bold yellow]")
235
191
 
236
192
  return response_text
237
193
 
@@ -242,10 +198,6 @@ class Agent:
242
198
  from .tools import ToolRegistry
243
199
  while turn < max_turns:
244
200
  turn += 1
245
- if self.memory.count_messages_tokens(self.messages) > (self.memory.max_tokens * 0.8):
246
- logger.info("Context full in headless mode. Distilling memory via LLM...")
247
- self.messages = await self._distill_memory()
248
-
249
201
  try:
250
202
  response_text = await self.client.chat_one_shot(self.messages)
251
203
  except Exception as e:
@@ -305,7 +257,7 @@ class Agent:
305
257
 
306
258
  # Compact context before calling API
307
259
  from rich.console import Console
308
- if self.memory.count_messages_tokens(self.messages) > (self.memory.max_tokens * 0.8):
260
+ if self.memory.count_messages_tokens(self.messages) > self.memory.max_tokens:
309
261
  logger.info("Context full. Distilling memory via LLM...")
310
262
  Console().print("[dim italic]🧠 Context full. Distilling memory into a summary to save tokens...[/dim italic]")
311
263
  self.messages = await self._distill_memory()
@@ -338,6 +290,9 @@ class Agent:
338
290
 
339
291
  if not tool_calls:
340
292
  # Model responded with plain text but didn't say TASK_COMPLETE.
293
+ if "[Error: Model API returned" in full_content or "[Error communicating with" in full_content:
294
+ break
295
+
341
296
  self.messages.append({"role": "user", "content": "Please continue. Use a tool or output TASK_COMPLETE."})
342
297
  continue
343
298
 
@@ -346,7 +301,7 @@ class Agent:
346
301
  tool_name = tool_call.get("tool", "Unknown")
347
302
 
348
303
  logger.info(f"Checking permissions for tool: {tool_name}")
349
- if not await self.permission_engine.check_permission(tool_call):
304
+ if not self.permission_engine.check_permission(tool_call):
350
305
  self.messages.append({
351
306
  "role": "user",
352
307
  "content": f"User explicitly denied permission to execute {tool_name}. Please choose another approach."
@@ -14,20 +14,13 @@ def run_login():
14
14
  console.print("Get your API key from [bold underline blue]https://build.nvidia.com/[/bold underline blue]")
15
15
 
16
16
  import getpass
17
- from .config import save_global_setting
18
-
19
- try:
20
- api_key = getpass.getpass("Enter your NVIDIA NIM API Key: ")
21
- except (EOFError, Exception):
22
- console.print("[red]Could not prompt for API key. Please set NIM_API_KEY environment variable or run 'nimcode login' in a regular terminal.[/red]")
23
- sys.exit(1)
24
-
17
+ api_key = getpass.getpass("Enter your NVIDIA NIM API Key: ")
25
18
  if not api_key.strip():
26
19
  console.print("[red]API Key cannot be empty.[/red]")
27
20
  return
28
21
 
29
22
  save_global_setting("api_key", api_key.strip())
30
- console.print("[green][OK] API Key saved securely to settings[/green]")
23
+ console.print("[green][OK] API Key saved successfully to ~/.nimcode/settings.json[/green]")
31
24
 
32
25
  def run_doctor():
33
26
  console.print("[bold cyan]NimCode Doctor[/bold cyan] - Diagnostics")
@@ -63,7 +56,19 @@ def install_hook():
63
56
  os.chmod(hook_path, os.stat(hook_path).st_mode | stat.S_IEXEC)
64
57
  console.print(f"[green][OK][/green] Git hook installed to {hook_path}")
65
58
 
59
+ def _silence_anyio_errors():
60
+ import sys
61
+ original_hook = sys.unraisablehook
62
+ def custom_unraisablehook(unraisable):
63
+ if unraisable.exc_type == RuntimeError and "exit cancel scope in a different task" in str(unraisable.exc_value):
64
+ return
65
+ if unraisable.exc_type == BaseExceptionGroup and "unhandled errors in a TaskGroup" in str(unraisable.exc_value):
66
+ return
67
+ original_hook(unraisable)
68
+ sys.unraisablehook = custom_unraisablehook
69
+
66
70
  def main():
71
+ _silence_anyio_errors()
67
72
  parser = argparse.ArgumentParser(description="NimCode: Autonomous Coding Agent for NVIDIA NIM APIs")
68
73
 
69
74
  # Check for doctor manually to avoid subparser conflict
@@ -74,25 +79,27 @@ def main():
74
79
  elif sys.argv[1] == "install-hook":
75
80
  install_hook()
76
81
  return
77
- elif sys.argv[1] == "login" or sys.argv[1] == "/login":
82
+ elif sys.argv[1] == "login":
78
83
  run_login()
79
84
  return
80
85
 
81
86
  # Main CLI arguments
82
87
  parser.add_argument("prompt", nargs="?", default=None, help="The task you want NimCode to accomplish. If omitted, starts interactive REPL.")
83
88
  parser.add_argument("--api-key", "-k", default=None, help="NVIDIA NIM API Key. Can also be set via NIM_API_KEY environment variable.")
84
- parser.add_argument("--model", "-m", default=None, help="Model ID to use from NIM (e.g. meta/llama-3.1-70b-instruct).")
89
+ parser.add_argument("--model", "-m", default="meta/llama-3.1-70b-instruct", help="Model ID to use from NIM.")
85
90
  parser.add_argument("--max-turns", "-t", type=int, default=30, help="Maximum number of turns the agent is allowed to run.")
86
91
  parser.add_argument("--permission-mode", "-p", type=PermissionMode, choices=list(PermissionMode), default=PermissionMode.DEFAULT, help="Permission mode for mutating tools.")
87
92
  parser.add_argument("--resume", "-r", action="store_true", help="Resume from the last session stored in NIMCODE.md.")
88
- parser.add_argument("--stdio", action="store_true", help="Run NimCode as a JSON-RPC server over stdio for IDE integration.")
89
93
 
90
94
  args = parser.parse_args()
91
95
 
92
96
  settings = load_settings()
97
+ api_base_url = settings.get("api_base_url", "https://integrate.api.nvidia.com/v1")
93
98
 
94
99
  final_key = args.api_key or os.environ.get("NIM_API_KEY") or settings.get("api_key")
95
- if not final_key:
100
+ is_local = "localhost" in api_base_url or "127.0.0.1" in api_base_url
101
+
102
+ if not final_key and not is_local:
96
103
  console.print("[yellow]No API Key found. Let's get you set up![/yellow]")
97
104
  run_login()
98
105
  settings = load_settings()
@@ -100,32 +107,12 @@ def main():
100
107
  if not final_key:
101
108
  console.print("[bold red]API Key is required to use NimCode. Exiting.[/bold red]")
102
109
  sys.exit(1)
103
-
104
- model = args.model or settings.get("model")
105
- if not model and not args.stdio:
106
- from rich.prompt import Prompt
107
- from .nim_client import NimClient
108
- client = NimClient(api_key=final_key)
109
- console.print("[yellow]Fetching available models from NVIDIA NIM...[/yellow]")
110
- models = asyncio.run(client.get_available_models())
111
- console.print("\n[bold cyan]Available Models:[/bold cyan]")
112
- for i, m in enumerate(models):
113
- console.print(f"[{i+1}] {m}")
114
-
115
- choice = Prompt.ask("Select a model number", choices=[str(i) for i in range(1, len(models)+1)], default="1")
116
- model = models[int(choice) - 1]
117
-
118
- from .config import save_global_setting
119
- if Prompt.ask("Save this model as default?", choices=["y", "n"], default="y") == "y":
120
- save_global_setting("model", model)
121
- console.print(f"[green]Saved {model} as default model.[/green]")
122
110
 
123
- if not model:
124
- model = "meta/llama-3.1-70b-instruct" # ultimate fallback
125
-
111
+ final_key = final_key or "local-dummy-key"
112
+
126
113
  agent = Agent(
127
114
  api_key=final_key,
128
- model=model,
115
+ model=args.model,
129
116
  max_turns=args.max_turns,
130
117
  permission_mode=args.permission_mode
131
118
  )
@@ -134,15 +121,29 @@ def main():
134
121
  # Load from history if possible
135
122
  agent.load_history()
136
123
 
137
- piped_input = None
138
- if not sys.stdin.isatty():
139
- import select
124
+ async def safe_start_repl(repl):
125
+ loop = asyncio.get_running_loop()
126
+ def custom_exception_handler(loop, context):
127
+ msg = context.get("message", "")
128
+ exception = context.get("exception", None)
129
+ if "unhandled errors in a TaskGroup" in str(msg) or "unhandled errors in a TaskGroup" in str(exception):
130
+ return
131
+ if exception and isinstance(exception, RuntimeError) and "exit cancel scope in a different task" in str(exception):
132
+ return
133
+ if "asynchronous generator" in str(msg):
134
+ return
135
+ loop.default_exception_handler(context)
136
+ loop.set_exception_handler(custom_exception_handler)
137
+
140
138
  try:
141
- # On Unix, we can check if stdin has data ready to read
142
- if sys.platform != "win32" and select.select([sys.stdin], [], [], 0.0)[0]:
143
- piped_input = sys.stdin.read().strip()
144
- except Exception:
139
+ await repl.start_repl()
140
+ except asyncio.CancelledError:
145
141
  pass
142
+
143
+ piped_input = None
144
+ if not sys.stdin.isatty():
145
+ piped_input = sys.stdin.read().strip()
146
+
146
147
  if piped_input:
147
148
  console.print(f"[bold green]Starting NimCode[/bold green] with model [cyan]{args.model}[/cyan]")
148
149
  prompt = f"{piped_input}\n\n{args.prompt or ''}".strip()
@@ -152,14 +153,10 @@ def main():
152
153
  console.print(f"[bold green]Starting NimCode[/bold green] with model [cyan]{args.model}[/cyan]")
153
154
  console.print(f"Task: {args.prompt}")
154
155
  asyncio.run(agent.run(args.prompt))
155
- elif args.stdio:
156
- from .stdio_server import StdioServer
157
- server = StdioServer(agent)
158
- asyncio.run(server.start())
159
156
  else:
160
157
  from .repl import NimcodeREPL
161
158
  repl = NimcodeREPL(agent)
162
- asyncio.run(repl.start_repl())
159
+ asyncio.run(safe_start_repl(repl))
163
160
 
164
161
  # We don't print "Done!" for REPL to keep it clean on exit
165
162
  if args.prompt or piped_input:
@@ -9,6 +9,7 @@ def load_settings() -> Dict[str, Any]:
9
9
  """Loads configuration from ~/.nimcode/settings.json and .nimcode/settings.json"""
10
10
  settings = {
11
11
  "model": "meta/llama-3.1-70b-instruct",
12
+ "api_base_url": "https://integrate.api.nvidia.com/v1",
12
13
  "mcp_servers": {}
13
14
  }
14
15
 
@@ -21,8 +22,7 @@ def load_settings() -> Dict[str, Any]:
21
22
  settings.update(global_settings)
22
23
  except Exception as e:
23
24
  logger.error(f"Failed to load global settings: {e}")
24
-
25
- # Keyring caused 30-second hangs on Windows, so we no longer scrub api_key from settings.
25
+
26
26
  # Local settings
27
27
  local_path = os.path.join(os.getcwd(), ".nimcode", "settings.json")
28
28
  if os.path.exists(local_path):
@@ -0,0 +1,69 @@
1
+ import os
2
+ from typing import List, Dict, Any
3
+
4
+ class MemoryManager:
5
+ def __init__(self, max_tokens: int = 4000):
6
+ # We assume 1 token ~= 4 chars roughly
7
+ self.max_tokens = max_tokens
8
+
9
+ @staticmethod
10
+ def count_tokens(text: str) -> int:
11
+ """Roughly count tokens in a string."""
12
+ if not text:
13
+ return 0
14
+ return len(text) // 4 + 1
15
+
16
+ @classmethod
17
+ def count_messages_tokens(cls, messages: List[Dict[str, Any]]) -> int:
18
+ total = 0
19
+ for msg in messages:
20
+ content = msg.get("content", "")
21
+ total += cls.count_tokens(content)
22
+ return total
23
+
24
+ def compact_context(self, messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
25
+ """
26
+ If messages exceed max_tokens, drops the oldest messages.
27
+ Always keeps the System prompt (first message).
28
+ Always keeps the most recent user prompt.
29
+ """
30
+ total_tokens = self.count_messages_tokens(messages)
31
+ if total_tokens <= self.max_tokens:
32
+ return messages
33
+
34
+ # We need to compact
35
+ if not messages:
36
+ return messages
37
+
38
+ compacted = [messages[0]] # System prompt
39
+ remaining_messages = messages[1:]
40
+
41
+ # We start from the end and add backwards until we hit the limit
42
+ # Reserved tokens for system prompt
43
+ current_tokens = self.count_tokens(messages[0].get("content", ""))
44
+
45
+ kept_messages = []
46
+ for msg in reversed(remaining_messages):
47
+ msg_tokens = self.count_tokens(msg.get("content", ""))
48
+ if current_tokens + msg_tokens > self.max_tokens:
49
+ # If we haven't even kept the most recent message, we MUST keep it and just truncate its content
50
+ if not kept_messages:
51
+ truncated_content = msg.get("content", "")[:(self.max_tokens - current_tokens) * 4]
52
+ kept_messages.insert(0, {"role": msg["role"], "content": truncated_content + "...[TRUNCATED]"})
53
+ break
54
+
55
+ kept_messages.insert(0, msg)
56
+ current_tokens += msg_tokens
57
+
58
+ compacted.extend(kept_messages)
59
+ return compacted
60
+
61
+ @staticmethod
62
+ def log_to_nimcode_md(turn: int, prompt: str, response: str, cwd: str = ".") -> None:
63
+ """Appends the interaction to NIMCODE.md for persistent session history."""
64
+ file_path = os.path.join(cwd, "NIMCODE.md")
65
+ with open(file_path, "a", encoding="utf-8") as f:
66
+ f.write(f"## Turn {turn}\n\n")
67
+ f.write(f"**User**: {prompt}\n\n")
68
+ f.write(f"**Agent**: {response}\n\n")
69
+ f.write("---\n\n")
@@ -8,58 +8,17 @@ from typing import List, Dict, Any, Optional, AsyncGenerator
8
8
  logger = logging.getLogger(__name__)
9
9
 
10
10
  class NimClient:
11
- def __init__(self, api_key: str, base_url: str = None, model: str = None, is_local: bool = False):
11
+ def __init__(self, api_key: str, base_url: str = "https://integrate.api.nvidia.com/v1", model: str = "meta/llama-3.1-70b-instruct"):
12
12
  self.api_key = api_key
13
- self.is_local = is_local
14
- self.base_url = (base_url or "https://integrate.api.nvidia.com/v1").rstrip("/")
15
- self.model = model or "meta/llama-3.1-70b-instruct"
13
+ self.base_url = base_url.rstrip("/")
14
+ self.model = model
16
15
  self.headers = {
17
16
  "Authorization": f"Bearer {self.api_key}",
18
17
  "Content-Type": "application/json",
19
18
  "Accept": "text/event-stream"
20
19
  }
21
20
 
22
- @staticmethod
23
- def get_model_context_length(model_name: str) -> int:
24
- """Heuristically determines the maximum token context length from the model name."""
25
- model_lower = model_name.lower()
26
- if "llama-3.1" in model_lower or "llama-3.2" in model_lower:
27
- return 128000
28
- elif "nemotron" in model_lower:
29
- return 128000 # Nemotron-4 is usually high context
30
- elif "mixtral" in model_lower:
31
- return 64000
32
- elif "phi" in model_lower:
33
- return 128000
34
- elif "gemma" in model_lower:
35
- return 8192
36
- # Default fallback
37
- return 32000
38
-
39
21
  async def get_available_models(self) -> List[str]:
40
- if self.is_local:
41
- try:
42
- async with httpx.AsyncClient() as client:
43
- response = await client.get(f"{self.base_url}/models", timeout=5.0)
44
- if response.status_code == 200:
45
- data = response.json()
46
- return [m["id"] for m in data.get("data", [])]
47
- except Exception as e:
48
- logger.error(f"Failed to fetch local models: {e}")
49
- return ["local-model"]
50
-
51
- try:
52
- async with httpx.AsyncClient() as client:
53
- response = await client.get(f"{self.base_url}/models", headers={"Authorization": f"Bearer {self.api_key}"}, timeout=10.0)
54
- if response.status_code == 200:
55
- data = response.json()
56
- return [m["id"] for m in data.get("data", [])]
57
- else:
58
- logger.error(f"Failed to fetch models from API: {response.status_code}")
59
- except Exception as e:
60
- logger.error(f"Failed to fetch models: {e}")
61
-
62
- # Fallback list if API fails
63
22
  return [
64
23
  "meta/llama-3.1-70b-instruct",
65
24
  "meta/llama-3.1-8b-instruct",
@@ -131,31 +90,57 @@ class NimClient:
131
90
  "stream": stream
132
91
  }
133
92
 
134
- async with httpx.AsyncClient() as client:
93
+ max_retries = 15
94
+ base_delay = 2.0
95
+ max_delay = 60.0
96
+
97
+ for attempt in range(max_retries):
98
+ chunk_yielded = False
135
99
  try:
136
- async with client.stream("POST", f"{self.base_url}/chat/completions", headers=self.headers, json=payload, timeout=60.0) as response:
137
- response.raise_for_status()
138
- async for line in response.aiter_lines():
139
- if line.startswith("data: ") and line != "data: [DONE]":
140
- data_str = line[6:]
141
- try:
142
- data_json = json.loads(data_str)
143
- chunk = data_json["choices"][0]["delta"].get("content", "")
144
- if chunk:
145
- yield chunk
146
- except json.JSONDecodeError:
147
- pass
100
+ async with httpx.AsyncClient() as client:
101
+ async with client.stream("POST", f"{self.base_url}/chat/completions", headers=self.headers, json=payload, timeout=120.0) as response:
102
+ if response.status_code in [408, 429, 500, 502, 503, 504, 529]:
103
+ raise httpx.HTTPStatusError(f"Temporary server error {response.status_code}", request=response.request, response=response)
104
+
105
+ response.raise_for_status()
106
+
107
+ async for line in response.aiter_lines():
108
+ if line.startswith("data: ") and line != "data: [DONE]":
109
+ data_str = line[6:]
110
+ try:
111
+ data_json = json.loads(data_str)
112
+ chunk = data_json["choices"][0]["delta"].get("content", "")
113
+ if chunk:
114
+ chunk_yielded = True
115
+ yield chunk
116
+ except json.JSONDecodeError:
117
+ pass
118
+ return # Success
148
119
  except httpx.HTTPStatusError as e:
149
- try:
150
- await e.response.aread()
151
- text = e.response.text
152
- except Exception:
153
- text = "<unread stream>"
154
- logger.error(f"API HTTP error: {e.response.status_code} - {text}")
155
- yield f"\\n\\n[Error: Model API returned {e.response.status_code}. Please check your NVIDIA API key.]"
120
+ transient_codes = [408, 429, 500, 502, 503, 504, 529]
121
+ if e.response.status_code not in transient_codes or chunk_yielded or attempt == max_retries - 1:
122
+ try:
123
+ await e.response.aread()
124
+ text = e.response.text
125
+ except Exception:
126
+ text = "<unread stream>"
127
+ logger.error(f"API HTTP error: {e.response.status_code} - {text}")
128
+ yield f"\n\n[Error: Model API returned {e.response.status_code}. Please check your API key if 401.]"
129
+ return
130
+
131
+ delay = min(max_delay, base_delay * (2 ** attempt))
132
+ logger.warning(f"API HTTP error {e.response.status_code}. Retrying in {delay}s...")
133
+ await asyncio.sleep(delay)
134
+
156
135
  except Exception as e:
157
- logger.error(f"API connection error: {e}")
158
- yield f"\\n\\n[Error communicating with NVIDIA API: {e}]"
136
+ if chunk_yielded or attempt == max_retries - 1:
137
+ logger.error(f"API connection error: {type(e).__name__} - {e}")
138
+ yield f"\n\n[Error communicating with NVIDIA API: {type(e).__name__} - {e}]"
139
+ return
140
+
141
+ delay = min(max_delay, base_delay * (2 ** attempt))
142
+ logger.warning(f"API connection error: {type(e).__name__} - {e}. Retrying in {delay}s...")
143
+ await asyncio.sleep(delay)
159
144
 
160
145
  def count_tokens_approx(self, messages: List[Dict[str, Any]]) -> int:
161
146
  total_chars = 0