nanoPyCodeAgent 0.8.0__py3-none-any.whl → 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
nanopycodeagent/agent.py CHANGED
@@ -11,18 +11,18 @@ or overwrite them, an ``edit`` tool to replace part of one, and a ``bash``
11
11
  tool to run shell commands; every call and its output are echoed to the
12
12
  terminal as they happen.
13
13
 
14
- The interactive loop handles only the happy path: anything unexpected — a
15
- network error, a Ctrl-C mid-turn — crashes the session, and restarting it is
16
- the recovery. That trade keeps the core flow readable; the hardened variant
17
- it replaced is preserved at the ``hardened-agent-loop`` tag. A headless run
18
- has no one to restart it, so it catches API errors, reports them verbatim,
19
- and turns them into an exit code.
14
+ Both modes retry interrupted response streams before changing conversation
15
+ history or running tools. Other unexpected failures and Ctrl-C mid-turn still
16
+ end an interactive session. Headless mode catches exhausted API/transport
17
+ errors, reports them verbatim, and turns them into an exit code.
20
18
  """
21
19
 
20
+ import json
22
21
  import os
23
22
  import sys
24
23
  import time
25
24
  import uuid
25
+ from copy import copy
26
26
  from importlib.metadata import PackageNotFoundError, version
27
27
  from pathlib import Path
28
28
 
@@ -37,34 +37,47 @@ except ImportError: # pragma: no cover - platform without readline
37
37
  pass
38
38
 
39
39
  import anthropic
40
- import httpx
41
40
  from anthropic.types import MessageParam, ToolResultBlockParam, ToolUseBlock
42
41
 
43
42
  from .atif import project_atif, write_atif
44
- from .bash_tool import BASH_TOOL, run_bash
43
+ from .bash_tool import run_bash
45
44
  from .cost import (
45
+ estimated_cost,
46
46
  pending_cost,
47
47
  resolve_generation_cost,
48
48
  usage_cost,
49
49
  )
50
- from .edit_tool import EDIT_TOOL, edit_preview, run_edit
50
+ from .edit_tool import edit_preview, run_edit
51
+ from .deadline import DeadlineExceeded, check_deadline_support, wall_clock_limit
51
52
  from .event_journal import (
52
53
  EventEmitter,
53
54
  EventJournal,
54
55
  JsonObject,
55
56
  JsonValue,
56
57
  NativeEvent,
58
+ RunOutcome,
57
59
  utc_now,
58
60
  )
59
- from .read_tool import READ_TOOL, run_read
60
- from .settings import load_settings_env
61
+ from .read_tool import run_read
62
+ from .settings import DEFAULT_MAX_TOKENS, load_settings_env, resolve_max_tokens
61
63
  from .terminal import Spinner, print_tool_output, print_tool_use
62
- from .write_tool import WRITE_TOOL, content_preview, run_write
64
+ from .tool_validation import TOOLS, tool_input_error
65
+ from .transport import HTTP_ERRORS, RETRYABLE_STREAM_ERRORS
66
+ from .write_tool import content_preview, run_write
63
67
 
64
68
  # The model used when ANTHROPIC_MODEL is set in neither the environment nor
65
69
  # the config file.
66
70
  DEFAULT_MODEL = "claude-sonnet-4-6"
67
- MAX_TOKENS = 8192
71
+
72
+ # The SDK retries failures before streaming begins. Recover interrupted response
73
+ # bodies here, before committing a reply to history or executing any of its tools.
74
+ STREAM_RETRY_DELAYS = (1.0, 2.0)
75
+ STREAM_RETRY_WINDOW_SECONDS = 300.0
76
+
77
+ _TRUNCATION_NOTICE = (
78
+ "[response truncated: reached max_tokens; stopped without finishing the task. "
79
+ "Tool calls from this response were not executed.]"
80
+ )
68
81
 
69
82
  # How many model replies one headless task may spend before the run stops on
70
83
  # its own. The interactive loop needs no such cap — a human watching the
@@ -73,6 +86,91 @@ MAX_TOKENS = 8192
73
86
  # refuses it.
74
87
  DEFAULT_MAX_TURNS = 50
75
88
 
89
+ # A headless run may also be given a wall-clock budget. When it is, the loop
90
+ # tells the model how much time is left and stops before the harness's own
91
+ # timeout can kill the process with nothing written. The last stretch is
92
+ # reserved so the model still has room to write the task's output file.
93
+ _FINALIZATION_RESERVE_SECONDS = 180
94
+ _FINALIZATION_RESERVE_FRACTION = 0.15
95
+ _FINALIZATION_RESERVE_TURNS = 10
96
+ _COST_RECONCILIATION_SECONDS = 30
97
+
98
+
99
+ def _format_duration(total_seconds: float) -> str:
100
+ total = max(0, int(total_seconds))
101
+ return f"{total // 60}:{total % 60:02d}"
102
+
103
+
104
+ def _time_budget_note(
105
+ *, turn: int, elapsed: float, budget: float | None, remaining: float | None,
106
+ max_turns: int | None = None,
107
+ ) -> str:
108
+ """Tell the model about both limits before either prevents finalization."""
109
+ replies_left = None if max_turns is None else max_turns - turn + 1
110
+ turn_note = f"Turn {turn}. " if max_turns is None else (
111
+ f"Turn {turn} of {max_turns}; {replies_left} replies remaining including "
112
+ "this one. The last reply must be a final summary: its tool calls "
113
+ "will not execute. "
114
+ )
115
+ time_low = remaining is not None and remaining <= max(
116
+ _FINALIZATION_RESERVE_FRACTION * budget, _FINALIZATION_RESERVE_SECONDS
117
+ )
118
+ note = "[runtime budget] " + turn_note
119
+ if remaining is not None:
120
+ note += (
121
+ f"Only {_format_duration(remaining)} of {_format_duration(budget)} left. "
122
+ if time_low else
123
+ f"Elapsed {_format_duration(elapsed)} of {_format_duration(budget)}; "
124
+ f"{_format_duration(remaining)} remaining. "
125
+ )
126
+ if time_low or (replies_left is not None and replies_left <= _FINALIZATION_RESERVE_TURNS):
127
+ note += (
128
+ "Finalize now. Stop investigating new approaches. Complete and save "
129
+ "the required deliverables, perform only the necessary checks, then "
130
+ "reply with a short summary and no further tool calls. If incomplete, "
131
+ "save useful progress and state the remaining limitation honestly."
132
+ )
133
+ else:
134
+ note += (
135
+ "Keep required deliverables up to date. Once the requirements are "
136
+ "satisfied and checked, finish immediately; unused budget is not "
137
+ "a reason to continue investigating."
138
+ )
139
+ return note
140
+
141
+
142
+ def _append_budget_note(
143
+ messages: list[MessageParam],
144
+ note: str,
145
+ *,
146
+ emitter: EventEmitter,
147
+ model_call_id: str,
148
+ reason: str = "time_budget",
149
+ ) -> None:
150
+ """Append a wall-clock reminder to the tail of the conversation.
151
+
152
+ The reminder goes into the most recent user message — the initial task, or
153
+ the tool results — rather than the system prompt. Rewriting the system
154
+ prompt each turn changes the very first tokens of every request and defeats
155
+ the provider's prefix cache; appending to the tail keeps each request an
156
+ extension of the previous one, so the cached prefix survives.
157
+ """
158
+ last = messages[-1]
159
+ content = last["content"]
160
+ if isinstance(content, str):
161
+ last["content"] = f"{content}\n\n{note}"
162
+ else:
163
+ content.append({"type": "text", "text": note})
164
+ emitter.emit(
165
+ "input.injected",
166
+ {
167
+ "model_call_id": model_call_id,
168
+ "content": note,
169
+ "reason": reason,
170
+ "source_timestamp": utc_now(),
171
+ },
172
+ )
173
+
76
174
  # Shared by both system prompts: which tool to reach for is the same question
77
175
  # whoever is asking.
78
176
  _TOOL_GUIDANCE = (
@@ -98,13 +196,12 @@ HEADLESS_SYSTEM_PROMPT = (
98
196
  "out. Work the task through to the end, then check the result with the "
99
197
  "tools instead of assuming it worked. When it is done, answer with a "
100
198
  "short summary and no further tool calls: that reply is what ends the "
101
- "run. "
199
+ "run. Runtime budget reminders report remaining time and model replies; "
200
+ "when either is running low, prioritize saving the required deliverables, "
201
+ "necessary verification, and a final summary. Do not start optional work "
202
+ "after the requirements are satisfied. "
102
203
  ) + _TOOL_GUIDANCE
103
204
 
104
- # Every tool offered to the model on each request.
105
- TOOLS = [READ_TOOL, WRITE_TOOL, EDIT_TOOL, BASH_TOOL]
106
-
107
-
108
205
  def _json_value(value: object) -> JsonValue:
109
206
  """Convert an SDK value into the provider-neutral event representation."""
110
207
  if value is None or isinstance(value, bool | int | float | str):
@@ -139,12 +236,19 @@ def _native_content_blocks(value: object) -> list[JsonValue]:
139
236
  if source_type == "text":
140
237
  content.append({"type": "text", "text": source_block.get("text", "")})
141
238
  elif source_type == "tool_use":
239
+ arguments = source_block.get("input")
142
240
  content.append(
143
241
  {
144
242
  "type": "tool_call",
145
243
  "tool_call_id": source_block.get("id"),
146
244
  "tool_name": source_block.get("name"),
147
- "input": source_block.get("input"),
245
+ # Journal/ATIF require object arguments. Keep rejected
246
+ # non-object input separately instead of discarding it.
247
+ "input": arguments if isinstance(arguments, dict) else {},
248
+ **(
249
+ {"raw_input": arguments}
250
+ if not isinstance(arguments, dict) else {}
251
+ ),
148
252
  }
149
253
  )
150
254
  else:
@@ -192,12 +296,25 @@ class _TextOutputProjector:
192
296
  model_call_id = str(event.payload["model_call_id"])
193
297
  if model_call_id in self._model_calls_with_text:
194
298
  print()
299
+ if event.payload["stop_reason"] == "max_tokens":
300
+ print(_TRUNCATION_NOTICE, file=sys.stderr)
301
+ elif event.type == "model.failed" and event.payload["will_retry"]:
302
+ if str(event.payload["model_call_id"]) in self._model_calls_with_text:
303
+ print()
304
+ print(
305
+ f"[response interrupted: {event.payload['error_type']}; "
306
+ f"retrying in {event.payload['retry_delay_seconds']:g}s]",
307
+ file=sys.stderr,
308
+ )
195
309
  elif event.type == "tool.started":
196
310
  tool_name = str(event.payload["tool_name"])
197
311
  arguments = event.payload["input"]
198
- if not isinstance(arguments, dict):
199
- raise TypeError("tool input event payload must be an object")
200
- if tool_name == "read":
312
+ error = event.payload.get("input_error") or tool_input_error(
313
+ tool_name, arguments
314
+ )
315
+ if error:
316
+ print_tool_use(f"[{tool_name}] (invalid arguments; not executed)")
317
+ elif tool_name == "read":
201
318
  print_tool_use(f"[read] {arguments['path']}")
202
319
  elif tool_name == "write":
203
320
  content = str(arguments["content"])
@@ -211,7 +328,7 @@ class _TextOutputProjector:
211
328
  f"[edit] {arguments['path']}\n"
212
329
  f"{edit_preview(old_text, new_text)}"
213
330
  )
214
- else:
331
+ elif tool_name == "bash":
215
332
  print_tool_use(f"[bash]$ {arguments['command']}")
216
333
  elif event.type == "tool.completed":
217
334
  result = event.payload["result"]
@@ -237,24 +354,29 @@ def _run_one_tool(
237
354
  block: ToolUseBlock,
238
355
  emitter: EventEmitter,
239
356
  model_call_id: str,
357
+ *,
358
+ input_error: str | None = None,
359
+ remaining_seconds: float | None = None,
240
360
  ) -> ToolResultBlockParam:
241
361
  """Execute one ``tool_use`` block and emit its runtime facts."""
242
362
  tool_input = _json_value(block.input)
243
- if not isinstance(tool_input, dict):
244
- raise TypeError("tool input must be an object")
363
+ input_error = input_error or tool_input_error(block.name, tool_input)
245
364
  emitter.emit(
246
365
  "tool.started",
247
366
  {
248
367
  "model_call_id": model_call_id,
249
368
  "tool_call_id": block.id,
250
369
  "tool_name": block.name,
251
- "input": tool_input,
370
+ "input": tool_input if isinstance(tool_input, dict) else {},
371
+ **({"input_error": input_error} if input_error else {}),
252
372
  "source_timestamp": utc_now(),
253
373
  },
254
374
  )
255
375
  tool_started_ns = time.perf_counter_ns()
256
376
  try:
257
- if block.name == "read":
377
+ if input_error:
378
+ output, is_error = input_error, True
379
+ elif block.name == "read":
258
380
  path = block.input["path"]
259
381
  output, is_error = run_read(
260
382
  path,
@@ -275,10 +397,13 @@ def _run_one_tool(
275
397
  new_text,
276
398
  replace_all=block.input.get("replace_all", False),
277
399
  )
278
- else: # bash — the only other tool offered
400
+ else: # bash; unknown names have already been rejected
279
401
  command = block.input["command"]
280
402
  with Spinner("Running..."):
281
- output, is_error = run_bash(command)
403
+ output, is_error = run_bash(command, **(
404
+ {"timeout_seconds": remaining_seconds}
405
+ if remaining_seconds is not None else {}
406
+ ))
282
407
  except BaseException as exc:
283
408
  emitter.emit(
284
409
  "tool.completed",
@@ -303,6 +428,10 @@ def _run_one_tool(
303
428
  "tool_name": block.name,
304
429
  "result": output,
305
430
  "is_error": is_error,
431
+ **(
432
+ {"error": {"type": "ToolInputError", "message": input_error}}
433
+ if input_error else {}
434
+ ),
306
435
  "duration_ms": (time.perf_counter_ns() - tool_started_ns) / 1_000_000,
307
436
  "source_timestamp": utc_now(),
308
437
  },
@@ -352,17 +481,21 @@ def _run_exchange(
352
481
  system: str,
353
482
  *,
354
483
  max_turns: int | None = None,
484
+ max_tokens: int = DEFAULT_MAX_TOKENS,
485
+ time_budget_seconds: int | None = None,
355
486
  reply_prefix: str = "\nAgent> ",
356
487
  trajectory_path: Path | None = None,
357
- ) -> bool:
488
+ ) -> RunOutcome:
358
489
  """Reply to the conversation so far, running tools until the model stops.
359
490
 
360
- Appends every assistant reply and tool result to ``messages`` in place.
361
- Returns True when the model ended a reply without asking for tools, and
362
- False when ``max_turns`` replies were spent while it was still calling
363
- them — the caller decides what an exhausted budget means.
491
+ Appends assistant replies and tool results to ``messages`` in place.
492
+ A truncated reply retains only its text and a notice in request history;
493
+ the original response is kept in the journal. Returns the stopping outcome,
494
+ distinguishing completion, turn-budget exhaustion, and response truncation.
364
495
  """
365
496
  run_id = f"run-{uuid.uuid4()}"
497
+ if time_budget_seconds is not None:
498
+ check_deadline_support()
366
499
  run_started_ns = time.perf_counter_ns()
367
500
  projector = _TextOutputProjector(reply_prefix)
368
501
  with EventJournal.create(run_id) as journal:
@@ -370,9 +503,15 @@ def _run_exchange(
370
503
  emitter.emit(
371
504
  "run.started",
372
505
  {
373
- "mode": "headless" if max_turns is not None else "interactive",
506
+ "mode": (
507
+ "headless"
508
+ if max_turns is not None or time_budget_seconds is not None
509
+ else "interactive"
510
+ ),
374
511
  "model": model,
375
512
  "max_turns": max_turns,
513
+ "max_tokens": max_tokens,
514
+ "time_budget_seconds": time_budget_seconds,
376
515
  "producer": {
377
516
  "name": "nanoPyCodeAgent",
378
517
  "version": _package_version(),
@@ -390,16 +529,21 @@ def _run_exchange(
390
529
  },
391
530
  )
392
531
  try:
393
- finished = _run_model_loop(
532
+ outcome = _run_model_loop(
394
533
  client,
395
534
  model,
396
535
  messages,
397
536
  system,
398
537
  emitter=emitter,
399
538
  max_turns=max_turns,
539
+ max_tokens=max_tokens,
540
+ time_budget_seconds=time_budget_seconds,
400
541
  )
401
542
  except BaseException as exc:
402
- cost_reconciliation = _reconcile_costs(client, journal, emitter)
543
+ cost_reconciliation = _reconcile_costs(
544
+ client, journal, emitter,
545
+ max_seconds=_COST_RECONCILIATION_SECONDS if time_budget_seconds else None,
546
+ )
403
547
  emitter.emit(
404
548
  "run.failed",
405
549
  {
@@ -417,11 +561,14 @@ def _run_exchange(
417
561
  )
418
562
  raise
419
563
  else:
420
- cost_reconciliation = _reconcile_costs(client, journal, emitter)
564
+ cost_reconciliation = _reconcile_costs(
565
+ client, journal, emitter,
566
+ max_seconds=_COST_RECONCILIATION_SECONDS if time_budget_seconds else None,
567
+ )
421
568
  emitter.emit(
422
569
  "run.completed",
423
570
  {
424
- "outcome": "completed" if finished else "max_turns_exhausted",
571
+ "outcome": outcome,
425
572
  "duration_ms": (time.perf_counter_ns() - run_started_ns)
426
573
  / 1_000_000,
427
574
  **(
@@ -438,7 +585,7 @@ def _run_exchange(
438
585
  project_atif(EventJournal.replay(journal.path)),
439
586
  trajectory_path,
440
587
  )
441
- return finished
588
+ return outcome
442
589
 
443
590
 
444
591
  def _run_model_loop(
@@ -449,14 +596,42 @@ def _run_model_loop(
449
596
  *,
450
597
  emitter: EventEmitter,
451
598
  max_turns: int | None,
452
- ) -> bool:
599
+ max_tokens: int,
600
+ time_budget_seconds: int | None = None,
601
+ ) -> RunOutcome:
453
602
  """Run model replies and tool calls for an already-started Agent Run."""
454
603
  turns = 0
604
+ retries = 0
605
+ retry_deadline = None
606
+ deadline = (
607
+ time.monotonic() + time_budget_seconds
608
+ if time_budget_seconds is not None and time_budget_seconds > 0
609
+ else None
610
+ )
455
611
  while True:
612
+ model_call_id = f"model-{uuid.uuid4()}"
613
+ remaining = None
614
+ if deadline is not None:
615
+ remaining = deadline - time.monotonic()
616
+ if remaining <= 0:
617
+ return "time_budget_exhausted"
618
+ if deadline is not None or (max_turns is not None and retries == 0):
619
+ _append_budget_note(
620
+ messages,
621
+ _time_budget_note(
622
+ turn=turns + 1,
623
+ elapsed=time_budget_seconds - remaining if deadline is not None else 0,
624
+ budget=time_budget_seconds,
625
+ remaining=remaining,
626
+ max_turns=max_turns,
627
+ ),
628
+ emitter=emitter,
629
+ model_call_id=model_call_id,
630
+ reason="time_budget" if deadline is not None else "turn_budget",
631
+ )
456
632
  # A spinner marks the wait for the reply; the first streamed
457
633
  # token replaces it with the reply prefix. A tool-only reply
458
634
  # streams no text, so the prefix is skipped for it entirely.
459
- model_call_id = f"model-{uuid.uuid4()}"
460
635
  emitter.emit(
461
636
  "model.started",
462
637
  {
@@ -468,28 +643,111 @@ def _run_model_loop(
468
643
  model_started_ns = time.perf_counter_ns()
469
644
  # Stream the reply so text shows up as it is generated, then grab
470
645
  # the accumulated message for the conversation history.
471
- with Spinner() as spinner, client.messages.stream(
472
- model=model,
473
- max_tokens=MAX_TOKENS,
474
- system=system,
475
- tools=TOOLS,
476
- messages=messages,
477
- ) as stream:
478
- for text in stream.text_stream:
479
- spinner.stop()
480
- emitter.emit(
481
- "model.output_delta",
482
- {
483
- "model_call_id": model_call_id,
484
- "delta": text,
485
- "source_timestamp": utc_now(),
486
- },
487
- )
488
- message = stream.get_final_message()
489
- generation_id = _response_header(stream, "x-generation-id")
490
- model_completed_ns = time.perf_counter_ns()
646
+ generation_id = None
647
+ stream_entered = False
648
+ try:
649
+ with wall_clock_limit(remaining), Spinner() as spinner, client.messages.stream(
650
+ model=model,
651
+ max_tokens=max_tokens,
652
+ system=system,
653
+ tools=TOOLS,
654
+ messages=messages,
655
+ **({"timeout": remaining} if remaining is not None else {}),
656
+ ) as stream:
657
+ stream_entered = True
658
+ generation_id = _response_header(stream, "x-generation-id")
659
+ input_json: dict[int, list[str]] = {}
660
+ for event in stream:
661
+ if event.type == "text":
662
+ spinner.stop()
663
+ emitter.emit(
664
+ "model.output_delta",
665
+ {
666
+ "model_call_id": model_call_id,
667
+ "delta": event.text,
668
+ "source_timestamp": utc_now(),
669
+ },
670
+ )
671
+ elif (
672
+ event.type == "content_block_delta"
673
+ and event.delta.type == "input_json_delta"
674
+ ):
675
+ input_json.setdefault(event.index, []).append(
676
+ event.delta.partial_json
677
+ )
678
+ message = stream.get_final_message()
679
+ model_completed_ns = time.perf_counter_ns()
680
+ except DeadlineExceeded as exc:
681
+ emitter.emit("model.failed", {
682
+ "model_call_id": model_call_id,
683
+ "error_type": type(exc).__name__, "message": str(exc),
684
+ "generation_id": generation_id,
685
+ "duration_ms": (time.perf_counter_ns() - model_started_ns) / 1_000_000,
686
+ "will_retry": False, "retry_delay_seconds": 0,
687
+ "source_timestamp": utc_now(),
688
+ })
689
+ return "time_budget_exhausted"
690
+ except (anthropic.APIError, *HTTP_ERRORS) as exc:
691
+ now = time.monotonic()
692
+ if retry_deadline is None:
693
+ retry_deadline = now + STREAM_RETRY_WINDOW_SECONDS
694
+ delay = STREAM_RETRY_DELAYS[retries] if retries < len(STREAM_RETRY_DELAYS) else 0
695
+ retryable = (
696
+ stream_entered
697
+ and isinstance(exc, RETRYABLE_STREAM_ERRORS)
698
+ and retries < len(STREAM_RETRY_DELAYS)
699
+ and now + delay <= retry_deadline
700
+ )
701
+ will_retry = retryable and (deadline is None or now + delay < deadline)
702
+ emitter.emit(
703
+ "model.failed",
704
+ {
705
+ "model_call_id": model_call_id,
706
+ "error_type": type(exc).__name__,
707
+ "message": str(exc),
708
+ "generation_id": generation_id,
709
+ "duration_ms": (time.perf_counter_ns() - model_started_ns) / 1_000_000,
710
+ "will_retry": will_retry,
711
+ "retry_delay_seconds": delay if will_retry else 0,
712
+ "source_timestamp": utc_now(),
713
+ },
714
+ )
715
+ if not will_retry:
716
+ if deadline is not None and (now >= deadline or (retryable and now + delay >= deadline)):
717
+ return "time_budget_exhausted"
718
+ raise
719
+ # The retry delay fits inside both the recovery window and budget.
720
+ time.sleep(delay)
721
+ retries += 1
722
+ continue
723
+
724
+ retries = 0
725
+ retry_deadline = None
491
726
 
492
727
  content = _native_content_blocks(message.content)
728
+ input_errors: dict[str, str] = {}
729
+ invalid_json_ids: set[str] = set()
730
+ for index, block in enumerate(message.content):
731
+ if block.type != "tool_use":
732
+ continue
733
+ error = tool_input_error(block.name, block.input)
734
+ if index in input_json:
735
+ raw_json = "".join(input_json[index])
736
+ try:
737
+ json.loads(raw_json)
738
+ except json.JSONDecodeError:
739
+ # The SDK parses partial JSON while streaming. Even a
740
+ # complete-looking dict is not permission to execute an
741
+ # unfinished call after a provider reports tool_use.
742
+ error = (
743
+ "Invalid tool argument JSON: incomplete or malformed. "
744
+ "Resend a complete JSON object."
745
+ )
746
+ invalid_json_ids.add(block.id)
747
+ content[index]["input_json"] = raw_json
748
+ if error:
749
+ input_errors[block.id] = error
750
+ content[index]["input_error"] = error
493
751
  tool_calls = [
494
752
  item
495
753
  for item in content
@@ -510,6 +768,10 @@ def _run_model_loop(
510
768
  ),
511
769
  "generation_id": generation_id,
512
770
  "cost": usage_cost(usage if isinstance(usage, dict) else None)
771
+ or estimated_cost(
772
+ str(getattr(message, "model", None) or model),
773
+ usage if isinstance(usage, dict) else None,
774
+ )
513
775
  or pending_cost(generation_id),
514
776
  "duration_ms": (model_completed_ns - model_started_ns) / 1_000_000,
515
777
  "source_timestamp": utc_now(),
@@ -517,20 +779,58 @@ def _run_model_loop(
517
779
  emitter.emit("model.completed", payload)
518
780
 
519
781
  turns += 1
520
- messages.append({"role": "assistant", "content": message.content})
782
+ if deadline is not None and time.monotonic() >= deadline:
783
+ return "time_budget_exhausted"
784
+ if message.stop_reason == "max_tokens":
785
+ # Do not execute partial tool calls or replay them without results.
786
+ # Thinking may also be cut off before its signature arrives. Keep
787
+ # visible text and an explicit notice for the next interactive turn.
788
+ text = "".join(
789
+ block.text for block in message.content if block.type == "text"
790
+ )
791
+ messages.append(
792
+ {
793
+ "role": "assistant",
794
+ "content": (f"{text}\n\n" if text else "") + _TRUNCATION_NOTICE,
795
+ }
796
+ )
797
+ return "response_truncated"
798
+ request_content = []
799
+ for block in message.content:
800
+ if block.type == "tool_use" and (
801
+ block.id in invalid_json_ids or not isinstance(block.input, dict)
802
+ ):
803
+ block = copy(block)
804
+ block.input = {}
805
+ request_content.append(block)
806
+ messages.append({"role": "assistant", "content": request_content})
521
807
  if message.stop_reason != "tool_use":
522
- return True
808
+ return "completed"
523
809
  if max_turns is not None and turns >= max_turns:
524
810
  # Stop before running the tools: their results would only be
525
811
  # useful to a reply this budget can no longer pay for.
526
- return False
812
+ return "max_turns_exhausted"
527
813
  # Every tool_use block needs a matching tool_result in the next
528
814
  # user message, or the API rejects the request.
529
- results = [
530
- _run_one_tool(block, emitter, model_call_id)
531
- for block in message.content
532
- if block.type == "tool_use"
533
- ]
815
+ results = []
816
+ for block in message.content:
817
+ if block.type != "tool_use":
818
+ continue
819
+ if deadline is not None and time.monotonic() >= deadline:
820
+ # The reply or a preceding tool consumed the remaining time.
821
+ # Stop the run without starting another tool or model call.
822
+ return "time_budget_exhausted"
823
+ remaining = None if deadline is None else deadline - time.monotonic()
824
+ try:
825
+ with wall_clock_limit(remaining):
826
+ result = _run_one_tool(
827
+ block, emitter, model_call_id,
828
+ input_error=input_errors.get(block.id),
829
+ remaining_seconds=remaining,
830
+ )
831
+ except DeadlineExceeded:
832
+ return "time_budget_exhausted"
833
+ results.append(result)
534
834
  messages.append({"role": "user", "content": results})
535
835
 
536
836
 
@@ -538,13 +838,16 @@ def _reconcile_costs(
538
838
  client: anthropic.Anthropic,
539
839
  journal: EventJournal,
540
840
  emitter: EventEmitter,
841
+ *,
842
+ max_seconds: float | None = None,
541
843
  ) -> list[JsonObject]:
542
- """Append resolved OpenRouter costs without affecting the run outcome."""
844
+ """Reconcile pending or estimated costs without affecting the run outcome."""
543
845
  base_url = getattr(client, "base_url", "")
544
846
  credential = client.api_key or client.auth_token
545
847
  if not isinstance(credential, str) or not credential:
546
848
  return []
547
849
  outcomes: list[JsonObject] = []
850
+ deadline = None if max_seconds is None else time.monotonic() + max_seconds
548
851
  entries = EventJournal.replay(journal.path)
549
852
  already_resolved = {
550
853
  str(entry.payload["generation_id"])
@@ -552,24 +855,33 @@ def _reconcile_costs(
552
855
  if entry.type == "model.cost_resolved"
553
856
  }
554
857
  for entry in entries:
555
- if entry.type != "model.completed":
858
+ if entry.type not in {"model.completed", "model.failed"}:
556
859
  continue
557
860
  generation_id = entry.payload.get("generation_id")
558
- cost = entry.payload.get("cost")
861
+ cost = (
862
+ pending_cost(generation_id)
863
+ if entry.type == "model.failed"
864
+ else entry.payload.get("cost")
865
+ )
559
866
  if (
560
867
  not isinstance(generation_id, str)
561
868
  or generation_id in already_resolved
562
869
  or not isinstance(cost, dict)
563
- or cost.get("status") != "pending"
870
+ or (cost.get("status") != "pending" and cost.get("kind") != "estimated")
564
871
  ):
565
872
  continue
566
873
  diagnostics: list[JsonObject] = []
567
- resolved = resolve_generation_cost(
568
- base_url,
569
- generation_id,
570
- credential,
571
- diagnostics=diagnostics,
572
- )
874
+ if deadline is not None and time.monotonic() >= deadline:
875
+ break
876
+ try:
877
+ with wall_clock_limit(None if deadline is None else deadline - time.monotonic()):
878
+ resolved = resolve_generation_cost(
879
+ base_url, generation_id, credential, diagnostics=diagnostics,
880
+ )
881
+ except DeadlineExceeded:
882
+ outcomes.append({"generation_id": generation_id, "status": "unresolved",
883
+ "attempts": diagnostics, "reason": "finalization_deadline"})
884
+ break
573
885
  if resolved is not None:
574
886
  resolved["source_timestamp"] = utc_now()
575
887
  emitter.emit("model.cost_resolved", resolved)
@@ -583,20 +895,22 @@ def _reconcile_costs(
583
895
  return outcomes
584
896
 
585
897
 
586
- def run() -> int:
898
+ def run(*, max_tokens: int | None = None) -> int:
587
899
  """Start the read → ask → answer loop until the user types ``/exit``.
588
900
 
589
901
  A reply may include tool calls; they are executed and their results
590
902
  fed back to the model until it finishes the turn without tool use.
591
903
  Returns the process exit code.
592
904
  """
905
+ max_tokens = resolve_max_tokens(max_tokens)
593
906
  client = _create_client()
594
907
  if client is None:
595
908
  return 1
596
909
 
597
910
  model = _resolve_model()
598
911
  print(
599
- f"nanoPyCodeAgent v{_package_version()} — model {model} "
912
+ f"nanoPyCodeAgent v{_package_version()} — model {model}, "
913
+ f"max tokens {max_tokens} "
600
914
  "(set ANTHROPIC_MODEL to override)."
601
915
  )
602
916
  print("Type a message to chat, or /exit to quit.")
@@ -618,7 +932,7 @@ def run() -> int:
618
932
  break
619
933
 
620
934
  messages.append({"role": "user", "content": user_input})
621
- _run_exchange(client, model, messages, SYSTEM_PROMPT)
935
+ _run_exchange(client, model, messages, SYSTEM_PROMPT, max_tokens=max_tokens)
622
936
 
623
937
  print("Bye!")
624
938
  return 0
@@ -628,6 +942,8 @@ def run_headless(
628
942
  task: str,
629
943
  *,
630
944
  max_turns: int = DEFAULT_MAX_TURNS,
945
+ max_tokens: int | None = None,
946
+ time_budget_seconds: int | None = None,
631
947
  trajectory_path: Path | None = None,
632
948
  ) -> int:
633
949
  """Work ``task`` to completion without a user, and return the exit code.
@@ -640,6 +956,7 @@ def run_headless(
640
956
  scores the result. Only a run that could not happen at all — no
641
957
  credentials, an API that keeps refusing — exits non-zero.
642
958
  """
959
+ max_tokens = resolve_max_tokens(max_tokens)
643
960
  client = _create_client()
644
961
  if client is None:
645
962
  return 1
@@ -649,32 +966,41 @@ def run_headless(
649
966
  # model's prose and the echoed tool calls, nothing else.
650
967
  print(
651
968
  f"nanoPyCodeAgent v{_package_version()} — model {model}, "
652
- f"max turns {max_turns}",
969
+ f"max turns {max_turns}, max tokens {max_tokens}, "
970
+ f"time budget {time_budget_seconds if time_budget_seconds else 'none'}",
653
971
  file=sys.stderr,
654
972
  )
655
973
 
656
974
  messages: list[MessageParam] = [{"role": "user", "content": task}]
657
975
  try:
658
- finished = _run_exchange(
976
+ outcome = _run_exchange(
659
977
  client,
660
978
  model,
661
979
  messages,
662
980
  HEADLESS_SYSTEM_PROMPT,
663
981
  max_turns=max_turns,
982
+ max_tokens=max_tokens,
983
+ time_budget_seconds=time_budget_seconds,
664
984
  reply_prefix="",
665
985
  trajectory_path=trajectory_path,
666
986
  )
667
- except (anthropic.APIError, httpx.HTTPError) as exc:
987
+ except (anthropic.APIError, *HTTP_ERRORS) as exc:
668
988
  # Printed verbatim on purpose: a harness classifies a failed run by
669
989
  # pattern-matching this text (rate limit, overloaded, context length,
670
990
  # …) to decide whether retrying is worth anything. Rewording it, or
671
991
  # swallowing it, throws that away.
672
992
  print(f"API error: {exc}", file=sys.stderr)
673
993
  return 1
674
- if not finished:
994
+ if outcome == "max_turns_exhausted":
675
995
  turns = "turn" if max_turns == 1 else "turns"
676
996
  print(
677
997
  f"[stopped after {max_turns} {turns} without finishing the task]",
678
998
  file=sys.stderr,
679
999
  )
1000
+ elif outcome == "time_budget_exhausted":
1001
+ print(
1002
+ f"[stopped after the {time_budget_seconds}s time budget without "
1003
+ "finishing the task]",
1004
+ file=sys.stderr,
1005
+ )
680
1006
  return 0