wrencode 0.1.4.3__tar.gz → 0.1.4.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: wrencode
3
- Version: 0.1.4.3
3
+ Version: 0.1.4.5
4
4
  Summary: A minimal agentic coding assistant in a single Python file
5
5
  Project-URL: Homepage, https://github.com/almostly/wrencode
6
6
  Project-URL: Repository, https://github.com/almostly/wrencode
@@ -72,9 +72,9 @@ for _dir in (os.path.dirname(os.path.abspath(__file__)), os.getcwd()):
72
72
  os.environ.setdefault(_k.strip(), _v)
73
73
 
74
74
  # -----------------------------------------------------------------------------------------------
75
- # Backend Configuration
75
+ # Backend configuration
76
76
  # -----------------------------------------------------------------------------------------------
77
- WRENCODE_VERSION = "0.1.4.3"
77
+ WRENCODE_VERSION = "0.1.4.5"
78
78
 
79
79
  # Per-backend defaults. "kind" controls how a backend is treated:
80
80
  # api - hosted HTTP API, needs an API key
@@ -220,9 +220,9 @@ def apply_backend(backend: str, model: str = "", api_key: str = "") -> None:
220
220
 
221
221
 
222
222
  # -----------------------------------------------------------------------------------------------
223
- # Constants & Environment Variables
223
+ # Constants & environment variables
224
224
  # -----------------------------------------------------------------------------------------------
225
- MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "4096"))
225
+ MAX_TOKENS = int(os.environ.get("MAX_TOKENS", "8192"))
226
226
  MAX_READ_BYTES = int(os.environ.get("MAX_READ_BYTES", str(4 * 1024 * 1024)))
227
227
  MAX_READ_LINES = int(os.environ.get("MAX_READ_LINES", "800"))
228
228
  GREP_MAX = int(os.environ.get("GREP_MAX_MATCHES", "80"))
@@ -239,7 +239,7 @@ _GLOB_SKIP: set[str] = {
239
239
  }
240
240
 
241
241
  # -----------------------------------------------------------------------------------------------
242
- # Terminal Colors
242
+ # Terminal colors
243
243
  # -----------------------------------------------------------------------------------------------
244
244
  RESET, BOLD, DIM = "\033[0m", "\033[1m", "\033[2m"
245
245
  BLUE, CYAN, GREEN, YELLOW, RED = (
@@ -505,7 +505,7 @@ WREN_BANNER = f"""{BRIGHT_CYAN}\u2588\u2588 \u2588\u2588 \u2588\u2588\u2588\
505
505
 
506
506
 
507
507
  # -----------------------------------------------------------------------------------------------
508
- # Path Helpers
508
+ # Path helpers
509
509
  # -----------------------------------------------------------------------------------------------
510
510
  def workspace_root() -> pathlib.Path:
511
511
  """Return the resolved workspace root path from env or cwd."""
@@ -533,7 +533,7 @@ def resolve_tool_path(raw: Any) -> pathlib.Path:
533
533
 
534
534
 
535
535
  # -----------------------------------------------------------------------------------------------
536
- # Input Validation
536
+ # Input validation
537
537
  # -----------------------------------------------------------------------------------------------
538
538
  def _require_str(args: dict[str, Any], key: str) -> str:
539
539
  """Require a non-empty string value from args dict by key."""
@@ -600,7 +600,9 @@ def read(args: dict[str, Any]) -> str:
600
600
  def write(args: dict[str, Any]) -> str:
601
601
  """Write content to a file, creating parent directories as needed."""
602
602
  path = resolve_tool_path(_require_str(args, "path"))
603
- content = args.get("content", "")
603
+ if "content" not in args:
604
+ return "error: 'content' is required (model response may have been truncated; raise MAX_TOKENS)"
605
+ content = args["content"]
604
606
  approval = confirm("write")
605
607
  if approval != "ok":
606
608
  return approval
@@ -1029,7 +1031,7 @@ def run_tool(name: str, args: dict[str, Any]) -> str:
1029
1031
 
1030
1032
 
1031
1033
  # -----------------------------------------------------------------------------------------------
1032
- # Message Formatting
1034
+ # Message formatting
1033
1035
  # -----------------------------------------------------------------------------------------------
1034
1036
  def flatten_content(content: Any) -> str:
1035
1037
  """Flatten Anthropic-style content list to plain string."""
@@ -1141,7 +1143,7 @@ def thinking_spinner() -> Any:
1141
1143
 
1142
1144
 
1143
1145
  # -----------------------------------------------------------------------------------------------
1144
- # Token & Output Cleaning
1146
+ # Token & output cleaning
1145
1147
  # -----------------------------------------------------------------------------------------------
1146
1148
  def strip_gptoss_tokens(text: str) -> str:
1147
1149
  """Strip GPT-OSS special tokens and channel markers from model output."""
@@ -1242,7 +1244,7 @@ def parse_tool_calls(text: str) -> list[dict[str, Any]]:
1242
1244
 
1243
1245
 
1244
1246
  # -----------------------------------------------------------------------------------------------
1245
- # Anthropic Native Tools
1247
+ # Anthropic native tools
1246
1248
  # -----------------------------------------------------------------------------------------------
1247
1249
  _TYPE_MAP: dict[str, str] = {
1248
1250
  "string": "string",
@@ -1276,8 +1278,54 @@ def _openai_headers() -> dict[str, str]:
1276
1278
  return {"Content-Type": "application/json", "Authorization": f"Bearer {API_KEY}"}
1277
1279
 
1278
1280
 
1281
+ def _warn_if_truncated(data: dict[str, Any]) -> None:
1282
+ """Print a stderr warning if the API truncated the response at max_tokens.
1283
+ A truncated `tool_use` may be missing required fields (e.g. `content` for
1284
+ `write`), so callers need to know to raise MAX_TOKENS rather than silently
1285
+ treating an empty operation as success.
1286
+ """
1287
+ if BACKEND == "anthropic":
1288
+ truncated = data.get("stop_reason") == "max_tokens"
1289
+ out_tokens = data.get("usage", {}).get("output_tokens")
1290
+ else:
1291
+ finish = (data.get("choices") or [{}])[0].get("finish_reason")
1292
+ truncated = finish == "length"
1293
+ out_tokens = data.get("usage", {}).get("completion_tokens")
1294
+ if truncated:
1295
+ print(
1296
+ f"{YELLOW}Warning: response truncated at MAX_TOKENS={MAX_TOKENS} "
1297
+ f"(output_tokens={out_tokens}). Tool calls may be incomplete — "
1298
+ f"raise MAX_TOKENS and retry.{RESET}",
1299
+ file=sys.stderr,
1300
+ )
1301
+
1302
+
1303
+ def _log_usage_debug(data: dict[str, Any]) -> None:
1304
+ """When WRENCODE_DEBUG=1, print one stderr line per turn with token usage.
1305
+ Surfaces cache_creation_input_tokens / cache_read_input_tokens so prompt-
1306
+ cache hit rate is observable without external tooling.
1307
+ """
1308
+ if not os.environ.get("WRENCODE_DEBUG"):
1309
+ return
1310
+ u = data.get("usage", {})
1311
+ if BACKEND == "anthropic":
1312
+ print(
1313
+ f"{DIM}usage: in={u.get('input_tokens')} out={u.get('output_tokens')} "
1314
+ f"cache_write={u.get('cache_creation_input_tokens', 0)} "
1315
+ f"cache_read={u.get('cache_read_input_tokens', 0)}{RESET}",
1316
+ file=sys.stderr,
1317
+ )
1318
+ else:
1319
+ print(
1320
+ f"{DIM}usage: in={u.get('prompt_tokens')} out={u.get('completion_tokens')}{RESET}",
1321
+ file=sys.stderr,
1322
+ )
1323
+
1324
+
1279
1325
  def _parse_native_response(data: dict[str, Any]) -> tuple[str, list["ToolCall"]]:
1280
1326
  """Parse a native (Anthropic / OpenAI) API response into display text and tool calls."""
1327
+ _warn_if_truncated(data)
1328
+ _log_usage_debug(data)
1281
1329
  if BACKEND == "anthropic":
1282
1330
  blocks = data.get("content", [])
1283
1331
  text = "\n".join(b["text"] for b in blocks if b.get("type") == "text").strip()
@@ -1361,7 +1409,7 @@ def _append_tool_results(
1361
1409
 
1362
1410
 
1363
1411
  # -----------------------------------------------------------------------------------------------
1364
- # HTTP Helper
1412
+ # HTTP helper
1365
1413
  # -----------------------------------------------------------------------------------------------
1366
1414
  def _http_post(
1367
1415
  url: str, payload: dict[str, Any], headers: dict[str, str]
@@ -1390,7 +1438,7 @@ def get_response(
1390
1438
  for m in messages
1391
1439
  ]
1392
1440
 
1393
- # OpenAI — native function calling
1441
+ # OpenAI - native function calling
1394
1442
  if BACKEND == "openai":
1395
1443
  data = _http_post(
1396
1444
  API_BASE,
@@ -1407,7 +1455,7 @@ def get_response(
1407
1455
  )
1408
1456
  return json.dumps(data) # return raw for agent loop to parse natively
1409
1457
 
1410
- # OpenRouter / Ollama — OpenAI-compatible chat completions (no native tools)
1458
+ # OpenRouter / Ollama - OpenAI-compatible chat completions (no native tools)
1411
1459
  if BACKEND in {"openrouter", "ollama"}:
1412
1460
  data = _http_post(
1413
1461
  API_BASE,
@@ -1422,22 +1470,30 @@ def get_response(
1422
1470
  )
1423
1471
  return str(data["choices"][0]["message"]["content"])
1424
1472
 
1425
- # Anthropic — native tool use API
1473
+ # Anthropic - native tool use API
1426
1474
  if BACKEND == "anthropic":
1475
+ # Prompt caching: mark the (large, static) system block + tool schemas as
1476
+ # ephemeral. Anthropic caches everything up to and including each marker
1477
+ # for ~5 minutes; subsequent turns in the same session read at ~10% of
1478
+ # normal input cost. Two breakpoints out of the four allowed per request.
1479
+ tools = _build_tool_schemas("anthropic")
1480
+ if tools:
1481
+ tools[-1] = {**tools[-1], "cache_control": {"type": "ephemeral"}}
1427
1482
  data = _http_post(
1428
1483
  API_BASE,
1429
1484
  {
1430
1485
  "model": MODEL,
1431
- "system": system_prompt,
1486
+ "system": [{"type": "text", "text": system_prompt,
1487
+ "cache_control": {"type": "ephemeral"}}],
1432
1488
  "messages": messages,
1433
1489
  "max_tokens": MAX_TOKENS,
1434
- "tools": _build_tool_schemas("anthropic"),
1490
+ "tools": tools,
1435
1491
  },
1436
1492
  _anthropic_headers(),
1437
1493
  )
1438
1494
  return json.dumps(data) # return raw for agent loop to parse natively
1439
1495
 
1440
- # Local proxy — Anthropic messages API; tool calls returned as XML <tool_call> tags in text
1496
+ # Local proxy - Anthropic messages API; tool calls returned as XML <tool_call> tags in text
1441
1497
  if BACKEND == "local":
1442
1498
  data = _http_post(
1443
1499
  API_BASE,
@@ -1509,7 +1565,7 @@ def get_response(
1509
1565
 
1510
1566
 
1511
1567
  # -----------------------------------------------------------------------------------------------
1512
- # History Management
1568
+ # History management
1513
1569
  # -----------------------------------------------------------------------------------------------
1514
1570
  def history_file_path() -> pathlib.Path:
1515
1571
  """Return the history file path from env override or user-level default."""
@@ -1609,7 +1665,7 @@ def compact_messages(
1609
1665
 
1610
1666
 
1611
1667
  # -----------------------------------------------------------------------------------------------
1612
- # Workspace & System Prompt
1668
+ # Workspace & system prompt
1613
1669
  # -----------------------------------------------------------------------------------------------
1614
1670
  def git_context() -> str:
1615
1671
  """Return a formatted git status string if inside a git repository."""
@@ -1661,7 +1717,7 @@ CRITICAL: You MUST use tools for file operations. Never say you can't access fil
1661
1717
 
1662
1718
 
1663
1719
  # -----------------------------------------------------------------------------------------------
1664
- # Agent Loop
1720
+ # Agentic loop
1665
1721
  # -----------------------------------------------------------------------------------------------
1666
1722
  def _track_error(
1667
1723
  result: str, last: Optional[str], count: int
@@ -1730,7 +1786,7 @@ def run_agent_turn(
1730
1786
 
1731
1787
 
1732
1788
  # -----------------------------------------------------------------------------------------------
1733
- # Slash Commands
1789
+ # Slash commands
1734
1790
  # -----------------------------------------------------------------------------------------------
1735
1791
  def handle_slash_command(
1736
1792
  cmd: str,
@@ -1781,7 +1837,7 @@ def handle_slash_command(
1781
1837
 
1782
1838
 
1783
1839
  # -----------------------------------------------------------------------------------------------
1784
- # Backend Selection & Config Persistence
1840
+ # Backend selection & config persistence
1785
1841
  # -----------------------------------------------------------------------------------------------
1786
1842
  def is_frozen() -> bool:
1787
1843
  """Set true when running as a PyInstaller standalone binary."""
@@ -2134,7 +2190,7 @@ def resolve_configuration(force_chooser: bool = False) -> None:
2134
2190
 
2135
2191
 
2136
2192
  # -----------------------------------------------------------------------------------------------
2137
- # Model Loading
2193
+ # Model loading
2138
2194
  # -----------------------------------------------------------------------------------------------
2139
2195
  def load_model() -> Optional[tuple[Any, Any]]:
2140
2196
  """Load model for the current backend and return mlx_state (or None for API backends)."""
@@ -2222,7 +2278,7 @@ def load_model() -> Optional[tuple[Any, Any]]:
2222
2278
 
2223
2279
 
2224
2280
  # -----------------------------------------------------------------------------------------------
2225
- # Entry Point
2281
+ # Entrypoint
2226
2282
  # -----------------------------------------------------------------------------------------------
2227
2283
  def print_help() -> None:
2228
2284
  """Print CLI usage."""
File without changes
File without changes
File without changes