@tiens.nguyen/gonext-local-worker 1.0.327 → 1.0.330

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/gonext_agent_chat.py +190 -13
  2. package/package.json +2 -1
@@ -1528,6 +1528,49 @@ def _is_backend_overloaded(err) -> bool:
1528
1528
  )
1529
1529
 
1530
1530
 
1531
+ def _is_auth_failure(err) -> bool:
1532
+ """True when the model server REJECTED OUR CREDENTIAL — a deterministic config error
1533
+ that must fail fast rather than retry (task #117 companion).
1534
+
1535
+ Deliberately narrow:
1536
+ • 401 is always an auth failure.
1537
+ • 403 is NOT, on its own. Several providers return 403 for QUOTA EXHAUSTED as well as
1538
+ for permission-denied, and reporting "you're out of credit" as "your API key is
1539
+ wrong" sends the user to fix the wrong thing. A 403 counts only when the body does
1540
+ NOT read like a quota/rate problem — those belong to _is_backend_overloaded.
1541
+ • 402 (payment required) is a billing problem, not a credential one — excluded here
1542
+ so it can carry its own message.
1543
+ Falls back to message text only when the exception carries no HTTP status (a bare
1544
+ string, or a wrapper that lost its cause)."""
1545
+ if _is_backend_overloaded(err):
1546
+ return False # a 429 (or a quota-flavoured 403) is never an auth failure
1547
+ s = str(err or "").lower()
1548
+ code = _http_status_of(err)
1549
+ if code is not None:
1550
+ if code == 401:
1551
+ return True
1552
+ if code == 403:
1553
+ return not re.search(r"quota|rate.?limit|billing|credit|capacity", s)
1554
+ return False
1555
+ return (
1556
+ "401" in s
1557
+ or "invalid_api_key" in s
1558
+ or "invalid api key" in s
1559
+ or "incorrect api key" in s
1560
+ or "unauthorized" in s
1561
+ or "authentication" in s and "fail" in s
1562
+ )
1563
+
1564
+
1565
+ def _is_billing_failure(err) -> bool:
1566
+ """True for 402 / explicit out-of-credit responses — deterministic like an auth failure
1567
+ but with a different fix, so it gets its own message instead of 'check your API key'."""
1568
+ if _http_status_of(err) == 402:
1569
+ return True
1570
+ s = str(err or "").lower()
1571
+ return "insufficient_quota" in s or "exceeded your current quota" in s
1572
+
1573
+
1531
1574
  def _retry_after_seconds(err):
1532
1575
  """The provider's Retry-After hint in seconds, or None. Reads the header off an
1533
1576
  openai-python APIStatusError (`err.response.headers`) and falls back to a
@@ -1585,12 +1628,15 @@ def _is_backend_unavailable(err) -> bool:
1585
1628
  # down". Checked FIRST because a 429 body can also contain the word "timeout".
1586
1629
  if _is_backend_overloaded(err):
1587
1630
  return False
1588
- # A structured 5xx IS a gateway/backend outage regardless of wording: a provider that
1589
- # returns `Error code: 502 - {"error": …}` carries no "bad gateway" text for the
1631
+ # A structured gateway/timeout status IS an outage regardless of wording: a provider
1632
+ # that returns `Error code: 502 - {"error": …}` carries no "bad gateway" text for the
1590
1633
  # phrase list below to match, and would otherwise be misfiled as a generic error.
1634
+ # 408 and Cloudflare's 522/524 are included so a real timeout STATUS keeps the same
1635
+ # answer the "timed out" text below has always given it — without them, adding the
1636
+ # structured check would have quietly demoted those.
1591
1637
  code = _http_status_of(err)
1592
1638
  if code is not None:
1593
- return code in (502, 503, 504)
1639
+ return code in (502, 503, 504, 408, 522, 524)
1594
1640
  return (
1595
1641
  "502 bad gateway" in s
1596
1642
  or "503 service" in s
@@ -1859,12 +1905,68 @@ def _looks_like_tool_code(code: str):
1859
1905
  return True, ""
1860
1906
 
1861
1907
 
1908
+ _CODE_PAIR_RE = r"<code[^>]*>[\s\S]*?</code>"
1909
+
1910
+
1911
+ def _close_dangling_code_block(text):
1912
+ """Close a code block the model OPENED but never CLOSED, when what follows it is a
1913
+ runnable tool call — the normal shape whenever the server honors `stop` (task #119).
1914
+
1915
+ smolagents appends its closing tag to the stop_sequences of every CodeAgent turn
1916
+ (agents.py: `stop_sequences.append(self.code_block_tags[1])`), so any backend that
1917
+ honors `stop` — i.e. every OpenAI-compatible API — returns the block WITHOUT its
1918
+ closer. Confirmed against the raw responses saved from a live run: what the model
1919
+ actually sends is `<code>\\nrun_command("npm run build", …)` and nothing after it.
1920
+ smolagents re-appends the tag itself once it has the text (agents.py, "This adds the
1921
+ end code sequence … to the history"), which is why a trace shows a tidy `…</code>`
1922
+ that never came off the wire — but every check WE run in between tested for the PAIR
1923
+ `<code>…</code>`, a delimiter this backend structurally cannot produce.
1924
+
1925
+ What that cost, live: a perfectly good turn was read as a "dangling code tag", burned
1926
+ a second model call on the CODE-NOW retry, and the retry's equally good opener-only
1927
+ reply was then DISCARDED for failing the same pair test — so every other step died on
1928
+ a parse error the model never caused, while the user watched "Error in code parsing"
1929
+ scroll past on steps that had in fact called the right tool.
1930
+
1931
+ So: canonicalize first, decide second. Everything after the LAST opener is taken as
1932
+ the block, a trailing "Thought:" line and stray tags are cut, and the result is
1933
+ substituted ONLY when _looks_like_tool_code passes (it compiles AND calls something)
1934
+ — the same #113 validity gate used for reasoning-channel recovery, so prose that
1935
+ merely trails a stray tag is never promoted into executable code.
1936
+
1937
+ No-op when the text already holds a pair, has no opener, or the tail isn't runnable;
1938
+ idempotent. Deliberately backend-agnostic: `stop` truncation is a property of the
1939
+ protocol, not of one vendor, and the gate makes it inert everywhere else."""
1940
+ if not isinstance(text, str) or not text:
1941
+ return text
1942
+ if re.search(_CODE_PAIR_RE, text, re.I):
1943
+ return text # a real block already — never touch it
1944
+ opens = list(re.finditer(r"<code[^>]*>", text, re.I))
1945
+ if not opens:
1946
+ return text # prose turn, or a lone stray closer (see #111)
1947
+ last = opens[-1]
1948
+ body = re.split(r"\n\s*Thought\s*:\s", text[last.end():], maxsplit=1)[0]
1949
+ body = re.sub(r"</?code[^>]*>", "", body, flags=re.I).strip()
1950
+ ok, _why = _looks_like_tool_code(body)
1951
+ if not ok:
1952
+ return text # not runnable — leave the step to fail honestly
1953
+ # Keep the Thought prose that preceded the block, minus any earlier stray tag (an
1954
+ # unbalanced opener left in the head would let the pair regex start from the wrong
1955
+ # place and swallow the prose into the code).
1956
+ head = re.sub(r"</?code[^>]*>", "", text[:last.start()], flags=re.I).rstrip()
1957
+ return (head + "\n" if head else "") + "<code>\n" + body + "\n</code>"
1958
+
1959
+
1862
1960
  def _has_runnable_block(text) -> bool:
1863
1961
  """True if `text` contains a <code>…</code> block whose python actually compiles —
1864
1962
  i.e. this step produced something that can run. Used by the #113 no-code loop-breaker
1865
- (openai backend only) to tell "the step worked" from "the model wrote prose again"."""
1963
+ (openai backend only) to tell "the step worked" from "the model wrote prose again".
1964
+
1965
+ Closes a stop-truncated block first (#119), so a step that DID call a tool is never
1966
+ counted toward the no-code streak just because the API ate the closing tag."""
1866
1967
  if not isinstance(text, str) or not text:
1867
1968
  return False
1969
+ text = _close_dangling_code_block(text)
1868
1970
  blk = re.search(r"<code[^>]*>([\s\S]*?)</code>", text, re.I)
1869
1971
  if not blk:
1870
1972
  return False
@@ -1881,7 +1983,10 @@ def _code_nudge_reason(content, openai_backend: bool):
1881
1983
  replayed offline against real traces (task #113).
1882
1984
 
1883
1985
  1. empty-content — the whole turn went to the reasoning channel, no message.
1884
- 2. dangling tag — a stray <code>/</code> with no valid pair.
1986
+ 2. dangling tag — a stray <code>/</code> with no valid pair. A block the server
1987
+ merely TRUNCATED at the `</code>` stop sequence is not that (task #119): it is
1988
+ closed first and accepted as-is, so a step that already called a tool no longer
1989
+ pays for a second model call to be told to do what it just did.
1885
1990
  3. prose-only plan — narration of what it is ABOUT to do, no tag at all. This one
1886
1991
  is openai-only and additionally requires _looks_like_scratchpad, so a FINISHED
1887
1992
  prose answer is never nudged into inventing another tool call. It exists because
@@ -1893,8 +1998,8 @@ def _code_nudge_reason(content, openai_backend: bool):
1893
1998
  text = (content or "").strip()
1894
1999
  if not text:
1895
2000
  return "empty-content"
1896
- if re.search(r"<code[^>]*>[\s\S]*?</code>", content, re.I):
1897
- return None # a real block — nothing to nudge
2001
+ if re.search(_CODE_PAIR_RE, _close_dangling_code_block(content), re.I):
2002
+ return None # a real block (closing a stop-truncated one first) — nothing to nudge
1898
2003
  if re.search(r"</?code[\s>/]", content, re.I):
1899
2004
  return "dangling code tag, no valid block"
1900
2005
  if openai_backend and _looks_like_scratchpad(text):
@@ -1919,9 +2024,12 @@ def _recover_code_from_reasoning(content, reasoning):
1919
2024
  Candidates are tried NEWEST-first and the first runnable one wins; if none is
1920
2025
  runnable we return `content` untouched and let the step fail normally, which
1921
2026
  is what happened before #111 existed."""
1922
- _pair = r"<code[^>]*>[\s\S]*?</code>"
1923
- if re.search(_pair, content or "", re.I):
1924
- return content # content already has a usable block — never override it
2027
+ _pair = _CODE_PAIR_RE
2028
+ if re.search(_pair, _close_dangling_code_block(content) or "", re.I):
2029
+ # Content already has a usable block — never override it. A block the server
2030
+ # truncated at the `</code>` stop sequence counts as usable (#119): the visible
2031
+ # channel is the model's actual answer and must win over a reasoning-channel draft.
2032
+ return content
1925
2033
  reasoning = reasoning or ""
1926
2034
  if not reasoning.strip():
1927
2035
  return content # nothing to recover from (Ollama / non-reasoning model)
@@ -5323,6 +5431,14 @@ def run_agent_chat(cfg):
5323
5431
  _log(f"streamed generate refused ({_clip(str(e), 200)}) → NOT falling "
5324
5432
  "back to non-streaming; letting the retry loop back off (#114)")
5325
5433
  raise
5434
+ if _is_auth_failure(e) or _is_billing_failure(e):
5435
+ # Same reasoning as the overload above: the credential was rejected, so
5436
+ # streaming was never the problem. Re-sending unstreamed would cost a
5437
+ # second identical request per attempt AND print "Streaming unavailable
5438
+ # — retrying without streaming…", which is simply untrue here.
5439
+ _log(f"streamed generate rejected ({_clip(str(e), 200)}) → NOT falling "
5440
+ "back to non-streaming; this is a credential/billing error")
5441
+ raise
5326
5442
  _log(f"streamed generate failed ({e}) → falling back to non-streaming")
5327
5443
  _emit({"type": "step",
5328
5444
  "text": "Streaming unavailable — retrying without streaming…"})
@@ -5454,6 +5570,12 @@ def run_agent_chat(cfg):
5454
5570
  _emit({"type": "step", "text": "Model output ran away — restarting this step…"})
5455
5571
  _log(f"streamed generate: discarded {len(content)} runaway chars → empty content")
5456
5572
  return "", role, in_tok, out_tok, reasoning_len
5573
+ # Task #119: smolagents passes its closing "</code>" as a stop sequence, so a
5574
+ # server that honors `stop` returns the block UNCLOSED. Close it here, before
5575
+ # anything downstream asks "did this turn produce a code block?" — otherwise a
5576
+ # step that called a tool correctly reads as broken. Gated on the code actually
5577
+ # compiling, so a genuinely dangling tag still falls through to the nudge below.
5578
+ content = _close_dangling_code_block(content)
5457
5579
  # Task #111: reasoning models on an OpenAI-compatible API (e.g. Kimi K3 @ Moonshot)
5458
5580
  # can emit the actionable <code>…</code> into the REASONING channel, leaving only
5459
5581
  # prose + a stray tag in `content` → smolagents' parser can't find the pair. When
@@ -5519,7 +5641,11 @@ def run_agent_chat(cfg):
5519
5641
  # in the trace is appended by SMOLAGENTS after we return (its stop-sequence
5520
5642
  # handling), so the content WE see here carries no tag at all. Openai-only.
5521
5643
  # See _code_nudge_reason for the full decision (kept pure so it can be replayed).
5522
- _pair_re = r"<code[^>]*>[\s\S]*?</code>"
5644
+ # NOTE (#119): none of these triggers may fire on a block the server merely
5645
+ # truncated at the "</code>" stop sequence — that is the SUCCESS shape for an
5646
+ # OpenAI-compatible backend, not a failure. _close_dangling_code_block runs in
5647
+ # _run_stream (and again inside _code_nudge_reason) so it is already excluded.
5648
+ _pair_re = _CODE_PAIR_RE
5523
5649
  why = _code_nudge_reason(content, self._openai_backend)
5524
5650
  if why:
5525
5651
  _log(f"{why} turn (reasoning={reasoning_len} chars) → retrying once with a "
@@ -5531,7 +5657,13 @@ def run_agent_chat(cfg):
5531
5657
  {"role": "user", "content": self._CODE_NOW_DIRECTIVE}
5532
5658
  ]
5533
5659
  r_content, r_role, r_in, r_out, _ = self._run_stream(retry_kwargs)
5534
- r_has_pair = re.search(_pair_re, r_content or "", re.I) is not None
5660
+ # Close a stop-truncated block before judging the retry (#119). This test
5661
+ # used to demand a literal pair, which the openai backend cannot return —
5662
+ # so the retry we had just paid for was thrown away and the broken original
5663
+ # was sent on to smolagents. (_run_stream already closed it; the call here
5664
+ # is idempotent and keeps the rule visible at the point of decision.)
5665
+ r_has_pair = re.search(
5666
+ _pair_re, _close_dangling_code_block(r_content) or "", re.I) is not None
5535
5667
  # Use the retry when it yields a real code block, or (empty original) any
5536
5668
  # content. If the retry is still broken and we HAD a Thought, keep the
5537
5669
  # original and let smolagents' own retry take it from here — no regression.
@@ -5542,7 +5674,13 @@ def run_agent_chat(cfg):
5542
5674
  out_tok += r_out
5543
5675
  if stop_sequences and not self.supports_stop_parameter:
5544
5676
  from smolagents.models import remove_content_after_stop_sequences
5677
+ # For a server that can't honor `stop`, smolagents emulates it by splitting
5678
+ # on each stop sequence and keeping split[0] — and "</code>" IS one of those
5679
+ # sequences, so this CUTS THE CLOSER back off a block that had one. Re-close
5680
+ # afterwards (#119) so both kinds of backend hand back the same canonical
5681
+ # shape; without it the emulated path re-creates the exact bug we just fixed.
5545
5682
  content = remove_content_after_stop_sequences(content, stop_sequences)
5683
+ content = _close_dangling_code_block(content)
5546
5684
  return ChatMessage(
5547
5685
  role=role,
5548
5686
  content=content,
@@ -5647,8 +5785,47 @@ def run_agent_chat(cfg):
5647
5785
  # for this server, e.g. an MLX-style name against Ollama) —
5648
5786
  # deterministic, so retrying is pointless. Fail fast with the
5649
5787
  # server's actual model list so the user can fix Settings.
5788
+ # A rejected CREDENTIAL is deterministic — retrying re-sends the
5789
+ # same bad key, and letting it fall through to the turn-level
5790
+ # handler would degrade to _plain_reply, i.e. the chat model
5791
+ # answering confidently about work that never ran (#94's failure
5792
+ # mode). Fail fast with a message that says WHICH key went out:
5793
+ # python uses `coding_api_key or agent_api_key`, so an unset coding
5794
+ # key silently authenticates a cloud endpoint with the LOCAL agent
5795
+ # key (the Auto+cloud path in #117) — a completely different fix
5796
+ # from "your coding key is wrong". The key itself is never echoed.
5797
+ if _is_auth_failure(e) or _is_billing_failure(e):
5798
+ _host = _host_of(coding_base_url)
5799
+ if _is_billing_failure(e):
5800
+ raise _AgentConfigError(
5801
+ f"The agent coding backend ({_host}) accepted the key "
5802
+ "but refused the request for BILLING reasons (out of "
5803
+ "credit / quota exhausted). Nothing ran. Top up the "
5804
+ "account, or point 'Agent coding model' at another "
5805
+ "backend in Settings → Agent."
5806
+ ) from e
5807
+ if coding_api_key:
5808
+ raise _AgentConfigError(
5809
+ f"The agent coding backend ({_host}) REJECTED the "
5810
+ "coding API key. Nothing ran. Re-enter 'Agent coding "
5811
+ "model API key' in Settings → Agent (the stored key is "
5812
+ "write-only, so a stale one can't be inspected — just "
5813
+ "save a fresh one)."
5814
+ ) from e
5815
+ raise _AgentConfigError(
5816
+ f"The agent coding backend ({_host}) rejected the request: "
5817
+ "NO coding API key was sent, so the local agent key was "
5818
+ "used instead. Set 'Coding backend' to OpenAI-compatible "
5819
+ "in Settings → Agent and enter the API key — the key field "
5820
+ "only appears once that backend is selected."
5821
+ ) from e
5650
5822
  if "404" in emsg and "not found" in emsg.lower():
5651
- ids = _list_model_ids(coding_base_url, agent_api_key)
5823
+ # Probe with the CODING key when there is one: using the agent
5824
+ # key here 401s against a cloud coder, so the "This server has:
5825
+ # …" hint — the only useful part of this error — silently
5826
+ # vanished exactly when it was needed.
5827
+ ids = _list_model_ids(coding_base_url,
5828
+ coding_api_key or agent_api_key)
5652
5829
  have = f" This server has: {', '.join(ids)}." if ids else ""
5653
5830
  raise _AgentConfigError(
5654
5831
  f"The agent coding model {coding_model_id!r} does not "
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tiens.nguyen/gonext-local-worker",
3
- "version": "1.0.327",
3
+ "version": "1.0.330",
4
4
  "description": "Polls GoNext cloud API for async local LLM jobs and runs them against Ollama/OpenAI-compatible servers on this Mac",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -17,6 +17,7 @@
17
17
  "gonext": "./gonext-repl.mjs"
18
18
  },
19
19
  "scripts": {
20
+ "test": "python3 -m unittest discover tests -v",
20
21
  "deploy:local": "npm install -g .",
21
22
  "run": "gonext-local-worker",
22
23
  "publish:org": "npm publish --access public"