@tiens.nguyen/gonext-local-worker 1.0.238 → 1.0.240

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1830,6 +1830,32 @@ async function runAgentChatJob(job) {
1830
1830
  (process.env.GONEXT_PROBE_PYTHON ?? process.env.GONEXT_MLX_LM_PYTHON ?? "")
1831
1831
  .trim() || "python3";
1832
1832
  const scriptPath = join(WORKER_DIR, "gonext_agent_chat.py");
1833
+
1834
+ // Workspace scope depends on WHERE the job came from (task #75). The `gonext`
1835
+ // terminal always sends its launch cwd as activeWorkspace; the web app never does —
1836
+ // that's the discriminator.
1837
+ // • Terminal: the real `gonext-local-worker workspace add` registry + that cwd.
1838
+ // • Web agent chat: ONE isolated temp scratch folder — NOT the Mac's registered
1839
+ // project folders. Inheriting them leaked their paths into the web prompt and
1840
+ // let terminal state drive web routing. run_command stays off (not a code repo).
1841
+ const terminalActiveWs = String(payload?.activeWorkspace ?? "").trim();
1842
+ let jobWorkspaces;
1843
+ let jobActiveWorkspace;
1844
+ if (terminalActiveWs) {
1845
+ jobWorkspaces = await readFile(join(homedir(), ".gonext", "workspaces.json"), "utf8")
1846
+ .then((raw) => {
1847
+ const parsed = JSON.parse(raw);
1848
+ return Array.isArray(parsed?.workspaces) ? parsed.workspaces : [];
1849
+ })
1850
+ .catch(() => []);
1851
+ jobActiveWorkspace = terminalActiveWs;
1852
+ } else {
1853
+ const webWork = join(homedir(), ".gonext", "web-agent-work");
1854
+ await mkdir(webWork, { recursive: true }).catch(() => {});
1855
+ jobWorkspaces = [{ name: "web", path: webWork, allowRun: false }];
1856
+ jobActiveWorkspace = webWork;
1857
+ }
1858
+
1833
1859
  const input = JSON.stringify({
1834
1860
  messages: payload?.messages ?? [],
1835
1861
  agentBaseURL: payload?.agentBaseURL ?? "",
@@ -1889,17 +1915,12 @@ async function runAgentChatJob(job) {
1889
1915
  ragEmbedModel: payload?.ragEmbedModel ?? "",
1890
1916
  ragEmbedUrl: payload?.ragEmbedUrl ?? "",
1891
1917
  ragTopK: payload?.ragTopK ?? 6,
1892
- // Workspaces the agent may read/edit/test code in read fresh each job so a
1893
- // `gonext-local-worker workspace add` applies without restarting the worker.
1894
- workspaces: await readFile(join(homedir(), ".gonext", "workspaces.json"), "utf8")
1895
- .then((raw) => {
1896
- const parsed = JSON.parse(raw);
1897
- return Array.isArray(parsed?.workspaces) ? parsed.workspaces : [];
1898
- })
1899
- .catch(() => []),
1900
- // The `gonext` terminal REPL's launch folder — download_file/unzip_file output
1901
- // lands here (must be a registered workspace; the python side re-validates).
1902
- activeWorkspace: payload?.activeWorkspace ?? "",
1918
+ // Workspaces the agent may read/edit/test code in (see the terminal-vs-web split
1919
+ // computed above): the real registry for the terminal, a lone temp folder for web.
1920
+ workspaces: jobWorkspaces,
1921
+ // download_file/unzip_file output lands here (a registered workspace; python
1922
+ // re-validates). Terminal = its launch cwd; web = the temp scratch folder.
1923
+ activeWorkspace: jobActiveWorkspace,
1903
1924
  });
1904
1925
  // 30 min max for an agent run: multi-step ReAct on a 14B/31B with a cold prompt
1905
1926
  // cache — or a remote Ollama box on slow GPU cold-loading a big model — can run
@@ -131,6 +131,12 @@ def _html_to_text(html_text, limit=3000):
131
131
  # Main menu / Donate / Create account…" before any article text, which both starves
132
132
  # the model of real content and bloats the per-step context (OOM risk on local MLX).
133
133
  text = re.sub(r"(?is)<(nav|header|footer|aside|form|button|menu)\b.*?</\1>", " ", text)
134
+ # Preserve TABLE structure before stripping tags: a cell boundary becomes " | "
135
+ # and a row becomes a newline, so a schedule/fixtures table survives as readable
136
+ # pipe-delimited rows instead of collapsing into a wall of words — a "make a table"
137
+ # task (task #75) is unanswerable if the source table is flattened on the way in.
138
+ text = re.sub(r"(?i)</(td|th)>", " | ", text)
139
+ text = re.sub(r"(?i)</tr>", "\n", text)
134
140
  # Turn block-ending tags into newlines so document structure survives stripping.
135
141
  text = re.sub(r"(?i)<(br|/p|/div|/li|/tr|/h[1-6]|/section|/article)\s*>", "\n", text)
136
142
  # Remove all remaining tags.
@@ -358,60 +364,90 @@ def _get_json(url, timeout=15):
358
364
  return None
359
365
 
360
366
 
361
- def _web_search_impl(query):
367
+ def _web_search_impl(query, max_results=5):
362
368
  """Look up factual info via free no-key JSON APIs (DuckDuckGo + Wikipedia).
363
369
 
364
- Returns a short text summary with a source URL, or a 'no results' message.
365
- Tries DuckDuckGo Instant Answer first, then falls back to a Wikipedia search +
366
- REST summary. Never fabricates callers should surface 'no results' honestly.
370
+ Returns a short SUMMARY (for a direct-fact question) followed by a numbered list
371
+ of the top matching PAGES (title snippet URL), so the model can fetch_url() a
372
+ real page when a one-line summary can't carry the data it needs (a schedule, a
373
+ table, a fixtures list). Returning only a single Wikipedia intro used to trap weak
374
+ models in a re-search loop — they never saw a candidate URL to open (task #75).
375
+ Never fabricates — returns an honest 'no results' when nothing is found.
367
376
  """
377
+ import html as _html
368
378
  from urllib.parse import quote
369
379
  q = (query or "").strip()
370
380
  if not q:
371
381
  return "web_search: empty query."
372
382
 
373
- # 1) DuckDuckGo Instant Answer API.
383
+ def _strip(s):
384
+ return _html.unescape(re.sub(r"<[^>]+>", "", s or "")).strip()
385
+
386
+ summary = ""
387
+ summary_src = ""
388
+ results = [] # (title, snippet, url)
389
+ seen = set()
390
+
391
+ def _add(title, snippet, url):
392
+ url = (url or "").strip()
393
+ if not url or url in seen or len(results) >= max_results:
394
+ return
395
+ seen.add(url)
396
+ results.append((_strip(title) or url, _strip(snippet), url))
397
+
398
+ # 1) DuckDuckGo Instant Answer — a direct abstract + related topics (as pages).
374
399
  ddg = _get_json(
375
400
  f"https://api.duckduckgo.com/?q={quote(q)}&format=json&no_html=1&skip_disambig=1"
376
401
  )
377
402
  if isinstance(ddg, dict):
378
403
  abstract = (ddg.get("AbstractText") or "").strip()
379
404
  if abstract:
380
- src = (ddg.get("AbstractURL") or "").strip()
381
- return f"{abstract[:1500]}\nSource: {src}" if src else abstract[:1500]
382
- # No abstract — use the first related topic that has text.
405
+ summary = abstract[:1200]
406
+ summary_src = (ddg.get("AbstractURL") or "").strip()
383
407
  for topic in ddg.get("RelatedTopics") or []:
384
- if isinstance(topic, dict) and topic.get("Text"):
385
- src = (topic.get("FirstURL") or "").strip()
386
- text = topic["Text"][:1500]
387
- return f"{text}\nSource: {src}" if src else text
408
+ if isinstance(topic, dict) and topic.get("Text") and topic.get("FirstURL"):
409
+ _add(topic["Text"][:80], topic["Text"], topic["FirstURL"])
388
410
 
389
- # 2) Wikipedia: find the best-matching title, then fetch its summary extract.
411
+ # 2) Wikipedia full-text search SEVERAL candidate pages with snippets, so the
412
+ # model can pick the specific one (e.g. a "…knockout stage" fixtures page).
390
413
  search = _get_json(
391
414
  "https://en.wikipedia.org/w/api.php?action=query&list=search"
392
- f"&srsearch={quote(q)}&format=json&srlimit=1"
415
+ f"&srsearch={quote(q)}&srlimit={max_results}&format=json"
393
416
  )
394
- title = ""
395
417
  try:
396
- title = search["query"]["search"][0]["title"]
418
+ hits = search["query"]["search"]
397
419
  except Exception: # noqa: BLE001
398
- title = ""
399
- if title:
400
- slug = quote(title.replace(" ", "_"))
401
- summary = _get_json("https://en.wikipedia.org/api/rest_v1/page/summary/" + slug)
402
- if isinstance(summary, dict):
403
- extract = (summary.get("extract") or "").strip()
404
- if extract:
405
- src = (
406
- (summary.get("content_urls") or {}).get("desktop", {}).get("page", "")
407
- or f"https://en.wikipedia.org/wiki/{slug}"
408
- )
409
- return f"{extract[:1500]}\nSource: {src}"
420
+ hits = []
421
+ for hit in hits:
422
+ slug = quote(hit.get("title", "").replace(" ", "_"))
423
+ _add(hit.get("title", ""), hit.get("snippet", ""),
424
+ f"https://en.wikipedia.org/wiki/{slug}")
425
+ # If no direct abstract, use the best-matching page's summary as the answer.
426
+ if not summary and hits:
427
+ slug = quote(hits[0]["title"].replace(" ", "_"))
428
+ s = _get_json("https://en.wikipedia.org/api/rest_v1/page/summary/" + slug)
429
+ if isinstance(s, dict):
430
+ summary = (s.get("extract") or "").strip()[:1200]
431
+ summary_src = f"https://en.wikipedia.org/wiki/{slug}"
432
+
433
+ if not summary and not results:
434
+ return (
435
+ f"No results found for '{q}'. Tell the user you couldn't find this — "
436
+ "do NOT invent an answer or a URL."
437
+ )
410
438
 
411
- return (
412
- f"No results found for '{q}'. Tell the user you couldn't find this — "
413
- "do NOT invent an answer or a URL."
414
- )
439
+ parts = []
440
+ if summary:
441
+ parts.append(summary + (f"\nSource: {summary_src}" if summary_src else ""))
442
+ if results:
443
+ lines = ["Top pages (call fetch_url on the most relevant to read it in full):"]
444
+ for i, (title, snippet, url) in enumerate(results, 1):
445
+ line = f"{i}. {title}"
446
+ if snippet:
447
+ line += f" — {snippet[:160]}"
448
+ lines.append(line + f"\n {url}")
449
+ parts.append("\n".join(lines))
450
+ return "\n\n".join(parts)
415
451
 
416
452
 
417
453
  class _AgentConfigError(RuntimeError):
@@ -3510,7 +3546,12 @@ def run_agent_chat(cfg):
3510
3546
  _log(f"fetch_url {url} → PDF, cannot read")
3511
3547
  _last_obs["text"] = msg
3512
3548
  return msg
3513
- text = _html_to_text(raw.decode("utf-8", errors="replace")) or "(page had no readable text)"
3549
+ # 8000, not the 3000 default: a research page's real content (fixture tables,
3550
+ # data) sits well past the lead, so the old cap returned only the intro/nav and
3551
+ # starved the model (task #75). Older large observations are trimmed by the #63
3552
+ # step_callback, so the extra size only costs context for the latest 2 steps.
3553
+ text = (_html_to_text(raw.decode("utf-8", errors="replace"), limit=8000)
3554
+ or "(page had no readable text)")
3514
3555
  out = f"{url}\n{text}"
3515
3556
  _log(f"fetch_url {url} → {len(text)} chars")
3516
3557
  _last_obs["text"] = out
@@ -3654,6 +3695,24 @@ def run_agent_chat(cfg):
3654
3695
  "text": f"PDF already created → {prev['title']}.pdf (reusing link)"})
3655
3696
  _last_obs["text"] = prev["msg"]
3656
3697
  return prev["msg"]
3698
+
3699
+ # Refuse a placeholder skeleton instead of shipping a confident-but-empty PDF
3700
+ # (task #75): a table of 'TBD vs TBD' rows means the agent never found the real
3701
+ # data. 3+ placeholder cells is the fingerprint of that failure and virtually
3702
+ # never appears in a genuine user document. Steer the model to answer honestly
3703
+ # rather than render a green "✅ ready" over an empty table.
3704
+ _placeholders = len(re.findall(
3705
+ r"\b(?:TBD|TBA|TBC|N/?A)\b|\?\?\?|—\s*(?:vs\.?)?\s*—", text or "", re.I))
3706
+ if _placeholders >= 3:
3707
+ msg = (
3708
+ f"This document is mostly placeholders ({_placeholders} TBD/TBA cells) — "
3709
+ "the real data was never found, so the PDF would be empty. Do NOT retry "
3710
+ "create_pdf. Call final_answer to tell the user honestly that you could "
3711
+ f"not find the actual {doc_title} data to fill the table."
3712
+ )
3713
+ _log(f"create_pdf refused: {_placeholders} placeholder cells")
3714
+ _last_obs["text"] = msg
3715
+ return msg
3657
3716
  # 1) Format the raw text into clean Markdown (extra model call, intentional).
3658
3717
  # CHAT model, not the coder — see the fast-path call site for why (task #74).
3659
3718
  _emit({"type": "step", "text": "Formatting document…"})
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@tiens.nguyen/gonext-local-worker",
3
- "version": "1.0.238",
3
+ "version": "1.0.240",
4
4
  "description": "Polls GoNext cloud API for async local LLM jobs and runs them against Ollama/OpenAI-compatible servers on this Mac",
5
5
  "type": "module",
6
6
  "license": "MIT",