@allansantos-dev/smart-tool 0.9.1 → 0.9.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -48,6 +48,7 @@ import index_views
48
48
  import index_inventory
49
49
  import usage_meter
50
50
  import document_text
51
+ import embedding_cache
51
52
  import model_defaults
52
53
  import project_identity
53
54
  import project_store
@@ -291,7 +292,7 @@ def _smart_search_job_payload(job_id):
291
292
  else:
292
293
  payload["error"] = job.get("error") or "Search did not complete."
293
294
  payload["resumable"] = bool(job.get("arguments"))
294
- if job.get("stats"):
295
+ if job.get("stats") and job["stats"] != payload.get("result"):
295
296
  payload["stats"] = job["stats"]
296
297
  if job.get("failure_phase"):
297
298
  payload["failure_phase"] = job["failure_phase"]
@@ -302,10 +303,10 @@ def _smart_search_job_payload(job_id):
302
303
  TOOLS = [
303
304
  {
304
305
  "name": "project_manage",
305
- "description": "Manages folders and indexing. list/status show project_id, storage, scope and jobs. graph reads the indexed view: coverage, references and static calls in Java, Angular, Python and JS/TS/JSX/TSX; includes HTML/CSS/Markdown references and TXT/DOCX coverage; with file_path, narrows to that file's symbols, imports and calls. inspect lists the files of a view_id; with file_path, returns only that file's chunks. usage shows consumption and reuse; duplicates lists duplicated functions (identical, near-identical and semantic bodies, min_similarity default 0.90; below that most pairs measured were false positives), excluding tests unless include_tests=true; fix real duplicates; only dismiss with duplicates_dismiss (finding_id, reason=false_positive|intentional, note) what is not a duplicate or is a copy kept on purpose: the finding stays hidden until the code changes, and duplicates_restore reopens it; integration shows the MCP client, hook and tool usage metrics; install_hook_preview shows the change that would install the hook redirecting Grep/Read to smart_search and WebSearch/WebFetch to web_search/web_fetch in this session's client and returns a plan_id; install_hook writes it with that plan_id and confirm=true, with a backup, only after the user approves the change; search_limits saves the project's default top_k per block ({code, test, doc}, each 1-20, total at most 20); storage shows disk usage and views; pin_view protects a view; cleanup_preview and cleanup_commit remove selected views after confirmation; compare compares two stored views, without checkout or embeddings. register adds a folder; preview checks the scope; index/rebuild update; pause/cancel keep checkpoints; resume continues; policy picks on_search/eager; scope adjusts the scope; remove deletes only the local index. Jobs via smart_search_result.",
306
+ "description": "Manages folders and indexing. Every action takes project_root (the folder) or project_id. list gives a short line per project; status shows storage, scope and jobs. graph with symbol (function, Class.method, class name, or path::name when the name repeats) answers what an edit touches: definition, callers with line, calls, tests that reach it through static calls (depth hops, default 3) and files importing it; use it before changing a function to know what to update. affected_tests answers which tests to run after editing: every test file importing a changed file (git diff against base, default HEAD, untracked included) plus changed tests, likeliest failures first, with the command to run them and run_all when config or unanalyzed code changed. graph without symbol reads the indexed view: coverage, references and static calls in Java, Angular, Python and JS/TS/JSX/TSX; includes HTML/CSS/Markdown references and TXT/DOCX coverage; with file_path, narrows to that file's symbols, imports and calls (lists capped by limit). inspect lists the files of a view_id; with file_path, returns only that file's chunks. usage shows consumption and reuse; docs lists public functions without a docstring (Python docstring, JSDoc, Javadoc; private, nested and override functions excluded) with coverage percent, excluding tests unless include_tests=true, for documenting a project that started without it; duplicates lists duplicated functions (identical, near-identical and semantic bodies, min_similarity default 0.90; below that most pairs measured were false positives), excluding tests unless include_tests=true; fix real duplicates; only dismiss with duplicates_dismiss (finding_id, reason=false_positive|intentional, note) what is not a duplicate or is a copy kept on purpose: the finding stays hidden until the code changes, and duplicates_restore reopens it; integration shows the MCP client, hook and tool usage metrics; install_hook_preview shows the change that would install the hook redirecting Grep/Read to smart_search and WebSearch/WebFetch to web_search/web_fetch in this session's client and returns a plan_id; install_hook writes it with that plan_id and confirm=true, with a backup, only after the user approves the change; search_limits saves the project's default top_k per block ({code, test, doc}, each 1-20, total at most 20); storage shows disk usage and views; pin_view protects a view; cleanup_preview and cleanup_commit remove selected views after confirmation; compare compares two stored views, without checkout or embeddings. register adds a folder; preview checks the scope; index/rebuild update; pause/cancel keep checkpoints; resume continues; policy picks on_search/eager; scope adjusts the scope; remove deletes only the local index. Jobs via smart_search_result.",
306
307
  "inputSchema": {
307
308
  "type": "object", "properties": {
308
- "action": {"type": "string", "enum": ["list", "status", "register", "preview", "index", "rebuild", "pause", "cancel", "resume", "watch", "scope", "remove", "relocate", "probe", "policy", "inspect", "graph", "usage", "storage", "pin_view", "cleanup_preview", "cleanup_commit", "compare", "search_limits", "integration", "install_hook_preview", "install_hook", "duplicates", "duplicates_dismiss", "duplicates_restore"]},
309
+ "action": {"type": "string", "enum": ["list", "status", "register", "preview", "index", "rebuild", "pause", "cancel", "resume", "watch", "scope", "remove", "relocate", "probe", "policy", "inspect", "graph", "usage", "storage", "pin_view", "cleanup_preview", "cleanup_commit", "compare", "search_limits", "integration", "install_hook_preview", "install_hook", "duplicates", "duplicates_dismiss", "duplicates_restore", "docs", "affected_tests"]},
309
310
  "finding_id": {"type": "string"},
310
311
  "reason": {"type": "string", "enum": ["false_positive", "intentional"]},
311
312
  "note": {"type": "string"},
@@ -324,6 +325,9 @@ TOOLS = [
324
325
  "left_view":{"type":"string"},"right_view":{"type":"string"},"relations":{"type":"boolean"},
325
326
  "update_mode": {"type": "string", "enum": ["on_search", "eager"]},
326
327
  "view_id": {"type": "string"}, "file_path": {"type": "string"},
328
+ "base": {"type": "string", "default": "HEAD", "description": "affected_tests: git revision the changes are compared with (HEAD = uncommitted work; main or origin/main for a branch)"},
329
+ "symbol": {"type": "string", "description": "graph: function, Class.method, class name or path::name to get its callers, calls, tests and importers"},
330
+ "depth": {"type": "integer", "minimum": 1, "maximum": 4, "default": 3, "description": "graph with symbol: call hops searched for tests"},
327
331
  "project_root": {"type": "string"}, "project_id": {"type": "string"},
328
332
  "job_id": {"type": "string"}, "watch": {"type": "boolean"},
329
333
  "manual": {"type": "boolean"}, "force_scope": {"type": "boolean"},
@@ -1414,27 +1418,69 @@ def _find_duplicates(root, arguments, dismissed):
1414
1418
  return duplicates.find(root, embed, configured_model, float(similarity), include_tests, limit, dismissed, include_dismissed)
1415
1419
 
1416
1420
 
1421
+ def _with_project_id(arguments):
1422
+ """Agents know the folder, not the id: project_root resolves to the registered project's project_id."""
1423
+ root = arguments.get("project_root")
1424
+ if arguments.get("project_id") or not isinstance(root, str) or arguments.get("action") in ("register", "relocate"):
1425
+ return arguments
1426
+ wanted = project_identity.canonical_root(root)
1427
+ match = next((p for p in project_store.all_projects() if project_identity.canonical_root(p["root"]) == wanted), None)
1428
+ if not match:
1429
+ raise ValueError(f"No registered project at {root}; register it first (action=register).")
1430
+ return {**arguments, "project_id": match["id"]}
1431
+
1432
+
1433
+ def _project_summary(project):
1434
+ data = _project_payload(project)
1435
+ index = data.get("index") or {}
1436
+ jobs = data.get("jobs") or []
1437
+ return {"project_id": data["id"], "name": data.get("name"), "root": data["root"], "status": data.get("status"),
1438
+ "enabled": data.get("enabled"), "paused": data.get("paused"), "update_mode": data.get("update_mode"),
1439
+ "files": index.get("files"), "chunks": index.get("chunks"), "last_indexed": data.get("last_indexed"),
1440
+ "view": (data.get("view") or {}).get("label"), "active_jobs": len(data.get("active_jobs") or []),
1441
+ "last_job": {k: jobs[0].get(k) for k in ("job_id", "status", "kind")} if jobs else None,
1442
+ "last_error": (data.get("last_error") or None) and str(data["last_error"])[:200]}
1443
+
1444
+
1417
1445
  def _project_action(arguments):
1418
1446
  if not isinstance(arguments, dict):
1419
1447
  raise ValueError("Pass the action in a JSON object.")
1448
+ arguments = _with_project_id(arguments)
1420
1449
  action = arguments.get("action", "list")
1421
1450
  if action == "probe":
1422
1451
  return _probe_index_models()
1423
- if action in ('graph','compare','duplicates','duplicates_dismiss','duplicates_restore'):
1452
+ if action in ('graph','compare','duplicates','duplicates_dismiss','duplicates_restore','docs','affected_tests'):
1424
1453
  # Análise somente leitura pode demorar; não prende o lock global de ações.
1425
1454
  key=arguments.get('project_id')
1426
1455
  if not isinstance(key,str) or not re.fullmatch(r'[a-f0-9]{16}',key):
1427
- raise ValueError('Pass a valid project_id.')
1456
+ raise ValueError('Pass project_root (the folder) or a valid project_id.')
1428
1457
  project=project_store.get(key)
1429
1458
  if not project:
1430
1459
  raise ValueError('Project not registered.')
1431
1460
  if action=='duplicates':
1432
1461
  return _find_duplicates(project['root'], arguments, project.get('dismissed_duplicates') or {})
1462
+ if action=='docs':
1463
+ import doc_check
1464
+ include_tests=arguments.get('include_tests',False)
1465
+ if type(include_tests) is not bool:
1466
+ raise ValueError('include_tests must be true or false.')
1467
+ limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
1468
+ return doc_check.coverage(project['root'],include_tests,limit,arguments.get('view_id'))
1469
+ if action=='affected_tests':
1470
+ import affected_tests
1471
+ limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
1472
+ return affected_tests.affected(project['root'],arguments.get('base','HEAD'),limit,arguments.get('view_id'))
1433
1473
  if action in ('duplicates_dismiss','duplicates_restore'):
1434
1474
  return _change_dismissal(key, project, arguments, action == 'duplicates_dismiss')
1435
1475
  if action=='graph':
1436
1476
  import code_graph
1437
- return code_graph.build(project['root'],arguments.get('view_id'),file_path=arguments.get('file_path'))
1477
+ import code_impact
1478
+ limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
1479
+ if arguments.get('symbol') is not None:
1480
+ return code_impact.impact(project['root'],arguments['symbol'],arguments.get('view_id'),
1481
+ arguments.get('depth',3),limit)
1482
+ graph=code_graph.build(project['root'],arguments.get('view_id'),file_path=arguments.get('file_path'))
1483
+ return code_impact.for_agent(graph,limit) if arguments.get('file_path') else graph
1438
1484
  import view_compare
1439
1485
  return view_compare.compare(project['root'],arguments.get('left_view'),arguments.get('right_view'),
1440
1486
  arguments.get('file_path'),arguments.get('relations',True))
@@ -1513,6 +1559,9 @@ def _project_action_locked(arguments):
1513
1559
  _metric("install_hook", target=result["client"], changed=result["changed"])
1514
1560
  return result
1515
1561
  if action == "list":
1562
+ if not arguments.get("full"):
1563
+ return {"projects": [_project_summary(p) for p in project_store.all_projects()],
1564
+ "detail": "Use action=status with project_id or project_root for scope, jobs and index details."}
1516
1565
  return {"projects": [_project_payload(p) for p in project_store.all_projects()],
1517
1566
  "storage_dir": indexer.INDEX_DIR,
1518
1567
  "gateway": dict(_GATEWAY_STATUS),
@@ -1526,7 +1575,7 @@ def _project_action_locked(arguments):
1526
1575
  return {"project": _project_payload(project)}
1527
1576
  key = arguments.get("project_id")
1528
1577
  if not isinstance(key, str) or not re.fullmatch(r"[a-f0-9]{16}", key):
1529
- raise ValueError("Pass a valid project_id; use the list action to look it up.")
1578
+ raise ValueError("Pass project_root (the folder) or a valid project_id.")
1530
1579
  project = project_store.get(key)
1531
1580
  if not project:
1532
1581
  raise ValueError("Project not registered.")
@@ -1995,10 +2044,8 @@ def _fan_out_web_search(query, num, budget_s=None, site_domains=()):
1995
2044
  try:
1996
2045
  fut.result()
1997
2046
  _record_tier_health(tier_name, success=True)
1998
- except (TierSkipped, browser_service_client.BrowserSkipped):
1999
- pass
2000
- except Exception:
2001
- _record_tier_health(tier_name, success=False)
2047
+ except Exception as exc:
2048
+ _record_tier_failure(tier_name, exc)
2002
2049
  return _cb
2003
2050
 
2004
2051
  for fut, tier_name in futures.items():
@@ -2243,8 +2290,7 @@ def _search_web_once(query, num, scope, deadline=None, assess_semantic=False,
2243
2290
  if not raw_results:
2244
2291
  raise RuntimeError("zero results in the requested domain")
2245
2292
  except Exception as exc:
2246
- if not isinstance(exc, TierSkipped):
2247
- _record_tier_health(tier_name, success=False)
2293
+ _record_tier_failure(tier_name, exc)
2248
2294
  errors.append(_sanitize_text(f"{tier_name}: {type(exc).__name__}: {exc}"))
2249
2295
  continue
2250
2296
  _record_tier_health(tier_name, success=True)
@@ -2278,8 +2324,7 @@ def _search_web_once(query, num, scope, deadline=None, assess_semantic=False,
2278
2324
  if not raw_results:
2279
2325
  raise RuntimeError("zero results in the requested domain")
2280
2326
  except Exception as exc:
2281
- if not isinstance(exc, (TierSkipped, browser_service_client.BrowserSkipped)):
2282
- _record_tier_health(tier_name, success=False)
2327
+ _record_tier_failure(tier_name, exc)
2283
2328
  errors.append(_sanitize_text(f"{tier_name}: {type(exc).__name__}: {exc}"))
2284
2329
  continue
2285
2330
  _record_tier_health(tier_name, success=True)
@@ -2588,13 +2633,26 @@ def _annotate_quick_results(rows, classification, cache=None):
2588
2633
  return result
2589
2634
 
2590
2635
 
2591
- def _record_tier_health(tier_name, success):
2636
+ def _record_tier_health(tier_name, success, outcome=None):
2592
2637
  try:
2593
- web_search_health.record(tier_name, success)
2638
+ web_search_health.record(tier_name, success, outcome)
2594
2639
  except Exception:
2595
2640
  pass # telemetria é best-effort — nunca pode derrubar uma busca já resolvida
2596
2641
 
2597
2642
 
2643
+ def _record_tier_failure(tier_name, exc):
2644
+ """Health entry for a failed tier call, keeping a 429, a captcha and a quota pause apart from other errors."""
2645
+ if isinstance(exc, (TierSkipped, browser_service_client.BrowserSkipped)):
2646
+ outcome = "paused"
2647
+ elif isinstance(exc, WebProviderRateLimited):
2648
+ outcome = "rate_limited"
2649
+ elif isinstance(exc, browser_service_client.BrowserCaptcha):
2650
+ outcome = "captcha"
2651
+ else:
2652
+ outcome = "failed"
2653
+ _record_tier_health(tier_name, False, outcome)
2654
+
2655
+
2598
2656
  _is_safe_crawl_target = web_fetch.is_safe_target
2599
2657
 
2600
2658
 
@@ -3073,6 +3131,7 @@ def _tool_result(text, is_error=False):
3073
3131
 
3074
3132
  WEB_FETCH_TIMEOUT_S = 90
3075
3133
  WEB_FETCH_RENDER_TIMEOUT_S = 60
3134
+ WEB_FETCH_VECTOR_ROOT = os.path.join(paths.DATA_DIR, "web-fetch")
3076
3135
 
3077
3136
 
3078
3137
  def _render_page(url, deadline):
@@ -3085,6 +3144,30 @@ def _render_page(url, deadline):
3085
3144
  return (pages[0].get("markdown") or "") if pages else ""
3086
3145
 
3087
3146
 
3147
+ def _web_fetch_embed(texts, deadline):
3148
+ """Vectors for the pieces of a long page and the prompt, with the pieces cached per embedding model. Fails loudly
3149
+ without an embedding model: a long page is never cut back to its start in silence."""
3150
+ model = local_embedder.resolve(config.load_config().get("embedding_model"), None)
3151
+ if not model or model == indexer.LEXICAL_MODEL:
3152
+ raise RuntimeError("This page is longer than 60,000 characters and choosing its relevant parts needs an "
3153
+ "embedding_model, which is not configured")
3154
+ try:
3155
+ token = model_client.get_token()
3156
+ probe = _embed(model, texts[-1:], token, deadline=deadline)
3157
+ cache = embedding_cache.VectorCache(WEB_FETCH_VECTOR_ROOT, indexer.INDEX_DIR, {
3158
+ "purpose": "web_fetch_pieces", "model": model, "dimensions": len(probe[0])})
3159
+ try:
3160
+ pieces = cache.embed(texts[:-1], lambda batch: _embed(model, batch, token, deadline=deadline),
3161
+ indexer.validate_vector, split=lambda unique: list(_embed_batches(unique)),
3162
+ workers=indexer.EMBED_WORKERS)
3163
+ finally:
3164
+ cache.db.close()
3165
+ except Exception as exc:
3166
+ raise RuntimeError(f"This page is longer than 60,000 characters and the embedding model ({model}) that picks "
3167
+ f"its relevant parts failed: {type(exc).__name__}: {exc}"[:400]) from None
3168
+ return pieces + probe
3169
+
3170
+
3088
3171
  def _web_fetch_unavailable():
3089
3172
  cfg = config.load_config()
3090
3173
  if not cfg.get("router_model"):
@@ -3105,18 +3188,22 @@ def _handle_web_fetch(arguments):
3105
3188
  raise RuntimeError(f"Gateway model unavailable ({problem}).")
3106
3189
  deadline = started + WEB_FETCH_TIMEOUT_S
3107
3190
  page = web_fetch.read_page(url, deadline, _render_page)
3108
- text = web_fetch.answer(config.load_config()["router_model"], prompt, page, deadline)
3191
+ selected, pieces = web_fetch.select(page["text"], prompt, lambda texts: _web_fetch_embed(texts, deadline))
3192
+ text = web_fetch.answer(config.load_config()["router_model"], prompt, {**page, "text": selected}, deadline)
3109
3193
  except Exception as exc:
3110
3194
  web_fetch.mark_failed(str(url))
3111
3195
  _metric("web_fetch", error=type(exc).__name__, elapsed_s=round(time.monotonic() - started, 2))
3112
3196
  raise RuntimeError(f"web_fetch failed: {str(exc).rstrip('.')}. The native WebFetch is allowed for this URL for "
3113
3197
  f"{web_fetch.FAILED_TTL_S // 60} min.") from None
3114
3198
  _metric("web_fetch", cached=page["cached"], rendered=page["rendered"], render_error=page.get("render_error"),
3115
- page_chars=len(page["text"]),
3199
+ page_chars=page["chars"], read_chars=len(selected), pieces=pieces,
3116
3200
  answer_chars=len(text), elapsed_s=round(time.monotonic() - started, 2))
3117
3201
  origin = "from cache" if page["cached"] else "rendered in the browser" if page["rendered"] else "downloaded now"
3118
3202
  read_at = datetime.datetime.fromtimestamp(page["fetched_at"]).strftime("%Y-%m-%d %H:%M")
3119
- cut = "; page cut to the first 60,000 characters" if page["truncated"] else ""
3203
+ cut = (f"; page of {page['chars']:,} characters: the {pieces} parts most related to the request were read"
3204
+ if pieces else "")
3205
+ if page["truncated"]:
3206
+ cut += f"; only the first {web_fetch.PAGE_STORE_MAX_CHARS:,} characters were kept"
3120
3207
  return _sanitize_text(f"Source: {page['final_url']} (read at {read_at}, {origin}{cut})\n\n{text}")
3121
3208
 
3122
3209
 
@@ -3395,7 +3482,7 @@ class MCPHandler(BaseHTTPRequestHandler):
3395
3482
  if self.path == "/setup/projects":
3396
3483
  try:
3397
3484
  setup_ui.require_token(self.headers)
3398
- self._write_json(200, _project_action({"action": "list"}))
3485
+ self._write_json(200, _project_action({"action": "list", "full": True}))
3399
3486
  except setup_ui.SetupError as exc:
3400
3487
  self._write_json(403, {"error": str(exc)})
3401
3488
  except Exception as exc:
package/version.py CHANGED
@@ -1,7 +1,7 @@
1
1
  """Single source of the Smart Tool version and of the User-Agent sent to public APIs."""
2
2
  import re
3
3
 
4
- VERSION = "0.9.1"
4
+ VERSION = "0.9.3"
5
5
  _CONTACT_RE = re.compile(r"^(?:[^\s@()<>;]+@[^\s@()<>;]+\.[^\s@()<>;]+|https?://[^\s()<>;]+)$")
6
6
 
7
7
 
@@ -1,16 +1,26 @@
1
- """`camoufox fetch` for the installer: a stalled download fails after 60 s instead of hanging, and progress is printed
2
- as plain lines (the rich bar stays invisible when the output is a pipe)."""
3
- import socket
1
+ """`camoufox fetch` for the installer. camoufox 0.5.6 prints a failed download and still exits 0, so the error is
2
+ caught here and turned into a failing exit, and the browser executable must exist afterwards. The 1.3 GB download
3
+ goes through ranged_download (Range requests that resume after a dropped or stalled connection; camoufox still checks
4
+ the sha256), its other requests get a timeout (requests passes None, which also overrides socket.setdefaulttimeout),
5
+ and progress is printed as plain lines (the rich bar stays invisible in a pipe)."""
6
+ import sys
7
+ from io import BytesIO
8
+ from pathlib import Path
4
9
 
10
+ import camoufox.__main__ as camoufox_cli
11
+ import requests
5
12
  from camoufox import pkgman
6
- from camoufox.__main__ import cli
7
13
 
8
- socket.setdefaulttimeout(60)
14
+ import ranged_download
15
+
16
+ TIMEOUT_S = (30, 60)
9
17
  REPORT_EVERY_MB = 50
10
- _webdl = pkgman.webdl
18
+ _update = camoufox_cli.CamoufoxUpdate.update
19
+ _get = requests.get
20
+ failures = []
11
21
 
12
22
 
13
- def _reporting_webdl(url, desc=None, buffer=None, bar=True, progress_callback=None):
23
+ def _reporter():
14
24
  reported = {"mb": -REPORT_EVERY_MB}
15
25
 
16
26
  def report(done, total):
@@ -19,10 +29,49 @@ def _reporting_webdl(url, desc=None, buffer=None, bar=True, progress_callback=No
19
29
  reported["mb"] = mb
20
30
  print(f"downloaded {mb} of {total // 1048576} MB", flush=True)
21
31
 
22
- return _webdl(url, desc=desc, buffer=buffer, bar=False, progress_callback=progress_callback or report)
32
+ return report
33
+
34
+
35
+ def _ranged_webdl(url, desc=None, buffer=None, bar=True, progress_callback=None):
36
+ buffer = BytesIO() if buffer is None else buffer
37
+ ranged_download.download(url, buffer, progress=progress_callback or _reporter())
38
+ return buffer
39
+
40
+
41
+ def _get_with_timeout(*args, **kwargs):
42
+ kwargs.setdefault("timeout", TIMEOUT_S)
43
+ return _get(*args, **kwargs)
44
+
45
+
46
+ def _recording_update(self, *args, **kwargs):
47
+ try:
48
+ return _update(self, *args, **kwargs)
49
+ except Exception as exc:
50
+ failures.append(exc)
51
+ raise
52
+
53
+
54
+ def installed_executable():
55
+ """The browser executable, without the download that launch_path() starts when it is missing."""
56
+ executable = Path(pkgman.camoufox_path(download_if_missing=False)) / pkgman.LAUNCH_FILE[pkgman.OS_NAME]
57
+ if not executable.is_file():
58
+ raise FileNotFoundError(executable)
59
+ return executable
60
+
23
61
 
62
+ def main():
63
+ requests.get = _get_with_timeout
64
+ pkgman.webdl = _ranged_webdl
65
+ camoufox_cli.CamoufoxUpdate.update = _recording_update
66
+ camoufox_cli.cli.main(["fetch"], standalone_mode=False)
67
+ if failures:
68
+ sys.exit(f"Camoufox download failed: {failures[-1]}")
69
+ try:
70
+ executable = installed_executable()
71
+ except Exception as exc:
72
+ sys.exit(f"Camoufox is not installed after the download: {exc}")
73
+ print(f"Camoufox {pkgman.installed_verstr()} ready: {executable}", flush=True)
24
74
 
25
- pkgman.webdl = _reporting_webdl
26
75
 
27
76
  if __name__ == "__main__":
28
- cli(["fetch"])
77
+ main()
@@ -0,0 +1,78 @@
1
+ """Downloads a file in HTTP Range requests of CHUNK_BYTES, each on a new connection, resuming from the last byte
2
+ received when a connection drops or stalls. Some networks (proxies, TLS inspection) cut every connection after a few
3
+ hundred MB, so a single long transfer can never finish there. It gives up only after MAX_IDLE_ATTEMPTS attempts in a
4
+ row without a single new byte. Standard library only: on Windows urllib trusts the system certificate store, including
5
+ a corporate inspection CA, and honors HTTPS_PROXY."""
6
+ import http.client
7
+ import time
8
+ import urllib.error
9
+ import urllib.request
10
+
11
+ CHUNK_BYTES = 32 * 1024 * 1024
12
+ READ_BYTES = 256 * 1024
13
+ TIMEOUT_S = 60
14
+ MAX_IDLE_ATTEMPTS = 6
15
+ MAX_FULL_RESTARTS = 3
16
+ RETRYABLE_HTTP = {408, 429}
17
+
18
+
19
+ class DownloadError(RuntimeError):
20
+ pass
21
+
22
+
23
+ def _open(url, start, end, timeout):
24
+ request = urllib.request.Request(url, headers={"Range": f"bytes={start}-{end}", "User-Agent": "SmartTool-installer"})
25
+ return urllib.request.urlopen(request, timeout=timeout)
26
+
27
+
28
+ def download(url, out, progress=None, chunk_bytes=CHUNK_BYTES, timeout=TIMEOUT_S, attempts=MAX_IDLE_ATTEMPTS,
29
+ sleep=time.sleep):
30
+ """Writes url into the seekable binary file out and returns its size; progress(done, total) after every read."""
31
+ done, total, idle, restarts = 0, None, 0, 0
32
+ while total is None or done < total:
33
+ end = done + chunk_bytes - 1 if total is None else min(done + chunk_bytes, total) - 1
34
+ before = done
35
+ try:
36
+ with _open(url, done, end, timeout) as response:
37
+ if response.status == 200:
38
+ if done:
39
+ restarts += 1
40
+ if restarts > MAX_FULL_RESTARTS:
41
+ raise DownloadError(f"{url} ignores Range requests and broke {restarts} times; giving up.")
42
+ out.seek(0)
43
+ out.truncate()
44
+ done = before = 0
45
+ length = response.headers.get("Content-Length")
46
+ total = int(length) if length else None
47
+ elif response.status == 206:
48
+ total = int(response.headers["Content-Range"].rsplit("/", 1)[1])
49
+ else:
50
+ raise DownloadError(f"{url} answered HTTP {response.status} to a Range request.")
51
+ out.seek(done)
52
+ while data := response.read1(READ_BYTES):
53
+ out.write(data)
54
+ done += len(data)
55
+ if progress:
56
+ progress(done, total or 0)
57
+ if response.status == 200 and total is None:
58
+ total = done
59
+ except urllib.error.HTTPError as exc:
60
+ exc.close()
61
+ if exc.code < 500 and exc.code not in RETRYABLE_HTTP:
62
+ raise DownloadError(f"{url} answered HTTP {exc.code}.") from exc
63
+ error = exc
64
+ except (OSError, http.client.HTTPException) as exc:
65
+ error = exc
66
+ else:
67
+ error = None
68
+ if done > before:
69
+ idle = 0
70
+ elif error is not None or done < (total or 0):
71
+ idle += 1
72
+ if idle >= attempts:
73
+ raise DownloadError(f"Download stopped at {done // 1048576} of {(total or 0) // 1048576} MB after "
74
+ f"{attempts} attempts without progress: {error or 'empty response'}")
75
+ sleep(min(2 ** idle, 30))
76
+ out.flush()
77
+ out.seek(0)
78
+ return done
package/web_fetch.py CHANGED
@@ -6,7 +6,14 @@ e do OpenCode (`packages/opencode/src/tool/webfetch.ts`: limite de 5 MB, User-Ag
6
6
  renderizada: melhor resposta 18 × 18 (1 empate), respondeu certo 22 × 16, caracteres devolvidos 35 mil × 72 mil;
7
7
  3,6 s de mediana, navegador em 7 de 37 (páginas montadas por JavaScript). Repetições da mesma URL nos transcripts:
8
8
  92 de 124 dentro de 24 h (PAGE_TTL_S).
9
+
10
+ Páginas acima de PAGE_MAX_CHARS (14,6% do cache, 10,4% das chamadas reais em 2026-10-08) não são mais cortadas no
11
+ começo: o texto inteiro fica no cache e o modelo lê os pedaços mais parecidos com o pedido, até SELECT_BUDGET_CHARS, na
12
+ ordem da página. Medido em 18 páginas reais e 41 perguntas: trechos depois do corte 0/28 → 13/28, cabeça 6/13 → 5/13,
13
+ latência mediana 3,3 s → 3,1 s. Abaixo do limite a página vai inteira (seleção ali empatou em qualidade e piorou a
14
+ cauda de latência; docs/BACKLOG.md).
9
15
  """
16
+ import math
10
17
  import codecs
11
18
  import http.client
12
19
  import ipaddress
@@ -25,6 +32,10 @@ import research_cache
25
32
  CACHE_NAMESPACE = "web_fetch"
26
33
  PAGE_TTL_S = 86400
27
34
  PAGE_MAX_CHARS = 60_000
35
+ PAGE_STORE_MAX_CHARS = 1_000_000
36
+ SELECT_BUDGET_CHARS = 15_000
37
+ CHUNK_CHARS = 2_000
38
+ PAGE_FORMAT = 2
28
39
  MAX_DOWNLOAD_BYTES = 5 * 1024 * 1024
29
40
  MAX_REDIRECTS = 5
30
41
  HTTP_TIMEOUT_S = 20
@@ -201,7 +212,8 @@ def read_page(url, deadline, render):
201
212
  """Page text (Markdown) from the 24 h cache, HTTP + trafilatura, or `render(url, deadline)` when HTTP comes thin."""
202
213
  url = normalize(url)
203
214
  for entry in research_cache.candidates(CACHE_NAMESPACE, url, limit=1):
204
- if entry.get("query") == url and time.time() - entry["saved_at"] < PAGE_TTL_S:
215
+ if (entry.get("query") == url and time.time() - entry["saved_at"] < PAGE_TTL_S
216
+ and entry["result"].get("format") == PAGE_FORMAT):
205
217
  return {**entry["result"], "cached": True, "fetched_at": entry["saved_at"]}
206
218
  final_url, kind, charset, body = fetch_http(url, deadline)
207
219
  rendered, render_error = False, None
@@ -223,12 +235,57 @@ def read_page(url, deadline, render):
223
235
  if len(text.strip()) < MIN_CHARS:
224
236
  detail = f"; browser failed: {render_error}" if render_error else ", not even when rendered in the browser"
225
237
  raise FetchError(f"Page without readable text ({final_url}){detail}.")
226
- page = {"url": url, "final_url": final_url, "text": text[:PAGE_MAX_CHARS], "truncated": len(text) > PAGE_MAX_CHARS,
227
- "rendered": rendered}
238
+ page = {"url": url, "final_url": final_url, "text": text[:PAGE_STORE_MAX_CHARS], "chars": len(text),
239
+ "truncated": len(text) > PAGE_STORE_MAX_CHARS, "rendered": rendered, "format": PAGE_FORMAT}
228
240
  saved_at = research_cache.put(CACHE_NAMESPACE, url, page, {})
229
241
  return {**page, "cached": False, "fetched_at": saved_at or time.time(), "render_error": render_error}
230
242
 
231
243
 
244
+ def chunks(text, size=CHUNK_CHARS):
245
+ """Consecutive pieces of about size characters, cut at a line break in the second half of each piece."""
246
+ parts, start = [], 0
247
+ while start < len(text):
248
+ end = min(len(text), start + size)
249
+ if end < len(text):
250
+ cut = text.rfind("\n", start + size // 2, end)
251
+ end = cut + 1 if cut > start else end
252
+ parts.append(text[start:end])
253
+ start = end
254
+ return parts
255
+
256
+
257
+ def _cosine(a, b):
258
+ norm = math.sqrt(sum(x * x for x in a)) * math.sqrt(sum(y * y for y in b))
259
+ return sum(x * y for x, y in zip(a, b)) / norm if norm else 0.0
260
+
261
+
262
+ def select(text, prompt, embed, budget=SELECT_BUDGET_CHARS):
263
+ """Text the model reads: the whole page up to PAGE_MAX_CHARS; above it, the pieces most similar to the prompt up
264
+ to budget characters, in page order, with [...] where pieces were skipped. embed(texts) returns one vector per
265
+ text and raises when the embedding model is unavailable. Returns (text, number of pieces kept or None)."""
266
+ if len(text) <= PAGE_MAX_CHARS:
267
+ return text, None
268
+ parts = chunks(text)
269
+ vectors = embed(parts + [prompt])
270
+ if len(vectors) != len(parts) + 1:
271
+ raise RuntimeError("The embedding model returned a different number of vectors than page pieces.")
272
+ query = vectors[-1]
273
+ ranked = sorted(range(len(parts)), key=lambda i: -_cosine(vectors[i], query))
274
+ kept, size = [], 0
275
+ for i in ranked:
276
+ if size + len(parts[i]) <= budget:
277
+ kept.append(i)
278
+ size += len(parts[i])
279
+ kept.sort()
280
+ pieces, last = [], None
281
+ for i in kept:
282
+ if last is not None and i != last + 1:
283
+ pieces.append("\n\n[...]\n\n")
284
+ pieces.append(parts[i])
285
+ last = i
286
+ return "".join(pieces), len(kept)
287
+
288
+
232
289
  def answer(model, prompt, page, deadline):
233
290
  response = model_client.fetch("/v1/chat/completions", model_client.get_token(), method="POST",
234
291
  timeout=min(90, _left(deadline)), body={
@@ -21,9 +21,17 @@ _MAX_LINES_BEFORE_ROTATE = 5000
21
21
  _LOCK = threading.Lock()
22
22
 
23
23
 
24
- def record(tier_name, success):
24
+ OUTCOMES = ("ok", "failed", "rate_limited", "captcha", "paused")
25
+
26
+
27
+ def record(tier_name, success, outcome=None):
28
+ """One tier call: outcome tells a provider 429 (rate_limited), a captcha and a self-imposed quota pause (paused,
29
+ left out of success rates) apart from an ordinary failure, so quota pressure can be read from the log."""
30
+ outcome = outcome or ("ok" if success else "failed")
31
+ if outcome not in OUTCOMES:
32
+ raise ValueError(f"Unknown web search outcome: {outcome}")
25
33
  os.makedirs(os.path.dirname(HEALTH_PATH), exist_ok=True)
26
- entry = {"ts": time.time(), "tier": tier_name, "success": bool(success)}
34
+ entry = {"ts": time.time(), "tier": tier_name, "success": bool(success), "outcome": outcome}
27
35
  with _LOCK:
28
36
  with open(HEALTH_PATH, "a", encoding="utf-8") as f:
29
37
  f.write(json.dumps(entry, ensure_ascii=False) + "\n")
@@ -67,7 +75,7 @@ def _read_recent_records():
67
75
 
68
76
 
69
77
  def recent_success_rate(tier_name, window=DEFAULT_WINDOW):
70
- records = [r for r in _read_recent_records() if r.get("tier") == tier_name]
78
+ records = [r for r in _read_recent_records() if r.get("tier") == tier_name and r.get("outcome") != "paused"]
71
79
  if not records:
72
80
  return None
73
81
  records = records[-window:]
@@ -79,7 +87,7 @@ def summary_for_tiers(tier_names, window=DEFAULT_WINDOW):
79
87
  all_records = _read_recent_records()
80
88
  summary = {}
81
89
  for name in tier_names:
82
- records = [r for r in all_records if r.get("tier") == name][-window:]
90
+ records = [r for r in all_records if r.get("tier") == name and r.get("outcome") != "paused"][-window:]
83
91
  if not records:
84
92
  summary[name] = None
85
93
  else: