@allansantos-dev/smart-tool 0.9.1 → 0.9.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +77 -0
- package/README.md +27 -2
- package/affected_tests.py +292 -0
- package/client_hooks.py +2 -2
- package/code_graph.py +75 -14
- package/code_graph_java.mjs +4 -1
- package/code_graph_js.cjs +15 -1
- package/code_impact.py +132 -0
- package/config.py +22 -0
- package/doc_check.py +256 -0
- package/duplicates.py +148 -13
- package/edit_preview.py +92 -0
- package/hook_decision.py +96 -2
- package/index_profile.py +1 -1
- package/install_runtime.py +37 -5
- package/package.json +1 -1
- package/router.py +4 -2
- package/setup_ui.py +113 -40
- package/smart_tool_daemon.py +108 -21
- package/version.py +1 -1
- package/web_adapters/browser/fetch_camoufox.py +59 -10
- package/web_adapters/browser/ranged_download.py +78 -0
- package/web_fetch.py +60 -3
- package/web_search_health.py +12 -4
package/smart_tool_daemon.py
CHANGED
|
@@ -48,6 +48,7 @@ import index_views
|
|
|
48
48
|
import index_inventory
|
|
49
49
|
import usage_meter
|
|
50
50
|
import document_text
|
|
51
|
+
import embedding_cache
|
|
51
52
|
import model_defaults
|
|
52
53
|
import project_identity
|
|
53
54
|
import project_store
|
|
@@ -291,7 +292,7 @@ def _smart_search_job_payload(job_id):
|
|
|
291
292
|
else:
|
|
292
293
|
payload["error"] = job.get("error") or "Search did not complete."
|
|
293
294
|
payload["resumable"] = bool(job.get("arguments"))
|
|
294
|
-
if job.get("stats"):
|
|
295
|
+
if job.get("stats") and job["stats"] != payload.get("result"):
|
|
295
296
|
payload["stats"] = job["stats"]
|
|
296
297
|
if job.get("failure_phase"):
|
|
297
298
|
payload["failure_phase"] = job["failure_phase"]
|
|
@@ -302,10 +303,10 @@ def _smart_search_job_payload(job_id):
|
|
|
302
303
|
TOOLS = [
|
|
303
304
|
{
|
|
304
305
|
"name": "project_manage",
|
|
305
|
-
"description": "Manages folders and indexing. list
|
|
306
|
+
"description": "Manages folders and indexing. Every action takes project_root (the folder) or project_id. list gives a short line per project; status shows storage, scope and jobs. graph with symbol (function, Class.method, class name, or path::name when the name repeats) answers what an edit touches: definition, callers with line, calls, tests that reach it through static calls (depth hops, default 3) and files importing it; use it before changing a function to know what to update. affected_tests answers which tests to run after editing: every test file importing a changed file (git diff against base, default HEAD, untracked included) plus changed tests, likeliest failures first, with the command to run them and run_all when config or unanalyzed code changed. graph without symbol reads the indexed view: coverage, references and static calls in Java, Angular, Python and JS/TS/JSX/TSX; includes HTML/CSS/Markdown references and TXT/DOCX coverage; with file_path, narrows to that file's symbols, imports and calls (lists capped by limit). inspect lists the files of a view_id; with file_path, returns only that file's chunks. usage shows consumption and reuse; docs lists public functions without a docstring (Python docstring, JSDoc, Javadoc; private, nested and override functions excluded) with coverage percent, excluding tests unless include_tests=true, for documenting a project that started without it; duplicates lists duplicated functions (identical, near-identical and semantic bodies, min_similarity default 0.90; below that most pairs measured were false positives), excluding tests unless include_tests=true; fix real duplicates; only dismiss with duplicates_dismiss (finding_id, reason=false_positive|intentional, note) what is not a duplicate or is a copy kept on purpose: the finding stays hidden until the code changes, and duplicates_restore reopens it; integration shows the MCP client, hook and tool usage metrics; install_hook_preview shows the change that would install the hook redirecting Grep/Read to smart_search and WebSearch/WebFetch to web_search/web_fetch in this session's client and returns a plan_id; install_hook writes it with that plan_id and confirm=true, with a backup, only after the user approves the change; search_limits saves the project's default top_k per block ({code, test, doc}, each 1-20, total at most 20); storage shows disk usage and views; pin_view protects a view; cleanup_preview and cleanup_commit remove selected views after confirmation; compare compares two stored views, without checkout or embeddings. register adds a folder; preview checks the scope; index/rebuild update; pause/cancel keep checkpoints; resume continues; policy picks on_search/eager; scope adjusts the scope; remove deletes only the local index. Jobs via smart_search_result.",
|
|
306
307
|
"inputSchema": {
|
|
307
308
|
"type": "object", "properties": {
|
|
308
|
-
"action": {"type": "string", "enum": ["list", "status", "register", "preview", "index", "rebuild", "pause", "cancel", "resume", "watch", "scope", "remove", "relocate", "probe", "policy", "inspect", "graph", "usage", "storage", "pin_view", "cleanup_preview", "cleanup_commit", "compare", "search_limits", "integration", "install_hook_preview", "install_hook", "duplicates", "duplicates_dismiss", "duplicates_restore"]},
|
|
309
|
+
"action": {"type": "string", "enum": ["list", "status", "register", "preview", "index", "rebuild", "pause", "cancel", "resume", "watch", "scope", "remove", "relocate", "probe", "policy", "inspect", "graph", "usage", "storage", "pin_view", "cleanup_preview", "cleanup_commit", "compare", "search_limits", "integration", "install_hook_preview", "install_hook", "duplicates", "duplicates_dismiss", "duplicates_restore", "docs", "affected_tests"]},
|
|
309
310
|
"finding_id": {"type": "string"},
|
|
310
311
|
"reason": {"type": "string", "enum": ["false_positive", "intentional"]},
|
|
311
312
|
"note": {"type": "string"},
|
|
@@ -324,6 +325,9 @@ TOOLS = [
|
|
|
324
325
|
"left_view":{"type":"string"},"right_view":{"type":"string"},"relations":{"type":"boolean"},
|
|
325
326
|
"update_mode": {"type": "string", "enum": ["on_search", "eager"]},
|
|
326
327
|
"view_id": {"type": "string"}, "file_path": {"type": "string"},
|
|
328
|
+
"base": {"type": "string", "default": "HEAD", "description": "affected_tests: git revision the changes are compared with (HEAD = uncommitted work; main or origin/main for a branch)"},
|
|
329
|
+
"symbol": {"type": "string", "description": "graph: function, Class.method, class name or path::name to get its callers, calls, tests and importers"},
|
|
330
|
+
"depth": {"type": "integer", "minimum": 1, "maximum": 4, "default": 3, "description": "graph with symbol: call hops searched for tests"},
|
|
327
331
|
"project_root": {"type": "string"}, "project_id": {"type": "string"},
|
|
328
332
|
"job_id": {"type": "string"}, "watch": {"type": "boolean"},
|
|
329
333
|
"manual": {"type": "boolean"}, "force_scope": {"type": "boolean"},
|
|
@@ -1414,27 +1418,69 @@ def _find_duplicates(root, arguments, dismissed):
|
|
|
1414
1418
|
return duplicates.find(root, embed, configured_model, float(similarity), include_tests, limit, dismissed, include_dismissed)
|
|
1415
1419
|
|
|
1416
1420
|
|
|
1421
|
+
def _with_project_id(arguments):
|
|
1422
|
+
"""Agents know the folder, not the id: project_root resolves to the registered project's project_id."""
|
|
1423
|
+
root = arguments.get("project_root")
|
|
1424
|
+
if arguments.get("project_id") or not isinstance(root, str) or arguments.get("action") in ("register", "relocate"):
|
|
1425
|
+
return arguments
|
|
1426
|
+
wanted = project_identity.canonical_root(root)
|
|
1427
|
+
match = next((p for p in project_store.all_projects() if project_identity.canonical_root(p["root"]) == wanted), None)
|
|
1428
|
+
if not match:
|
|
1429
|
+
raise ValueError(f"No registered project at {root}; register it first (action=register).")
|
|
1430
|
+
return {**arguments, "project_id": match["id"]}
|
|
1431
|
+
|
|
1432
|
+
|
|
1433
|
+
def _project_summary(project):
|
|
1434
|
+
data = _project_payload(project)
|
|
1435
|
+
index = data.get("index") or {}
|
|
1436
|
+
jobs = data.get("jobs") or []
|
|
1437
|
+
return {"project_id": data["id"], "name": data.get("name"), "root": data["root"], "status": data.get("status"),
|
|
1438
|
+
"enabled": data.get("enabled"), "paused": data.get("paused"), "update_mode": data.get("update_mode"),
|
|
1439
|
+
"files": index.get("files"), "chunks": index.get("chunks"), "last_indexed": data.get("last_indexed"),
|
|
1440
|
+
"view": (data.get("view") or {}).get("label"), "active_jobs": len(data.get("active_jobs") or []),
|
|
1441
|
+
"last_job": {k: jobs[0].get(k) for k in ("job_id", "status", "kind")} if jobs else None,
|
|
1442
|
+
"last_error": (data.get("last_error") or None) and str(data["last_error"])[:200]}
|
|
1443
|
+
|
|
1444
|
+
|
|
1417
1445
|
def _project_action(arguments):
|
|
1418
1446
|
if not isinstance(arguments, dict):
|
|
1419
1447
|
raise ValueError("Pass the action in a JSON object.")
|
|
1448
|
+
arguments = _with_project_id(arguments)
|
|
1420
1449
|
action = arguments.get("action", "list")
|
|
1421
1450
|
if action == "probe":
|
|
1422
1451
|
return _probe_index_models()
|
|
1423
|
-
if action in ('graph','compare','duplicates','duplicates_dismiss','duplicates_restore'):
|
|
1452
|
+
if action in ('graph','compare','duplicates','duplicates_dismiss','duplicates_restore','docs','affected_tests'):
|
|
1424
1453
|
# Análise somente leitura pode demorar; não prende o lock global de ações.
|
|
1425
1454
|
key=arguments.get('project_id')
|
|
1426
1455
|
if not isinstance(key,str) or not re.fullmatch(r'[a-f0-9]{16}',key):
|
|
1427
|
-
raise ValueError('Pass a valid project_id.')
|
|
1456
|
+
raise ValueError('Pass project_root (the folder) or a valid project_id.')
|
|
1428
1457
|
project=project_store.get(key)
|
|
1429
1458
|
if not project:
|
|
1430
1459
|
raise ValueError('Project not registered.')
|
|
1431
1460
|
if action=='duplicates':
|
|
1432
1461
|
return _find_duplicates(project['root'], arguments, project.get('dismissed_duplicates') or {})
|
|
1462
|
+
if action=='docs':
|
|
1463
|
+
import doc_check
|
|
1464
|
+
include_tests=arguments.get('include_tests',False)
|
|
1465
|
+
if type(include_tests) is not bool:
|
|
1466
|
+
raise ValueError('include_tests must be true or false.')
|
|
1467
|
+
limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
|
|
1468
|
+
return doc_check.coverage(project['root'],include_tests,limit,arguments.get('view_id'))
|
|
1469
|
+
if action=='affected_tests':
|
|
1470
|
+
import affected_tests
|
|
1471
|
+
limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
|
|
1472
|
+
return affected_tests.affected(project['root'],arguments.get('base','HEAD'),limit,arguments.get('view_id'))
|
|
1433
1473
|
if action in ('duplicates_dismiss','duplicates_restore'):
|
|
1434
1474
|
return _change_dismissal(key, project, arguments, action == 'duplicates_dismiss')
|
|
1435
1475
|
if action=='graph':
|
|
1436
1476
|
import code_graph
|
|
1437
|
-
|
|
1477
|
+
import code_impact
|
|
1478
|
+
limit=_parse_bounded_int(arguments.get('limit'),default=30,min_v=1,max_v=200,field_name='limit')
|
|
1479
|
+
if arguments.get('symbol') is not None:
|
|
1480
|
+
return code_impact.impact(project['root'],arguments['symbol'],arguments.get('view_id'),
|
|
1481
|
+
arguments.get('depth',3),limit)
|
|
1482
|
+
graph=code_graph.build(project['root'],arguments.get('view_id'),file_path=arguments.get('file_path'))
|
|
1483
|
+
return code_impact.for_agent(graph,limit) if arguments.get('file_path') else graph
|
|
1438
1484
|
import view_compare
|
|
1439
1485
|
return view_compare.compare(project['root'],arguments.get('left_view'),arguments.get('right_view'),
|
|
1440
1486
|
arguments.get('file_path'),arguments.get('relations',True))
|
|
@@ -1513,6 +1559,9 @@ def _project_action_locked(arguments):
|
|
|
1513
1559
|
_metric("install_hook", target=result["client"], changed=result["changed"])
|
|
1514
1560
|
return result
|
|
1515
1561
|
if action == "list":
|
|
1562
|
+
if not arguments.get("full"):
|
|
1563
|
+
return {"projects": [_project_summary(p) for p in project_store.all_projects()],
|
|
1564
|
+
"detail": "Use action=status with project_id or project_root for scope, jobs and index details."}
|
|
1516
1565
|
return {"projects": [_project_payload(p) for p in project_store.all_projects()],
|
|
1517
1566
|
"storage_dir": indexer.INDEX_DIR,
|
|
1518
1567
|
"gateway": dict(_GATEWAY_STATUS),
|
|
@@ -1526,7 +1575,7 @@ def _project_action_locked(arguments):
|
|
|
1526
1575
|
return {"project": _project_payload(project)}
|
|
1527
1576
|
key = arguments.get("project_id")
|
|
1528
1577
|
if not isinstance(key, str) or not re.fullmatch(r"[a-f0-9]{16}", key):
|
|
1529
|
-
raise ValueError("Pass
|
|
1578
|
+
raise ValueError("Pass project_root (the folder) or a valid project_id.")
|
|
1530
1579
|
project = project_store.get(key)
|
|
1531
1580
|
if not project:
|
|
1532
1581
|
raise ValueError("Project not registered.")
|
|
@@ -1995,10 +2044,8 @@ def _fan_out_web_search(query, num, budget_s=None, site_domains=()):
|
|
|
1995
2044
|
try:
|
|
1996
2045
|
fut.result()
|
|
1997
2046
|
_record_tier_health(tier_name, success=True)
|
|
1998
|
-
except
|
|
1999
|
-
|
|
2000
|
-
except Exception:
|
|
2001
|
-
_record_tier_health(tier_name, success=False)
|
|
2047
|
+
except Exception as exc:
|
|
2048
|
+
_record_tier_failure(tier_name, exc)
|
|
2002
2049
|
return _cb
|
|
2003
2050
|
|
|
2004
2051
|
for fut, tier_name in futures.items():
|
|
@@ -2243,8 +2290,7 @@ def _search_web_once(query, num, scope, deadline=None, assess_semantic=False,
|
|
|
2243
2290
|
if not raw_results:
|
|
2244
2291
|
raise RuntimeError("zero results in the requested domain")
|
|
2245
2292
|
except Exception as exc:
|
|
2246
|
-
|
|
2247
|
-
_record_tier_health(tier_name, success=False)
|
|
2293
|
+
_record_tier_failure(tier_name, exc)
|
|
2248
2294
|
errors.append(_sanitize_text(f"{tier_name}: {type(exc).__name__}: {exc}"))
|
|
2249
2295
|
continue
|
|
2250
2296
|
_record_tier_health(tier_name, success=True)
|
|
@@ -2278,8 +2324,7 @@ def _search_web_once(query, num, scope, deadline=None, assess_semantic=False,
|
|
|
2278
2324
|
if not raw_results:
|
|
2279
2325
|
raise RuntimeError("zero results in the requested domain")
|
|
2280
2326
|
except Exception as exc:
|
|
2281
|
-
|
|
2282
|
-
_record_tier_health(tier_name, success=False)
|
|
2327
|
+
_record_tier_failure(tier_name, exc)
|
|
2283
2328
|
errors.append(_sanitize_text(f"{tier_name}: {type(exc).__name__}: {exc}"))
|
|
2284
2329
|
continue
|
|
2285
2330
|
_record_tier_health(tier_name, success=True)
|
|
@@ -2588,13 +2633,26 @@ def _annotate_quick_results(rows, classification, cache=None):
|
|
|
2588
2633
|
return result
|
|
2589
2634
|
|
|
2590
2635
|
|
|
2591
|
-
def _record_tier_health(tier_name, success):
|
|
2636
|
+
def _record_tier_health(tier_name, success, outcome=None):
|
|
2592
2637
|
try:
|
|
2593
|
-
web_search_health.record(tier_name, success)
|
|
2638
|
+
web_search_health.record(tier_name, success, outcome)
|
|
2594
2639
|
except Exception:
|
|
2595
2640
|
pass # telemetria é best-effort — nunca pode derrubar uma busca já resolvida
|
|
2596
2641
|
|
|
2597
2642
|
|
|
2643
|
+
def _record_tier_failure(tier_name, exc):
|
|
2644
|
+
"""Health entry for a failed tier call, keeping a 429, a captcha and a quota pause apart from other errors."""
|
|
2645
|
+
if isinstance(exc, (TierSkipped, browser_service_client.BrowserSkipped)):
|
|
2646
|
+
outcome = "paused"
|
|
2647
|
+
elif isinstance(exc, WebProviderRateLimited):
|
|
2648
|
+
outcome = "rate_limited"
|
|
2649
|
+
elif isinstance(exc, browser_service_client.BrowserCaptcha):
|
|
2650
|
+
outcome = "captcha"
|
|
2651
|
+
else:
|
|
2652
|
+
outcome = "failed"
|
|
2653
|
+
_record_tier_health(tier_name, False, outcome)
|
|
2654
|
+
|
|
2655
|
+
|
|
2598
2656
|
_is_safe_crawl_target = web_fetch.is_safe_target
|
|
2599
2657
|
|
|
2600
2658
|
|
|
@@ -3073,6 +3131,7 @@ def _tool_result(text, is_error=False):
|
|
|
3073
3131
|
|
|
3074
3132
|
WEB_FETCH_TIMEOUT_S = 90
|
|
3075
3133
|
WEB_FETCH_RENDER_TIMEOUT_S = 60
|
|
3134
|
+
WEB_FETCH_VECTOR_ROOT = os.path.join(paths.DATA_DIR, "web-fetch")
|
|
3076
3135
|
|
|
3077
3136
|
|
|
3078
3137
|
def _render_page(url, deadline):
|
|
@@ -3085,6 +3144,30 @@ def _render_page(url, deadline):
|
|
|
3085
3144
|
return (pages[0].get("markdown") or "") if pages else ""
|
|
3086
3145
|
|
|
3087
3146
|
|
|
3147
|
+
def _web_fetch_embed(texts, deadline):
|
|
3148
|
+
"""Vectors for the pieces of a long page and the prompt, with the pieces cached per embedding model. Fails loudly
|
|
3149
|
+
without an embedding model: a long page is never cut back to its start in silence."""
|
|
3150
|
+
model = local_embedder.resolve(config.load_config().get("embedding_model"), None)
|
|
3151
|
+
if not model or model == indexer.LEXICAL_MODEL:
|
|
3152
|
+
raise RuntimeError("This page is longer than 60,000 characters and choosing its relevant parts needs an "
|
|
3153
|
+
"embedding_model, which is not configured")
|
|
3154
|
+
try:
|
|
3155
|
+
token = model_client.get_token()
|
|
3156
|
+
probe = _embed(model, texts[-1:], token, deadline=deadline)
|
|
3157
|
+
cache = embedding_cache.VectorCache(WEB_FETCH_VECTOR_ROOT, indexer.INDEX_DIR, {
|
|
3158
|
+
"purpose": "web_fetch_pieces", "model": model, "dimensions": len(probe[0])})
|
|
3159
|
+
try:
|
|
3160
|
+
pieces = cache.embed(texts[:-1], lambda batch: _embed(model, batch, token, deadline=deadline),
|
|
3161
|
+
indexer.validate_vector, split=lambda unique: list(_embed_batches(unique)),
|
|
3162
|
+
workers=indexer.EMBED_WORKERS)
|
|
3163
|
+
finally:
|
|
3164
|
+
cache.db.close()
|
|
3165
|
+
except Exception as exc:
|
|
3166
|
+
raise RuntimeError(f"This page is longer than 60,000 characters and the embedding model ({model}) that picks "
|
|
3167
|
+
f"its relevant parts failed: {type(exc).__name__}: {exc}"[:400]) from None
|
|
3168
|
+
return pieces + probe
|
|
3169
|
+
|
|
3170
|
+
|
|
3088
3171
|
def _web_fetch_unavailable():
|
|
3089
3172
|
cfg = config.load_config()
|
|
3090
3173
|
if not cfg.get("router_model"):
|
|
@@ -3105,18 +3188,22 @@ def _handle_web_fetch(arguments):
|
|
|
3105
3188
|
raise RuntimeError(f"Gateway model unavailable ({problem}).")
|
|
3106
3189
|
deadline = started + WEB_FETCH_TIMEOUT_S
|
|
3107
3190
|
page = web_fetch.read_page(url, deadline, _render_page)
|
|
3108
|
-
|
|
3191
|
+
selected, pieces = web_fetch.select(page["text"], prompt, lambda texts: _web_fetch_embed(texts, deadline))
|
|
3192
|
+
text = web_fetch.answer(config.load_config()["router_model"], prompt, {**page, "text": selected}, deadline)
|
|
3109
3193
|
except Exception as exc:
|
|
3110
3194
|
web_fetch.mark_failed(str(url))
|
|
3111
3195
|
_metric("web_fetch", error=type(exc).__name__, elapsed_s=round(time.monotonic() - started, 2))
|
|
3112
3196
|
raise RuntimeError(f"web_fetch failed: {str(exc).rstrip('.')}. The native WebFetch is allowed for this URL for "
|
|
3113
3197
|
f"{web_fetch.FAILED_TTL_S // 60} min.") from None
|
|
3114
3198
|
_metric("web_fetch", cached=page["cached"], rendered=page["rendered"], render_error=page.get("render_error"),
|
|
3115
|
-
page_chars=
|
|
3199
|
+
page_chars=page["chars"], read_chars=len(selected), pieces=pieces,
|
|
3116
3200
|
answer_chars=len(text), elapsed_s=round(time.monotonic() - started, 2))
|
|
3117
3201
|
origin = "from cache" if page["cached"] else "rendered in the browser" if page["rendered"] else "downloaded now"
|
|
3118
3202
|
read_at = datetime.datetime.fromtimestamp(page["fetched_at"]).strftime("%Y-%m-%d %H:%M")
|
|
3119
|
-
cut = "; page
|
|
3203
|
+
cut = (f"; page of {page['chars']:,} characters: the {pieces} parts most related to the request were read"
|
|
3204
|
+
if pieces else "")
|
|
3205
|
+
if page["truncated"]:
|
|
3206
|
+
cut += f"; only the first {web_fetch.PAGE_STORE_MAX_CHARS:,} characters were kept"
|
|
3120
3207
|
return _sanitize_text(f"Source: {page['final_url']} (read at {read_at}, {origin}{cut})\n\n{text}")
|
|
3121
3208
|
|
|
3122
3209
|
|
|
@@ -3395,7 +3482,7 @@ class MCPHandler(BaseHTTPRequestHandler):
|
|
|
3395
3482
|
if self.path == "/setup/projects":
|
|
3396
3483
|
try:
|
|
3397
3484
|
setup_ui.require_token(self.headers)
|
|
3398
|
-
self._write_json(200, _project_action({"action": "list"}))
|
|
3485
|
+
self._write_json(200, _project_action({"action": "list", "full": True}))
|
|
3399
3486
|
except setup_ui.SetupError as exc:
|
|
3400
3487
|
self._write_json(403, {"error": str(exc)})
|
|
3401
3488
|
except Exception as exc:
|
package/version.py
CHANGED
|
@@ -1,16 +1,26 @@
|
|
|
1
|
-
"""`camoufox fetch` for the installer
|
|
2
|
-
|
|
3
|
-
|
|
1
|
+
"""`camoufox fetch` for the installer. camoufox 0.5.6 prints a failed download and still exits 0, so the error is
|
|
2
|
+
caught here and turned into a failing exit, and the browser executable must exist afterwards. The 1.3 GB download
|
|
3
|
+
goes through ranged_download (Range requests that resume after a dropped or stalled connection; camoufox still checks
|
|
4
|
+
the sha256), its other requests get a timeout (requests passes None, which also overrides socket.setdefaulttimeout),
|
|
5
|
+
and progress is printed as plain lines (the rich bar stays invisible in a pipe)."""
|
|
6
|
+
import sys
|
|
7
|
+
from io import BytesIO
|
|
8
|
+
from pathlib import Path
|
|
4
9
|
|
|
10
|
+
import camoufox.__main__ as camoufox_cli
|
|
11
|
+
import requests
|
|
5
12
|
from camoufox import pkgman
|
|
6
|
-
from camoufox.__main__ import cli
|
|
7
13
|
|
|
8
|
-
|
|
14
|
+
import ranged_download
|
|
15
|
+
|
|
16
|
+
TIMEOUT_S = (30, 60)
|
|
9
17
|
REPORT_EVERY_MB = 50
|
|
10
|
-
|
|
18
|
+
_update = camoufox_cli.CamoufoxUpdate.update
|
|
19
|
+
_get = requests.get
|
|
20
|
+
failures = []
|
|
11
21
|
|
|
12
22
|
|
|
13
|
-
def
|
|
23
|
+
def _reporter():
|
|
14
24
|
reported = {"mb": -REPORT_EVERY_MB}
|
|
15
25
|
|
|
16
26
|
def report(done, total):
|
|
@@ -19,10 +29,49 @@ def _reporting_webdl(url, desc=None, buffer=None, bar=True, progress_callback=No
|
|
|
19
29
|
reported["mb"] = mb
|
|
20
30
|
print(f"downloaded {mb} of {total // 1048576} MB", flush=True)
|
|
21
31
|
|
|
22
|
-
return
|
|
32
|
+
return report
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _ranged_webdl(url, desc=None, buffer=None, bar=True, progress_callback=None):
|
|
36
|
+
buffer = BytesIO() if buffer is None else buffer
|
|
37
|
+
ranged_download.download(url, buffer, progress=progress_callback or _reporter())
|
|
38
|
+
return buffer
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _get_with_timeout(*args, **kwargs):
|
|
42
|
+
kwargs.setdefault("timeout", TIMEOUT_S)
|
|
43
|
+
return _get(*args, **kwargs)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _recording_update(self, *args, **kwargs):
|
|
47
|
+
try:
|
|
48
|
+
return _update(self, *args, **kwargs)
|
|
49
|
+
except Exception as exc:
|
|
50
|
+
failures.append(exc)
|
|
51
|
+
raise
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def installed_executable():
|
|
55
|
+
"""The browser executable, without the download that launch_path() starts when it is missing."""
|
|
56
|
+
executable = Path(pkgman.camoufox_path(download_if_missing=False)) / pkgman.LAUNCH_FILE[pkgman.OS_NAME]
|
|
57
|
+
if not executable.is_file():
|
|
58
|
+
raise FileNotFoundError(executable)
|
|
59
|
+
return executable
|
|
60
|
+
|
|
23
61
|
|
|
62
|
+
def main():
|
|
63
|
+
requests.get = _get_with_timeout
|
|
64
|
+
pkgman.webdl = _ranged_webdl
|
|
65
|
+
camoufox_cli.CamoufoxUpdate.update = _recording_update
|
|
66
|
+
camoufox_cli.cli.main(["fetch"], standalone_mode=False)
|
|
67
|
+
if failures:
|
|
68
|
+
sys.exit(f"Camoufox download failed: {failures[-1]}")
|
|
69
|
+
try:
|
|
70
|
+
executable = installed_executable()
|
|
71
|
+
except Exception as exc:
|
|
72
|
+
sys.exit(f"Camoufox is not installed after the download: {exc}")
|
|
73
|
+
print(f"Camoufox {pkgman.installed_verstr()} ready: {executable}", flush=True)
|
|
24
74
|
|
|
25
|
-
pkgman.webdl = _reporting_webdl
|
|
26
75
|
|
|
27
76
|
if __name__ == "__main__":
|
|
28
|
-
|
|
77
|
+
main()
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
"""Downloads a file in HTTP Range requests of CHUNK_BYTES, each on a new connection, resuming from the last byte
|
|
2
|
+
received when a connection drops or stalls. Some networks (proxies, TLS inspection) cut every connection after a few
|
|
3
|
+
hundred MB, so a single long transfer can never finish there. It gives up only after MAX_IDLE_ATTEMPTS attempts in a
|
|
4
|
+
row without a single new byte. Standard library only: on Windows urllib trusts the system certificate store, including
|
|
5
|
+
a corporate inspection CA, and honors HTTPS_PROXY."""
|
|
6
|
+
import http.client
|
|
7
|
+
import time
|
|
8
|
+
import urllib.error
|
|
9
|
+
import urllib.request
|
|
10
|
+
|
|
11
|
+
CHUNK_BYTES = 32 * 1024 * 1024
|
|
12
|
+
READ_BYTES = 256 * 1024
|
|
13
|
+
TIMEOUT_S = 60
|
|
14
|
+
MAX_IDLE_ATTEMPTS = 6
|
|
15
|
+
MAX_FULL_RESTARTS = 3
|
|
16
|
+
RETRYABLE_HTTP = {408, 429}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class DownloadError(RuntimeError):
|
|
20
|
+
pass
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _open(url, start, end, timeout):
|
|
24
|
+
request = urllib.request.Request(url, headers={"Range": f"bytes={start}-{end}", "User-Agent": "SmartTool-installer"})
|
|
25
|
+
return urllib.request.urlopen(request, timeout=timeout)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def download(url, out, progress=None, chunk_bytes=CHUNK_BYTES, timeout=TIMEOUT_S, attempts=MAX_IDLE_ATTEMPTS,
|
|
29
|
+
sleep=time.sleep):
|
|
30
|
+
"""Writes url into the seekable binary file out and returns its size; progress(done, total) after every read."""
|
|
31
|
+
done, total, idle, restarts = 0, None, 0, 0
|
|
32
|
+
while total is None or done < total:
|
|
33
|
+
end = done + chunk_bytes - 1 if total is None else min(done + chunk_bytes, total) - 1
|
|
34
|
+
before = done
|
|
35
|
+
try:
|
|
36
|
+
with _open(url, done, end, timeout) as response:
|
|
37
|
+
if response.status == 200:
|
|
38
|
+
if done:
|
|
39
|
+
restarts += 1
|
|
40
|
+
if restarts > MAX_FULL_RESTARTS:
|
|
41
|
+
raise DownloadError(f"{url} ignores Range requests and broke {restarts} times; giving up.")
|
|
42
|
+
out.seek(0)
|
|
43
|
+
out.truncate()
|
|
44
|
+
done = before = 0
|
|
45
|
+
length = response.headers.get("Content-Length")
|
|
46
|
+
total = int(length) if length else None
|
|
47
|
+
elif response.status == 206:
|
|
48
|
+
total = int(response.headers["Content-Range"].rsplit("/", 1)[1])
|
|
49
|
+
else:
|
|
50
|
+
raise DownloadError(f"{url} answered HTTP {response.status} to a Range request.")
|
|
51
|
+
out.seek(done)
|
|
52
|
+
while data := response.read1(READ_BYTES):
|
|
53
|
+
out.write(data)
|
|
54
|
+
done += len(data)
|
|
55
|
+
if progress:
|
|
56
|
+
progress(done, total or 0)
|
|
57
|
+
if response.status == 200 and total is None:
|
|
58
|
+
total = done
|
|
59
|
+
except urllib.error.HTTPError as exc:
|
|
60
|
+
exc.close()
|
|
61
|
+
if exc.code < 500 and exc.code not in RETRYABLE_HTTP:
|
|
62
|
+
raise DownloadError(f"{url} answered HTTP {exc.code}.") from exc
|
|
63
|
+
error = exc
|
|
64
|
+
except (OSError, http.client.HTTPException) as exc:
|
|
65
|
+
error = exc
|
|
66
|
+
else:
|
|
67
|
+
error = None
|
|
68
|
+
if done > before:
|
|
69
|
+
idle = 0
|
|
70
|
+
elif error is not None or done < (total or 0):
|
|
71
|
+
idle += 1
|
|
72
|
+
if idle >= attempts:
|
|
73
|
+
raise DownloadError(f"Download stopped at {done // 1048576} of {(total or 0) // 1048576} MB after "
|
|
74
|
+
f"{attempts} attempts without progress: {error or 'empty response'}")
|
|
75
|
+
sleep(min(2 ** idle, 30))
|
|
76
|
+
out.flush()
|
|
77
|
+
out.seek(0)
|
|
78
|
+
return done
|
package/web_fetch.py
CHANGED
|
@@ -6,7 +6,14 @@ e do OpenCode (`packages/opencode/src/tool/webfetch.ts`: limite de 5 MB, User-Ag
|
|
|
6
6
|
renderizada: melhor resposta 18 × 18 (1 empate), respondeu certo 22 × 16, caracteres devolvidos 35 mil × 72 mil;
|
|
7
7
|
3,6 s de mediana, navegador em 7 de 37 (páginas montadas por JavaScript). Repetições da mesma URL nos transcripts:
|
|
8
8
|
92 de 124 dentro de 24 h (PAGE_TTL_S).
|
|
9
|
+
|
|
10
|
+
Páginas acima de PAGE_MAX_CHARS (14,6% do cache, 10,4% das chamadas reais em 2026-10-08) não são mais cortadas no
|
|
11
|
+
começo: o texto inteiro fica no cache e o modelo lê os pedaços mais parecidos com o pedido, até SELECT_BUDGET_CHARS, na
|
|
12
|
+
ordem da página. Medido em 18 páginas reais e 41 perguntas: trechos depois do corte 0/28 → 13/28, cabeça 6/13 → 5/13,
|
|
13
|
+
latência mediana 3,3 s → 3,1 s. Abaixo do limite a página vai inteira (seleção ali empatou em qualidade e piorou a
|
|
14
|
+
cauda de latência; docs/BACKLOG.md).
|
|
9
15
|
"""
|
|
16
|
+
import math
|
|
10
17
|
import codecs
|
|
11
18
|
import http.client
|
|
12
19
|
import ipaddress
|
|
@@ -25,6 +32,10 @@ import research_cache
|
|
|
25
32
|
CACHE_NAMESPACE = "web_fetch"
|
|
26
33
|
PAGE_TTL_S = 86400
|
|
27
34
|
PAGE_MAX_CHARS = 60_000
|
|
35
|
+
PAGE_STORE_MAX_CHARS = 1_000_000
|
|
36
|
+
SELECT_BUDGET_CHARS = 15_000
|
|
37
|
+
CHUNK_CHARS = 2_000
|
|
38
|
+
PAGE_FORMAT = 2
|
|
28
39
|
MAX_DOWNLOAD_BYTES = 5 * 1024 * 1024
|
|
29
40
|
MAX_REDIRECTS = 5
|
|
30
41
|
HTTP_TIMEOUT_S = 20
|
|
@@ -201,7 +212,8 @@ def read_page(url, deadline, render):
|
|
|
201
212
|
"""Page text (Markdown) from the 24 h cache, HTTP + trafilatura, or `render(url, deadline)` when HTTP comes thin."""
|
|
202
213
|
url = normalize(url)
|
|
203
214
|
for entry in research_cache.candidates(CACHE_NAMESPACE, url, limit=1):
|
|
204
|
-
if entry.get("query") == url and time.time() - entry["saved_at"] < PAGE_TTL_S
|
|
215
|
+
if (entry.get("query") == url and time.time() - entry["saved_at"] < PAGE_TTL_S
|
|
216
|
+
and entry["result"].get("format") == PAGE_FORMAT):
|
|
205
217
|
return {**entry["result"], "cached": True, "fetched_at": entry["saved_at"]}
|
|
206
218
|
final_url, kind, charset, body = fetch_http(url, deadline)
|
|
207
219
|
rendered, render_error = False, None
|
|
@@ -223,12 +235,57 @@ def read_page(url, deadline, render):
|
|
|
223
235
|
if len(text.strip()) < MIN_CHARS:
|
|
224
236
|
detail = f"; browser failed: {render_error}" if render_error else ", not even when rendered in the browser"
|
|
225
237
|
raise FetchError(f"Page without readable text ({final_url}){detail}.")
|
|
226
|
-
page = {"url": url, "final_url": final_url, "text": text[:
|
|
227
|
-
"rendered": rendered}
|
|
238
|
+
page = {"url": url, "final_url": final_url, "text": text[:PAGE_STORE_MAX_CHARS], "chars": len(text),
|
|
239
|
+
"truncated": len(text) > PAGE_STORE_MAX_CHARS, "rendered": rendered, "format": PAGE_FORMAT}
|
|
228
240
|
saved_at = research_cache.put(CACHE_NAMESPACE, url, page, {})
|
|
229
241
|
return {**page, "cached": False, "fetched_at": saved_at or time.time(), "render_error": render_error}
|
|
230
242
|
|
|
231
243
|
|
|
244
|
+
def chunks(text, size=CHUNK_CHARS):
|
|
245
|
+
"""Consecutive pieces of about size characters, cut at a line break in the second half of each piece."""
|
|
246
|
+
parts, start = [], 0
|
|
247
|
+
while start < len(text):
|
|
248
|
+
end = min(len(text), start + size)
|
|
249
|
+
if end < len(text):
|
|
250
|
+
cut = text.rfind("\n", start + size // 2, end)
|
|
251
|
+
end = cut + 1 if cut > start else end
|
|
252
|
+
parts.append(text[start:end])
|
|
253
|
+
start = end
|
|
254
|
+
return parts
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _cosine(a, b):
|
|
258
|
+
norm = math.sqrt(sum(x * x for x in a)) * math.sqrt(sum(y * y for y in b))
|
|
259
|
+
return sum(x * y for x, y in zip(a, b)) / norm if norm else 0.0
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def select(text, prompt, embed, budget=SELECT_BUDGET_CHARS):
|
|
263
|
+
"""Text the model reads: the whole page up to PAGE_MAX_CHARS; above it, the pieces most similar to the prompt up
|
|
264
|
+
to budget characters, in page order, with [...] where pieces were skipped. embed(texts) returns one vector per
|
|
265
|
+
text and raises when the embedding model is unavailable. Returns (text, number of pieces kept or None)."""
|
|
266
|
+
if len(text) <= PAGE_MAX_CHARS:
|
|
267
|
+
return text, None
|
|
268
|
+
parts = chunks(text)
|
|
269
|
+
vectors = embed(parts + [prompt])
|
|
270
|
+
if len(vectors) != len(parts) + 1:
|
|
271
|
+
raise RuntimeError("The embedding model returned a different number of vectors than page pieces.")
|
|
272
|
+
query = vectors[-1]
|
|
273
|
+
ranked = sorted(range(len(parts)), key=lambda i: -_cosine(vectors[i], query))
|
|
274
|
+
kept, size = [], 0
|
|
275
|
+
for i in ranked:
|
|
276
|
+
if size + len(parts[i]) <= budget:
|
|
277
|
+
kept.append(i)
|
|
278
|
+
size += len(parts[i])
|
|
279
|
+
kept.sort()
|
|
280
|
+
pieces, last = [], None
|
|
281
|
+
for i in kept:
|
|
282
|
+
if last is not None and i != last + 1:
|
|
283
|
+
pieces.append("\n\n[...]\n\n")
|
|
284
|
+
pieces.append(parts[i])
|
|
285
|
+
last = i
|
|
286
|
+
return "".join(pieces), len(kept)
|
|
287
|
+
|
|
288
|
+
|
|
232
289
|
def answer(model, prompt, page, deadline):
|
|
233
290
|
response = model_client.fetch("/v1/chat/completions", model_client.get_token(), method="POST",
|
|
234
291
|
timeout=min(90, _left(deadline)), body={
|
package/web_search_health.py
CHANGED
|
@@ -21,9 +21,17 @@ _MAX_LINES_BEFORE_ROTATE = 5000
|
|
|
21
21
|
_LOCK = threading.Lock()
|
|
22
22
|
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
OUTCOMES = ("ok", "failed", "rate_limited", "captcha", "paused")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def record(tier_name, success, outcome=None):
|
|
28
|
+
"""One tier call: outcome tells a provider 429 (rate_limited), a captcha and a self-imposed quota pause (paused,
|
|
29
|
+
left out of success rates) apart from an ordinary failure, so quota pressure can be read from the log."""
|
|
30
|
+
outcome = outcome or ("ok" if success else "failed")
|
|
31
|
+
if outcome not in OUTCOMES:
|
|
32
|
+
raise ValueError(f"Unknown web search outcome: {outcome}")
|
|
25
33
|
os.makedirs(os.path.dirname(HEALTH_PATH), exist_ok=True)
|
|
26
|
-
entry = {"ts": time.time(), "tier": tier_name, "success": bool(success)}
|
|
34
|
+
entry = {"ts": time.time(), "tier": tier_name, "success": bool(success), "outcome": outcome}
|
|
27
35
|
with _LOCK:
|
|
28
36
|
with open(HEALTH_PATH, "a", encoding="utf-8") as f:
|
|
29
37
|
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
|
|
@@ -67,7 +75,7 @@ def _read_recent_records():
|
|
|
67
75
|
|
|
68
76
|
|
|
69
77
|
def recent_success_rate(tier_name, window=DEFAULT_WINDOW):
|
|
70
|
-
records = [r for r in _read_recent_records() if r.get("tier") == tier_name]
|
|
78
|
+
records = [r for r in _read_recent_records() if r.get("tier") == tier_name and r.get("outcome") != "paused"]
|
|
71
79
|
if not records:
|
|
72
80
|
return None
|
|
73
81
|
records = records[-window:]
|
|
@@ -79,7 +87,7 @@ def summary_for_tiers(tier_names, window=DEFAULT_WINDOW):
|
|
|
79
87
|
all_records = _read_recent_records()
|
|
80
88
|
summary = {}
|
|
81
89
|
for name in tier_names:
|
|
82
|
-
records = [r for r in all_records if r.get("tier") == name][-window:]
|
|
90
|
+
records = [r for r in all_records if r.get("tier") == name and r.get("outcome") != "paused"][-window:]
|
|
83
91
|
if not records:
|
|
84
92
|
summary[name] = None
|
|
85
93
|
else:
|