pi-unsloth-webtools 0.9.1 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -38,13 +38,15 @@ Both tools display their target in the TUI tool row: `web_search "query"` and `w
38
38
 
39
39
  Mirrors Unsloth Studio's `web_search` tool:
40
40
 
41
- - Searches exactly like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the same seven engines
42
- (duckduckgo, brave, google, mojeek, yahoo, yandex, wikipedia; bing is disabled upstream),
43
- the same provider deduplication, href-dedupe aggregator with frequency ordering (hrefs are
41
+ - Searches like Studio's pinned `ddgs==9.14.4` `DDGS.text()`: the Studio engines that still work
42
+ here (duckduckgo, yandex; the others are behind bot walls, see
43
+ [Known differences from Studio](#known-differences-from-studio)) plus startpage,
44
+ the same provider deduplication, and an href-dedupe aggregator (hrefs are
44
45
  canonicalized first — `utm_*`/tracking parameters and fragments are dropped and the URL is
45
46
  re-serialized, collapsing host-case, default-port, and trailing-slash variants — so the
46
- same page found via different tracking links collapses), and the same `SimpleFilterRanker`
47
- re-ranking. Formats results identically: `Title:` / `URL:` /
47
+ same page found via different tracking links collapses). Results are fused with weighted
48
+ reciprocal-rank fusion: each engine's own ordering, not a keyword guess, decides relevance,
49
+ and the fused list is capped per registrable domain. Formats results identically:
48
50
  `Snippet:` blocks separated by `---`, ending with the hint to call `web_fetch` to
49
51
  read a full page.
50
52
  - Rate-limit, timeout, and empty-result messages mirror Studio's `_search_failure_message`.
@@ -53,6 +55,11 @@ Mirrors Unsloth Studio's `web_search` tool:
53
55
  skipped); timeouts and cancellations are never retried.
54
56
  - Sweeps stop as soon as enough results are gathered: engines still in flight are aborted
55
57
  instead of being allowed to run to their timeout.
58
+ - Engine requests go through the browser-fingerprint transport first (`webFetch.transport`,
59
+ default `tls-first`) and fall back to the plain Node transport; a refusal on one transport is
60
+ retried on the other. Redirects are followed manually, so every hop is re-checked against the
61
+ website policy and private-address literals are refused. When a SOCKS5 proxy is configured the
62
+ sweep stays on the plain transport, keeping agent and Tor routing intact.
56
63
 
57
64
  ### web_fetch
58
65
 
@@ -218,11 +225,31 @@ third-party rendering service.
218
225
  any numeric-only line at a fixed edge position where page numbers appear on at least
219
226
  half the pages (so a one-off number sharing that position is dropped too, while fused
220
227
  labels like `Page 3 of 12` survive). Studio and pymupdf4llm return them verbatim.
221
- - Search engines: Node's `fetch` TLS fingerprint differs from ddgs's `primp`
222
- impersonation, so Google/Brave/Yahoo/Yandex may block or serve consent pages more
223
- aggressively (a blocked engine simply contributes no results). User agents are a
224
- fixed browser set plus ddgs's Android Google UA generator, not `fake_useragent`'s
225
- database.
228
+ - Search engines: the sweep speaks a browser's TLS/HTTP2 shape through `wreq-js` first (the same
229
+ transport `web_fetch` uses), so engine fingerprints match Chrome rather than Node's `fetch`;
230
+ the plain Node transport is the fallback. User agents on that fallback are a fixed browser set,
231
+ not `fake_useragent`'s database.
232
+ - Startpage: an extra engine beyond Studio's set, matching newer ddgs (its Google-backed index
233
+ means it fills the `google` provider slot). It answers a plain `GET /sp/search?query=` and
234
+ serves an Anubis proof-of-work challenge to clients that do not look like a browser, so it works
235
+ through the browser-fingerprint transport and the fallback's browser headers, but not a bare
236
+ Node `fetch`. Startpage's POST endpoint and its safesearch parameter are challenge-gated and are
237
+ not used.
238
+ - Unused engines: mojeek, yahoo, google, brave and wikipedia are not part of the sweep. A
239
+ non-JavaScript client cannot get past mojeek (JavaScript challenge) or yahoo (`_bv` bot beacon),
240
+ google and brave answer rate limits and bot interstitials instead of results (HTTP 429 in
241
+ testing), and wikipedia is an API for a source the other engines already return — so the sweep
242
+ no longer spends a slot on it, and `wikipedia.org` results are ranked like any other instead of
243
+ being forced to the top. The remaining three engines can be narrowed further per machine with
244
+ `webSearch.engines` (see Configuration).
245
+ - Ranking: Studio's `SimpleFilterRanker` (keyword buckets with `wikipedia.org` pinned first) is
246
+ replaced by weighted reciprocal-rank fusion over each engine's own ordering. On the
247
+ `npm run engine:eval` query set that lifts mean precision@5 from 0.28 (keyword buckets) to 0.35 —
248
+ the fusion keeps the ranking signal the engines already computed instead of guessing from query
249
+ substrings. Weights default to uniform because the three engines overlap so little (mean Jaccard
250
+ 0.07–0.22) that weighting mostly decides which engine dominates the list rather than which result
251
+ is better. A per-domain cap exists (`webSearch.maxPerHost`) but is off by default: it measurably
252
+ costs precision, and the uncapped fused top-5 already spans 4.3 of 5 distinct domains.
226
253
  - Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
227
254
  timeout whenever any engine timed out, so the timeout message is not masked by later
228
255
  generic engine failures. The timeout budget bounds the entire sweep: per-engine
@@ -231,16 +258,16 @@ third-party rendering service.
231
258
  - Proxies: Studio routes through environment proxies; this port resolves and pins the target IP and
232
259
  tunnels that connection through `HTTPS_PROXY` / `HTTP_PROXY` / `ALL_PROXY` when the proxy is a
233
260
  SOCKS5 proxy (`NO_PROXY` exclusions respected; DNS stays local for the guard). Other proxy
234
- schemes fall back to a direct connection. The search path uses the process-wide `fetch`, so an
235
- agent-level proxy dispatcher applies there too — see
236
- [Companion: rotating exit IPs](#companion-rotating-exit-ips).
261
+ schemes fall back to a direct connection. The search sweep stays on the process-wide `fetch`
262
+ whenever a SOCKS5 proxy is configured, so an agent-level proxy dispatcher still applies there —
263
+ see [Companion: rotating exit IPs](#companion-rotating-exit-ips).
237
264
  - Dedup and titles: the aggregator keys on canonicalized hrefs (`utm_*`/tracking parameters
238
265
  and fragments stripped, then the URL re-serialized); fetched HTML pages are prefixed with
239
266
  the document `<title>`. Studio keys on raw hrefs and returns the converted body alone.
240
267
  - Upstream drift: current ddgs ships ten backends (adding bing, startpage, grokipedia),
241
268
  requires a `vqd` token for DuckDuckGo, and exposes an `extract()` mode. This port
242
- deliberately pins the Studio snapshot — seven engines, bing disabled upstream, no vqd,
243
- no pagination — so engine behavior matches Studio rather than ddgs head.
269
+ uses duckduckgo and yandex from the Studio snapshot plus startpage — bing stays
270
+ disabled, no vqd, no pagination — so engine behavior matches Studio rather than ddgs head.
244
271
 
245
272
  ## When to use alternatives
246
273
 
@@ -266,8 +293,9 @@ choose the best tool per URL. No need to fork this package to add those features
266
293
 
267
294
  [`pi-tor-proxy`](https://github.com/YuGiMob/pi-tor-proxy) routes pi's in-process `fetch` traffic
268
295
  through Tor (it downloads and manages its own Tor binary) and gives each pi instance its own
269
- circuit and exit IP. The search sweep uses the process-wide `fetch`, so it leaves through the
270
- current Tor exit, and many search engines rate-limit or challenge per outgoing IP —
296
+ circuit and exit IP. With a SOCKS5 proxy configured, the search sweep stays on the process-wide
297
+ `fetch`, so it leaves through the current Tor exit, and many search engines rate-limit or
298
+ challenge per outgoing IP —
271
299
  `/tor-cycle` swaps the exit those limits are counted against, while `/tor-country` and
272
300
  `/tor-exclude` constrain which exits are used.
273
301
 
@@ -310,7 +338,10 @@ Optional settings in `~/.pi/agent/settings.json` or `.pi/settings.json` (project
310
338
  | `websitePolicy` | none | Not read from settings. Tools run unrestricted by default; `websitePolicy` is a programmatic option the host passes to `webSearch` / `fetchPageText` |
311
339
  | `unslothWebTools.allowPrivateAddresses` / `webFetch.allowPrivateAddresses` | `true` | Opt out to restore the resolved-IP SSRF guard: private/loopback/link-local hosts (localhost, LAN IPs) are refused again. Non-canonical numeric IP encodings stay blocked either way |
312
340
  | `unslothWebTools.allowLocalFiles` / `webFetch.allowLocalFiles` | `true` | Opt out to refuse local files in `web_fetch` (`file://` URLs, absolute, `~/`, or `./` paths); when enabled, PDFs are extracted and HTML converted |
313
- | `webFetch.transport` / `unslothWebTools.transport` | `tls-first` | Fetch transport order: `tls-first` (default), `direct-first`, or `off` to disable the browser-fingerprint transport entirely |
341
+ | `webFetch.transport` / `unslothWebTools.transport` | `tls-first` | Transport order for `web_fetch` and `web_search` engine requests: `tls-first` (default), `direct-first`, or `off` to disable the browser-fingerprint transport entirely |
342
+ | `webSearch.engines` / `unslothWebTools.engines` | all three (duckduckgo, yandex, startpage) | Restrict `web_search` to a subset of engine names, e.g. `["duckduckgo", "yandex"]`; unknown names are ignored, and a list that matches nothing falls back to every engine |
343
+ | `webSearch.engineWeights` / `unslothWebTools.engineWeights` | all `1` | Per-engine fusion weight, e.g. `{"startpage": 2, "yandex": 0.5}`; only positive numbers are read |
344
+ | `webSearch.maxPerHost` / `unslothWebTools.maxPerHost` | `0` (no cap) | Most results one registrable domain may contribute; `0` disables the cap |
314
345
  | `webRender.lightpandaEnabled` / `unslothWebTools.lightpandaEnabled` | `true` | Opt out to disable local Lightpanda rendering |
315
346
  | `webRender.lightpandaPath` / `unslothWebTools.lightpandaPath` | launcher installed by `scripts/install-lightpanda.sh`, else `lightpanda` on `PATH` | Path to the Lightpanda binary used for local rendering |
316
347
  | `webRender.lightpandaCommand` / `unslothWebTools.lightpandaCommand` | none | Command prefix that launches the renderer, for WSL (`["wsl.exe","-e","<path>"]`) or containers; overrides `lightpandaPath`. The fetch flags are appended to it |
@@ -358,6 +389,7 @@ npm run test:unit
358
389
  npm run test:smoke
359
390
  bash scripts/install-lightpanda.sh
360
391
  npm run camoufox:warmup
392
+ npm run engine:eval
361
393
  ```
362
394
 
363
395
  `npm run camoufox:warmup` measures what a warm Camoufox costs and buys: launch time, idle CPU and
@@ -366,6 +398,11 @@ RSS, per-fetch latency with the browser already running, and whether a persisten
366
398
  `--virtual` for a virtual display, `--pin=0` to let Camoufox rotate fingerprints, `--seconds=N`,
367
399
  `--idle=N`.
368
400
 
401
+ `npm run engine:eval` measures each search engine against a small hand-labelled developer query
402
+ set: yield, latency and failures, pairwise URL overlap, unique contribution, and offline fusion
403
+ simulations with candidate weight vectors and per-host caps. `--cache=FILE` reuses a previous run
404
+ without touching the network; `--delay`, `--timeout`, `--max` and `--transport` tune the sweep.
405
+
369
406
  ## Tests
370
407
 
371
408
  `npm test` runs the full suite. `npm run test:unit` skips the live-network smoke tests,
@@ -386,7 +423,7 @@ The suite ports Unsloth Studio's own tests for these tools:
386
423
  - `test/fetch-flow.test.ts`: GitHub README rewrite, deadline/cancellation, HTML sniffing
387
424
  (from `test_web_fetch_extraction.py`; the fetch client is injected via seams)
388
425
  - `test/engines.test.ts`: the ddgs engine port, normalizers, the XPath subset, the
389
- aggregator, the ranker, and the Wikipedia engine with a stubbed fetch
426
+ reciprocal-rank aggregator, the per-host cap, and the Startpage engine with a stubbed fetch
390
427
  - `test/pdf-parity.test.ts`: MuPDF engine capabilities, PDF 1.5 object streams,
391
428
  ASCII85Decode, font `/Differences` encodings, pymupdf4llm-style headings/links/tables
392
429
  - `test/entities.test.ts`: `decodeHtmlEntities` parity with CPython `html.unescape`,
package/engines.ts CHANGED
@@ -1,11 +1,12 @@
1
- import { randomBytes } from "node:crypto";
2
1
  import { appendFile, chmod, mkdir } from "node:fs/promises";
3
2
  import { dirname, isAbsolute, join } from "node:path";
4
3
  import { collapseWhitespace, decodeHtmlEntities, feedHtml } from "./html-to-md.ts";
5
4
  import type { AttrDict } from "./html-to-md.ts";
6
5
  import { randomUserAgent } from "./user-agents.ts";
7
6
  import { agentDir } from "./agent-dir.ts";
8
- import { MAX_SIGNAL_TIMEOUT_MS } from "./web-access.ts";
7
+ import { impersonatedRequest, type FetchTransport, type TlsHopOptions, type TlsHopResponse } from "./tls-fetch.ts";
8
+ import { checkUrlAccess, isPublicIp, MAX_SIGNAL_TIMEOUT_MS, type WebsitePolicy } from "./web-access.ts";
9
+ import { socksProxyForUrl } from "./proxy.ts";
9
10
  export class EmptySweepError extends Error {
10
11
  constructor() {
11
12
  super("No results found");
@@ -247,6 +248,13 @@ function parsePredExpr(input: string): Pred {
247
248
  const name = word();
248
249
  return { op: "desc", tag: name };
249
250
  }
251
+ if (input.startsWith("./", pos)) {
252
+ pos += 2;
253
+ const name = word();
254
+ const { preds, next } = parsePredicateBlocks(input, pos);
255
+ pos = next;
256
+ return { op: "child", tag: name, preds };
257
+ }
250
258
  const name = word();
251
259
  const { preds, next } = parsePredicateBlocks(input, pos);
252
260
  pos = next;
@@ -471,42 +479,18 @@ export function extractResults(
471
479
  return results;
472
480
  }
473
481
 
474
- function googleUserAgent(): string {
475
- const devices: [string, string, number, number][] = [
476
- ["5.0", "SM-G900P Build/LRX21T", 39, 60],
477
- ["6.0", "Nexus 5 Build/MRA58N", 39, 60],
478
- ["8.0", "Pixel 2 Build/OPD3.170816.012", 39, 60],
479
- ];
480
- const [androidVer, device, chromeMin, chromeMax] = devices[Math.floor(Math.random() * devices.length)];
481
- const chromeMajor = chromeMin + Math.floor(Math.random() * (chromeMax - chromeMin + 1));
482
- const chromeBuild = 1000 + Math.floor(Math.random() * 9000);
483
- const chromePatch = 1000 + Math.floor(Math.random() * 1000);
484
- return (
485
- `Mozilla/5.0 (Linux; Android ${androidVer}; ${device}) ` +
486
- `AppleWebKit/537.36 (KHTML, like Gecko) ` +
487
- `Chrome/${chromeMajor}.0.${chromeBuild}.${chromePatch} Mobile Safari/537.36`
488
- );
489
- }
490
-
491
- function tokenUrlSafe(byteLength: number): string {
492
- return randomBytes(byteLength).toString("base64url");
493
- }
494
-
495
- function unquotePlus(value: string): string {
496
- try {
497
- return decodeURIComponent(value.replace(/\+/g, "%20"));
498
- } catch {
499
- return value.replace(/\+/g, " ");
500
- }
501
- }
482
+ export type EngineImpersonation = (options: TlsHopOptions) => Promise<TlsHopResponse | null>;
502
483
 
503
- function yahooExtractUrl(raw: string): string {
504
- const afterRu = raw.split("/RU=", 2)[1] ?? "";
505
- const t = afterRu.split("/RK=", 1)[0].split("/RS=", 1)[0];
506
- return unquotePlus(t);
484
+ export interface SearchEngineOptions {
485
+ transport?: FetchTransport;
486
+ policy?: WebsitePolicy | null;
487
+ impersonate?: EngineImpersonation;
488
+ engines?: string[];
489
+ engineWeights?: Record<string, number>;
490
+ maxPerHost?: number;
507
491
  }
508
492
 
509
- interface EngineContext {
493
+ export interface EngineContext extends SearchEngineOptions {
510
494
  region: string;
511
495
  safesearch: string;
512
496
  }
@@ -514,7 +498,6 @@ interface EngineContext {
514
498
  export interface Engine {
515
499
  name: string;
516
500
  provider: string;
517
- priority?: number;
518
501
  search(
519
502
  query: string,
520
503
  ctx: EngineContext,
@@ -528,6 +511,7 @@ interface HttpRequestOptions {
528
511
  cookies?: Record<string, string>;
529
512
  timeoutMs: number;
530
513
  signal?: AbortSignal;
514
+ ctx?: EngineContext;
531
515
  }
532
516
 
533
517
  interface HttpOptions extends HttpRequestOptions {
@@ -555,6 +539,8 @@ async function httpPost(
555
539
 
556
540
  const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
557
541
  const ENGINE_RETRY_BACKOFF_MS = 250;
542
+ const MAX_ENGINE_HOPS = 5;
543
+ const DEFAULT_ENGINE_TRANSPORT: FetchTransport = "off";
558
544
 
559
545
  const sleep = (ms: number) => new Promise<void>((resolve) => setTimeout(resolve, ms));
560
546
 
@@ -578,10 +564,84 @@ async function readBodyCapped(response: Response): Promise<string | null> {
578
564
  return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
579
565
  }
580
566
 
581
- function mapFetchError(err: unknown): never {
582
- if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
583
- if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
584
- throw err;
567
+ interface EngineHopResponse {
568
+ status: number;
569
+ location: string | null;
570
+ body: string | null;
571
+ }
572
+
573
+ type EngineTransportKind = "direct" | "tls";
574
+
575
+ function engineTransportOrder(transport: FetchTransport, target: URL): EngineTransportKind[] {
576
+ if (transport === "off" || socksProxyForUrl(target) !== null) return ["direct"];
577
+ return transport === "direct-first" ? ["direct", "tls"] : ["tls", "direct"];
578
+ }
579
+
580
+ function engineTargetAllowed(target: string, policy: WebsitePolicy | null): boolean {
581
+ const [allowed, , hostname] = checkUrlAccess(target, policy);
582
+ if (!allowed) return false;
583
+ const isLiteral = hostname.includes(":") || /^\d+\.\d+\.\d+\.\d+$/.test(hostname);
584
+ return !isLiteral || isPublicIp(hostname);
585
+ }
586
+
587
+ function classifyRequestError(
588
+ err: unknown,
589
+ caller: AbortSignal | undefined,
590
+ hopSignal: AbortSignal,
591
+ ): Error | null {
592
+ if (caller?.aborted) return new SearchCancelled();
593
+ if (hopSignal.aborted) return new SearchTimeoutError();
594
+ if (err instanceof DOMException && err.name === "TimeoutError") return new SearchTimeoutError();
595
+ if (err instanceof DOMException && err.name === "AbortError") return new SearchTimeoutError();
596
+ if (err instanceof Error && err.message === "timed out") return new SearchTimeoutError();
597
+ if (err instanceof Error && err.message === "cancelled") return new SearchCancelled();
598
+ return null;
599
+ }
600
+
601
+ async function directEngineHop(
602
+ target: URL,
603
+ headers: Record<string, string>,
604
+ method: string,
605
+ body: string | undefined,
606
+ signal: AbortSignal,
607
+ ): Promise<EngineHopResponse | null> {
608
+ const response = await fetch(target.toString(), { method, headers, body, signal, redirect: "manual" });
609
+ if (response.status === 200) {
610
+ const text = await readBodyCapped(response);
611
+ return text === null ? null : { status: 200, location: null, body: text };
612
+ }
613
+ try {
614
+ await response.body?.cancel();
615
+ } catch {}
616
+ return { status: response.status, location: response.headers.get("location"), body: null };
617
+ }
618
+
619
+ function tlsEngineHop(
620
+ impersonate: EngineImpersonation,
621
+ target: URL,
622
+ headers: Record<string, string>,
623
+ method: string,
624
+ body: string | undefined,
625
+ caller: AbortSignal | undefined,
626
+ timeoutMs: number,
627
+ ): Promise<EngineHopResponse | null> {
628
+ return impersonate({
629
+ url: target,
630
+ timeoutMs,
631
+ signal: caller,
632
+ maxBytes: MAX_ENGINE_RESPONSE_BYTES,
633
+ maxPdfBytes: MAX_ENGINE_RESPONSE_BYTES,
634
+ method,
635
+ body,
636
+ extraHeaders: headers,
637
+ }).then((response) => {
638
+ if (response === null || response.truncated) return null;
639
+ return {
640
+ status: response.status,
641
+ location: response.headers.location ?? null,
642
+ body: response.status === 200 ? new TextDecoder("utf-8").decode(response.body) : null,
643
+ };
644
+ });
585
645
  }
586
646
 
587
647
  async function httpFetch(
@@ -601,32 +661,67 @@ async function httpFetch(
601
661
  : null;
602
662
  if (cookie) headers["Cookie"] = cookie;
603
663
  const timeoutMs = Math.min(MAX_SIGNAL_TIMEOUT_MS, Math.max(1, options.timeoutMs));
604
- const signals: AbortSignal[] = [AbortSignal.timeout(timeoutMs)];
605
- if (options.signal) signals.push(options.signal);
606
- let response: Response;
607
- try {
608
- response = await fetch(url, {
609
- method: options.method ?? "GET",
610
- headers,
611
- body: options.method === "POST" ? options.body : undefined,
612
- signal: AbortSignal.any(signals),
613
- });
614
- } catch (err) {
615
- throw mapFetchError(err);
616
- }
617
- if (response.status !== 200) {
618
- try {
619
- await response.body?.cancel();
620
- } catch {
664
+ const deadline = Date.now() + timeoutMs;
665
+ const caller = options.signal;
666
+ const transport = options.ctx?.transport ?? DEFAULT_ENGINE_TRANSPORT;
667
+ const policy = options.ctx?.policy ?? null;
668
+ const impersonate = options.ctx?.impersonate ?? (transport === "off" ? null : impersonatedRequest);
669
+ let method = options.method ?? "GET";
670
+ let body = method === "POST" ? options.body : undefined;
671
+ let target = url;
672
+ for (let hop = 0; hop < MAX_ENGINE_HOPS; hop++) {
673
+ if (caller?.aborted) throw new SearchCancelled();
674
+ const remaining = deadline - Date.now();
675
+ if (remaining <= 0) throw new SearchTimeoutError();
676
+ const targetUrl = new URL(target);
677
+ const hopSignal = caller
678
+ ? AbortSignal.any([caller, AbortSignal.timeout(remaining)])
679
+ : AbortSignal.timeout(remaining);
680
+ let response: EngineHopResponse | null = null;
681
+ let failure: unknown = null;
682
+ for (const kind of engineTransportOrder(transport, targetUrl)) {
683
+ try {
684
+ let attempt: EngineHopResponse | null = null;
685
+ if (kind === "tls") {
686
+ if (impersonate !== null) {
687
+ attempt = await tlsEngineHop(impersonate, targetUrl, headers, method, body, caller, remaining);
688
+ }
689
+ } else {
690
+ attempt = await directEngineHop(targetUrl, headers, method, body, hopSignal);
691
+ }
692
+ if (attempt === null) continue;
693
+ if (response === null || (attempt.status < 400 && response.status >= 400)) response = attempt;
694
+ if (attempt.status < 400) break;
695
+ } catch (err) {
696
+ const mapped = classifyRequestError(err, caller, hopSignal);
697
+ if (mapped) throw mapped;
698
+ failure = err;
699
+ }
700
+ }
701
+ if (response === null) {
702
+ if (failure) throw failure;
621
703
  return null;
622
704
  }
623
- return null;
624
- }
625
- try {
626
- return await readBodyCapped(response);
627
- } catch (err) {
628
- throw mapFetchError(err);
705
+ if (response.status >= 300 && response.status < 400) {
706
+ if (!response.location) return null;
707
+ let next: string;
708
+ try {
709
+ next = new URL(response.location, target).toString();
710
+ } catch {
711
+ return null;
712
+ }
713
+ if (!engineTargetAllowed(next, policy)) return null;
714
+ if (response.status !== 307 && response.status !== 308) {
715
+ method = "GET";
716
+ body = undefined;
717
+ }
718
+ target = next;
719
+ continue;
720
+ }
721
+ if (response.status !== 200) return null;
722
+ return response.body ?? "";
629
723
  }
724
+ return null;
630
725
  }
631
726
 
632
727
  const DUCKDUCKGO: Engine = {
@@ -636,7 +731,7 @@ const DUCKDUCKGO: Engine = {
636
731
  const html = await httpPost(
637
732
  "https://html.duckduckgo.com/html/",
638
733
  { q: query, b: "", l: ctx.region },
639
- { headers: { "User-Agent": randomUserAgent() }, timeoutMs, signal },
734
+ { headers: { "User-Agent": randomUserAgent() }, timeoutMs, signal, ctx },
640
735
  );
641
736
  if (!html) return null;
642
737
  const results = extractResults(html, "//div[contains(@class, 'body')]", {
@@ -648,126 +743,15 @@ const DUCKDUCKGO: Engine = {
648
743
  },
649
744
  };
650
745
 
651
- const BRAVE: Engine = {
652
- name: "brave",
653
- provider: "brave",
654
- async search(query, ctx, timeoutMs, signal) {
655
- const country = ctx.region.toLowerCase().split("-")[0];
656
- const cookies: Record<string, string> = { [country]: country, useLocation: "0" };
657
- if (ctx.safesearch !== "moderate") {
658
- cookies["safesearch"] = ctx.safesearch === "on" ? "strict" : "off";
659
- }
660
- const html = await httpGet(
661
- "https://search.brave.com/search",
662
- { q: query, source: "web" },
663
- { cookies, timeoutMs, signal },
664
- );
665
- if (!html) return null;
666
- return extractResults(html, "//div[@data-type='web']", {
667
- title:
668
- ".//div[(contains(@class,'title') or contains(@class,'sitename-container')) and position()=last()]//text()",
669
- href: ".//a[div[contains(@class, 'title')]]/@href",
670
- body: ".//div[contains(@class, 'snippet')]//div[contains(@class, 'content')]//text()",
671
- });
672
- },
673
- };
674
-
675
- const GOOGLE: Engine = {
676
- name: "google",
677
- provider: "google",
678
- async search(query, ctx, timeoutMs, signal) {
679
- const [country, lang] = ctx.region.split("-");
680
- const safesearchBase: Record<string, string> = { on: "2", moderate: "1", off: "0" };
681
- const html = await httpGet(
682
- "https://www.google.com/search",
683
- {
684
- q: query,
685
- filter: safesearchBase[ctx.safesearch.toLowerCase()] ?? "1",
686
- start: "0",
687
- hl: `${lang}-${country.toUpperCase()}`,
688
- lr: `lang_${lang}`,
689
- cr: `country${country.toUpperCase()}`,
690
- },
691
- {
692
- headers: { "User-Agent": googleUserAgent() },
693
- cookies: { CONSENT: "YES+" },
694
- timeoutMs,
695
- signal,
696
- },
697
- );
698
- if (!html) return null;
699
- const results = extractResults(html, "//div[@data-hveid][.//h3]", {
700
- title: ".//h3//text()",
701
- href: ".//a[.//h3]/@href",
702
- body: "./div/div[last()]//text()",
703
- });
704
- return results
705
- .map((r) => {
706
- if (r.href.startsWith("/url?q=")) {
707
- r.href = r.href.split("?q=")[1].split("&")[0];
708
- }
709
- return r;
710
- })
711
- .filter((r) => r.title && r.href.startsWith("http"));
712
- },
713
- };
714
-
715
- const MOJEEK: Engine = {
716
- name: "mojeek",
717
- provider: "mojeek",
718
- async search(query, ctx, timeoutMs, signal) {
719
- const [country, lang] = ctx.region.toLowerCase().split("-");
720
- const params: Record<string, string> = { q: query };
721
- if (ctx.safesearch === "on") params["safe"] = "1";
722
- const html = await httpGet(
723
- "https://www.mojeek.com/search",
724
- params,
725
- { cookies: { arc: country, lb: lang }, timeoutMs, signal },
726
- );
727
- if (!html) return null;
728
- return extractResults(html, "//ul[contains(@class, 'results')]/li", {
729
- title: ".//h2//text()",
730
- href: ".//h2/a/@href",
731
- body: ".//p[@class='s']//text()",
732
- });
733
- },
734
- };
735
-
736
- const YAHOO: Engine = {
737
- name: "yahoo",
738
- provider: "bing",
739
- async search(query, _ctx, timeoutMs, signal) {
740
- const ylt = tokenUrlSafe(18);
741
- const ylu = tokenUrlSafe(35);
742
- const html = await httpGet(
743
- `https://search.yahoo.com/search;_ylt=${ylt};_ylu=${ylu}`,
744
- { p: query },
745
- { timeoutMs, signal },
746
- );
747
- if (!html) return null;
748
- const results = extractResults(html, "//div[contains(@class, 'relsrch')]", {
749
- title: ".//div[contains(@class, 'Title')]//h3//text()",
750
- href: ".//div[contains(@class, 'Title')]//a/@href",
751
- body: ".//div[contains(@class, 'Text')]//text()",
752
- });
753
- return results
754
- .filter((r) => !r.href.startsWith("https://www.bing.com/aclick?"))
755
- .map((r) => {
756
- if (r.href.includes("/RU=")) r.href = yahooExtractUrl(r.href);
757
- return r;
758
- });
759
- },
760
- };
761
-
762
746
  const YANDEX: Engine = {
763
747
  name: "yandex",
764
748
  provider: "yandex",
765
- async search(query, _ctx, timeoutMs, signal) {
749
+ async search(query, ctx, timeoutMs, signal) {
766
750
  const searchid = 1000000 + Math.floor(Math.random() * 9000000);
767
751
  const html = await httpGet(
768
752
  "https://yandex.com/search/site/",
769
753
  { text: query, web: "1", searchid: String(searchid) },
770
- { timeoutMs, signal },
754
+ { timeoutMs, signal, ctx },
771
755
  );
772
756
  if (!html) return null;
773
757
  return extractResults(html, "//li[contains(@class, 'serp-item')]", {
@@ -778,60 +762,36 @@ const YANDEX: Engine = {
778
762
  },
779
763
  };
780
764
 
781
- const WIKIPEDIA: Engine = {
782
- name: "wikipedia",
783
- provider: "wikipedia",
784
- priority: 2,
765
+ const START_PAGE: Engine = {
766
+ name: "startpage",
767
+ provider: "google",
785
768
  async search(query, ctx, timeoutMs, signal) {
786
- const started = Date.now();
787
- const lang = ctx.region.toLowerCase().split("-")[1] ?? "en";
788
- const encoded = encodeURIComponent(query);
789
- const opensearchUrl =
790
- `https://${lang}.wikipedia.org/w/api.php?action=opensearch&profile=fuzzy&limit=1&search=${encoded}`;
791
- const opensearch = await httpGet(opensearchUrl, {}, { timeoutMs, signal });
792
- if (!opensearch) return null;
793
- let data: unknown;
794
- try {
795
- data = JSON.parse(opensearch);
796
- } catch {
797
- return null;
798
- }
799
- const payload = data as [string, string[], string[], string[]];
800
- if (!payload[1] || !payload[1].length) return [];
801
- const title = payload[1][0];
802
- const href = payload[3][0];
803
- let body = "";
804
- const extractUrl =
805
- `https://${lang}.wikipedia.org/w/api.php?action=query&format=json&prop=extracts` +
806
- `&titles=${encodeURIComponent(title)}&explaintext=0&exintro=0&redirects=1`;
807
- const extract = await httpGet(extractUrl, {}, { timeoutMs: Math.max(1, timeoutMs - (Date.now() - started)), signal });
808
- if (extract) {
809
- try {
810
- const pageData = JSON.parse(extract) as {
811
- query: { pages: Record<string, { extract?: string }> };
812
- };
813
- const pages = Object.values(pageData.query.pages);
814
- if (pages.length) body = pages[0].extract ?? "";
815
- } catch {
816
- body = "";
817
- }
818
- }
819
- if (body.includes("may refer to:")) return [];
820
- return [{ title: normalizeText(title), href: normalizeUrl(href), body: normalizeText(body) }];
769
+ const [country, lang] = ctx.region.toLowerCase().split("-");
770
+ const html = await httpGet(
771
+ "https://www.startpage.com/sp/search",
772
+ { query, qsr: `${lang}_${country.toUpperCase()}` },
773
+ { headers: { Referer: "https://www.startpage.com/" }, timeoutMs, signal, ctx },
774
+ );
775
+ if (!html) return null;
776
+ return extractResults(html, "//div[contains(@class, 'result')][./a]", {
777
+ title: ".//h2//text()",
778
+ href: "./a/@href",
779
+ body: ".//p//text()",
780
+ });
821
781
  },
822
782
  };
823
783
 
824
- export const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, BRAVE, GOOGLE, MOJEEK, YAHOO, YANDEX, WIKIPEDIA];
784
+ export const TEXT_ENGINES: Engine[] = [DUCKDUCKGO, YANDEX, START_PAGE];
825
785
 
826
786
  export class ResultsAggregator {
827
787
  private cache = new Map<string, SearchResult>();
828
- private counter = new Map<string, number>();
788
+ private scores = new Map<string, number>();
829
789
 
830
790
  get size(): number {
831
791
  return this.cache.size;
832
792
  }
833
793
 
834
- append(item: SearchResult): void {
794
+ append(item: SearchResult, weight = 1, rank = 1): void {
835
795
  if (typeof item.href !== "string" || !item.href.trim()) return;
836
796
  const key = canonicalizeHref(item.href);
837
797
  if (!key) return;
@@ -839,54 +799,66 @@ export class ResultsAggregator {
839
799
  if (!existing || item.body.length > existing.body.length) {
840
800
  this.cache.set(key, { ...item, href: key });
841
801
  }
842
- this.counter.set(key, (this.counter.get(key) ?? 0) + 1);
802
+ this.scores.set(key, (this.scores.get(key) ?? 0) + weight / (RRF_RANK_CONSTANT + rank));
843
803
  }
844
804
 
845
- extend(items: SearchResult[]): void {
846
- for (const item of items) this.append(item);
805
+ extend(items: SearchResult[], weight = 1): void {
806
+ const seen = new Set<string>();
807
+ items.forEach((item, index) => {
808
+ const key = canonicalizeHref(item.href);
809
+ if (!key) return;
810
+ const first = !seen.has(key);
811
+ seen.add(key);
812
+ this.append(item, first ? weight : 0, index + 1);
813
+ });
847
814
  }
848
815
 
849
- extractDicts(): SearchResult[] {
850
- return [...this.counter.entries()]
851
- .sort((a, b) => b[1] - a[1])
852
- .map(([key]) => this.cache.get(key)!);
816
+ ranked(): SearchResult[] {
817
+ return [...this.cache.entries()]
818
+ .filter(([, doc]) => !isWikimediaCategory(doc))
819
+ .map(([key, doc]) => ({ doc, score: this.scores.get(key) ?? 0 }))
820
+ .sort((a, b) => b.score - a.score || a.doc.href.localeCompare(b.doc.href))
821
+ .map((entry) => entry.doc);
853
822
  }
854
823
  }
855
824
 
856
- function extractTokens(query: string): Set<string> {
857
- return new Set(query.toLowerCase().split(/\W+/u).filter((t) => t.length >= 3));
858
- }
825
+ const RRF_RANK_CONSTANT = 60;
826
+ const DEFAULT_MAX_PER_HOST = 0;
827
+ const MULTI_PART_SUFFIXES = new Set(["co.uk", "org.uk", "com.au", "co.jp", "co.nz", "com.br", "co.in"]);
859
828
 
860
- function hasAnyToken(text: string, tokens: Set<string>): boolean {
861
- const lower = text.toLowerCase();
862
- for (const token of tokens) {
863
- if (lower.includes(token)) return true;
829
+ export function registrableDomain(url: string): string {
830
+ let hostname = "";
831
+ try {
832
+ hostname = new URL(url).hostname.toLowerCase().replace(/^www\./, "");
833
+ } catch {
834
+ return "";
864
835
  }
865
- return false;
836
+ const parts = hostname.split(".");
837
+ if (parts.length < 2) return hostname;
838
+ const lastTwo = parts.slice(-2).join(".");
839
+ return MULTI_PART_SUFFIXES.has(lastTwo) && parts.length >= 3 ? parts.slice(-3).join(".") : lastTwo;
866
840
  }
867
841
 
868
- export function rankResults(docs: SearchResult[], query: string): SearchResult[] {
869
- const tokens = extractTokens(query);
870
- const wiki: SearchResult[] = [];
871
- const both: SearchResult[] = [];
872
- const titleOnly: SearchResult[] = [];
873
- const bodyOnly: SearchResult[] = [];
874
- const neither: SearchResult[] = [];
842
+ export function capByHost(docs: SearchResult[], limit: number, maxPerHost: number): SearchResult[] {
843
+ if (limit <= 0) return [];
844
+ const perHost = new Map<string, number>();
845
+ const capped: SearchResult[] = [];
846
+ const allowed = maxPerHost > 0 ? maxPerHost : Number.POSITIVE_INFINITY;
875
847
  for (const doc of docs) {
876
- if (doc.title.includes("Category:") && doc.title.includes("Wikimedia")) continue;
877
- if (doc.href.includes("wikipedia.org")) {
878
- wiki.push(doc);
879
- continue;
880
- }
881
- const hitTitle = hasAnyToken(doc.title, tokens);
882
- const hitBody = hasAnyToken(doc.body, tokens);
883
- if (hitTitle && hitBody) both.push(doc);
884
- else if (hitTitle) titleOnly.push(doc);
885
- else if (hitBody) bodyOnly.push(doc);
886
- else neither.push(doc);
848
+ const domain = registrableDomain(doc.href);
849
+ const used = perHost.get(domain) ?? 0;
850
+ if (used >= allowed) continue;
851
+ perHost.set(domain, used + 1);
852
+ capped.push(doc);
853
+ if (capped.length >= limit) break;
887
854
  }
888
- return [...wiki, ...both, ...titleOnly, ...bodyOnly, ...neither];
855
+ return capped;
889
856
  }
857
+
858
+ function isWikimediaCategory(doc: SearchResult): boolean {
859
+ return doc.title.includes("Category:") && doc.title.includes("Wikimedia");
860
+ }
861
+
890
862
  async function recordSweepStats(query: string, maxResults: number, started: number, timedOutProviders: string[], resultCount: number): Promise<void> {
891
863
  const flag = process.env.PI_UNSLOTH_WEBTOOLS_STATS?.trim();
892
864
  if (!flag) return;
@@ -922,15 +894,20 @@ async function recordSweepStats(query: string, maxResults: number, started: numb
922
894
  } catch {}
923
895
  }
924
896
 
925
- function shuffledEngines(): Engine[] {
926
- const shuffled = [...TEXT_ENGINES];
897
+ function selectedEngines(names: string[] | undefined): Engine[] {
898
+ if (!names?.length) return TEXT_ENGINES;
899
+ const wanted = new Set(names.map((name) => name.trim().toLowerCase()));
900
+ const chosen = TEXT_ENGINES.filter((engine) => wanted.has(engine.name));
901
+ return chosen.length ? chosen : TEXT_ENGINES;
902
+ }
903
+
904
+ function shuffledEngines(engines: Engine[] = TEXT_ENGINES): Engine[] {
905
+ const shuffled = [...engines];
927
906
  for (let i = shuffled.length - 1; i > 0; i--) {
928
907
  const j = Math.floor(Math.random() * (i + 1));
929
908
  [shuffled[i], shuffled[j]] = [shuffled[j], shuffled[i]];
930
909
  }
931
- const wikipedia = shuffled.find((e) => e.priority === 2);
932
- const rest = shuffled.filter((e) => e.priority !== 2);
933
- return wikipedia ? [wikipedia, ...rest] : shuffled;
910
+ return shuffled;
934
911
  }
935
912
 
936
913
  export async function autoTextSearch(
@@ -938,13 +915,23 @@ export async function autoTextSearch(
938
915
  maxResults: number,
939
916
  timeoutMs: number,
940
917
  signal?: AbortSignal,
918
+ options: SearchEngineOptions = {},
941
919
  ): Promise<SearchResult[]> {
942
920
  const started = Date.now();
943
- const engines = shuffledEngines();
921
+ const engines = shuffledEngines(selectedEngines(options.engines));
944
922
  const deadline = started + timeoutMs;
945
923
  const seenProviders = new Set<string>();
946
924
  const aggregator = new ResultsAggregator();
947
- const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
925
+ const engineWeights = options.engineWeights ?? {};
926
+ const maxPerHost = options.maxPerHost ?? DEFAULT_MAX_PER_HOST;
927
+ const enough = () => capByHost(aggregator.ranked(), maxResults, maxPerHost).length >= maxResults;
928
+ const ctx: EngineContext = {
929
+ region: "us-en",
930
+ safesearch: "moderate",
931
+ transport: options.transport ?? DEFAULT_ENGINE_TRANSPORT,
932
+ policy: options.policy ?? null,
933
+ impersonate: options.impersonate,
934
+ };
948
935
  const controller = new AbortController();
949
936
  let onAbort: (() => void) | undefined;
950
937
  if (signal) {
@@ -998,18 +985,18 @@ export async function autoTextSearch(
998
985
  }
999
986
  }
1000
987
  if (results && results.length) {
1001
- aggregator.extend(results);
988
+ aggregator.extend(results, engineWeights[engine.name] ?? 1);
1002
989
  seenProviders.add(engine.provider);
1003
- if (aggregator.size >= maxResults) controller.abort();
990
+ if (enough()) controller.abort();
1004
991
  }
1005
992
  };
1006
993
  while (i < engines.length || pending.size > 0) {
1007
- if (aggregator.size >= maxResults || cancelled) {
994
+ if (enough() || cancelled) {
1008
995
  controller.abort();
1009
996
  break;
1010
997
  }
1011
998
  while (i < engines.length && pending.size < maxWorkers) {
1012
- if (aggregator.size >= maxResults || cancelled) {
999
+ if (enough() || cancelled) {
1013
1000
  controller.abort();
1014
1001
  break;
1015
1002
  }
@@ -1027,7 +1014,7 @@ export async function autoTextSearch(
1027
1014
  );
1028
1015
  }
1029
1016
  if (pending.size === 0) break;
1030
- if (aggregator.size >= maxResults || cancelled) {
1017
+ if (enough() || cancelled) {
1031
1018
  controller.abort();
1032
1019
  break;
1033
1020
  }
@@ -1036,10 +1023,10 @@ export async function autoTextSearch(
1036
1023
  await Promise.allSettled(pending);
1037
1024
  if (onAbort && signal) signal.removeEventListener("abort", onAbort);
1038
1025
  if (cancelled) throw new SearchCancelled();
1039
- const results = rankResults(aggregator.extractDicts(), query);
1026
+ const results = capByHost(aggregator.ranked(), maxResults, maxPerHost);
1040
1027
  if (results.length) {
1041
1028
  void recordSweepStats(query, maxResults, started, [...timedOutProviders], results.length);
1042
- return results.slice(0, maxResults);
1029
+ return results;
1043
1030
  }
1044
1031
  if (timedOutProviders.size) {
1045
1032
  const sorted = [...timedOutProviders].sort();
package/index.ts CHANGED
@@ -10,7 +10,13 @@ import {
10
10
  type FetchPageOutcome,
11
11
  } from "./web-fetch.ts";
12
12
  import { renderPageWithLightpanda as defaultRenderLocalPageText } from "./lightpanda.ts";
13
- import { loadDefaultFetchSettings, loadDefaultFetchTimeoutMs, loadLightpandaSettings } from "./settings.ts";
13
+ import {
14
+ loadDefaultEngines,
15
+ loadDefaultEngineWeights,
16
+ loadDefaultFetchSettings,
17
+ loadDefaultMaxPerHost,
18
+ loadLightpandaSettings,
19
+ } from "./settings.ts";
14
20
 
15
21
  function toolCallLine(theme: Theme, name: string, detail: string) {
16
22
  const line = theme.fg("toolTitle", theme.bold(name)) + (detail ? ` ${theme.fg("accent", detail)}` : "");
@@ -179,12 +185,20 @@ export function createWebTools(deps: WebToolsDeps = {}) {
179
185
  onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
180
186
  const timeoutParam = positiveNumber(params.timeoutMs);
181
187
  const searchCwd = (_ctx as ExtensionContext | undefined)?.cwd;
182
- const searchTimeoutMs = timeoutParam ?? (await loadDefaultFetchTimeoutMs(searchCwd)) ?? SEARCH_TIMEOUT_MS;
188
+ const searchSettings = await loadDefaultFetchSettings(searchCwd);
189
+ const searchEngines = await loadDefaultEngines(searchCwd);
190
+ const searchEngineWeights = await loadDefaultEngineWeights(searchCwd);
191
+ const searchMaxPerHost = await loadDefaultMaxPerHost(searchCwd);
192
+ const searchTimeoutMs = timeoutParam ?? searchSettings.timeoutMs ?? SEARCH_TIMEOUT_MS;
183
193
  const text = await webSearch(params.query, {
184
194
  signal: signal ?? undefined,
185
195
  timeoutMs: searchTimeoutMs,
186
196
  maxResults: positiveNumber(params.maxResults),
187
197
  cwd: searchCwd,
198
+ transport: searchSettings.transport,
199
+ engines: searchEngines,
200
+ engineWeights: searchEngineWeights,
201
+ maxPerHost: searchMaxPerHost,
188
202
  });
189
203
  return { content: [{ type: "text", text }], details: {} };
190
204
  },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.9.1",
3
+ "version": "0.10.0",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools that began as a port of the Unsloth Studio codebase and now diverge from it (multi-engine search, opt-in SSRF guard, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
@@ -60,6 +60,7 @@
60
60
  "test:smoke": "vitest run test/smoke.test.ts",
61
61
  "test:lightpanda": "vitest run test/lightpanda-live.test.ts",
62
62
  "camoufox:warmup": "node scripts/camoufox-warmup.ts",
63
+ "engine:eval": "node scripts/engine-eval.ts",
63
64
  "typecheck": "tsc --noEmit",
64
65
  "check:package": "node -e \"const fs=require('fs');const pkg=require('./package.json');const listed=new Set(pkg.files);const bad=[];for(const f of pkg.files){if(!fs.existsSync(f))bad.push('missing file: '+f)}for(const f of fs.readdirSync('.').filter(f=>f.endsWith('.ts'))){if(!listed.has(f))bad.push('unlisted source: '+f)}if(bad.length){console.error(bad.join('\\n'));process.exit(1)}\"",
65
66
  "prepublishOnly": "npm run typecheck && npm run lint && npm run test:unit && npm run check:package && npm run check:publint",
package/settings.ts CHANGED
@@ -50,6 +50,26 @@ function pickBoolean(data: Record<string, unknown>, paths: string[][]): boolean
50
50
  return undefined;
51
51
  }
52
52
 
53
+ function pickNumberRecord(data: Record<string, unknown>, paths: string[][]): Record<string, number> | undefined {
54
+ for (const path of paths) {
55
+ let cur: unknown = data;
56
+ for (const key of path) {
57
+ if (cur && typeof cur === "object" && !Array.isArray(cur)) cur = (cur as Record<string, unknown>)[key];
58
+ else {
59
+ cur = undefined;
60
+ break;
61
+ }
62
+ }
63
+ if (!cur || typeof cur !== "object" || Array.isArray(cur)) continue;
64
+ const record: Record<string, number> = {};
65
+ for (const [key, value] of Object.entries(cur as Record<string, unknown>)) {
66
+ if (typeof value === "number" && Number.isFinite(value) && value > 0) record[key] = value;
67
+ }
68
+ if (Object.keys(record).length) return record;
69
+ }
70
+ return undefined;
71
+ }
72
+
53
73
  function pickStringArray(data: Record<string, unknown>, paths: string[][]): string[] | undefined {
54
74
  for (const path of paths) {
55
75
  let cur: unknown = data;
@@ -133,6 +153,21 @@ const LIGHTPANDA_COMMAND_PATHS: string[][] = [
133
153
  ["unslothWebTools", "lightpandaCommand"],
134
154
  ["webRender", "lightpandaCommand"],
135
155
  ];
156
+
157
+ const ENGINE_WEIGHTS_PATHS: string[][] = [
158
+ ["unslothWebTools", "engineWeights"],
159
+ ["webSearch", "engineWeights"],
160
+ ];
161
+
162
+ const MAX_PER_HOST_PATHS: string[][] = [
163
+ ["unslothWebTools", "maxPerHost"],
164
+ ["webSearch", "maxPerHost"],
165
+ ];
166
+
167
+ const ENGINE_PATHS: string[][] = [
168
+ ["unslothWebTools", "engines"],
169
+ ["webSearch", "engines"],
170
+ ];
136
171
  function clampMaxResults(value: number): number {
137
172
  return Math.min(20, Math.max(1, value));
138
173
  }
@@ -163,6 +198,28 @@ export async function loadDefaultMaxResults(cwd?: string): Promise<number> {
163
198
  return result;
164
199
  }
165
200
 
201
+ export async function loadDefaultEngines(cwd?: string): Promise<string[] | undefined> {
202
+ let result: string[] | undefined;
203
+ for (const data of await settingsEntries(cwd)) {
204
+ const candidate = pickStringArray(data, ENGINE_PATHS);
205
+ if (candidate !== undefined) result = candidate;
206
+ }
207
+ return result;
208
+ }
209
+
210
+ export async function loadDefaultEngineWeights(cwd?: string): Promise<Record<string, number> | undefined> {
211
+ let result: Record<string, number> | undefined;
212
+ for (const data of await settingsEntries(cwd)) {
213
+ const candidate = pickNumberRecord(data, ENGINE_WEIGHTS_PATHS);
214
+ if (candidate !== undefined) result = candidate;
215
+ }
216
+ return result;
217
+ }
218
+
219
+ export async function loadDefaultMaxPerHost(cwd?: string): Promise<number | undefined> {
220
+ return loadFetchSetting(cwd, MAX_PER_HOST_PATHS, (n) => n >= 0);
221
+ }
222
+
166
223
  async function loadFetchSetting(cwd: string | undefined, paths: string[][], valid: (n: number) => boolean): Promise<number | undefined> {
167
224
  let result: number | undefined;
168
225
  for (const data of await settingsEntries(cwd)) {
package/tls-fetch.ts CHANGED
@@ -49,15 +49,19 @@ interface ImpersonationModule {
49
49
 
50
50
  export type ImpersonationLoader = () => Promise<ImpersonationModule | null>;
51
51
 
52
+ export type FetchTransport = "tls-first" | "direct-first" | "off";
53
+
52
54
  export interface TlsHopOptions {
53
55
  url: URL;
54
- pinnedIp: string;
55
- family: number;
56
+ pinnedIp?: string;
57
+ family?: number;
56
58
  timeoutMs: number;
57
59
  signal?: AbortSignal;
58
60
  maxBytes: number;
59
61
  maxPdfBytes?: number;
60
62
  profile?: string;
63
+ method?: string;
64
+ body?: string;
61
65
  extraHeaders?: Record<string, string>;
62
66
  loader?: ImpersonationLoader;
63
67
  }
@@ -85,7 +89,7 @@ async function loadImpersonationModule(): Promise<ImpersonationModule | null> {
85
89
  const transports = new Map<string, Promise<ImpersonationTransport>>();
86
90
 
87
91
  function transportKey(options: TlsHopOptions): string {
88
- return `${options.profile ?? DEFAULT_PROFILE}|${options.url.hostname}|${options.pinnedIp}|${options.family}`;
92
+ return `${options.profile ?? DEFAULT_PROFILE}|${options.url.hostname}|${options.pinnedIp ?? ""}|${options.family ?? 0}`;
89
93
  }
90
94
 
91
95
  function evictTransports(): void {
@@ -107,12 +111,13 @@ async function transportFor(
107
111
  const key = transportKey(options);
108
112
  const existing = transports.get(key);
109
113
  if (existing) return existing;
114
+ const transportOptions: Record<string, unknown> = {
115
+ browser: options.profile ?? DEFAULT_PROFILE,
116
+ os: DEFAULT_OS,
117
+ };
118
+ if (options.pinnedIp) transportOptions.resolve = { [options.url.hostname]: options.pinnedIp };
110
119
  const created = module
111
- .createTransport({
112
- resolve: { [options.url.hostname]: options.pinnedIp },
113
- browser: options.profile ?? DEFAULT_PROFILE,
114
- os: DEFAULT_OS,
115
- })
120
+ .createTransport(transportOptions)
116
121
  .catch((error: unknown) => {
117
122
  transports.delete(key);
118
123
  throw error;
@@ -179,12 +184,15 @@ export async function impersonatedRequest(options: TlsHopOptions): Promise<TlsHo
179
184
  const signal = options.signal ? AbortSignal.any([options.signal, timeoutSignal]) : timeoutSignal;
180
185
  let response: ImpersonationResponse;
181
186
  try {
182
- response = await module.fetch(options.url.toString(), {
187
+ const init: Record<string, unknown> = {
183
188
  transport,
189
+ method: options.method ?? "GET",
184
190
  redirect: "manual",
185
191
  signal,
186
192
  headers: hopHeaders(options.extraHeaders),
187
- });
193
+ };
194
+ if (options.body !== undefined) init.body = options.body;
195
+ response = await module.fetch(options.url.toString(), init);
188
196
  } catch (error) {
189
197
  throw new Error(failureMessage(error, options.signal, signal));
190
198
  }
package/web-fetch.ts CHANGED
@@ -23,7 +23,7 @@ import {
23
23
  stripIpv6Brackets,
24
24
  type WebsitePolicy,
25
25
  } from "./web-access.ts";
26
- import { impersonatedRequest, type TlsHopOptions, type TlsHopResponse } from "./tls-fetch.ts";
26
+ import { impersonatedRequest, type FetchTransport, type TlsHopOptions, type TlsHopResponse } from "./tls-fetch.ts";
27
27
  import { collapseWhitespace, decodeHtmlEntities, feedHtml, htmlToMarkdown, visibleChars } from "./html-to-md.ts";
28
28
  import type { AttrDict } from "./html-to-md.ts";
29
29
  import { INVALID_CHARREFS } from "./entities.ts";
@@ -210,7 +210,7 @@ export interface ResolvedHost {
210
210
  alternates?: { ip: string; family: number }[];
211
211
  }
212
212
 
213
- export type FetchTransport = "tls-first" | "direct-first" | "off";
213
+ export type { FetchTransport };
214
214
 
215
215
  type TransportKind = "direct" | "tls";
216
216
  export interface FetchSeams {
package/web-search.ts CHANGED
@@ -6,8 +6,10 @@ import {
6
6
  EmptySweepError,
7
7
  SearchCancelled,
8
8
  SearchTimeoutError,
9
+ type SearchEngineOptions,
9
10
  type SearchResult,
10
11
  } from "./engines.ts";
12
+ import type { FetchTransport } from "./tls-fetch.ts";
11
13
 
12
14
  export { EmptySweepError, SearchCancelled, SearchTimeoutError } from "./engines.ts";
13
15
 
@@ -24,15 +26,20 @@ export type SearchClient = (
24
26
  maxResults: number,
25
27
  signal?: AbortSignal,
26
28
  timeoutMs?: number,
29
+ options?: SearchEngineOptions,
27
30
  ) => Promise<SearchResult[]>;
28
31
  export async function ddgSearch(
29
32
  query: string,
30
33
  maxResults = MAX_RESULTS,
31
34
  signal?: AbortSignal,
32
35
  timeoutMs = SEARCH_TIMEOUT_MS,
36
+ options: SearchEngineOptions = {},
33
37
  ): Promise<SearchResult[]> {
34
38
  if (signal?.aborted) throw new SearchCancelled();
35
- return autoTextSearch(query, maxResults, timeoutMs, signal);
39
+ return autoTextSearch(query, maxResults, timeoutMs, signal, {
40
+ ...options,
41
+ transport: options.transport ?? "tls-first",
42
+ });
36
43
  }
37
44
 
38
45
  const POLICY_OVERFETCH = 4;
@@ -44,6 +51,10 @@ export interface WebSearchOptions {
44
51
  websitePolicy?: WebsitePolicy | null;
45
52
  client?: SearchClient;
46
53
  cwd?: string;
54
+ transport?: FetchTransport;
55
+ engines?: string[];
56
+ engineWeights?: Record<string, number>;
57
+ maxPerHost?: number;
47
58
  }
48
59
 
49
60
  export { loadDefaultMaxResults };
@@ -71,7 +82,13 @@ export async function webSearch(
71
82
  (policy?.blockedDomains?.length ?? 0) > 0,
72
83
  );
73
84
  const wanted = restricted ? maxResults * POLICY_OVERFETCH : maxResults;
74
- const results = await client(effectiveQuery, wanted, signal, timeoutMs);
85
+ const results = await client(effectiveQuery, wanted, signal, timeoutMs, {
86
+ transport: options.transport,
87
+ policy,
88
+ engines: options.engines,
89
+ engineWeights: options.engineWeights,
90
+ maxPerHost: options.maxPerHost,
91
+ });
75
92
  if (signal?.aborted) return "Search cancelled.";
76
93
  if (!results.length) return EMPTY_SEARCH_RESULTS[0];
77
94
  const allowed: SearchResult[] = [];