pi-unsloth-webtools 0.2.4 → 0.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -45,6 +45,8 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
45
45
  validation and fetch.
46
46
  - GitHub repo root pages are rewritten to the unauthenticated README API
47
47
  (`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
48
+ - HTTP 4xx/5xx responses are reported as errors with the status reason instead of
49
+ returning error-page content.
48
50
  - Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
49
51
  - 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
50
52
  (`signal` cancels mid-flight).
@@ -53,7 +55,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
53
55
  pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
54
56
  with Studio's corrupted/incomplete fallback to plain text.
55
57
  - Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
56
- decoding (declared charset, BOM sniffing for UTF-8/16/32, `<meta charset>` sniffing for
58
+ decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
57
59
  CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
58
60
  - HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
59
61
  links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
@@ -78,7 +80,9 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
78
80
  database.
79
81
  - Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
80
82
  timeout whenever any engine timed out, so the timeout message is not masked by later
81
- generic engine failures.
83
+ generic engine failures. The timeout budget bounds the entire sweep: per-engine
84
+ timeouts shrink as the budget is consumed, so the reported timeout matches the
85
+ worst-case wall time.
82
86
  - Proxies: Studio routes through environment proxies; this port always connects
83
87
  directly with DNS pinning (deliberately out of scope).
84
88
 
package/engines.ts CHANGED
@@ -510,6 +510,7 @@ async function httpFetch(
510
510
  Accept: "*/*",
511
511
  ...options.headers,
512
512
  };
513
+ if (options.method === "POST") headers["Content-Type"] = "application/x-www-form-urlencoded";
513
514
  const cookie = options.cookies
514
515
  ? Object.entries(options.cookies)
515
516
  .map(([key, value]) => `${key}=${value}`)
@@ -695,6 +696,7 @@ const WIKIPEDIA: Engine = {
695
696
  provider: "wikipedia",
696
697
  priority: 2,
697
698
  async search(query, ctx, timeoutMs, signal) {
699
+ const started = Date.now();
698
700
  const lang = ctx.region.toLowerCase().split("-")[1] ?? "en";
699
701
  const encoded = encodeURIComponent(query);
700
702
  const opensearchUrl =
@@ -715,7 +717,7 @@ const WIKIPEDIA: Engine = {
715
717
  const extractUrl =
716
718
  `https://${lang}.wikipedia.org/w/api.php?action=query&format=json&prop=extracts` +
717
719
  `&titles=${encodeURIComponent(title)}&explaintext=0&exintro=0&redirects=1`;
718
- const extract = await httpGet(extractUrl, {}, { timeoutMs, signal });
720
+ const extract = await httpGet(extractUrl, {}, { timeoutMs: Math.max(1, timeoutMs - (Date.now() - started)), signal });
719
721
  if (extract) {
720
722
  try {
721
723
  const pageData = JSON.parse(extract) as {
@@ -815,6 +817,7 @@ export async function autoTextSearch(
815
817
  signal?: AbortSignal,
816
818
  ): Promise<SearchResult[]> {
817
819
  const engines = shuffledEngines();
820
+ const deadline = Date.now() + timeoutMs;
818
821
  const seenProviders = new Set<string>();
819
822
  const aggregator = new ResultsAggregator();
820
823
  const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
@@ -825,8 +828,9 @@ export async function autoTextSearch(
825
828
  let i = 0;
826
829
  let pending: Promise<void>[] = [];
827
830
  const run = async (engine: Engine) => {
831
+ const remaining = Math.max(1, deadline - Date.now());
828
832
  try {
829
- const results = await engine.search(query, ctx, timeoutMs, signal);
833
+ const results = await engine.search(query, ctx, remaining, signal);
830
834
  if (results && results.length) {
831
835
  aggregator.extend(results);
832
836
  seenProviders.add(engine.provider);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.2.4",
3
+ "version": "0.2.5",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
@@ -51,7 +51,8 @@
51
51
  "test:unit": "vitest run --exclude **/smoke.test.ts",
52
52
  "test:smoke": "vitest run test/smoke.test.ts",
53
53
  "typecheck": "tsc --noEmit",
54
- "prepublishOnly": "npm run typecheck && npm run test:unit"
54
+ "check:package": "node -e \"const fs=require('fs');const pkg=require('./package.json');const listed=new Set(pkg.files);const bad=[];for(const f of pkg.files){if(!fs.existsSync(f))bad.push('missing file: '+f)}for(const f of fs.readdirSync('.').filter(f=>f.endsWith('.ts'))){if(!listed.has(f))bad.push('unlisted source: '+f)}if(bad.length){console.error(bad.join('\\n'));process.exit(1)}\"",
55
+ "prepublishOnly": "npm run typecheck && npm run test:unit && npm run check:package"
55
56
  },
56
57
  "devDependencies": {
57
58
  "@earendil-works/pi-coding-agent": "^0.84.0",
package/web-fetch.ts CHANGED
@@ -462,6 +462,7 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
462
462
  const request = transport.request(options, (res: IncomingMessage) => {
463
463
  const chunks: Buffer[] = [];
464
464
  let total = 0;
465
+ let head = Buffer.alloc(0);
465
466
  const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
466
467
  let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
467
468
  let extendedForPdf = false;
@@ -488,8 +489,12 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
488
489
  finish("timed out", Buffer.concat(chunks));
489
490
  return;
490
491
  }
492
+ if (head.length < 1024) {
493
+ const need = 1024 - head.length;
494
+ head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
495
+ }
491
496
  if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
492
- if (hasPdfMagic(Buffer.concat(chunks))) {
497
+ if (hasPdfMagic(head)) {
493
498
  limit = opts.maxPdfBytes;
494
499
  extendedForPdf = true;
495
500
  }
@@ -628,6 +633,15 @@ export async function fetchUrlRaw(
628
633
  budgetError = fetchBudgetExceeded(deadline, signal, now);
629
634
  if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
630
635
 
636
+ if (response.status >= 400) {
637
+ const reason = http.STATUS_CODES[response.status] ?? "";
638
+ return {
639
+ error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
640
+ body: "",
641
+ contentType: "",
642
+ };
643
+ }
644
+
631
645
  const contentTypeHeader = response.headers["content-type"];
632
646
  const contentType = contentTypeHeader
633
647
  ? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
@@ -679,7 +693,7 @@ export async function fetchUrlRaw(
679
693
  const bomCodec = bomCodecFor(response.body);
680
694
  const rawHtml = decodeWithCodec(
681
695
  response.body,
682
- declaredCodec ?? bomCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
696
+ bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
683
697
  );
684
698
 
685
699
  if (looksBinary(rawHtml)) {
package/web-search.ts CHANGED
@@ -93,7 +93,7 @@ export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS
93
93
  export function formatSearchResults(results: SearchResult[]): string {
94
94
  const parts = results.map((result) => {
95
95
  const title = String(result.title ?? "").replace(/\s+/g, " ");
96
- const href = String(result.href ?? "").trim();
96
+ const href = String(result.href ?? "").replace(/\s+/g, " ").trim();
97
97
  const snippet = String(result.body ?? "").replace(/\s+/g, " ");
98
98
  return `Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`;
99
99
  });