pi-unsloth-webtools 0.2.3 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -2
- package/engines.ts +41 -7
- package/package.json +3 -2
- package/web-fetch.ts +60 -24
- package/web-search.ts +1 -1
package/README.md
CHANGED
|
@@ -45,6 +45,8 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
45
45
|
validation and fetch.
|
|
46
46
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
47
47
|
(`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
|
|
48
|
+
- HTTP 4xx/5xx responses are reported as errors with the status reason instead of
|
|
49
|
+
returning error-page content.
|
|
48
50
|
- Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
|
|
49
51
|
- 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
|
|
50
52
|
(`signal` cancels mid-flight).
|
|
@@ -53,7 +55,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
53
55
|
pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
|
|
54
56
|
with Studio's corrupted/incomplete fallback to plain text.
|
|
55
57
|
- Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
|
|
56
|
-
decoding (
|
|
58
|
+
decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
|
|
57
59
|
CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
|
|
58
60
|
- HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
|
|
59
61
|
links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
|
|
@@ -78,7 +80,9 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
78
80
|
database.
|
|
79
81
|
- Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
|
|
80
82
|
timeout whenever any engine timed out, so the timeout message is not masked by later
|
|
81
|
-
generic engine failures.
|
|
83
|
+
generic engine failures. The timeout budget bounds the entire sweep: per-engine
|
|
84
|
+
timeouts shrink as the budget is consumed, so the reported timeout matches the
|
|
85
|
+
worst-case wall time.
|
|
82
86
|
- Proxies: Studio routes through environment proxies; this port always connects
|
|
83
87
|
directly with DNS pinning (deliberately out of scope).
|
|
84
88
|
|
package/engines.ts
CHANGED
|
@@ -472,6 +472,28 @@ async function httpPost(
|
|
|
472
472
|
return httpFetch(url, { ...options, method: "POST", body: new URLSearchParams(data).toString() });
|
|
473
473
|
}
|
|
474
474
|
|
|
475
|
+
const MAX_ENGINE_RESPONSE_BYTES = 5 * 1024 * 1024;
|
|
476
|
+
|
|
477
|
+
async function readBodyCapped(response: Response): Promise<string | null> {
|
|
478
|
+
const declared = Number(response.headers.get("content-length") ?? "0");
|
|
479
|
+
if (declared > MAX_ENGINE_RESPONSE_BYTES) return null;
|
|
480
|
+
if (!response.body) return "";
|
|
481
|
+
const reader = response.body.getReader();
|
|
482
|
+
const chunks: Uint8Array[] = [];
|
|
483
|
+
let total = 0;
|
|
484
|
+
while (true) {
|
|
485
|
+
const { done, value } = await reader.read();
|
|
486
|
+
if (done) break;
|
|
487
|
+
total += value.length;
|
|
488
|
+
if (total > MAX_ENGINE_RESPONSE_BYTES) {
|
|
489
|
+
await reader.cancel();
|
|
490
|
+
return null;
|
|
491
|
+
}
|
|
492
|
+
chunks.push(value);
|
|
493
|
+
}
|
|
494
|
+
return new TextDecoder("utf-8").decode(Buffer.concat(chunks));
|
|
495
|
+
}
|
|
496
|
+
|
|
475
497
|
async function httpFetch(
|
|
476
498
|
url: string,
|
|
477
499
|
options: {
|
|
@@ -488,6 +510,7 @@ async function httpFetch(
|
|
|
488
510
|
Accept: "*/*",
|
|
489
511
|
...options.headers,
|
|
490
512
|
};
|
|
513
|
+
if (options.method === "POST") headers["Content-Type"] = "application/x-www-form-urlencoded";
|
|
491
514
|
const cookie = options.cookies
|
|
492
515
|
? Object.entries(options.cookies)
|
|
493
516
|
.map(([key, value]) => `${key}=${value}`)
|
|
@@ -505,13 +528,18 @@ async function httpFetch(
|
|
|
505
528
|
signal: AbortSignal.any(signals),
|
|
506
529
|
});
|
|
507
530
|
} catch (err) {
|
|
508
|
-
if (err instanceof DOMException && err.name === "TimeoutError")
|
|
509
|
-
|
|
510
|
-
}
|
|
531
|
+
if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
|
|
532
|
+
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
511
533
|
throw err;
|
|
512
534
|
}
|
|
513
535
|
if (response.status !== 200) return null;
|
|
514
|
-
|
|
536
|
+
try {
|
|
537
|
+
return await readBodyCapped(response);
|
|
538
|
+
} catch (err) {
|
|
539
|
+
if (err instanceof DOMException && err.name === "TimeoutError") throw new SearchTimeoutError();
|
|
540
|
+
if (err instanceof DOMException && err.name === "AbortError") throw new SearchCancelled();
|
|
541
|
+
throw err;
|
|
542
|
+
}
|
|
515
543
|
}
|
|
516
544
|
|
|
517
545
|
const DUCKDUCKGO: Engine = {
|
|
@@ -668,6 +696,7 @@ const WIKIPEDIA: Engine = {
|
|
|
668
696
|
provider: "wikipedia",
|
|
669
697
|
priority: 2,
|
|
670
698
|
async search(query, ctx, timeoutMs, signal) {
|
|
699
|
+
const started = Date.now();
|
|
671
700
|
const lang = ctx.region.toLowerCase().split("-")[1] ?? "en";
|
|
672
701
|
const encoded = encodeURIComponent(query);
|
|
673
702
|
const opensearchUrl =
|
|
@@ -688,7 +717,7 @@ const WIKIPEDIA: Engine = {
|
|
|
688
717
|
const extractUrl =
|
|
689
718
|
`https://${lang}.wikipedia.org/w/api.php?action=query&format=json&prop=extracts` +
|
|
690
719
|
`&titles=${encodeURIComponent(title)}&explaintext=0&exintro=0&redirects=1`;
|
|
691
|
-
const extract = await httpGet(extractUrl, {}, { timeoutMs, signal });
|
|
720
|
+
const extract = await httpGet(extractUrl, {}, { timeoutMs: Math.max(1, timeoutMs - (Date.now() - started)), signal });
|
|
692
721
|
if (extract) {
|
|
693
722
|
try {
|
|
694
723
|
const pageData = JSON.parse(extract) as {
|
|
@@ -788,23 +817,27 @@ export async function autoTextSearch(
|
|
|
788
817
|
signal?: AbortSignal,
|
|
789
818
|
): Promise<SearchResult[]> {
|
|
790
819
|
const engines = shuffledEngines();
|
|
820
|
+
const deadline = Date.now() + timeoutMs;
|
|
791
821
|
const seenProviders = new Set<string>();
|
|
792
822
|
const aggregator = new ResultsAggregator();
|
|
793
823
|
const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
|
|
794
824
|
let timedOut = false;
|
|
825
|
+
let cancelled = false;
|
|
795
826
|
const uniqueProviders = new Set(engines.map((e) => e.provider)).size;
|
|
796
827
|
const maxWorkers = Math.min(uniqueProviders, Math.ceil(maxResults / 10) + 1);
|
|
797
828
|
let i = 0;
|
|
798
829
|
let pending: Promise<void>[] = [];
|
|
799
830
|
const run = async (engine: Engine) => {
|
|
831
|
+
const remaining = Math.max(1, deadline - Date.now());
|
|
800
832
|
try {
|
|
801
|
-
const results = await engine.search(query, ctx,
|
|
833
|
+
const results = await engine.search(query, ctx, remaining, signal);
|
|
802
834
|
if (results && results.length) {
|
|
803
835
|
aggregator.extend(results);
|
|
804
836
|
seenProviders.add(engine.provider);
|
|
805
837
|
}
|
|
806
838
|
} catch (e) {
|
|
807
|
-
if (e instanceof
|
|
839
|
+
if (e instanceof SearchCancelled) cancelled = true;
|
|
840
|
+
if (e instanceof SearchTimeoutError) timedOut = true;
|
|
808
841
|
}
|
|
809
842
|
};
|
|
810
843
|
while (i < engines.length) {
|
|
@@ -818,6 +851,7 @@ export async function autoTextSearch(
|
|
|
818
851
|
}
|
|
819
852
|
}
|
|
820
853
|
await Promise.allSettled(pending);
|
|
854
|
+
if (cancelled) throw new SearchCancelled();
|
|
821
855
|
const results = rankResults(aggregator.extractDicts(), query);
|
|
822
856
|
if (results.length) return results.slice(0, maxResults);
|
|
823
857
|
if (timedOut) throw new SearchTimeoutError();
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.5",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -51,7 +51,8 @@
|
|
|
51
51
|
"test:unit": "vitest run --exclude **/smoke.test.ts",
|
|
52
52
|
"test:smoke": "vitest run test/smoke.test.ts",
|
|
53
53
|
"typecheck": "tsc --noEmit",
|
|
54
|
-
"
|
|
54
|
+
"check:package": "node -e \"const fs=require('fs');const pkg=require('./package.json');const listed=new Set(pkg.files);const bad=[];for(const f of pkg.files){if(!fs.existsSync(f))bad.push('missing file: '+f)}for(const f of fs.readdirSync('.').filter(f=>f.endsWith('.ts'))){if(!listed.has(f))bad.push('unlisted source: '+f)}if(bad.length){console.error(bad.join('\\n'));process.exit(1)}\"",
|
|
55
|
+
"prepublishOnly": "npm run typecheck && npm run test:unit && npm run check:package"
|
|
55
56
|
},
|
|
56
57
|
"devDependencies": {
|
|
57
58
|
"@earendil-works/pi-coding-agent": "^0.84.0",
|
package/web-fetch.ts
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
type WebsitePolicy,
|
|
11
11
|
} from "./web-access.ts";
|
|
12
12
|
import { htmlToMarkdown } from "./html-to-md.ts";
|
|
13
|
+
import { INVALID_CHARREFS } from "./entities.ts";
|
|
13
14
|
import { extractPdfText, PdfParseError } from "./pdf.ts";
|
|
14
15
|
import { randomUserAgent } from "./user-agents.ts";
|
|
15
16
|
|
|
@@ -97,14 +98,17 @@ const ASCII_TEXT_BYTES = new Set<number>([
|
|
|
97
98
|
0x1b,
|
|
98
99
|
]);
|
|
99
100
|
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
101
|
+
export class FetchCancelledError extends Error {
|
|
102
|
+
constructor() {
|
|
103
|
+
super("cancelled");
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
export class FetchTimeoutError extends Error {
|
|
108
|
+
constructor() {
|
|
109
|
+
super("timed out");
|
|
110
|
+
}
|
|
111
|
+
}
|
|
108
112
|
|
|
109
113
|
export interface FetchPageOptions {
|
|
110
114
|
timeoutMs?: number;
|
|
@@ -145,6 +149,8 @@ export interface HopOptions {
|
|
|
145
149
|
maxBytes: number;
|
|
146
150
|
maxPdfBytes: number;
|
|
147
151
|
inactivityMs: number;
|
|
152
|
+
deadlineMs?: number;
|
|
153
|
+
nowMs?: () => number;
|
|
148
154
|
signal?: AbortSignal;
|
|
149
155
|
}
|
|
150
156
|
|
|
@@ -339,8 +345,8 @@ function decodeSingleByte(bytes: Buffer, cp1252: boolean): string {
|
|
|
339
345
|
for (const byte of bytes) {
|
|
340
346
|
if (byte < 0x80) {
|
|
341
347
|
out += String.fromCharCode(byte);
|
|
342
|
-
} else if (cp1252 && byte in
|
|
343
|
-
out +=
|
|
348
|
+
} else if (cp1252 && byte in INVALID_CHARREFS) {
|
|
349
|
+
out += INVALID_CHARREFS[byte];
|
|
344
350
|
} else {
|
|
345
351
|
out += String.fromCharCode(byte);
|
|
346
352
|
}
|
|
@@ -431,7 +437,7 @@ function fetchBudgetExceeded(
|
|
|
431
437
|
}
|
|
432
438
|
|
|
433
439
|
|
|
434
|
-
function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
440
|
+
export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
435
441
|
return new Promise((resolve, reject) => {
|
|
436
442
|
const url = opts.url;
|
|
437
443
|
const transport = url.protocol === "https:" ? https : http;
|
|
@@ -446,27 +452,49 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
446
452
|
lookup: (_host, _opts, callback) =>
|
|
447
453
|
callback(null, [{ address: opts.pinnedIp, family: opts.family }]),
|
|
448
454
|
};
|
|
455
|
+
let settled = false;
|
|
456
|
+
const settle = (action: () => void) => {
|
|
457
|
+
if (settled) return;
|
|
458
|
+
settled = true;
|
|
459
|
+
opts.signal?.removeEventListener("abort", onAbort);
|
|
460
|
+
action();
|
|
461
|
+
};
|
|
449
462
|
const request = transport.request(options, (res: IncomingMessage) => {
|
|
450
463
|
const chunks: Buffer[] = [];
|
|
451
464
|
let total = 0;
|
|
465
|
+
let head = Buffer.alloc(0);
|
|
452
466
|
const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
|
|
453
467
|
let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
|
|
454
468
|
let extendedForPdf = false;
|
|
455
469
|
const finish = (err: string | null, body: Buffer) => {
|
|
456
470
|
settle(() => {
|
|
457
|
-
if (err)
|
|
458
|
-
|
|
471
|
+
if (err) {
|
|
472
|
+
if (err === "cancelled") reject(new FetchCancelledError());
|
|
473
|
+
else if (err === "timed out") reject(new FetchTimeoutError());
|
|
474
|
+
else reject(new Error(err));
|
|
475
|
+
} else {
|
|
459
476
|
resolve({
|
|
460
477
|
status: res.statusCode ?? 0,
|
|
461
478
|
headers: res.headers as Record<string, string | string[] | undefined>,
|
|
462
479
|
body,
|
|
463
480
|
});
|
|
481
|
+
}
|
|
464
482
|
});
|
|
465
483
|
};
|
|
466
484
|
res.on("data", (chunk: Buffer) => {
|
|
467
485
|
if (settled) return;
|
|
486
|
+
const now = opts.nowMs ?? Date.now;
|
|
487
|
+
if (opts.deadlineMs !== undefined && now() >= opts.deadlineMs) {
|
|
488
|
+
res.destroy();
|
|
489
|
+
finish("timed out", Buffer.concat(chunks));
|
|
490
|
+
return;
|
|
491
|
+
}
|
|
492
|
+
if (head.length < 1024) {
|
|
493
|
+
const need = 1024 - head.length;
|
|
494
|
+
head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
|
|
495
|
+
}
|
|
468
496
|
if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
|
|
469
|
-
if (hasPdfMagic(
|
|
497
|
+
if (hasPdfMagic(head)) {
|
|
470
498
|
limit = opts.maxPdfBytes;
|
|
471
499
|
extendedForPdf = true;
|
|
472
500
|
}
|
|
@@ -488,16 +516,9 @@ function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
488
516
|
res.on("end", () => finish(null, Buffer.concat(chunks)));
|
|
489
517
|
res.on("error", (err) => finish(err.message, Buffer.concat(chunks)));
|
|
490
518
|
});
|
|
491
|
-
const onAbort = () => request.destroy(new
|
|
519
|
+
const onAbort = () => request.destroy(new FetchCancelledError());
|
|
492
520
|
opts.signal?.addEventListener("abort", onAbort, { once: true });
|
|
493
|
-
|
|
494
|
-
const settle = (action: () => void) => {
|
|
495
|
-
if (settled) return;
|
|
496
|
-
settled = true;
|
|
497
|
-
opts.signal?.removeEventListener("abort", onAbort);
|
|
498
|
-
action();
|
|
499
|
-
};
|
|
500
|
-
request.on("timeout", () => request.destroy(new Error("timed out")));
|
|
521
|
+
request.on("timeout", () => request.destroy(new FetchTimeoutError()));
|
|
501
522
|
request.on("error", (err) => settle(() => reject(err)));
|
|
502
523
|
request.end();
|
|
503
524
|
});
|
|
@@ -557,9 +578,15 @@ export async function fetchUrlRaw(
|
|
|
557
578
|
maxBytes,
|
|
558
579
|
maxPdfBytes,
|
|
559
580
|
inactivityMs: inactivity,
|
|
581
|
+
deadlineMs: deadline,
|
|
582
|
+
nowMs: now,
|
|
560
583
|
signal,
|
|
561
584
|
});
|
|
562
585
|
} catch (err) {
|
|
586
|
+
if (err instanceof FetchCancelledError)
|
|
587
|
+
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
588
|
+
if (err instanceof FetchTimeoutError)
|
|
589
|
+
return { error: "Failed to fetch URL: timed out.", body: "", contentType: "" };
|
|
563
590
|
const message = err instanceof Error ? err.message : String(err);
|
|
564
591
|
if (message === "cancelled")
|
|
565
592
|
return { error: "Failed to fetch URL: cancelled.", body: "", contentType: "" };
|
|
@@ -606,6 +633,15 @@ export async function fetchUrlRaw(
|
|
|
606
633
|
budgetError = fetchBudgetExceeded(deadline, signal, now);
|
|
607
634
|
if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
|
|
608
635
|
|
|
636
|
+
if (response.status >= 400) {
|
|
637
|
+
const reason = http.STATUS_CODES[response.status] ?? "";
|
|
638
|
+
return {
|
|
639
|
+
error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
|
|
640
|
+
body: "",
|
|
641
|
+
contentType: "",
|
|
642
|
+
};
|
|
643
|
+
}
|
|
644
|
+
|
|
609
645
|
const contentTypeHeader = response.headers["content-type"];
|
|
610
646
|
const contentType = contentTypeHeader
|
|
611
647
|
? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
|
|
@@ -657,7 +693,7 @@ export async function fetchUrlRaw(
|
|
|
657
693
|
const bomCodec = bomCodecFor(response.body);
|
|
658
694
|
const rawHtml = decodeWithCodec(
|
|
659
695
|
response.body,
|
|
660
|
-
|
|
696
|
+
bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
|
|
661
697
|
);
|
|
662
698
|
|
|
663
699
|
if (looksBinary(rawHtml)) {
|
package/web-search.ts
CHANGED
|
@@ -93,7 +93,7 @@ export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS
|
|
|
93
93
|
export function formatSearchResults(results: SearchResult[]): string {
|
|
94
94
|
const parts = results.map((result) => {
|
|
95
95
|
const title = String(result.title ?? "").replace(/\s+/g, " ");
|
|
96
|
-
const href = String(result.href ?? "").trim();
|
|
96
|
+
const href = String(result.href ?? "").replace(/\s+/g, " ").trim();
|
|
97
97
|
const snippet = String(result.body ?? "").replace(/\s+/g, " ");
|
|
98
98
|
return `Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`;
|
|
99
99
|
});
|