pi-unsloth-webtools 0.2.4 → 0.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -2
- package/engines.ts +6 -2
- package/package.json +3 -2
- package/web-fetch.ts +16 -2
- package/web-search.ts +1 -1
package/README.md
CHANGED
|
@@ -45,6 +45,8 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
45
45
|
validation and fetch.
|
|
46
46
|
- GitHub repo root pages are rewritten to the unauthenticated README API
|
|
47
47
|
(`Accept: application/vnd.github.raw+json`), falling back to the HTML page on failure.
|
|
48
|
+
- HTTP 4xx/5xx responses are reported as errors with the status reason instead of
|
|
49
|
+
returning error-page content.
|
|
48
50
|
- Up to 4 redirect hops, each re-validated and re-resolved against the same rules.
|
|
49
51
|
- 512 KiB download cap (10 MiB for PDFs), overall deadline + per-hop socket timeouts, abort-aware
|
|
50
52
|
(`signal` cancels mid-flight).
|
|
@@ -53,7 +55,7 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
53
55
|
pymupdf4llm-style markdown layer (headings, bold/italic, code fences, links, tables)
|
|
54
56
|
with Studio's corrupted/incomplete fallback to plain text.
|
|
55
57
|
- Content sniffing: MIME allow/deny, binary magic signatures, PDF magic detection, and charset
|
|
56
|
-
decoding (
|
|
58
|
+
decoding (BOM sniffing for UTF-8/16/32 first, then the declared charset, `<meta charset>` sniffing for
|
|
57
59
|
CJK and Windows/ISO encodings, cp1252 rescue for mislabeled single-byte pages).
|
|
58
60
|
- HTML → Markdown conversion ported from Studio's dependency-free `_html_to_md.py`: headings,
|
|
59
61
|
links, emphasis, lists, tables, blockquotes, code fences, entity decoding; hidden-element
|
|
@@ -78,7 +80,9 @@ Port of Studio's `_fetch_page_text` / `_fetch_url_raw` pipeline:
|
|
|
78
80
|
database.
|
|
79
81
|
- Empty sweeps: ddgs 9.14.4 raises the last engine exception; this port reports a
|
|
80
82
|
timeout whenever any engine timed out, so the timeout message is not masked by later
|
|
81
|
-
generic engine failures.
|
|
83
|
+
generic engine failures. The timeout budget bounds the entire sweep: per-engine
|
|
84
|
+
timeouts shrink as the budget is consumed, so the reported timeout matches the
|
|
85
|
+
worst-case wall time.
|
|
82
86
|
- Proxies: Studio routes through environment proxies; this port always connects
|
|
83
87
|
directly with DNS pinning (deliberately out of scope).
|
|
84
88
|
|
package/engines.ts
CHANGED
|
@@ -510,6 +510,7 @@ async function httpFetch(
|
|
|
510
510
|
Accept: "*/*",
|
|
511
511
|
...options.headers,
|
|
512
512
|
};
|
|
513
|
+
if (options.method === "POST") headers["Content-Type"] = "application/x-www-form-urlencoded";
|
|
513
514
|
const cookie = options.cookies
|
|
514
515
|
? Object.entries(options.cookies)
|
|
515
516
|
.map(([key, value]) => `${key}=${value}`)
|
|
@@ -695,6 +696,7 @@ const WIKIPEDIA: Engine = {
|
|
|
695
696
|
provider: "wikipedia",
|
|
696
697
|
priority: 2,
|
|
697
698
|
async search(query, ctx, timeoutMs, signal) {
|
|
699
|
+
const started = Date.now();
|
|
698
700
|
const lang = ctx.region.toLowerCase().split("-")[1] ?? "en";
|
|
699
701
|
const encoded = encodeURIComponent(query);
|
|
700
702
|
const opensearchUrl =
|
|
@@ -715,7 +717,7 @@ const WIKIPEDIA: Engine = {
|
|
|
715
717
|
const extractUrl =
|
|
716
718
|
`https://${lang}.wikipedia.org/w/api.php?action=query&format=json&prop=extracts` +
|
|
717
719
|
`&titles=${encodeURIComponent(title)}&explaintext=0&exintro=0&redirects=1`;
|
|
718
|
-
const extract = await httpGet(extractUrl, {}, { timeoutMs, signal });
|
|
720
|
+
const extract = await httpGet(extractUrl, {}, { timeoutMs: Math.max(1, timeoutMs - (Date.now() - started)), signal });
|
|
719
721
|
if (extract) {
|
|
720
722
|
try {
|
|
721
723
|
const pageData = JSON.parse(extract) as {
|
|
@@ -815,6 +817,7 @@ export async function autoTextSearch(
|
|
|
815
817
|
signal?: AbortSignal,
|
|
816
818
|
): Promise<SearchResult[]> {
|
|
817
819
|
const engines = shuffledEngines();
|
|
820
|
+
const deadline = Date.now() + timeoutMs;
|
|
818
821
|
const seenProviders = new Set<string>();
|
|
819
822
|
const aggregator = new ResultsAggregator();
|
|
820
823
|
const ctx: EngineContext = { region: "us-en", safesearch: "moderate" };
|
|
@@ -825,8 +828,9 @@ export async function autoTextSearch(
|
|
|
825
828
|
let i = 0;
|
|
826
829
|
let pending: Promise<void>[] = [];
|
|
827
830
|
const run = async (engine: Engine) => {
|
|
831
|
+
const remaining = Math.max(1, deadline - Date.now());
|
|
828
832
|
try {
|
|
829
|
-
const results = await engine.search(query, ctx,
|
|
833
|
+
const results = await engine.search(query, ctx, remaining, signal);
|
|
830
834
|
if (results && results.length) {
|
|
831
835
|
aggregator.extend(results);
|
|
832
836
|
seenProviders.add(engine.provider);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.5",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -51,7 +51,8 @@
|
|
|
51
51
|
"test:unit": "vitest run --exclude **/smoke.test.ts",
|
|
52
52
|
"test:smoke": "vitest run test/smoke.test.ts",
|
|
53
53
|
"typecheck": "tsc --noEmit",
|
|
54
|
-
"
|
|
54
|
+
"check:package": "node -e \"const fs=require('fs');const pkg=require('./package.json');const listed=new Set(pkg.files);const bad=[];for(const f of pkg.files){if(!fs.existsSync(f))bad.push('missing file: '+f)}for(const f of fs.readdirSync('.').filter(f=>f.endsWith('.ts'))){if(!listed.has(f))bad.push('unlisted source: '+f)}if(bad.length){console.error(bad.join('\\n'));process.exit(1)}\"",
|
|
55
|
+
"prepublishOnly": "npm run typecheck && npm run test:unit && npm run check:package"
|
|
55
56
|
},
|
|
56
57
|
"devDependencies": {
|
|
57
58
|
"@earendil-works/pi-coding-agent": "^0.84.0",
|
package/web-fetch.ts
CHANGED
|
@@ -462,6 +462,7 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
462
462
|
const request = transport.request(options, (res: IncomingMessage) => {
|
|
463
463
|
const chunks: Buffer[] = [];
|
|
464
464
|
let total = 0;
|
|
465
|
+
let head = Buffer.alloc(0);
|
|
465
466
|
const declaredPdf = String(res.headers["content-type"] ?? "").toLowerCase().includes("pdf");
|
|
466
467
|
let limit = declaredPdf ? opts.maxPdfBytes : opts.maxBytes;
|
|
467
468
|
let extendedForPdf = false;
|
|
@@ -488,8 +489,12 @@ export function requestHop(opts: HopOptions): Promise<HopResponse> {
|
|
|
488
489
|
finish("timed out", Buffer.concat(chunks));
|
|
489
490
|
return;
|
|
490
491
|
}
|
|
492
|
+
if (head.length < 1024) {
|
|
493
|
+
const need = 1024 - head.length;
|
|
494
|
+
head = Buffer.concat([head, chunk.subarray(0, Math.min(need, chunk.length))]);
|
|
495
|
+
}
|
|
491
496
|
if (!declaredPdf && !extendedForPdf && total + chunk.length > opts.maxBytes) {
|
|
492
|
-
if (hasPdfMagic(
|
|
497
|
+
if (hasPdfMagic(head)) {
|
|
493
498
|
limit = opts.maxPdfBytes;
|
|
494
499
|
extendedForPdf = true;
|
|
495
500
|
}
|
|
@@ -628,6 +633,15 @@ export async function fetchUrlRaw(
|
|
|
628
633
|
budgetError = fetchBudgetExceeded(deadline, signal, now);
|
|
629
634
|
if (budgetError !== null) return { error: budgetError, body: "", contentType: "" };
|
|
630
635
|
|
|
636
|
+
if (response.status >= 400) {
|
|
637
|
+
const reason = http.STATUS_CODES[response.status] ?? "";
|
|
638
|
+
return {
|
|
639
|
+
error: `Failed to fetch URL: HTTP ${response.status}${reason ? ` ${reason}` : ""}`,
|
|
640
|
+
body: "",
|
|
641
|
+
contentType: "",
|
|
642
|
+
};
|
|
643
|
+
}
|
|
644
|
+
|
|
631
645
|
const contentTypeHeader = response.headers["content-type"];
|
|
632
646
|
const contentType = contentTypeHeader
|
|
633
647
|
? (/^[\w.+-]+\/[\w.+-]+/.exec(String(contentTypeHeader).toLowerCase()) ?? [""])[0]
|
|
@@ -679,7 +693,7 @@ export async function fetchUrlRaw(
|
|
|
679
693
|
const bomCodec = bomCodecFor(response.body);
|
|
680
694
|
const rawHtml = decodeWithCodec(
|
|
681
695
|
response.body,
|
|
682
|
-
|
|
696
|
+
bomCodec ?? declaredCodec ?? sniffMetaCharsetForHtml(response.body, contentType) ?? "utf-8",
|
|
683
697
|
);
|
|
684
698
|
|
|
685
699
|
if (looksBinary(rawHtml)) {
|
package/web-search.ts
CHANGED
|
@@ -93,7 +93,7 @@ export function searchFailureMessage(exc: unknown, timeoutMs = SEARCH_TIMEOUT_MS
|
|
|
93
93
|
export function formatSearchResults(results: SearchResult[]): string {
|
|
94
94
|
const parts = results.map((result) => {
|
|
95
95
|
const title = String(result.title ?? "").replace(/\s+/g, " ");
|
|
96
|
-
const href = String(result.href ?? "").trim();
|
|
96
|
+
const href = String(result.href ?? "").replace(/\s+/g, " ").trim();
|
|
97
97
|
const snippet = String(result.body ?? "").replace(/\s+/g, " ");
|
|
98
98
|
return `Title: ${title}\nURL: ${href}\nSnippet: ${snippet}`;
|
|
99
99
|
});
|