pi-unsloth-webtools 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -2
- package/agent-dir.ts +19 -0
- package/cache.ts +154 -0
- package/engines.ts +82 -9
- package/html-to-md.ts +84 -52
- package/index.ts +117 -63
- package/package.json +4 -1
- package/pdf.ts +6 -1
- package/settings.ts +114 -0
- package/web-access.ts +53 -27
- package/web-fetch.ts +296 -115
- package/web-search.ts +7 -1
package/index.ts
CHANGED
|
@@ -1,10 +1,23 @@
|
|
|
1
|
-
import type
|
|
1
|
+
import { defineTool, type ExtensionAPI, type ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import { Type } from "typebox";
|
|
3
|
-
import { webSearch } from "./web-search.ts";
|
|
4
|
-
import { DEFAULT_FETCH_TIMEOUT_MS, fetchPageText } from "./web-fetch.ts";
|
|
3
|
+
import { webSearch as defaultWebSearch } from "./web-search.ts";
|
|
4
|
+
import { DEFAULT_FETCH_TIMEOUT_MS, fetchPageText as defaultFetchPageText } from "./web-fetch.ts";
|
|
5
|
+
import { loadDefaultFetchSettings } from "./settings.ts";
|
|
5
6
|
|
|
6
|
-
function
|
|
7
|
-
|
|
7
|
+
function positiveNumber(value: unknown): number | undefined {
|
|
8
|
+
if (typeof value !== "number" || !Number.isFinite(value)) return undefined;
|
|
9
|
+
const n = Math.floor(value);
|
|
10
|
+
return n > 0 ? n : undefined;
|
|
11
|
+
}
|
|
12
|
+
async function fetchDefaults(cwd: string | undefined, params: { timeoutMs?: unknown; maxChars?: unknown }) {
|
|
13
|
+
const timeoutParam = positiveNumber(params.timeoutMs);
|
|
14
|
+
const maxCharsParam = positiveNumber(params.maxChars);
|
|
15
|
+
if (timeoutParam !== undefined && maxCharsParam !== undefined) return { timeoutMs: timeoutParam, maxChars: maxCharsParam };
|
|
16
|
+
const defaults = await loadDefaultFetchSettings(cwd);
|
|
17
|
+
return {
|
|
18
|
+
timeoutMs: timeoutParam ?? defaults.timeoutMs ?? DEFAULT_FETCH_TIMEOUT_MS,
|
|
19
|
+
maxChars: maxCharsParam ?? defaults.maxChars,
|
|
20
|
+
};
|
|
8
21
|
}
|
|
9
22
|
|
|
10
23
|
const WebSearchParams = Type.Object({
|
|
@@ -23,6 +36,19 @@ const WebSearchParams = Type.Object({
|
|
|
23
36
|
"Truncate the fetched page to this many characters (only used with the url parameter)",
|
|
24
37
|
}),
|
|
25
38
|
),
|
|
39
|
+
maxResults: Type.Optional(
|
|
40
|
+
Type.Number({
|
|
41
|
+
minimum: 1,
|
|
42
|
+
maximum: 20,
|
|
43
|
+
description: "Maximum number of search results to return (default: 5)",
|
|
44
|
+
}),
|
|
45
|
+
),
|
|
46
|
+
timeoutMs: Type.Optional(
|
|
47
|
+
Type.Number({
|
|
48
|
+
minimum: 1000,
|
|
49
|
+
description: "Overall timeout in milliseconds for the search or fetch",
|
|
50
|
+
}),
|
|
51
|
+
),
|
|
26
52
|
});
|
|
27
53
|
|
|
28
54
|
const WebFetchParams = Type.Object({
|
|
@@ -32,65 +58,93 @@ const WebFetchParams = Type.Object({
|
|
|
32
58
|
description: "Truncate the returned content to this many characters (default: no limit)",
|
|
33
59
|
}),
|
|
34
60
|
),
|
|
61
|
+
timeoutMs: Type.Optional(
|
|
62
|
+
Type.Number({
|
|
63
|
+
minimum: 1000,
|
|
64
|
+
description: "Overall timeout in milliseconds (default: 60000)",
|
|
65
|
+
}),
|
|
66
|
+
),
|
|
35
67
|
});
|
|
36
68
|
|
|
37
|
-
export
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
description:
|
|
42
|
-
"Search the web and fetch page content. Returns snippets for all results. " +
|
|
43
|
-
"Use the url parameter to fetch full page text from a specific URL.",
|
|
44
|
-
promptSnippet: "Search the web and fetch page content",
|
|
45
|
-
promptGuidelines: [
|
|
46
|
-
"Use web_search with the url parameter (e.g. {\"url\": \"<URL>\"}) to read the full text of a page found in search results.",
|
|
47
|
-
],
|
|
48
|
-
parameters: WebSearchParams,
|
|
49
|
-
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
50
|
-
if (params.url?.trim()) {
|
|
51
|
-
const url = params.url.trim();
|
|
52
|
-
onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
|
|
53
|
-
return {
|
|
54
|
-
content: [
|
|
55
|
-
{
|
|
56
|
-
type: "text",
|
|
57
|
-
text: await fetchPageText(url, {
|
|
58
|
-
timeoutMs: DEFAULT_FETCH_TIMEOUT_MS,
|
|
59
|
-
signal: signal ?? undefined,
|
|
60
|
-
maxChars: positiveMaxChars(params.maxChars),
|
|
61
|
-
}),
|
|
62
|
-
},
|
|
63
|
-
],
|
|
64
|
-
details: {},
|
|
65
|
-
};
|
|
66
|
-
}
|
|
67
|
-
onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
|
|
68
|
-
const text = await webSearch(params.query, { signal: signal ?? undefined });
|
|
69
|
-
return { content: [{ type: "text", text }], details: {} };
|
|
70
|
-
},
|
|
71
|
-
});
|
|
69
|
+
export interface WebToolsDeps {
|
|
70
|
+
fetchPageText?: typeof defaultFetchPageText;
|
|
71
|
+
webSearch?: typeof defaultWebSearch;
|
|
72
|
+
}
|
|
72
73
|
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
"
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
74
|
+
export function createWebTools(deps: WebToolsDeps = {}) {
|
|
75
|
+
const fetchPageText = deps.fetchPageText ?? defaultFetchPageText;
|
|
76
|
+
const webSearch = deps.webSearch ?? defaultWebSearch;
|
|
77
|
+
return {
|
|
78
|
+
webSearchTool: defineTool({
|
|
79
|
+
name: "web_search",
|
|
80
|
+
label: "Web Search",
|
|
81
|
+
description:
|
|
82
|
+
"Search the web and fetch page content. Returns snippets for all results. " +
|
|
83
|
+
"Use the url parameter to fetch full page text from a specific URL.",
|
|
84
|
+
promptSnippet: "Search the web and fetch page content",
|
|
85
|
+
promptGuidelines: [
|
|
86
|
+
'Use web_search with the url parameter (e.g. {"url": "<URL>"}) to read the full text of a page found in search results.',
|
|
87
|
+
],
|
|
88
|
+
parameters: WebSearchParams,
|
|
89
|
+
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
90
|
+
if (params.url?.trim()) {
|
|
91
|
+
const url = params.url.trim();
|
|
92
|
+
onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
|
|
93
|
+
const cwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
94
|
+
const { timeoutMs, maxChars } = await fetchDefaults(cwd, params);
|
|
95
|
+
return {
|
|
96
|
+
content: [
|
|
97
|
+
{
|
|
98
|
+
type: "text",
|
|
99
|
+
text: await fetchPageText(url, {
|
|
100
|
+
timeoutMs,
|
|
101
|
+
signal: signal ?? undefined,
|
|
102
|
+
maxChars,
|
|
103
|
+
}),
|
|
104
|
+
},
|
|
105
|
+
],
|
|
106
|
+
details: {},
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
|
|
110
|
+
const text = await webSearch(params.query, {
|
|
111
|
+
signal: signal ?? undefined,
|
|
112
|
+
timeoutMs: positiveNumber(params.timeoutMs),
|
|
113
|
+
maxResults: positiveNumber(params.maxResults),
|
|
114
|
+
cwd: (_ctx as ExtensionContext | undefined)?.cwd,
|
|
115
|
+
});
|
|
116
|
+
return { content: [{ type: "text", text }], details: {} };
|
|
117
|
+
},
|
|
118
|
+
}),
|
|
119
|
+
webFetchTool: defineTool({
|
|
120
|
+
name: "web_fetch",
|
|
121
|
+
label: "Web Fetch",
|
|
122
|
+
description:
|
|
123
|
+
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
124
|
+
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
125
|
+
"stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
126
|
+
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
127
|
+
"Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
|
|
128
|
+
"is capped.",
|
|
129
|
+
promptSnippet: "Fetch a web page and return readable text content",
|
|
130
|
+
parameters: WebFetchParams,
|
|
131
|
+
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
132
|
+
onUpdate?.({ content: [{ type: "text", text: `Fetching ${params.url}...` }], details: {} });
|
|
133
|
+
const cwd = (_ctx as ExtensionContext | undefined)?.cwd;
|
|
134
|
+
const { timeoutMs, maxChars } = await fetchDefaults(cwd, params);
|
|
135
|
+
const text = await fetchPageText(params.url, {
|
|
136
|
+
timeoutMs,
|
|
137
|
+
signal: signal ?? undefined,
|
|
138
|
+
maxChars,
|
|
139
|
+
});
|
|
140
|
+
return { content: [{ type: "text", text }], details: {} };
|
|
141
|
+
},
|
|
142
|
+
}),
|
|
143
|
+
};
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
export default function (pi: ExtensionAPI) {
|
|
147
|
+
const { webSearchTool, webFetchTool } = createWebTools();
|
|
148
|
+
pi.registerTool(webSearchTool);
|
|
149
|
+
pi.registerTool(webFetchTool);
|
|
96
150
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.4.0",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
|
@@ -31,6 +31,9 @@
|
|
|
31
31
|
"entities.ts",
|
|
32
32
|
"pdf.ts",
|
|
33
33
|
"user-agents.ts",
|
|
34
|
+
"cache.ts",
|
|
35
|
+
"settings.ts",
|
|
36
|
+
"agent-dir.ts",
|
|
34
37
|
"README.md",
|
|
35
38
|
"LICENSE"
|
|
36
39
|
],
|
package/pdf.ts
CHANGED
|
@@ -594,7 +594,7 @@ function assemblePages(
|
|
|
594
594
|
pageLimitOverride?: boolean,
|
|
595
595
|
): string {
|
|
596
596
|
const parts: string[] = [];
|
|
597
|
-
const pageLimitReached = pageLimitOverride ?? pages.length
|
|
597
|
+
const pageLimitReached = pageLimitOverride ?? pages.length > MAX_WEB_PDF_PAGES;
|
|
598
598
|
for (const page of pages) {
|
|
599
599
|
const pageText = page.text.trim();
|
|
600
600
|
if (!pageText) continue;
|
|
@@ -697,6 +697,11 @@ function extractTextOps(bytes: Buffer): string {
|
|
|
697
697
|
}
|
|
698
698
|
}
|
|
699
699
|
}
|
|
700
|
+
if (c === "'" || c === '"') {
|
|
701
|
+
if (inText) out.push("\n");
|
|
702
|
+
i++;
|
|
703
|
+
continue;
|
|
704
|
+
}
|
|
700
705
|
const token = text.slice(i, i + 2);
|
|
701
706
|
if (token === "BT") {
|
|
702
707
|
inText = true;
|
package/settings.ts
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import { join } from "node:path";
|
|
3
|
+
import { agentDir } from "./agent-dir.ts";
|
|
4
|
+
|
|
5
|
+
async function readJson(file: string): Promise<Record<string, unknown> | undefined> {
|
|
6
|
+
try {
|
|
7
|
+
const raw = await readFile(file, "utf-8");
|
|
8
|
+
const data = JSON.parse(raw);
|
|
9
|
+
if (data && typeof data === "object" && !Array.isArray(data)) return data as Record<string, unknown>;
|
|
10
|
+
return undefined;
|
|
11
|
+
} catch {
|
|
12
|
+
return undefined;
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function toNumber(value: unknown): number | undefined {
|
|
17
|
+
if (typeof value !== "number" || !Number.isFinite(value)) return undefined;
|
|
18
|
+
return Math.floor(value);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
function pickNumber(data: Record<string, unknown>, paths: string[][]): number | undefined {
|
|
22
|
+
for (const path of paths) {
|
|
23
|
+
let cur: unknown = data;
|
|
24
|
+
for (const key of path) {
|
|
25
|
+
if (cur && typeof cur === "object" && !Array.isArray(cur)) cur = (cur as Record<string, unknown>)[key];
|
|
26
|
+
else {
|
|
27
|
+
cur = undefined;
|
|
28
|
+
break;
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
const n = toNumber(cur);
|
|
32
|
+
if (n !== undefined) return n;
|
|
33
|
+
}
|
|
34
|
+
return undefined;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const MAX_RESULTS = 5;
|
|
38
|
+
|
|
39
|
+
const MAX_RESULTS_PATHS: string[][] = [
|
|
40
|
+
["unslothWebTools", "maxResults"],
|
|
41
|
+
["webSearch", "maxResults"],
|
|
42
|
+
["smartWebSearch", "resultsPerQuery"],
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
const FETCH_MAX_CHARS_PATHS: string[][] = [
|
|
46
|
+
["unslothWebTools", "maxChars"],
|
|
47
|
+
["webFetch", "maxChars"],
|
|
48
|
+
["smartFetchDefaultMaxChars"],
|
|
49
|
+
];
|
|
50
|
+
|
|
51
|
+
const FETCH_TIMEOUT_PATHS: string[][] = [
|
|
52
|
+
["unslothWebTools", "timeoutMs"],
|
|
53
|
+
["webFetch", "timeoutMs"],
|
|
54
|
+
["smartFetchDefaultTimeoutMs"],
|
|
55
|
+
];
|
|
56
|
+
|
|
57
|
+
function clampMaxResults(value: number): number {
|
|
58
|
+
return Math.min(20, Math.max(1, value));
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function settingsFiles(cwd?: string): string[] {
|
|
62
|
+
const base = agentDir();
|
|
63
|
+
const globalFile = base ? join(base, "settings.json") : "";
|
|
64
|
+
const files = globalFile ? [globalFile] : [];
|
|
65
|
+
if (cwd) files.push(join(cwd, ".pi", "settings.json"));
|
|
66
|
+
return files;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
async function settingsEntries(cwd?: string): Promise<Record<string, unknown>[]> {
|
|
70
|
+
const entries: Record<string, unknown>[] = [];
|
|
71
|
+
for (const file of settingsFiles(cwd)) {
|
|
72
|
+
const data = await readJson(file);
|
|
73
|
+
if (data) entries.push(data);
|
|
74
|
+
}
|
|
75
|
+
return entries;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
export async function loadDefaultMaxResults(cwd?: string): Promise<number> {
|
|
79
|
+
let result = MAX_RESULTS;
|
|
80
|
+
for (const data of await settingsEntries(cwd)) {
|
|
81
|
+
const candidate = pickNumber(data, MAX_RESULTS_PATHS);
|
|
82
|
+
if (candidate !== undefined) result = clampMaxResults(candidate);
|
|
83
|
+
}
|
|
84
|
+
return result;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
async function loadFetchSetting(cwd: string | undefined, paths: string[][], valid: (n: number) => boolean): Promise<number | undefined> {
|
|
88
|
+
let result: number | undefined;
|
|
89
|
+
for (const data of await settingsEntries(cwd)) {
|
|
90
|
+
const candidate = pickNumber(data, paths);
|
|
91
|
+
if (candidate !== undefined && valid(candidate)) result = candidate;
|
|
92
|
+
}
|
|
93
|
+
return result;
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
export async function loadDefaultFetchMaxChars(cwd?: string): Promise<number | undefined> {
|
|
97
|
+
return loadFetchSetting(cwd, FETCH_MAX_CHARS_PATHS, (n) => n > 0);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export async function loadDefaultFetchTimeoutMs(cwd?: string): Promise<number | undefined> {
|
|
101
|
+
return loadFetchSetting(cwd, FETCH_TIMEOUT_PATHS, (n) => n >= 1000);
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export async function loadDefaultFetchSettings(cwd?: string): Promise<{ maxChars?: number; timeoutMs?: number }> {
|
|
105
|
+
let maxChars: number | undefined;
|
|
106
|
+
let timeoutMs: number | undefined;
|
|
107
|
+
for (const data of await settingsEntries(cwd)) {
|
|
108
|
+
const c = pickNumber(data, FETCH_MAX_CHARS_PATHS);
|
|
109
|
+
if (c !== undefined && c > 0) maxChars = c;
|
|
110
|
+
const t = pickNumber(data, FETCH_TIMEOUT_PATHS);
|
|
111
|
+
if (t !== undefined && t >= 1000) timeoutMs = t;
|
|
112
|
+
}
|
|
113
|
+
return { maxChars, timeoutMs };
|
|
114
|
+
}
|
package/web-access.ts
CHANGED
|
@@ -6,7 +6,13 @@ const MAX_CACHEABLE_DOMAIN_LEN = 253;
|
|
|
6
6
|
const SITE_FILTER_LIMIT = 8;
|
|
7
7
|
const DOTTED_HOST_RE = /^[A-Za-z0-9-]+(\.[A-Za-z0-9-]+)+$/;
|
|
8
8
|
const PORT_RE = /^[0-9]{1,5}$/;
|
|
9
|
+
export const MAX_SIGNAL_TIMEOUT_MS = 2 ** 31 - 1;
|
|
9
10
|
|
|
11
|
+
function throwInvalidDomain(value: unknown): never {
|
|
12
|
+
throw new Error(`Invalid website domain: ${String(value)}`);
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
const INVALID_HOST_REASON = "Blocked: the URL has an invalid hostname or port.";
|
|
10
16
|
export interface WebsitePolicy {
|
|
11
17
|
allowedDomains: string[];
|
|
12
18
|
blockedDomains: string[];
|
|
@@ -19,24 +25,30 @@ export function normalizeDomain(value: unknown): string {
|
|
|
19
25
|
[...domain].some((char) => char.charCodeAt(0) < 32) ||
|
|
20
26
|
["\\", "/", "@", "?", "#"].some((char) => domain.includes(char))
|
|
21
27
|
) {
|
|
22
|
-
|
|
28
|
+
throwInvalidDomain(value);
|
|
23
29
|
}
|
|
24
|
-
const
|
|
25
|
-
|
|
26
|
-
|
|
30
|
+
const startsBracket = domain.startsWith("[");
|
|
31
|
+
const endsBracket = domain.endsWith("]");
|
|
32
|
+
if (startsBracket !== endsBracket) {
|
|
33
|
+
throwInvalidDomain(value);
|
|
27
34
|
}
|
|
28
|
-
const stripped = (
|
|
35
|
+
const stripped = (startsBracket ? domain.slice(1, -1) : domain).replace(/\.+$/, "");
|
|
29
36
|
if (/^[0-9a-fA-F:.]+$/.test(stripped) && stripped.includes(":")) {
|
|
30
37
|
try {
|
|
31
38
|
return compressIpv6(stripped);
|
|
32
39
|
} catch {
|
|
33
|
-
|
|
40
|
+
throwInvalidDomain(value);
|
|
34
41
|
}
|
|
35
42
|
}
|
|
36
43
|
if (stripped.includes(":")) {
|
|
37
44
|
throw new Error("Website limits must contain domains without schemes or ports");
|
|
38
45
|
}
|
|
39
46
|
const numericParts = stripped.split(".");
|
|
47
|
+
const canonicalIpv4 =
|
|
48
|
+
numericParts.length === 4 &&
|
|
49
|
+
numericParts.every((part) => /^(?:0|[1-9][0-9]{0,2})$/.test(part)) &&
|
|
50
|
+
numericParts.every((part) => Number(part) <= 255);
|
|
51
|
+
if (canonicalIpv4) return stripped;
|
|
40
52
|
if (
|
|
41
53
|
numericParts.length <= 4 &&
|
|
42
54
|
numericParts.every((part) => /^(?:0x[0-9a-f]+|[0-9]+)$/.test(part))
|
|
@@ -47,13 +59,13 @@ export function normalizeDomain(value: unknown): string {
|
|
|
47
59
|
try {
|
|
48
60
|
asciiDomain = domainToASCII(stripped).toLowerCase();
|
|
49
61
|
} catch {
|
|
50
|
-
|
|
62
|
+
throwInvalidDomain(value);
|
|
51
63
|
}
|
|
52
64
|
if (
|
|
53
65
|
asciiDomain.length > 253 ||
|
|
54
66
|
!asciiDomain.split(".").every((label) => DOMAIN_LABEL_RE.test(label))
|
|
55
67
|
) {
|
|
56
|
-
|
|
68
|
+
throwInvalidDomain(value);
|
|
57
69
|
}
|
|
58
70
|
return asciiDomain;
|
|
59
71
|
}
|
|
@@ -189,7 +201,7 @@ export function checkUrlAccess(
|
|
|
189
201
|
try {
|
|
190
202
|
parsed = new URL(candidate);
|
|
191
203
|
} catch {
|
|
192
|
-
return [false,
|
|
204
|
+
return [false, INVALID_HOST_REASON, ""];
|
|
193
205
|
}
|
|
194
206
|
const scheme = parsed.protocol.replace(/:$/, "").toLowerCase();
|
|
195
207
|
if (scheme !== "http" && scheme !== "https") {
|
|
@@ -199,20 +211,16 @@ export function checkUrlAccess(
|
|
|
199
211
|
return [false, "Blocked: URLs with credentials or encoded hostnames are not allowed.", ""];
|
|
200
212
|
}
|
|
201
213
|
if (!parsed.hostname) {
|
|
202
|
-
return [false,
|
|
214
|
+
return [false, INVALID_HOST_REASON, ""];
|
|
203
215
|
}
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
207
|
-
}
|
|
208
|
-
} catch {
|
|
209
|
-
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
216
|
+
if (parsed.port && !(PORT_RE.test(parsed.port) && Number(parsed.port) >= 1 && Number(parsed.port) <= 65535)) {
|
|
217
|
+
return [false, INVALID_HOST_REASON, ""];
|
|
210
218
|
}
|
|
211
219
|
let hostname: string;
|
|
212
220
|
try {
|
|
213
221
|
hostname = normalizeDomain(parsed.hostname);
|
|
214
222
|
} catch {
|
|
215
|
-
return [false,
|
|
223
|
+
return [false, INVALID_HOST_REASON, ""];
|
|
216
224
|
}
|
|
217
225
|
if (!policyAllows(hostname, policy)) {
|
|
218
226
|
return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
|
|
@@ -318,7 +326,7 @@ const GITHUB_NON_OWNER_SEGMENTS = new Set([
|
|
|
318
326
|
|
|
319
327
|
const GITHUB_NAME_RE = /^[A-Za-z0-9_.\-]{1,100}$/;
|
|
320
328
|
|
|
321
|
-
function
|
|
329
|
+
function parseGithubRepo(url: string): { owner: string; repo: string; rest: string[] } | null {
|
|
322
330
|
let parsed: URL;
|
|
323
331
|
try {
|
|
324
332
|
parsed = new URL(url);
|
|
@@ -328,22 +336,40 @@ function githubRepoOwnerRepo(url: string): [string, string] | null {
|
|
|
328
336
|
const host = (parsed.hostname ?? "").toLowerCase();
|
|
329
337
|
if (host !== "github.com" && host !== "www.github.com") return null;
|
|
330
338
|
const parts = parsed.pathname.split("/").filter((part) => part.length > 0);
|
|
331
|
-
if (parts.length
|
|
332
|
-
const
|
|
339
|
+
if (parts.length < 2) return null;
|
|
340
|
+
const owner = parts[0];
|
|
341
|
+
const repoRaw = parts[1];
|
|
333
342
|
if (GITHUB_NON_OWNER_SEGMENTS.has(owner.toLowerCase())) return null;
|
|
334
|
-
const
|
|
335
|
-
if (!GITHUB_NAME_RE.test(owner) || !GITHUB_NAME_RE.test(
|
|
336
|
-
return
|
|
343
|
+
const repo = repoRaw.endsWith(".git") ? repoRaw.slice(0, -4) : repoRaw;
|
|
344
|
+
if (!GITHUB_NAME_RE.test(owner) || !GITHUB_NAME_RE.test(repo)) return null;
|
|
345
|
+
return { owner, repo, rest: parts.slice(2) };
|
|
346
|
+
}
|
|
347
|
+
function githubRepoRoot(url: string): { owner: string; repo: string; rest: string[] } | null {
|
|
348
|
+
const parsed = parseGithubRepo(url);
|
|
349
|
+
if (!parsed || parsed.rest.length !== 0) return null;
|
|
350
|
+
return parsed;
|
|
337
351
|
}
|
|
338
352
|
|
|
339
353
|
export function githubRepoReadmeApiUrl(url: string): string | null {
|
|
340
|
-
const
|
|
341
|
-
|
|
354
|
+
const parsed = githubRepoRoot(url);
|
|
355
|
+
if (!parsed) return null;
|
|
356
|
+
return `https://api.github.com/repos/${parsed.owner}/${parsed.repo}/readme`;
|
|
342
357
|
}
|
|
343
358
|
|
|
344
359
|
export function githubRepoRawReadmeUrl(url: string): string | null {
|
|
345
|
-
const
|
|
346
|
-
|
|
360
|
+
const parsed = githubRepoRoot(url);
|
|
361
|
+
if (!parsed) return null;
|
|
362
|
+
return `https://raw.githubusercontent.com/${parsed.owner}/${parsed.repo}/HEAD/README.md`;
|
|
363
|
+
}
|
|
364
|
+
export function githubRawContentUrl(url: string): string | null {
|
|
365
|
+
const parsed = parseGithubRepo(url);
|
|
366
|
+
if (!parsed || parsed.rest.length < 2) return null;
|
|
367
|
+
const [kind, ref] = parsed.rest;
|
|
368
|
+
if (kind !== "blob" && kind !== "raw") return null;
|
|
369
|
+
if (!ref) return null;
|
|
370
|
+
const filePath = parsed.rest.slice(2).join("/");
|
|
371
|
+
if (!filePath) return null;
|
|
372
|
+
return `https://raw.githubusercontent.com/${parsed.owner}/${parsed.repo}/${ref}/${filePath}`;
|
|
347
373
|
}
|
|
348
374
|
|
|
349
375
|
function ipv4Octets(ip: string): number[] | null {
|