pi-unsloth-webtools 0.2.5 → 0.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +57 -11
- package/engines.ts +118 -45
- package/html-to-md.ts +93 -66
- package/index.ts +106 -60
- package/package.json +1 -1
- package/pdf.ts +97 -11
- package/web-access.ts +73 -50
- package/web-fetch.ts +401 -95
- package/web-search.ts +5 -4
package/index.ts
CHANGED
|
@@ -1,9 +1,11 @@
|
|
|
1
|
-
import type
|
|
1
|
+
import { defineTool, type ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import { Type } from "typebox";
|
|
3
|
-
import { webSearch } from "./web-search.ts";
|
|
4
|
-
import { fetchPageText } from "./web-fetch.ts";
|
|
3
|
+
import { webSearch as defaultWebSearch } from "./web-search.ts";
|
|
4
|
+
import { DEFAULT_FETCH_TIMEOUT_MS, fetchPageText as defaultFetchPageText } from "./web-fetch.ts";
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
function positiveNumber(value: unknown): number | undefined {
|
|
7
|
+
return typeof value === "number" && value > 0 ? value : undefined;
|
|
8
|
+
}
|
|
7
9
|
|
|
8
10
|
const WebSearchParams = Type.Object({
|
|
9
11
|
query: Type.Optional(
|
|
@@ -15,6 +17,25 @@ const WebSearchParams = Type.Object({
|
|
|
15
17
|
"A URL to fetch full page content from (instead of searching). Use this to read a page found in search results.",
|
|
16
18
|
}),
|
|
17
19
|
),
|
|
20
|
+
maxChars: Type.Optional(
|
|
21
|
+
Type.Number({
|
|
22
|
+
description:
|
|
23
|
+
"Truncate the fetched page to this many characters (only used with the url parameter)",
|
|
24
|
+
}),
|
|
25
|
+
),
|
|
26
|
+
maxResults: Type.Optional(
|
|
27
|
+
Type.Number({
|
|
28
|
+
minimum: 1,
|
|
29
|
+
maximum: 20,
|
|
30
|
+
description: "Maximum number of search results to return (default: 5)",
|
|
31
|
+
}),
|
|
32
|
+
),
|
|
33
|
+
timeoutMs: Type.Optional(
|
|
34
|
+
Type.Number({
|
|
35
|
+
minimum: 1000,
|
|
36
|
+
description: "Overall timeout in milliseconds for the search or fetch",
|
|
37
|
+
}),
|
|
38
|
+
),
|
|
18
39
|
});
|
|
19
40
|
|
|
20
41
|
const WebFetchParams = Type.Object({
|
|
@@ -24,63 +45,88 @@ const WebFetchParams = Type.Object({
|
|
|
24
45
|
description: "Truncate the returned content to this many characters (default: no limit)",
|
|
25
46
|
}),
|
|
26
47
|
),
|
|
48
|
+
timeoutMs: Type.Optional(
|
|
49
|
+
Type.Number({
|
|
50
|
+
minimum: 1000,
|
|
51
|
+
description: "Overall timeout in milliseconds (default: 60000)",
|
|
52
|
+
}),
|
|
53
|
+
),
|
|
27
54
|
});
|
|
28
55
|
|
|
29
|
-
export
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
],
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
56
|
+
export interface WebToolsDeps {
|
|
57
|
+
fetchPageText?: typeof defaultFetchPageText;
|
|
58
|
+
webSearch?: typeof defaultWebSearch;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function createWebTools(deps: WebToolsDeps = {}) {
|
|
62
|
+
const fetchPageText = deps.fetchPageText ?? defaultFetchPageText;
|
|
63
|
+
const webSearch = deps.webSearch ?? defaultWebSearch;
|
|
64
|
+
return {
|
|
65
|
+
webSearchTool: defineTool({
|
|
66
|
+
name: "web_search",
|
|
67
|
+
label: "Web Search",
|
|
68
|
+
description:
|
|
69
|
+
"Search the web and fetch page content. Returns snippets for all results. " +
|
|
70
|
+
"Use the url parameter to fetch full page text from a specific URL.",
|
|
71
|
+
promptSnippet: "Search the web and fetch page content",
|
|
72
|
+
promptGuidelines: [
|
|
73
|
+
'Use web_search with the url parameter (e.g. {"url": "<URL>"}) to read the full text of a page found in search results.',
|
|
74
|
+
],
|
|
75
|
+
parameters: WebSearchParams,
|
|
76
|
+
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
77
|
+
if (params.url?.trim()) {
|
|
78
|
+
const url = params.url.trim();
|
|
79
|
+
onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
|
|
80
|
+
return {
|
|
81
|
+
content: [
|
|
82
|
+
{
|
|
83
|
+
type: "text",
|
|
84
|
+
text: await fetchPageText(url, {
|
|
85
|
+
timeoutMs: positiveNumber(params.timeoutMs) ?? DEFAULT_FETCH_TIMEOUT_MS,
|
|
86
|
+
signal: signal ?? undefined,
|
|
87
|
+
maxChars: positiveNumber(params.maxChars),
|
|
88
|
+
}),
|
|
89
|
+
},
|
|
90
|
+
],
|
|
91
|
+
details: {},
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
|
|
95
|
+
const text = await webSearch(params.query, {
|
|
96
|
+
signal: signal ?? undefined,
|
|
97
|
+
timeoutMs: positiveNumber(params.timeoutMs),
|
|
98
|
+
maxResults: positiveNumber(params.maxResults),
|
|
99
|
+
});
|
|
100
|
+
return { content: [{ type: "text", text }], details: {} };
|
|
101
|
+
},
|
|
102
|
+
}),
|
|
103
|
+
webFetchTool: defineTool({
|
|
104
|
+
name: "web_fetch",
|
|
105
|
+
label: "Web Fetch",
|
|
106
|
+
description:
|
|
107
|
+
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
108
|
+
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
109
|
+
"stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
110
|
+
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
111
|
+
"Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
|
|
112
|
+
"is capped.",
|
|
113
|
+
promptSnippet: "Fetch a web page and return readable text content",
|
|
114
|
+
parameters: WebFetchParams,
|
|
115
|
+
async execute(_toolCallId, params, signal, onUpdate, _ctx) {
|
|
116
|
+
onUpdate?.({ content: [{ type: "text", text: `Fetching ${params.url}...` }], details: {} });
|
|
117
|
+
const text = await fetchPageText(params.url, {
|
|
118
|
+
timeoutMs: positiveNumber(params.timeoutMs) ?? DEFAULT_FETCH_TIMEOUT_MS,
|
|
119
|
+
signal: signal ?? undefined,
|
|
120
|
+
maxChars: positiveNumber(params.maxChars),
|
|
121
|
+
});
|
|
122
|
+
return { content: [{ type: "text", text }], details: {} };
|
|
123
|
+
},
|
|
124
|
+
}),
|
|
125
|
+
};
|
|
126
|
+
}
|
|
60
127
|
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
"Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
|
|
66
|
-
"main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
|
|
67
|
-
"stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
|
|
68
|
-
"README API, so the README is returned instead of the repo page's UI chrome. " +
|
|
69
|
-
"Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
|
|
70
|
-
"is capped.",
|
|
71
|
-
promptSnippet: "Fetch a web page and return readable text content",
|
|
72
|
-
parameters: WebFetchParams,
|
|
73
|
-
async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
|
|
74
|
-
const maxChars =
|
|
75
|
-
typeof params.maxChars === "number" && params.maxChars > 0
|
|
76
|
-
? params.maxChars
|
|
77
|
-
: undefined;
|
|
78
|
-
const text = await fetchPageText(params.url, {
|
|
79
|
-
timeoutMs: FETCH_TIMEOUT_MS,
|
|
80
|
-
signal: signal ?? undefined,
|
|
81
|
-
maxChars,
|
|
82
|
-
});
|
|
83
|
-
return { content: [{ type: "text", text }], details: {} };
|
|
84
|
-
},
|
|
85
|
-
});
|
|
128
|
+
export default function (pi: ExtensionAPI) {
|
|
129
|
+
const { webSearchTool, webFetchTool } = createWebTools();
|
|
130
|
+
pi.registerTool(webSearchTool);
|
|
131
|
+
pi.registerTool(webFetchTool);
|
|
86
132
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-unsloth-webtools",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.1",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
|
|
6
6
|
"main": "index.ts",
|
package/pdf.ts
CHANGED
|
@@ -26,6 +26,7 @@ interface MupdfDocument {
|
|
|
26
26
|
|
|
27
27
|
interface MupdfPage {
|
|
28
28
|
toStructuredText(options?: string): MupdfStructuredText;
|
|
29
|
+
getBounds(): [number, number, number, number];
|
|
29
30
|
getLinks(): { getBounds(): [number, number, number, number]; getURI(): string }[];
|
|
30
31
|
destroy(): void;
|
|
31
32
|
}
|
|
@@ -92,7 +93,6 @@ interface LinkInfo {
|
|
|
92
93
|
interface TableBand {
|
|
93
94
|
markdown: string;
|
|
94
95
|
firstLineIndex: number;
|
|
95
|
-
lineIndexes: Set<number>;
|
|
96
96
|
}
|
|
97
97
|
|
|
98
98
|
interface HeaderInfo {
|
|
@@ -137,10 +137,13 @@ function markdownCorrupted(text: string): boolean {
|
|
|
137
137
|
return shaped > threshold || (text.match(/\ufffd/g) ?? []).length > threshold;
|
|
138
138
|
}
|
|
139
139
|
|
|
140
|
-
function
|
|
141
|
-
|
|
140
|
+
function countLetters(text: string): number {
|
|
141
|
+
return [...text].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
function markdownIncomplete(markdown: string, plainLetters: number): boolean {
|
|
142
145
|
if (plainLetters < PDF_INCOMPLETE_MIN_LETTERS) return false;
|
|
143
|
-
const markdownLetters =
|
|
146
|
+
const markdownLetters = countLetters(markdown);
|
|
144
147
|
return markdownLetters < PDF_INCOMPLETE_RATIO * plainLetters;
|
|
145
148
|
}
|
|
146
149
|
|
|
@@ -256,6 +259,84 @@ function getRawLines(json: JsonBlock[]): MergedLine[] {
|
|
|
256
259
|
return nlines;
|
|
257
260
|
}
|
|
258
261
|
|
|
262
|
+
const RUNNING_EDGE_FRACTION = 0.12;
|
|
263
|
+
const RUNNING_MIN_PAGES = 2;
|
|
264
|
+
const RUNNING_MIN_FRACTION = 0.5;
|
|
265
|
+
const RUNNING_POSITION_TOLERANCE = 5;
|
|
266
|
+
const PAGE_NUMBER_RE = /^\d{1,4}$/;
|
|
267
|
+
|
|
268
|
+
function lineText(line: MergedLine): string {
|
|
269
|
+
return line.spans.map((s) => s.text).join(" ").trim();
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
function lineAtEdge(line: MergedLine, pageHeight: number): boolean {
|
|
273
|
+
return (
|
|
274
|
+
line.lrect.y0 < pageHeight * RUNNING_EDGE_FRACTION ||
|
|
275
|
+
line.lrect.y1 > pageHeight * (1 - RUNNING_EDGE_FRACTION)
|
|
276
|
+
);
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
function linePositionKey(y0: number): number {
|
|
280
|
+
return Math.round(y0 / RUNNING_POSITION_TOLERANCE);
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
function countRunningLines(
|
|
284
|
+
pages: PageData[],
|
|
285
|
+
heights: number[],
|
|
286
|
+
): { textCounts: Map<string, number>; numericPositionCounts: Map<number, number> } {
|
|
287
|
+
const textCounts = new Map<string, number>();
|
|
288
|
+
const numericPositionCounts = new Map<number, number>();
|
|
289
|
+
for (let i = 0; i < pages.length; i++) {
|
|
290
|
+
const height = heights[i];
|
|
291
|
+
const seenTexts = new Set<string>();
|
|
292
|
+
const seenPositions = new Set<number>();
|
|
293
|
+
for (const line of pages[i].lines) {
|
|
294
|
+
if (!lineAtEdge(line, height)) continue;
|
|
295
|
+
const text = lineText(line);
|
|
296
|
+
if (!text) continue;
|
|
297
|
+
const position = linePositionKey(line.lrect.y0);
|
|
298
|
+
const key = `${text}@${position}`;
|
|
299
|
+
if (!seenTexts.has(key)) {
|
|
300
|
+
seenTexts.add(key);
|
|
301
|
+
textCounts.set(key, (textCounts.get(key) ?? 0) + 1);
|
|
302
|
+
}
|
|
303
|
+
if (!seenPositions.has(position)) {
|
|
304
|
+
seenPositions.add(position);
|
|
305
|
+
if (PAGE_NUMBER_RE.test(text)) {
|
|
306
|
+
numericPositionCounts.set(position, (numericPositionCounts.get(position) ?? 0) + 1);
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
return { textCounts, numericPositionCounts };
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
function stripRunningLines(pages: PageData[], heights: number[]): number[] {
|
|
315
|
+
const { textCounts, numericPositionCounts } = countRunningLines(pages, heights);
|
|
316
|
+
const threshold = Math.max(RUNNING_MIN_PAGES, Math.ceil(pages.length * RUNNING_MIN_FRACTION));
|
|
317
|
+
const removedLetters: number[] = [];
|
|
318
|
+
for (let i = 0; i < pages.length; i++) {
|
|
319
|
+
const height = heights[i];
|
|
320
|
+
let removed = 0;
|
|
321
|
+
pages[i].lines = pages[i].lines.filter((line) => {
|
|
322
|
+
if (!lineAtEdge(line, height)) return true;
|
|
323
|
+
const text = lineText(line);
|
|
324
|
+
if (!text) return true;
|
|
325
|
+
const position = linePositionKey(line.lrect.y0);
|
|
326
|
+
const repeated = (textCounts.get(`${text}@${position}`) ?? 0) >= threshold;
|
|
327
|
+
const pageNumber =
|
|
328
|
+
PAGE_NUMBER_RE.test(text) && (numericPositionCounts.get(position) ?? 0) >= threshold;
|
|
329
|
+
if (repeated || pageNumber) {
|
|
330
|
+
removed += countLetters(text);
|
|
331
|
+
return false;
|
|
332
|
+
}
|
|
333
|
+
return true;
|
|
334
|
+
});
|
|
335
|
+
removedLetters.push(removed);
|
|
336
|
+
}
|
|
337
|
+
return removedLetters;
|
|
338
|
+
}
|
|
339
|
+
|
|
259
340
|
function findLink(links: LinkInfo[], span: SpanData): string | null {
|
|
260
341
|
const midX = (span.x0 + span.x1) / 2;
|
|
261
342
|
const midY = (span.y0 + span.y1) / 2;
|
|
@@ -317,8 +398,7 @@ function detectTableBands(lines: MergedLine[]): TableBand[] {
|
|
|
317
398
|
for (const row of rows.slice(1)) {
|
|
318
399
|
output += "|" + row.join("|") + "|\n";
|
|
319
400
|
}
|
|
320
|
-
|
|
321
|
-
bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...indexes), lineIndexes: indexes });
|
|
401
|
+
bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...band.map((l) => lines.indexOf(l))) });
|
|
322
402
|
}
|
|
323
403
|
}
|
|
324
404
|
}
|
|
@@ -436,10 +516,6 @@ function writeText(
|
|
|
436
516
|
}
|
|
437
517
|
out += "\n";
|
|
438
518
|
}
|
|
439
|
-
while (emittedBands < tableBands.length) {
|
|
440
|
-
out += "\n" + tableBands[emittedBands].markdown;
|
|
441
|
-
emittedBands++;
|
|
442
|
-
}
|
|
443
519
|
out += "\n";
|
|
444
520
|
if (code) out += "```\n";
|
|
445
521
|
out += "\n\n";
|
|
@@ -473,8 +549,11 @@ export async function extractPdfPages(
|
|
|
473
549
|
const total = doc.countPages();
|
|
474
550
|
const count = Math.min(total, MAX_WEB_PDF_PAGES);
|
|
475
551
|
const pages: PageData[] = [];
|
|
552
|
+
const heights: number[] = [];
|
|
476
553
|
for (let i = 0; i < count; i++) {
|
|
477
554
|
const page = doc.loadPage(i);
|
|
555
|
+
const bounds = page.getBounds();
|
|
556
|
+
heights.push(bounds[3] - bounds[1]);
|
|
478
557
|
let st: MupdfStructuredText | null = null;
|
|
479
558
|
try {
|
|
480
559
|
st = page.toStructuredText("");
|
|
@@ -496,10 +575,12 @@ export async function extractPdfPages(
|
|
|
496
575
|
page.destroy();
|
|
497
576
|
}
|
|
498
577
|
}
|
|
578
|
+
const removedLetters = stripRunningLines(pages, heights);
|
|
499
579
|
const info = identifyHeaders(pages.map((p) => p.lines));
|
|
500
580
|
const extracted = pages.map((p, i) => {
|
|
501
581
|
const markdown = renderPageMarkdown(p.lines, info, p.links);
|
|
502
|
-
const
|
|
582
|
+
const plainLetters = Math.max(0, countLetters(p.plain) - removedLetters[i]);
|
|
583
|
+
const text = markdownCorrupted(markdown) || markdownIncomplete(markdown, plainLetters) ? p.plain : markdown;
|
|
503
584
|
return { text, pageNumber: i + 1 };
|
|
504
585
|
});
|
|
505
586
|
return { pages: extracted, totalPages: total };
|
|
@@ -616,6 +697,11 @@ function extractTextOps(bytes: Buffer): string {
|
|
|
616
697
|
}
|
|
617
698
|
}
|
|
618
699
|
}
|
|
700
|
+
if (c === "'" || c === '"') {
|
|
701
|
+
if (inText) out.push("\n");
|
|
702
|
+
i++;
|
|
703
|
+
continue;
|
|
704
|
+
}
|
|
619
705
|
const token = text.slice(i, i + 2);
|
|
620
706
|
if (token === "BT") {
|
|
621
707
|
inText = true;
|
package/web-access.ts
CHANGED
|
@@ -37,6 +37,11 @@ export function normalizeDomain(value: unknown): string {
|
|
|
37
37
|
throw new Error("Website limits must contain domains without schemes or ports");
|
|
38
38
|
}
|
|
39
39
|
const numericParts = stripped.split(".");
|
|
40
|
+
const canonicalIpv4 =
|
|
41
|
+
numericParts.length === 4 &&
|
|
42
|
+
numericParts.every((part) => /^(?:0|[1-9][0-9]{0,2})$/.test(part)) &&
|
|
43
|
+
numericParts.every((part) => Number(part) <= 255);
|
|
44
|
+
if (canonicalIpv4) return stripped;
|
|
40
45
|
if (
|
|
41
46
|
numericParts.length <= 4 &&
|
|
42
47
|
numericParts.every((part) => /^(?:0x[0-9a-f]+|[0-9]+)$/.test(part))
|
|
@@ -77,11 +82,12 @@ function compressIpv6(ip: string): string {
|
|
|
77
82
|
}
|
|
78
83
|
|
|
79
84
|
function compressGroups(groups: string[]): string {
|
|
85
|
+
const normalized = groups.map((g) => (/^0+$/.test(g) ? "0" : g.replace(/^0+(?=[0-9a-f])/, "")));
|
|
80
86
|
let bestStart = -1;
|
|
81
87
|
let bestLen = 0;
|
|
82
88
|
let runStart = -1;
|
|
83
|
-
for (let i = 0; i <=
|
|
84
|
-
if (i <
|
|
89
|
+
for (let i = 0; i <= normalized.length; i++) {
|
|
90
|
+
if (i < normalized.length && normalized[i] === "0") {
|
|
85
91
|
if (runStart === -1) runStart = i;
|
|
86
92
|
} else if (runStart !== -1) {
|
|
87
93
|
const runLen = i - runStart;
|
|
@@ -92,10 +98,20 @@ function compressGroups(groups: string[]): string {
|
|
|
92
98
|
runStart = -1;
|
|
93
99
|
}
|
|
94
100
|
}
|
|
95
|
-
if (bestLen < 2) return
|
|
96
|
-
const head =
|
|
97
|
-
const tail =
|
|
98
|
-
return
|
|
101
|
+
if (bestLen < 2) return normalized.join(":");
|
|
102
|
+
const head = normalized.slice(0, bestStart);
|
|
103
|
+
const tail = normalized.slice(bestStart + bestLen);
|
|
104
|
+
return `${head.join(":")}::${tail.join(":")}`;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
const PCP_ANYCAST = new Set(["2001:1::1", "2001:1::2"]);
|
|
108
|
+
|
|
109
|
+
function isPcpAnycast(lower: string): boolean {
|
|
110
|
+
try {
|
|
111
|
+
return PCP_ANYCAST.has(compressIpv6(lower));
|
|
112
|
+
} catch {
|
|
113
|
+
return false;
|
|
114
|
+
}
|
|
99
115
|
}
|
|
100
116
|
|
|
101
117
|
export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
|
|
@@ -119,15 +135,7 @@ export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
|
|
|
119
135
|
if (rawDomains.length > MAX_DOMAINS_PER_LIST) {
|
|
120
136
|
throw new Error(`${key} supports at most ${MAX_DOMAINS_PER_LIST} domains`);
|
|
121
137
|
}
|
|
122
|
-
|
|
123
|
-
for (const rawDomain of rawDomains) {
|
|
124
|
-
if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
|
|
125
|
-
throw new Error(`${key} must contain only strings`);
|
|
126
|
-
}
|
|
127
|
-
const domain = normalizeDomain(rawDomain);
|
|
128
|
-
if (!domains.includes(domain)) domains.push(domain);
|
|
129
|
-
}
|
|
130
|
-
normalized[key] = domains;
|
|
138
|
+
normalized[key] = normalizeDomainList(rawDomains, key);
|
|
131
139
|
}
|
|
132
140
|
return normalized;
|
|
133
141
|
}
|
|
@@ -136,12 +144,22 @@ function matchesDomain(hostname: string, domain: string): boolean {
|
|
|
136
144
|
return hostname === domain || hostname.endsWith(`.${domain}`);
|
|
137
145
|
}
|
|
138
146
|
|
|
139
|
-
|
|
140
|
-
|
|
147
|
+
function normalizeDomainList(domains: unknown[], listName: string): string[] {
|
|
148
|
+
const out: string[] = [];
|
|
149
|
+
for (const rawDomain of domains) {
|
|
150
|
+
if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
|
|
151
|
+
throw new Error(`${listName} must contain only strings`);
|
|
152
|
+
}
|
|
153
|
+
const domain = normalizeDomain(rawDomain);
|
|
154
|
+
if (!out.includes(domain)) out.push(domain);
|
|
155
|
+
}
|
|
156
|
+
return out;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
function policyAllows(host: string, policy: WebsitePolicy | null): boolean {
|
|
141
160
|
let normalized: WebsitePolicy;
|
|
142
161
|
try {
|
|
143
|
-
|
|
144
|
-
normalized = normalizePolicyMaybe(policy);
|
|
162
|
+
normalized = normalizeWebsitePolicy(policy);
|
|
145
163
|
} catch {
|
|
146
164
|
return false;
|
|
147
165
|
}
|
|
@@ -150,22 +168,12 @@ export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null):
|
|
|
150
168
|
return allowed.length === 0 || allowed.some((domain) => matchesDomain(host, domain));
|
|
151
169
|
}
|
|
152
170
|
|
|
153
|
-
function
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
const domain = normalizeDomain(rawDomain);
|
|
159
|
-
if (!domains.includes(domain)) domains.push(domain);
|
|
160
|
-
}
|
|
161
|
-
normalized[key] = domains;
|
|
171
|
+
export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null): boolean {
|
|
172
|
+
try {
|
|
173
|
+
return policyAllows(normalizeDomain(hostname), policy);
|
|
174
|
+
} catch {
|
|
175
|
+
return false;
|
|
162
176
|
}
|
|
163
|
-
return normalized;
|
|
164
|
-
}
|
|
165
|
-
|
|
166
|
-
function normalizePolicyMaybe(policy: WebsitePolicy | null | undefined): WebsitePolicy {
|
|
167
|
-
if (!policy) return { allowedDomains: [], blockedDomains: [] };
|
|
168
|
-
return normalizePolicyObject(policy);
|
|
169
177
|
}
|
|
170
178
|
|
|
171
179
|
export function checkUrlAccess(
|
|
@@ -211,7 +219,7 @@ export function checkUrlAccess(
|
|
|
211
219
|
} catch {
|
|
212
220
|
return [false, "Blocked: the URL has an invalid hostname or port.", ""];
|
|
213
221
|
}
|
|
214
|
-
if (!
|
|
222
|
+
if (!policyAllows(hostname, policy)) {
|
|
215
223
|
return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
|
|
216
224
|
}
|
|
217
225
|
return [true, "", hostname];
|
|
@@ -315,7 +323,7 @@ const GITHUB_NON_OWNER_SEGMENTS = new Set([
|
|
|
315
323
|
|
|
316
324
|
const GITHUB_NAME_RE = /^[A-Za-z0-9_.\-]{1,100}$/;
|
|
317
325
|
|
|
318
|
-
|
|
326
|
+
function githubRepoOwnerRepo(url: string): [string, string] | null {
|
|
319
327
|
let parsed: URL;
|
|
320
328
|
try {
|
|
321
329
|
parsed = new URL(url);
|
|
@@ -330,7 +338,17 @@ export function githubRepoReadmeApiUrl(url: string): string | null {
|
|
|
330
338
|
if (GITHUB_NON_OWNER_SEGMENTS.has(owner.toLowerCase())) return null;
|
|
331
339
|
const cleanRepo = repo.endsWith(".git") ? repo.slice(0, -4) : repo;
|
|
332
340
|
if (!GITHUB_NAME_RE.test(owner) || !GITHUB_NAME_RE.test(cleanRepo)) return null;
|
|
333
|
-
return
|
|
341
|
+
return [owner, cleanRepo];
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
export function githubRepoReadmeApiUrl(url: string): string | null {
|
|
345
|
+
const pair = githubRepoOwnerRepo(url);
|
|
346
|
+
return pair ? `https://api.github.com/repos/${pair[0]}/${pair[1]}/readme` : null;
|
|
347
|
+
}
|
|
348
|
+
|
|
349
|
+
export function githubRepoRawReadmeUrl(url: string): string | null {
|
|
350
|
+
const pair = githubRepoOwnerRepo(url);
|
|
351
|
+
return pair ? `https://raw.githubusercontent.com/${pair[0]}/${pair[1]}/HEAD/README.md` : null;
|
|
334
352
|
}
|
|
335
353
|
|
|
336
354
|
function ipv4Octets(ip: string): number[] | null {
|
|
@@ -353,6 +371,7 @@ export function isPublicIp(ip: string): boolean {
|
|
|
353
371
|
if (o[0] === 172 && o[1] >= 16 && o[1] <= 31) return false;
|
|
354
372
|
if (o[0] === 192 && o[1] === 0 && o[2] === 0) return false;
|
|
355
373
|
if (o[0] === 192 && o[1] === 0 && o[2] === 2) return false;
|
|
374
|
+
if (o[0] === 192 && o[1] === 88 && o[2] === 99) return false;
|
|
356
375
|
if (o[0] === 192 && o[1] === 168) return false;
|
|
357
376
|
if (o[0] === 198 && (o[1] === 18 || o[1] === 19)) return false;
|
|
358
377
|
if (o[0] === 198 && o[1] === 51 && o[2] === 100) return false;
|
|
@@ -360,18 +379,22 @@ export function isPublicIp(ip: string): boolean {
|
|
|
360
379
|
if (o[0] >= 224) return false;
|
|
361
380
|
return true;
|
|
362
381
|
}
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
if (
|
|
370
|
-
if (
|
|
371
|
-
if (
|
|
372
|
-
if (
|
|
373
|
-
|
|
374
|
-
if (
|
|
375
|
-
if (
|
|
382
|
+
let canonical: string;
|
|
383
|
+
try {
|
|
384
|
+
canonical = compressIpv6(ip);
|
|
385
|
+
} catch {
|
|
386
|
+
return false;
|
|
387
|
+
}
|
|
388
|
+
if (canonical === "::1" || canonical.startsWith("::")) return false;
|
|
389
|
+
if (canonical.startsWith("fc") || canonical.startsWith("fd")) return false;
|
|
390
|
+
if (/^fe[89ab][0-9a-f]:/.test(canonical)) return false;
|
|
391
|
+
if (canonical.startsWith("ff")) return false;
|
|
392
|
+
if (canonical.startsWith("2001:db8")) return false;
|
|
393
|
+
if (canonical.startsWith("64:ff9b:")) return false;
|
|
394
|
+
if (canonical.startsWith("2001:10:")) return false;
|
|
395
|
+
if (canonical.startsWith("2002:")) return false;
|
|
396
|
+
if (canonical.startsWith("2001:0:") || canonical.startsWith("2001::")) return false;
|
|
397
|
+
if (canonical.startsWith("2001:2:")) return false;
|
|
398
|
+
if (isPcpAnycast(canonical)) return false;
|
|
376
399
|
return true;
|
|
377
400
|
}
|