pi-unsloth-webtools 0.2.5 → 0.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.ts CHANGED
@@ -1,9 +1,11 @@
1
- import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
1
+ import { defineTool, type ExtensionAPI } from "@earendil-works/pi-coding-agent";
2
2
  import { Type } from "typebox";
3
- import { webSearch } from "./web-search.ts";
4
- import { fetchPageText } from "./web-fetch.ts";
3
+ import { webSearch as defaultWebSearch } from "./web-search.ts";
4
+ import { DEFAULT_FETCH_TIMEOUT_MS, fetchPageText as defaultFetchPageText } from "./web-fetch.ts";
5
5
 
6
- const FETCH_TIMEOUT_MS = 60_000;
6
+ function positiveNumber(value: unknown): number | undefined {
7
+ return typeof value === "number" && value > 0 ? value : undefined;
8
+ }
7
9
 
8
10
  const WebSearchParams = Type.Object({
9
11
  query: Type.Optional(
@@ -15,6 +17,25 @@ const WebSearchParams = Type.Object({
15
17
  "A URL to fetch full page content from (instead of searching). Use this to read a page found in search results.",
16
18
  }),
17
19
  ),
20
+ maxChars: Type.Optional(
21
+ Type.Number({
22
+ description:
23
+ "Truncate the fetched page to this many characters (only used with the url parameter)",
24
+ }),
25
+ ),
26
+ maxResults: Type.Optional(
27
+ Type.Number({
28
+ minimum: 1,
29
+ maximum: 20,
30
+ description: "Maximum number of search results to return (default: 5)",
31
+ }),
32
+ ),
33
+ timeoutMs: Type.Optional(
34
+ Type.Number({
35
+ minimum: 1000,
36
+ description: "Overall timeout in milliseconds for the search or fetch",
37
+ }),
38
+ ),
18
39
  });
19
40
 
20
41
  const WebFetchParams = Type.Object({
@@ -24,63 +45,88 @@ const WebFetchParams = Type.Object({
24
45
  description: "Truncate the returned content to this many characters (default: no limit)",
25
46
  }),
26
47
  ),
48
+ timeoutMs: Type.Optional(
49
+ Type.Number({
50
+ minimum: 1000,
51
+ description: "Overall timeout in milliseconds (default: 60000)",
52
+ }),
53
+ ),
27
54
  });
28
55
 
29
- export default function (pi: ExtensionAPI) {
30
- pi.registerTool({
31
- name: "web_search",
32
- label: "Web Search",
33
- description:
34
- "Search the web and fetch page content. Returns snippets for all results. " +
35
- "Use the url parameter to fetch full page text from a specific URL.",
36
- promptSnippet: "Search the web and fetch page content",
37
- promptGuidelines: [
38
- "Use web_search with the url parameter (e.g. {\"url\": \"<URL>\"}) to read the full text of a page found in search results.",
39
- ],
40
- parameters: WebSearchParams,
41
- async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
42
- if (params.url?.trim()) {
43
- return {
44
- content: [
45
- {
46
- type: "text",
47
- text: await fetchPageText(params.url.trim(), {
48
- timeoutMs: FETCH_TIMEOUT_MS,
49
- signal: signal ?? undefined,
50
- }),
51
- },
52
- ],
53
- details: {},
54
- };
55
- }
56
- const text = await webSearch(params.query, { signal: signal ?? undefined });
57
- return { content: [{ type: "text", text }], details: {} };
58
- },
59
- });
56
+ export interface WebToolsDeps {
57
+ fetchPageText?: typeof defaultFetchPageText;
58
+ webSearch?: typeof defaultWebSearch;
59
+ }
60
+
61
+ export function createWebTools(deps: WebToolsDeps = {}) {
62
+ const fetchPageText = deps.fetchPageText ?? defaultFetchPageText;
63
+ const webSearch = deps.webSearch ?? defaultWebSearch;
64
+ return {
65
+ webSearchTool: defineTool({
66
+ name: "web_search",
67
+ label: "Web Search",
68
+ description:
69
+ "Search the web and fetch page content. Returns snippets for all results. " +
70
+ "Use the url parameter to fetch full page text from a specific URL.",
71
+ promptSnippet: "Search the web and fetch page content",
72
+ promptGuidelines: [
73
+ 'Use web_search with the url parameter (e.g. {"url": "<URL>"}) to read the full text of a page found in search results.',
74
+ ],
75
+ parameters: WebSearchParams,
76
+ async execute(_toolCallId, params, signal, onUpdate, _ctx) {
77
+ if (params.url?.trim()) {
78
+ const url = params.url.trim();
79
+ onUpdate?.({ content: [{ type: "text", text: `Fetching ${url}...` }], details: {} });
80
+ return {
81
+ content: [
82
+ {
83
+ type: "text",
84
+ text: await fetchPageText(url, {
85
+ timeoutMs: positiveNumber(params.timeoutMs) ?? DEFAULT_FETCH_TIMEOUT_MS,
86
+ signal: signal ?? undefined,
87
+ maxChars: positiveNumber(params.maxChars),
88
+ }),
89
+ },
90
+ ],
91
+ details: {},
92
+ };
93
+ }
94
+ onUpdate?.({ content: [{ type: "text", text: "Searching the web..." }], details: {} });
95
+ const text = await webSearch(params.query, {
96
+ signal: signal ?? undefined,
97
+ timeoutMs: positiveNumber(params.timeoutMs),
98
+ maxResults: positiveNumber(params.maxResults),
99
+ });
100
+ return { content: [{ type: "text", text }], details: {} };
101
+ },
102
+ }),
103
+ webFetchTool: defineTool({
104
+ name: "web_fetch",
105
+ label: "Web Fetch",
106
+ description:
107
+ "Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
108
+ "main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
109
+ "stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
110
+ "README API, so the README is returned instead of the repo page's UI chrome. " +
111
+ "Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
112
+ "is capped.",
113
+ promptSnippet: "Fetch a web page and return readable text content",
114
+ parameters: WebFetchParams,
115
+ async execute(_toolCallId, params, signal, onUpdate, _ctx) {
116
+ onUpdate?.({ content: [{ type: "text", text: `Fetching ${params.url}...` }], details: {} });
117
+ const text = await fetchPageText(params.url, {
118
+ timeoutMs: positiveNumber(params.timeoutMs) ?? DEFAULT_FETCH_TIMEOUT_MS,
119
+ signal: signal ?? undefined,
120
+ maxChars: positiveNumber(params.maxChars),
121
+ });
122
+ return { content: [{ type: "text", text }], details: {} };
123
+ },
124
+ }),
125
+ };
126
+ }
60
127
 
61
- pi.registerTool({
62
- name: "web_fetch",
63
- label: "Web Fetch",
64
- description:
65
- "Fetch a URL and return its readable text. HTML pages are converted to Markdown using a " +
66
- "main-content heuristic: article/main scoping plus hidden-element and boilerplate " +
67
- "stripping. Non-HTML text is returned as-is. GitHub repo root pages are rewritten to the " +
68
- "README API, so the README is returned instead of the repo page's UI chrome. " +
69
- "Private/loopback/link-local targets are blocked (SSRF protection), and the download size " +
70
- "is capped.",
71
- promptSnippet: "Fetch a web page and return readable text content",
72
- parameters: WebFetchParams,
73
- async execute(_toolCallId, params, signal, _onUpdate, _ctx) {
74
- const maxChars =
75
- typeof params.maxChars === "number" && params.maxChars > 0
76
- ? params.maxChars
77
- : undefined;
78
- const text = await fetchPageText(params.url, {
79
- timeoutMs: FETCH_TIMEOUT_MS,
80
- signal: signal ?? undefined,
81
- maxChars,
82
- });
83
- return { content: [{ type: "text", text }], details: {} };
84
- },
85
- });
128
+ export default function (pi: ExtensionAPI) {
129
+ const { webSearchTool, webFetchTool } = createWebTools();
130
+ pi.registerTool(webSearchTool);
131
+ pi.registerTool(webFetchTool);
86
132
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-unsloth-webtools",
3
- "version": "0.2.5",
3
+ "version": "0.3.1",
4
4
  "type": "module",
5
5
  "description": "Pi extension: web_search and web_fetch tools ported from the Unsloth Studio codebase (DuckDuckGo search, SSRF-safe direct fetching, HTML-to-Markdown extraction)",
6
6
  "main": "index.ts",
package/pdf.ts CHANGED
@@ -26,6 +26,7 @@ interface MupdfDocument {
26
26
 
27
27
  interface MupdfPage {
28
28
  toStructuredText(options?: string): MupdfStructuredText;
29
+ getBounds(): [number, number, number, number];
29
30
  getLinks(): { getBounds(): [number, number, number, number]; getURI(): string }[];
30
31
  destroy(): void;
31
32
  }
@@ -92,7 +93,6 @@ interface LinkInfo {
92
93
  interface TableBand {
93
94
  markdown: string;
94
95
  firstLineIndex: number;
95
- lineIndexes: Set<number>;
96
96
  }
97
97
 
98
98
  interface HeaderInfo {
@@ -137,10 +137,13 @@ function markdownCorrupted(text: string): boolean {
137
137
  return shaped > threshold || (text.match(/\ufffd/g) ?? []).length > threshold;
138
138
  }
139
139
 
140
- function markdownIncomplete(markdown: string, plain: string): boolean {
141
- const plainLetters = [...plain].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
140
+ function countLetters(text: string): number {
141
+ return [...text].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
142
+ }
143
+
144
+ function markdownIncomplete(markdown: string, plainLetters: number): boolean {
142
145
  if (plainLetters < PDF_INCOMPLETE_MIN_LETTERS) return false;
143
- const markdownLetters = [...markdown].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
146
+ const markdownLetters = countLetters(markdown);
144
147
  return markdownLetters < PDF_INCOMPLETE_RATIO * plainLetters;
145
148
  }
146
149
 
@@ -256,6 +259,84 @@ function getRawLines(json: JsonBlock[]): MergedLine[] {
256
259
  return nlines;
257
260
  }
258
261
 
262
+ const RUNNING_EDGE_FRACTION = 0.12;
263
+ const RUNNING_MIN_PAGES = 2;
264
+ const RUNNING_MIN_FRACTION = 0.5;
265
+ const RUNNING_POSITION_TOLERANCE = 5;
266
+ const PAGE_NUMBER_RE = /^\d{1,4}$/;
267
+
268
+ function lineText(line: MergedLine): string {
269
+ return line.spans.map((s) => s.text).join(" ").trim();
270
+ }
271
+
272
+ function lineAtEdge(line: MergedLine, pageHeight: number): boolean {
273
+ return (
274
+ line.lrect.y0 < pageHeight * RUNNING_EDGE_FRACTION ||
275
+ line.lrect.y1 > pageHeight * (1 - RUNNING_EDGE_FRACTION)
276
+ );
277
+ }
278
+
279
+ function linePositionKey(y0: number): number {
280
+ return Math.round(y0 / RUNNING_POSITION_TOLERANCE);
281
+ }
282
+
283
+ function countRunningLines(
284
+ pages: PageData[],
285
+ heights: number[],
286
+ ): { textCounts: Map<string, number>; numericPositionCounts: Map<number, number> } {
287
+ const textCounts = new Map<string, number>();
288
+ const numericPositionCounts = new Map<number, number>();
289
+ for (let i = 0; i < pages.length; i++) {
290
+ const height = heights[i];
291
+ const seenTexts = new Set<string>();
292
+ const seenPositions = new Set<number>();
293
+ for (const line of pages[i].lines) {
294
+ if (!lineAtEdge(line, height)) continue;
295
+ const text = lineText(line);
296
+ if (!text) continue;
297
+ const position = linePositionKey(line.lrect.y0);
298
+ const key = `${text}@${position}`;
299
+ if (!seenTexts.has(key)) {
300
+ seenTexts.add(key);
301
+ textCounts.set(key, (textCounts.get(key) ?? 0) + 1);
302
+ }
303
+ if (!seenPositions.has(position)) {
304
+ seenPositions.add(position);
305
+ if (PAGE_NUMBER_RE.test(text)) {
306
+ numericPositionCounts.set(position, (numericPositionCounts.get(position) ?? 0) + 1);
307
+ }
308
+ }
309
+ }
310
+ }
311
+ return { textCounts, numericPositionCounts };
312
+ }
313
+
314
+ function stripRunningLines(pages: PageData[], heights: number[]): number[] {
315
+ const { textCounts, numericPositionCounts } = countRunningLines(pages, heights);
316
+ const threshold = Math.max(RUNNING_MIN_PAGES, Math.ceil(pages.length * RUNNING_MIN_FRACTION));
317
+ const removedLetters: number[] = [];
318
+ for (let i = 0; i < pages.length; i++) {
319
+ const height = heights[i];
320
+ let removed = 0;
321
+ pages[i].lines = pages[i].lines.filter((line) => {
322
+ if (!lineAtEdge(line, height)) return true;
323
+ const text = lineText(line);
324
+ if (!text) return true;
325
+ const position = linePositionKey(line.lrect.y0);
326
+ const repeated = (textCounts.get(`${text}@${position}`) ?? 0) >= threshold;
327
+ const pageNumber =
328
+ PAGE_NUMBER_RE.test(text) && (numericPositionCounts.get(position) ?? 0) >= threshold;
329
+ if (repeated || pageNumber) {
330
+ removed += countLetters(text);
331
+ return false;
332
+ }
333
+ return true;
334
+ });
335
+ removedLetters.push(removed);
336
+ }
337
+ return removedLetters;
338
+ }
339
+
259
340
  function findLink(links: LinkInfo[], span: SpanData): string | null {
260
341
  const midX = (span.x0 + span.x1) / 2;
261
342
  const midY = (span.y0 + span.y1) / 2;
@@ -317,8 +398,7 @@ function detectTableBands(lines: MergedLine[]): TableBand[] {
317
398
  for (const row of rows.slice(1)) {
318
399
  output += "|" + row.join("|") + "|\n";
319
400
  }
320
- const indexes = new Set(band.map((l) => lines.indexOf(l)));
321
- bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...indexes), lineIndexes: indexes });
401
+ bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...band.map((l) => lines.indexOf(l))) });
322
402
  }
323
403
  }
324
404
  }
@@ -436,10 +516,6 @@ function writeText(
436
516
  }
437
517
  out += "\n";
438
518
  }
439
- while (emittedBands < tableBands.length) {
440
- out += "\n" + tableBands[emittedBands].markdown;
441
- emittedBands++;
442
- }
443
519
  out += "\n";
444
520
  if (code) out += "```\n";
445
521
  out += "\n\n";
@@ -473,8 +549,11 @@ export async function extractPdfPages(
473
549
  const total = doc.countPages();
474
550
  const count = Math.min(total, MAX_WEB_PDF_PAGES);
475
551
  const pages: PageData[] = [];
552
+ const heights: number[] = [];
476
553
  for (let i = 0; i < count; i++) {
477
554
  const page = doc.loadPage(i);
555
+ const bounds = page.getBounds();
556
+ heights.push(bounds[3] - bounds[1]);
478
557
  let st: MupdfStructuredText | null = null;
479
558
  try {
480
559
  st = page.toStructuredText("");
@@ -496,10 +575,12 @@ export async function extractPdfPages(
496
575
  page.destroy();
497
576
  }
498
577
  }
578
+ const removedLetters = stripRunningLines(pages, heights);
499
579
  const info = identifyHeaders(pages.map((p) => p.lines));
500
580
  const extracted = pages.map((p, i) => {
501
581
  const markdown = renderPageMarkdown(p.lines, info, p.links);
502
- const text = markdownCorrupted(markdown) || markdownIncomplete(markdown, p.plain) ? p.plain : markdown;
582
+ const plainLetters = Math.max(0, countLetters(p.plain) - removedLetters[i]);
583
+ const text = markdownCorrupted(markdown) || markdownIncomplete(markdown, plainLetters) ? p.plain : markdown;
503
584
  return { text, pageNumber: i + 1 };
504
585
  });
505
586
  return { pages: extracted, totalPages: total };
@@ -616,6 +697,11 @@ function extractTextOps(bytes: Buffer): string {
616
697
  }
617
698
  }
618
699
  }
700
+ if (c === "'" || c === '"') {
701
+ if (inText) out.push("\n");
702
+ i++;
703
+ continue;
704
+ }
619
705
  const token = text.slice(i, i + 2);
620
706
  if (token === "BT") {
621
707
  inText = true;
package/web-access.ts CHANGED
@@ -37,6 +37,11 @@ export function normalizeDomain(value: unknown): string {
37
37
  throw new Error("Website limits must contain domains without schemes or ports");
38
38
  }
39
39
  const numericParts = stripped.split(".");
40
+ const canonicalIpv4 =
41
+ numericParts.length === 4 &&
42
+ numericParts.every((part) => /^(?:0|[1-9][0-9]{0,2})$/.test(part)) &&
43
+ numericParts.every((part) => Number(part) <= 255);
44
+ if (canonicalIpv4) return stripped;
40
45
  if (
41
46
  numericParts.length <= 4 &&
42
47
  numericParts.every((part) => /^(?:0x[0-9a-f]+|[0-9]+)$/.test(part))
@@ -77,11 +82,12 @@ function compressIpv6(ip: string): string {
77
82
  }
78
83
 
79
84
  function compressGroups(groups: string[]): string {
85
+ const normalized = groups.map((g) => (/^0+$/.test(g) ? "0" : g.replace(/^0+(?=[0-9a-f])/, "")));
80
86
  let bestStart = -1;
81
87
  let bestLen = 0;
82
88
  let runStart = -1;
83
- for (let i = 0; i <= groups.length; i++) {
84
- if (i < groups.length && groups[i] === "0") {
89
+ for (let i = 0; i <= normalized.length; i++) {
90
+ if (i < normalized.length && normalized[i] === "0") {
85
91
  if (runStart === -1) runStart = i;
86
92
  } else if (runStart !== -1) {
87
93
  const runLen = i - runStart;
@@ -92,10 +98,20 @@ function compressGroups(groups: string[]): string {
92
98
  runStart = -1;
93
99
  }
94
100
  }
95
- if (bestLen < 2) return groups.map((g) => g.replace(/^0+(?=[0-9a-f])/, "")).join(":");
96
- const head = groups.slice(0, bestStart).map((g) => g.replace(/^0+(?=[0-9a-f])/, ""));
97
- const tail = groups.slice(bestStart + bestLen).map((g) => g.replace(/^0+(?=[0-9a-f])/, ""));
98
- return [...head, "", ...tail].join(":");
101
+ if (bestLen < 2) return normalized.join(":");
102
+ const head = normalized.slice(0, bestStart);
103
+ const tail = normalized.slice(bestStart + bestLen);
104
+ return `${head.join(":")}::${tail.join(":")}`;
105
+ }
106
+
107
+ const PCP_ANYCAST = new Set(["2001:1::1", "2001:1::2"]);
108
+
109
+ function isPcpAnycast(lower: string): boolean {
110
+ try {
111
+ return PCP_ANYCAST.has(compressIpv6(lower));
112
+ } catch {
113
+ return false;
114
+ }
99
115
  }
100
116
 
101
117
  export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
@@ -119,15 +135,7 @@ export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
119
135
  if (rawDomains.length > MAX_DOMAINS_PER_LIST) {
120
136
  throw new Error(`${key} supports at most ${MAX_DOMAINS_PER_LIST} domains`);
121
137
  }
122
- const domains: string[] = [];
123
- for (const rawDomain of rawDomains) {
124
- if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
125
- throw new Error(`${key} must contain only strings`);
126
- }
127
- const domain = normalizeDomain(rawDomain);
128
- if (!domains.includes(domain)) domains.push(domain);
129
- }
130
- normalized[key] = domains;
138
+ normalized[key] = normalizeDomainList(rawDomains, key);
131
139
  }
132
140
  return normalized;
133
141
  }
@@ -136,12 +144,22 @@ function matchesDomain(hostname: string, domain: string): boolean {
136
144
  return hostname === domain || hostname.endsWith(`.${domain}`);
137
145
  }
138
146
 
139
- export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null): boolean {
140
- let host: string;
147
+ function normalizeDomainList(domains: unknown[], listName: string): string[] {
148
+ const out: string[] = [];
149
+ for (const rawDomain of domains) {
150
+ if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
151
+ throw new Error(`${listName} must contain only strings`);
152
+ }
153
+ const domain = normalizeDomain(rawDomain);
154
+ if (!out.includes(domain)) out.push(domain);
155
+ }
156
+ return out;
157
+ }
158
+
159
+ function policyAllows(host: string, policy: WebsitePolicy | null): boolean {
141
160
  let normalized: WebsitePolicy;
142
161
  try {
143
- host = normalizeDomain(hostname);
144
- normalized = normalizePolicyMaybe(policy);
162
+ normalized = normalizeWebsitePolicy(policy);
145
163
  } catch {
146
164
  return false;
147
165
  }
@@ -150,22 +168,12 @@ export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null):
150
168
  return allowed.length === 0 || allowed.some((domain) => matchesDomain(host, domain));
151
169
  }
152
170
 
153
- function normalizePolicyObject(policy: WebsitePolicy): WebsitePolicy {
154
- const normalized: WebsitePolicy = { allowedDomains: [], blockedDomains: [] };
155
- for (const key of ["allowedDomains", "blockedDomains"] as const) {
156
- const domains: string[] = [];
157
- for (const rawDomain of policy[key]) {
158
- const domain = normalizeDomain(rawDomain);
159
- if (!domains.includes(domain)) domains.push(domain);
160
- }
161
- normalized[key] = domains;
171
+ export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null): boolean {
172
+ try {
173
+ return policyAllows(normalizeDomain(hostname), policy);
174
+ } catch {
175
+ return false;
162
176
  }
163
- return normalized;
164
- }
165
-
166
- function normalizePolicyMaybe(policy: WebsitePolicy | null | undefined): WebsitePolicy {
167
- if (!policy) return { allowedDomains: [], blockedDomains: [] };
168
- return normalizePolicyObject(policy);
169
177
  }
170
178
 
171
179
  export function checkUrlAccess(
@@ -211,7 +219,7 @@ export function checkUrlAccess(
211
219
  } catch {
212
220
  return [false, "Blocked: the URL has an invalid hostname or port.", ""];
213
221
  }
214
- if (!hostnameAllowed(hostname, policy)) {
222
+ if (!policyAllows(hostname, policy)) {
215
223
  return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
216
224
  }
217
225
  return [true, "", hostname];
@@ -315,7 +323,7 @@ const GITHUB_NON_OWNER_SEGMENTS = new Set([
315
323
 
316
324
  const GITHUB_NAME_RE = /^[A-Za-z0-9_.\-]{1,100}$/;
317
325
 
318
- export function githubRepoReadmeApiUrl(url: string): string | null {
326
+ function githubRepoOwnerRepo(url: string): [string, string] | null {
319
327
  let parsed: URL;
320
328
  try {
321
329
  parsed = new URL(url);
@@ -330,7 +338,17 @@ export function githubRepoReadmeApiUrl(url: string): string | null {
330
338
  if (GITHUB_NON_OWNER_SEGMENTS.has(owner.toLowerCase())) return null;
331
339
  const cleanRepo = repo.endsWith(".git") ? repo.slice(0, -4) : repo;
332
340
  if (!GITHUB_NAME_RE.test(owner) || !GITHUB_NAME_RE.test(cleanRepo)) return null;
333
- return `https://api.github.com/repos/${owner}/${cleanRepo}/readme`;
341
+ return [owner, cleanRepo];
342
+ }
343
+
344
+ export function githubRepoReadmeApiUrl(url: string): string | null {
345
+ const pair = githubRepoOwnerRepo(url);
346
+ return pair ? `https://api.github.com/repos/${pair[0]}/${pair[1]}/readme` : null;
347
+ }
348
+
349
+ export function githubRepoRawReadmeUrl(url: string): string | null {
350
+ const pair = githubRepoOwnerRepo(url);
351
+ return pair ? `https://raw.githubusercontent.com/${pair[0]}/${pair[1]}/HEAD/README.md` : null;
334
352
  }
335
353
 
336
354
  function ipv4Octets(ip: string): number[] | null {
@@ -353,6 +371,7 @@ export function isPublicIp(ip: string): boolean {
353
371
  if (o[0] === 172 && o[1] >= 16 && o[1] <= 31) return false;
354
372
  if (o[0] === 192 && o[1] === 0 && o[2] === 0) return false;
355
373
  if (o[0] === 192 && o[1] === 0 && o[2] === 2) return false;
374
+ if (o[0] === 192 && o[1] === 88 && o[2] === 99) return false;
356
375
  if (o[0] === 192 && o[1] === 168) return false;
357
376
  if (o[0] === 198 && (o[1] === 18 || o[1] === 19)) return false;
358
377
  if (o[0] === 198 && o[1] === 51 && o[2] === 100) return false;
@@ -360,18 +379,22 @@ export function isPublicIp(ip: string): boolean {
360
379
  if (o[0] >= 224) return false;
361
380
  return true;
362
381
  }
363
- const lower = ip.toLowerCase();
364
- if (lower === "::" || lower === "::1") return false;
365
- if (lower.startsWith("fc") || lower.startsWith("fd")) return false;
366
- if (/^fe[89ab][0-9a-f]:/.test(lower)) return false;
367
- if (lower.startsWith("ff")) return false;
368
- if (lower.startsWith("2001:db8")) return false;
369
- if (lower.startsWith("64:ff9b:")) return false;
370
- if (lower.startsWith("2001:10:")) return false;
371
- if (lower.startsWith("2002:")) return false;
372
- if (lower.startsWith("2001:0:") || lower.startsWith("2001::")) return false;
373
- const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(lower);
374
- if (mapped) return isPublicIp(mapped[1]);
375
- if (lower.startsWith("::ffff:")) return false;
382
+ let canonical: string;
383
+ try {
384
+ canonical = compressIpv6(ip);
385
+ } catch {
386
+ return false;
387
+ }
388
+ if (canonical === "::1" || canonical.startsWith("::")) return false;
389
+ if (canonical.startsWith("fc") || canonical.startsWith("fd")) return false;
390
+ if (/^fe[89ab][0-9a-f]:/.test(canonical)) return false;
391
+ if (canonical.startsWith("ff")) return false;
392
+ if (canonical.startsWith("2001:db8")) return false;
393
+ if (canonical.startsWith("64:ff9b:")) return false;
394
+ if (canonical.startsWith("2001:10:")) return false;
395
+ if (canonical.startsWith("2002:")) return false;
396
+ if (canonical.startsWith("2001:0:") || canonical.startsWith("2001::")) return false;
397
+ if (canonical.startsWith("2001:2:")) return false;
398
+ if (isPcpAnycast(canonical)) return false;
376
399
  return true;
377
400
  }