pi-unsloth-webtools 0.2.4 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/pdf.ts CHANGED
@@ -26,6 +26,7 @@ interface MupdfDocument {
26
26
 
27
27
  interface MupdfPage {
28
28
  toStructuredText(options?: string): MupdfStructuredText;
29
+ getBounds(): [number, number, number, number];
29
30
  getLinks(): { getBounds(): [number, number, number, number]; getURI(): string }[];
30
31
  destroy(): void;
31
32
  }
@@ -92,7 +93,6 @@ interface LinkInfo {
92
93
  interface TableBand {
93
94
  markdown: string;
94
95
  firstLineIndex: number;
95
- lineIndexes: Set<number>;
96
96
  }
97
97
 
98
98
  interface HeaderInfo {
@@ -137,10 +137,13 @@ function markdownCorrupted(text: string): boolean {
137
137
  return shaped > threshold || (text.match(/\ufffd/g) ?? []).length > threshold;
138
138
  }
139
139
 
140
- function markdownIncomplete(markdown: string, plain: string): boolean {
141
- const plainLetters = [...plain].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
140
+ function countLetters(text: string): number {
141
+ return [...text].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
142
+ }
143
+
144
+ function markdownIncomplete(markdown: string, plainLetters: number): boolean {
142
145
  if (plainLetters < PDF_INCOMPLETE_MIN_LETTERS) return false;
143
- const markdownLetters = [...markdown].filter((c) => /[\p{L}\p{N}]/u.test(c)).length;
146
+ const markdownLetters = countLetters(markdown);
144
147
  return markdownLetters < PDF_INCOMPLETE_RATIO * plainLetters;
145
148
  }
146
149
 
@@ -256,6 +259,84 @@ function getRawLines(json: JsonBlock[]): MergedLine[] {
256
259
  return nlines;
257
260
  }
258
261
 
262
+ const RUNNING_EDGE_FRACTION = 0.12;
263
+ const RUNNING_MIN_PAGES = 2;
264
+ const RUNNING_MIN_FRACTION = 0.5;
265
+ const RUNNING_POSITION_TOLERANCE = 5;
266
+ const PAGE_NUMBER_RE = /^\d{1,4}$/;
267
+
268
+ function lineText(line: MergedLine): string {
269
+ return line.spans.map((s) => s.text).join(" ").trim();
270
+ }
271
+
272
+ function lineAtEdge(line: MergedLine, pageHeight: number): boolean {
273
+ return (
274
+ line.lrect.y0 < pageHeight * RUNNING_EDGE_FRACTION ||
275
+ line.lrect.y1 > pageHeight * (1 - RUNNING_EDGE_FRACTION)
276
+ );
277
+ }
278
+
279
+ function linePositionKey(y0: number): number {
280
+ return Math.round(y0 / RUNNING_POSITION_TOLERANCE);
281
+ }
282
+
283
+ function countRunningLines(
284
+ pages: PageData[],
285
+ heights: number[],
286
+ ): { textCounts: Map<string, number>; numericPositionCounts: Map<number, number> } {
287
+ const textCounts = new Map<string, number>();
288
+ const numericPositionCounts = new Map<number, number>();
289
+ for (let i = 0; i < pages.length; i++) {
290
+ const height = heights[i];
291
+ const seenTexts = new Set<string>();
292
+ const seenPositions = new Set<number>();
293
+ for (const line of pages[i].lines) {
294
+ if (!lineAtEdge(line, height)) continue;
295
+ const text = lineText(line);
296
+ if (!text) continue;
297
+ const position = linePositionKey(line.lrect.y0);
298
+ const key = `${text}@${position}`;
299
+ if (!seenTexts.has(key)) {
300
+ seenTexts.add(key);
301
+ textCounts.set(key, (textCounts.get(key) ?? 0) + 1);
302
+ }
303
+ if (!seenPositions.has(position)) {
304
+ seenPositions.add(position);
305
+ if (PAGE_NUMBER_RE.test(text)) {
306
+ numericPositionCounts.set(position, (numericPositionCounts.get(position) ?? 0) + 1);
307
+ }
308
+ }
309
+ }
310
+ }
311
+ return { textCounts, numericPositionCounts };
312
+ }
313
+
314
+ function stripRunningLines(pages: PageData[], heights: number[]): number[] {
315
+ const { textCounts, numericPositionCounts } = countRunningLines(pages, heights);
316
+ const threshold = Math.max(RUNNING_MIN_PAGES, Math.ceil(pages.length * RUNNING_MIN_FRACTION));
317
+ const removedLetters: number[] = [];
318
+ for (let i = 0; i < pages.length; i++) {
319
+ const height = heights[i];
320
+ let removed = 0;
321
+ pages[i].lines = pages[i].lines.filter((line) => {
322
+ if (!lineAtEdge(line, height)) return true;
323
+ const text = lineText(line);
324
+ if (!text) return true;
325
+ const position = linePositionKey(line.lrect.y0);
326
+ const repeated = (textCounts.get(`${text}@${position}`) ?? 0) >= threshold;
327
+ const pageNumber =
328
+ PAGE_NUMBER_RE.test(text) && (numericPositionCounts.get(position) ?? 0) >= threshold;
329
+ if (repeated || pageNumber) {
330
+ removed += countLetters(text);
331
+ return false;
332
+ }
333
+ return true;
334
+ });
335
+ removedLetters.push(removed);
336
+ }
337
+ return removedLetters;
338
+ }
339
+
259
340
  function findLink(links: LinkInfo[], span: SpanData): string | null {
260
341
  const midX = (span.x0 + span.x1) / 2;
261
342
  const midY = (span.y0 + span.y1) / 2;
@@ -317,8 +398,7 @@ function detectTableBands(lines: MergedLine[]): TableBand[] {
317
398
  for (const row of rows.slice(1)) {
318
399
  output += "|" + row.join("|") + "|\n";
319
400
  }
320
- const indexes = new Set(band.map((l) => lines.indexOf(l)));
321
- bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...indexes), lineIndexes: indexes });
401
+ bands.push({ markdown: output + "\n", firstLineIndex: Math.min(...band.map((l) => lines.indexOf(l))) });
322
402
  }
323
403
  }
324
404
  }
@@ -436,10 +516,6 @@ function writeText(
436
516
  }
437
517
  out += "\n";
438
518
  }
439
- while (emittedBands < tableBands.length) {
440
- out += "\n" + tableBands[emittedBands].markdown;
441
- emittedBands++;
442
- }
443
519
  out += "\n";
444
520
  if (code) out += "```\n";
445
521
  out += "\n\n";
@@ -473,8 +549,11 @@ export async function extractPdfPages(
473
549
  const total = doc.countPages();
474
550
  const count = Math.min(total, MAX_WEB_PDF_PAGES);
475
551
  const pages: PageData[] = [];
552
+ const heights: number[] = [];
476
553
  for (let i = 0; i < count; i++) {
477
554
  const page = doc.loadPage(i);
555
+ const bounds = page.getBounds();
556
+ heights.push(bounds[3] - bounds[1]);
478
557
  let st: MupdfStructuredText | null = null;
479
558
  try {
480
559
  st = page.toStructuredText("");
@@ -496,10 +575,12 @@ export async function extractPdfPages(
496
575
  page.destroy();
497
576
  }
498
577
  }
578
+ const removedLetters = stripRunningLines(pages, heights);
499
579
  const info = identifyHeaders(pages.map((p) => p.lines));
500
580
  const extracted = pages.map((p, i) => {
501
581
  const markdown = renderPageMarkdown(p.lines, info, p.links);
502
- const text = markdownCorrupted(markdown) || markdownIncomplete(markdown, p.plain) ? p.plain : markdown;
582
+ const plainLetters = Math.max(0, countLetters(p.plain) - removedLetters[i]);
583
+ const text = markdownCorrupted(markdown) || markdownIncomplete(markdown, plainLetters) ? p.plain : markdown;
503
584
  return { text, pageNumber: i + 1 };
504
585
  });
505
586
  return { pages: extracted, totalPages: total };
package/web-access.ts CHANGED
@@ -77,11 +77,12 @@ function compressIpv6(ip: string): string {
77
77
  }
78
78
 
79
79
  function compressGroups(groups: string[]): string {
80
+ const normalized = groups.map((g) => (/^0+$/.test(g) ? "0" : g.replace(/^0+(?=[0-9a-f])/, "")));
80
81
  let bestStart = -1;
81
82
  let bestLen = 0;
82
83
  let runStart = -1;
83
- for (let i = 0; i <= groups.length; i++) {
84
- if (i < groups.length && groups[i] === "0") {
84
+ for (let i = 0; i <= normalized.length; i++) {
85
+ if (i < normalized.length && normalized[i] === "0") {
85
86
  if (runStart === -1) runStart = i;
86
87
  } else if (runStart !== -1) {
87
88
  const runLen = i - runStart;
@@ -92,10 +93,20 @@ function compressGroups(groups: string[]): string {
92
93
  runStart = -1;
93
94
  }
94
95
  }
95
- if (bestLen < 2) return groups.map((g) => g.replace(/^0+(?=[0-9a-f])/, "")).join(":");
96
- const head = groups.slice(0, bestStart).map((g) => g.replace(/^0+(?=[0-9a-f])/, ""));
97
- const tail = groups.slice(bestStart + bestLen).map((g) => g.replace(/^0+(?=[0-9a-f])/, ""));
98
- return [...head, "", ...tail].join(":");
96
+ if (bestLen < 2) return normalized.join(":");
97
+ const head = normalized.slice(0, bestStart);
98
+ const tail = normalized.slice(bestStart + bestLen);
99
+ return `${head.join(":")}::${tail.join(":")}`;
100
+ }
101
+
102
+ const PCP_ANYCAST = new Set(["2001:1::1", "2001:1::2"]);
103
+
104
+ function isPcpAnycast(lower: string): boolean {
105
+ try {
106
+ return PCP_ANYCAST.has(compressIpv6(lower));
107
+ } catch {
108
+ return false;
109
+ }
99
110
  }
100
111
 
101
112
  export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
@@ -119,15 +130,7 @@ export function normalizeWebsitePolicy(value: unknown): WebsitePolicy {
119
130
  if (rawDomains.length > MAX_DOMAINS_PER_LIST) {
120
131
  throw new Error(`${key} supports at most ${MAX_DOMAINS_PER_LIST} domains`);
121
132
  }
122
- const domains: string[] = [];
123
- for (const rawDomain of rawDomains) {
124
- if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
125
- throw new Error(`${key} must contain only strings`);
126
- }
127
- const domain = normalizeDomain(rawDomain);
128
- if (!domains.includes(domain)) domains.push(domain);
129
- }
130
- normalized[key] = domains;
133
+ normalized[key] = normalizeDomainList(rawDomains, key);
131
134
  }
132
135
  return normalized;
133
136
  }
@@ -136,12 +139,22 @@ function matchesDomain(hostname: string, domain: string): boolean {
136
139
  return hostname === domain || hostname.endsWith(`.${domain}`);
137
140
  }
138
141
 
139
- export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null): boolean {
140
- let host: string;
142
+ function normalizeDomainList(domains: unknown[], listName: string): string[] {
143
+ const out: string[] = [];
144
+ for (const rawDomain of domains) {
145
+ if (typeof rawDomain !== "string" || rawDomain.length > MAX_CACHEABLE_DOMAIN_LEN) {
146
+ throw new Error(`${listName} must contain only strings`);
147
+ }
148
+ const domain = normalizeDomain(rawDomain);
149
+ if (!out.includes(domain)) out.push(domain);
150
+ }
151
+ return out;
152
+ }
153
+
154
+ function policyAllows(host: string, policy: WebsitePolicy | null): boolean {
141
155
  let normalized: WebsitePolicy;
142
156
  try {
143
- host = normalizeDomain(hostname);
144
- normalized = normalizePolicyMaybe(policy);
157
+ normalized = normalizeWebsitePolicy(policy);
145
158
  } catch {
146
159
  return false;
147
160
  }
@@ -150,22 +163,12 @@ export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null):
150
163
  return allowed.length === 0 || allowed.some((domain) => matchesDomain(host, domain));
151
164
  }
152
165
 
153
- function normalizePolicyObject(policy: WebsitePolicy): WebsitePolicy {
154
- const normalized: WebsitePolicy = { allowedDomains: [], blockedDomains: [] };
155
- for (const key of ["allowedDomains", "blockedDomains"] as const) {
156
- const domains: string[] = [];
157
- for (const rawDomain of policy[key]) {
158
- const domain = normalizeDomain(rawDomain);
159
- if (!domains.includes(domain)) domains.push(domain);
160
- }
161
- normalized[key] = domains;
166
+ export function hostnameAllowed(hostname: string, policy: WebsitePolicy | null): boolean {
167
+ try {
168
+ return policyAllows(normalizeDomain(hostname), policy);
169
+ } catch {
170
+ return false;
162
171
  }
163
- return normalized;
164
- }
165
-
166
- function normalizePolicyMaybe(policy: WebsitePolicy | null | undefined): WebsitePolicy {
167
- if (!policy) return { allowedDomains: [], blockedDomains: [] };
168
- return normalizePolicyObject(policy);
169
172
  }
170
173
 
171
174
  export function checkUrlAccess(
@@ -211,7 +214,7 @@ export function checkUrlAccess(
211
214
  } catch {
212
215
  return [false, "Blocked: the URL has an invalid hostname or port.", ""];
213
216
  }
214
- if (!hostnameAllowed(hostname, policy)) {
217
+ if (!policyAllows(hostname, policy)) {
215
218
  return [false, `Blocked: the website access policy disallows ${hostname}.`, hostname];
216
219
  }
217
220
  return [true, "", hostname];
@@ -315,7 +318,7 @@ const GITHUB_NON_OWNER_SEGMENTS = new Set([
315
318
 
316
319
  const GITHUB_NAME_RE = /^[A-Za-z0-9_.\-]{1,100}$/;
317
320
 
318
- export function githubRepoReadmeApiUrl(url: string): string | null {
321
+ function githubRepoOwnerRepo(url: string): [string, string] | null {
319
322
  let parsed: URL;
320
323
  try {
321
324
  parsed = new URL(url);
@@ -330,7 +333,17 @@ export function githubRepoReadmeApiUrl(url: string): string | null {
330
333
  if (GITHUB_NON_OWNER_SEGMENTS.has(owner.toLowerCase())) return null;
331
334
  const cleanRepo = repo.endsWith(".git") ? repo.slice(0, -4) : repo;
332
335
  if (!GITHUB_NAME_RE.test(owner) || !GITHUB_NAME_RE.test(cleanRepo)) return null;
333
- return `https://api.github.com/repos/${owner}/${cleanRepo}/readme`;
336
+ return [owner, cleanRepo];
337
+ }
338
+
339
+ export function githubRepoReadmeApiUrl(url: string): string | null {
340
+ const pair = githubRepoOwnerRepo(url);
341
+ return pair ? `https://api.github.com/repos/${pair[0]}/${pair[1]}/readme` : null;
342
+ }
343
+
344
+ export function githubRepoRawReadmeUrl(url: string): string | null {
345
+ const pair = githubRepoOwnerRepo(url);
346
+ return pair ? `https://raw.githubusercontent.com/${pair[0]}/${pair[1]}/HEAD/README.md` : null;
334
347
  }
335
348
 
336
349
  function ipv4Octets(ip: string): number[] | null {
@@ -353,6 +366,7 @@ export function isPublicIp(ip: string): boolean {
353
366
  if (o[0] === 172 && o[1] >= 16 && o[1] <= 31) return false;
354
367
  if (o[0] === 192 && o[1] === 0 && o[2] === 0) return false;
355
368
  if (o[0] === 192 && o[1] === 0 && o[2] === 2) return false;
369
+ if (o[0] === 192 && o[1] === 88 && o[2] === 99) return false;
356
370
  if (o[0] === 192 && o[1] === 168) return false;
357
371
  if (o[0] === 198 && (o[1] === 18 || o[1] === 19)) return false;
358
372
  if (o[0] === 198 && o[1] === 51 && o[2] === 100) return false;
@@ -360,18 +374,22 @@ export function isPublicIp(ip: string): boolean {
360
374
  if (o[0] >= 224) return false;
361
375
  return true;
362
376
  }
363
- const lower = ip.toLowerCase();
364
- if (lower === "::" || lower === "::1") return false;
365
- if (lower.startsWith("fc") || lower.startsWith("fd")) return false;
366
- if (/^fe[89ab][0-9a-f]:/.test(lower)) return false;
367
- if (lower.startsWith("ff")) return false;
368
- if (lower.startsWith("2001:db8")) return false;
369
- if (lower.startsWith("64:ff9b:")) return false;
370
- if (lower.startsWith("2001:10:")) return false;
371
- if (lower.startsWith("2002:")) return false;
372
- if (lower.startsWith("2001:0:") || lower.startsWith("2001::")) return false;
373
- const mapped = /^::ffff:(\d+\.\d+\.\d+\.\d+)$/.exec(lower);
374
- if (mapped) return isPublicIp(mapped[1]);
375
- if (lower.startsWith("::ffff:")) return false;
377
+ let canonical: string;
378
+ try {
379
+ canonical = compressIpv6(ip);
380
+ } catch {
381
+ return false;
382
+ }
383
+ if (canonical === "::1" || canonical.startsWith("::")) return false;
384
+ if (canonical.startsWith("fc") || canonical.startsWith("fd")) return false;
385
+ if (/^fe[89ab][0-9a-f]:/.test(canonical)) return false;
386
+ if (canonical.startsWith("ff")) return false;
387
+ if (canonical.startsWith("2001:db8")) return false;
388
+ if (canonical.startsWith("64:ff9b:")) return false;
389
+ if (canonical.startsWith("2001:10:")) return false;
390
+ if (canonical.startsWith("2002:")) return false;
391
+ if (canonical.startsWith("2001:0:") || canonical.startsWith("2001::")) return false;
392
+ if (canonical.startsWith("2001:2:")) return false;
393
+ if (isPcpAnycast(canonical)) return false;
376
394
  return true;
377
395
  }