@vierratale/ai 0.1.0-beta.8 → 0.1.0-fixfetch-beta9.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +120 -16
- package/bin/vierrataleai.js +74 -1
- package/package.json +8 -2
- package/src/catalog.js +78 -49
- package/src/cli.js +864 -103
- package/src/cmd/agent.js +1695 -0
- package/src/cmd/executor.js +257 -0
- package/src/cmd/tools.js +131 -0
- package/src/config.js +24 -10
- package/src/index.js +1 -0
- package/src/installer.js +130 -2
- package/src/prompts/system.json +2 -2
- package/src/prompts/system.md +42 -0
- package/src/providers/anthropic.js +65 -34
- package/src/providers/base.js +10 -0
- package/src/providers/cortex.js +223 -40
- package/src/providers/gemini.js +62 -31
- package/src/providers/openai.js +64 -33
- package/src/session.js +195 -15
- package/src/ui/branding.js +1 -1
- package/src/ui/chatbox.js +271 -0
- package/src/utils/filewriter.js +97 -0
- package/src/utils/intents.js +61 -1
- package/src/utils/logger.js +67 -0
- package/src/utils/math.js +142 -0
- package/src/utils/webfetch.js +159 -11
- package/build.sh +0 -17
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
import { existsSync, mkdirSync, statSync, writeFileSync } from 'fs';
|
|
2
|
+
import { dirname, resolve, sep } from 'path';
|
|
3
|
+
|
|
4
|
+
const FILE_RE = /^FILE:\s*(.+?)\s*$/i;
|
|
5
|
+
const FOLDER_RE = /^FOLDER:\s*(.+?)\s*$/i;
|
|
6
|
+
|
|
7
|
+
export class FileWriter {
|
|
8
|
+
static parse(text) {
|
|
9
|
+
const files = [];
|
|
10
|
+
const folders = [];
|
|
11
|
+
const lines = String(text || '').split('\n');
|
|
12
|
+
let i = 0;
|
|
13
|
+
while (i < lines.length) {
|
|
14
|
+
let m = FILE_RE.exec(lines[i]);
|
|
15
|
+
if (m) {
|
|
16
|
+
const path = m[1].trim();
|
|
17
|
+
let j = i + 1;
|
|
18
|
+
while (j < lines.length && lines[j].trim() === '') j++;
|
|
19
|
+
if (j < lines.length && lines[j].startsWith('```')) {
|
|
20
|
+
const start = j + 1;
|
|
21
|
+
let end = start;
|
|
22
|
+
while (end < lines.length && !lines[end].startsWith('```')) end++;
|
|
23
|
+
if (end < lines.length) {
|
|
24
|
+
files.push({
|
|
25
|
+
path,
|
|
26
|
+
language: lines[j].slice(3).trim(),
|
|
27
|
+
content: lines.slice(start, end).join('\n'),
|
|
28
|
+
});
|
|
29
|
+
i = end + 1;
|
|
30
|
+
continue;
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
m = FOLDER_RE.exec(lines[i]);
|
|
35
|
+
if (m) {
|
|
36
|
+
folders.push(m[1].trim());
|
|
37
|
+
}
|
|
38
|
+
i++;
|
|
39
|
+
}
|
|
40
|
+
return { files, folders };
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
static resolveTarget(path) {
|
|
44
|
+
const base = resolve(process.cwd());
|
|
45
|
+
const target = resolve(base, path);
|
|
46
|
+
if (target !== base && !target.startsWith(base + sep)) {
|
|
47
|
+
throw new Error(`Refusing to write outside the working directory: ${path}`);
|
|
48
|
+
}
|
|
49
|
+
return target;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
static async write({ files = [], folders = [], overwrite = false, confirm } = {}) {
|
|
53
|
+
const results = [];
|
|
54
|
+
|
|
55
|
+
for (const p of folders) {
|
|
56
|
+
try {
|
|
57
|
+
mkdirSync(this.resolveTarget(p), { recursive: true });
|
|
58
|
+
results.push({ kind: 'folder', path: p, target: this.resolveTarget(p), status: 'created' });
|
|
59
|
+
} catch (err) {
|
|
60
|
+
results.push({ kind: 'folder', path: p, status: 'error', error: err.message });
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
for (const f of files) {
|
|
65
|
+
let target;
|
|
66
|
+
try {
|
|
67
|
+
target = this.resolveTarget(f.path);
|
|
68
|
+
} catch (err) {
|
|
69
|
+
results.push({ kind: 'file', path: f.path, status: 'error', error: err.message });
|
|
70
|
+
continue;
|
|
71
|
+
}
|
|
72
|
+
if (existsSync(target) && statSync(target).isDirectory()) {
|
|
73
|
+
results.push({ kind: 'file', path: f.path, target, status: 'error', error: 'Target is a directory' });
|
|
74
|
+
continue;
|
|
75
|
+
}
|
|
76
|
+
if (existsSync(target) && confirm) {
|
|
77
|
+
const ok = await confirm(`Overwrite ${f.path}?`);
|
|
78
|
+
if (!ok) {
|
|
79
|
+
results.push({ kind: 'file', path: f.path, target, status: 'skipped', error: 'Declined' });
|
|
80
|
+
continue;
|
|
81
|
+
}
|
|
82
|
+
} else if (existsSync(target) && !overwrite) {
|
|
83
|
+
results.push({ kind: 'file', path: f.path, target, status: 'skipped', error: 'File already exists' });
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
86
|
+
try {
|
|
87
|
+
mkdirSync(dirname(target), { recursive: true });
|
|
88
|
+
writeFileSync(target, f.content, 'utf-8');
|
|
89
|
+
results.push({ kind: 'file', path: f.path, target, status: 'written' });
|
|
90
|
+
} catch (err) {
|
|
91
|
+
results.push({ kind: 'file', path: f.path, target, status: 'error', error: err.message });
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
return results;
|
|
96
|
+
}
|
|
97
|
+
}
|
package/src/utils/intents.js
CHANGED
|
@@ -19,14 +19,74 @@ const SELF_PHRASES = [
|
|
|
19
19
|
'what engine', 'which engine', 'are you gpt', 'are you chatgpt',
|
|
20
20
|
];
|
|
21
21
|
|
|
22
|
+
const ACTION_VERBS = 'create|make|write|save|generate|build|add|new|update|edit|install|put|set';
|
|
23
|
+
const FILE_THINGS = 'file|folder|directory|project|code|script|config|\\.json|app|readme|package\\.json|website|html|css';
|
|
24
|
+
const INFO_ASK = /^(how|what|why|when|where|which)\b/i;
|
|
25
|
+
const FILE_REQUEST = new RegExp(`\\b(${ACTION_VERBS})\\b[^.!?]{0,80}\\b(${FILE_THINGS})\\b`, 'i');
|
|
26
|
+
|
|
27
|
+
const STORY_WORDS = [
|
|
28
|
+
'story', 'stories', 'tale', 'tales', 'legend', 'legends',
|
|
29
|
+
'myth', 'myths', 'fable', 'fables', 'folklore', 'folktale',
|
|
30
|
+
'folk tale', 'fairytale', 'fairy tale',
|
|
31
|
+
].join('|');
|
|
32
|
+
const STORY_LINKS = 'about|of|regarding|recounting|called|named';
|
|
33
|
+
const TRIM_TAIL = '\\s+(?:and|in|into)\\b|\\s*[.!?]|$';
|
|
34
|
+
|
|
35
|
+
const STORY_SUBJECT_PATTERN = new RegExp(
|
|
36
|
+
`\\b(?:${STORY_WORDS})\\s+(?:${STORY_LINKS})\\s+[a-z0-9]{3,}`, 'i',
|
|
37
|
+
);
|
|
38
|
+
const STORY_ASK_PATTERN = new RegExp(
|
|
39
|
+
`\\b(?:tell|write|narrate|recite|make|create|generate|give|share|hear|listen to)\\b[^.!?\\n]{0,60}\\b(?:${STORY_WORDS})\\b`, 'i',
|
|
40
|
+
);
|
|
41
|
+
const STORY_CAPTURE = new RegExp(
|
|
42
|
+
`\\b(?:${STORY_WORDS})\\s+(?:${STORY_LINKS})\\s+([^.!?\\n]{2,}?)\\s*(?=${TRIM_TAIL})`, 'i',
|
|
43
|
+
);
|
|
44
|
+
const ABOUT_CAPTURE = /\b(?:tell me|talk|read|learn|research|hear|know|find out|look up|speak|write|give)\b[^.!?\n]{0,30}\babout\s+([^.!?\n]{2,}?)\s*(?=\s+(?:and|in|into)\b|\s*[.!?]|$)/i;
|
|
45
|
+
const PLACE_PATTERN = /\bfrom\s+[a-z][a-z\s'.-]{1,40}$/i;
|
|
46
|
+
|
|
47
|
+
const GREETINGS = new Set([
|
|
48
|
+
'hi', 'hello', 'hey', 'hii', 'heyy', 'howdy', 'yo', 'hai',
|
|
49
|
+
'halo', 'hallo', 'hola', 'hiya', 'sup', 'good morning',
|
|
50
|
+
'good afternoon', 'good evening', 'good night', 'selamat pagi',
|
|
51
|
+
'selamat siang', 'selamat sore', 'selamat malam', 'assalamualaikum',
|
|
52
|
+
]);
|
|
53
|
+
|
|
54
|
+
export function looksLikeStoryRequest(text) {
|
|
55
|
+
return STORY_ASK_PATTERN.test(text) || STORY_SUBJECT_PATTERN.test(text);
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export function extractSearchTopic(text) {
|
|
59
|
+
const s = String(text).trim();
|
|
60
|
+
const story = STORY_CAPTURE.exec(s);
|
|
61
|
+
if (story) return story[1].trim();
|
|
62
|
+
const about = ABOUT_CAPTURE.exec(s);
|
|
63
|
+
return about ? about[1].trim() : null;
|
|
64
|
+
}
|
|
65
|
+
|
|
22
66
|
export function detectIntent(text) {
|
|
23
67
|
const lower = text.toLowerCase().trim();
|
|
24
68
|
if (lower.startsWith('/')) return { type: 'command' };
|
|
25
69
|
|
|
70
|
+
if (GREETINGS.has(lower)) {
|
|
71
|
+
return { type: 'greeting' };
|
|
72
|
+
}
|
|
73
|
+
|
|
26
74
|
if (SELF_PHRASES.some((p) => lower.includes(p))) {
|
|
27
75
|
return { type: 'self' };
|
|
28
76
|
}
|
|
29
77
|
|
|
78
|
+
if (STORY_SUBJECT_PATTERN.test(lower) || PLACE_PATTERN.test(lower)) {
|
|
79
|
+
return { type: 'knowledge' };
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
const isKnowledgePrefixed = KNOWLEDGE_PREFIXES.some(
|
|
83
|
+
(p) => lower.startsWith(p + ' ') || lower === p,
|
|
84
|
+
);
|
|
85
|
+
|
|
86
|
+
if (!INFO_ASK.test(lower) && !isKnowledgePrefixed && FILE_REQUEST.test(lower)) {
|
|
87
|
+
return { type: 'chat' };
|
|
88
|
+
}
|
|
89
|
+
|
|
30
90
|
const isQuestion = /\?$/.test(lower);
|
|
31
91
|
const firstWord = lower.split(/\s+/)[0] || '';
|
|
32
92
|
|
|
@@ -34,7 +94,7 @@ export function detectIntent(text) {
|
|
|
34
94
|
if (isQuestion && QUESTION_PREFIXES.includes(firstWord)) {
|
|
35
95
|
needsWeb = true;
|
|
36
96
|
}
|
|
37
|
-
if (
|
|
97
|
+
if (isKnowledgePrefixed) {
|
|
38
98
|
needsWeb = true;
|
|
39
99
|
}
|
|
40
100
|
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { readFileSync, existsSync, mkdirSync, appendFileSync } from 'fs';
|
|
2
|
+
import { join } from 'path';
|
|
3
|
+
import { Config } from '../config.js';
|
|
4
|
+
|
|
5
|
+
const MAX_LINE = 2000;
|
|
6
|
+
|
|
7
|
+
function pad(n) {
|
|
8
|
+
return String(n).padStart(2, '0');
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
function dateStamp(d) {
|
|
12
|
+
return `${d.getFullYear()}${pad(d.getMonth() + 1)}${pad(d.getDate())}`;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
function timeStamp(d) {
|
|
16
|
+
return `${d.getFullYear()}-${pad(d.getMonth() + 1)}-${pad(d.getDate())} ${pad(d.getHours())}:${pad(d.getMinutes())}:${pad(d.getSeconds())}`;
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function cap(text) {
|
|
20
|
+
const s = String(text ?? '');
|
|
21
|
+
return s.length > MAX_LINE ? `${s.slice(0, MAX_LINE)}…` : s;
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
// Automatic activity logger. Writes one timestamped file per day into the
|
|
25
|
+
// config directory (logs/). A no-op until init() is called, so tests and
|
|
26
|
+
// library consumers never accidentally write into the real logs folder.
|
|
27
|
+
class Logger {
|
|
28
|
+
constructor() {
|
|
29
|
+
this.dir = null;
|
|
30
|
+
this.file = null;
|
|
31
|
+
this._initialized = false;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
init(opts = {}) {
|
|
35
|
+
if (this._initialized) return this;
|
|
36
|
+
this.dir = opts.baseDir ? join(opts.baseDir, 'logs') : join(Config.getConfigDir(), 'logs');
|
|
37
|
+
mkdirSync(this.dir, { recursive: true });
|
|
38
|
+
this.file = join(this.dir, `vierrataleai-${dateStamp(new Date())}.log`);
|
|
39
|
+
this._initialized = true;
|
|
40
|
+
return this;
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
log(event, message = '') {
|
|
44
|
+
if (!this._initialized) return '';
|
|
45
|
+
const line = `${timeStamp(new Date())} [${event}] ${cap(message)}`;
|
|
46
|
+
try {
|
|
47
|
+
appendFileSync(this.file, line + '\n');
|
|
48
|
+
} catch {}
|
|
49
|
+
return line;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
getPath() {
|
|
53
|
+
return this.file;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
tail(count = 30) {
|
|
57
|
+
if (!this._initialized || !existsSync(this.file)) return [];
|
|
58
|
+
try {
|
|
59
|
+
return readFileSync(this.file, 'utf-8').split('\n').filter(Boolean).slice(-count);
|
|
60
|
+
} catch {
|
|
61
|
+
return [];
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export { Logger };
|
|
67
|
+
export const logger = new Logger();
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
// utils/math.js - Deterministic, safe math evaluator.
|
|
2
|
+
// Parses arithmetic ( + - * / ^ % parentheses, decimals, scientific notation )
|
|
3
|
+
// WITHOUT using eval(), so it can never execute arbitrary code. Returns null
|
|
4
|
+
// when the input is not a clean math expression, so callers fall back to the
|
|
5
|
+
// model.
|
|
6
|
+
|
|
7
|
+
// Extract a bare arithmetic expression from a line of chat text.
|
|
8
|
+
export function extractMathExpression(text) {
|
|
9
|
+
const s = String(text || '').trim();
|
|
10
|
+
// Strip a leading question/math verb phrase if present.
|
|
11
|
+
const cleaned = s.replace(
|
|
12
|
+
/^(?:what(?:'s| is| are)|what's|compute|calculate|solve|evaluate|equals?)\b\s*/i,
|
|
13
|
+
''
|
|
14
|
+
).replace(/\??$/, '').trim();
|
|
15
|
+
const expr = cleaned.replace(/[?,!.]$/, '').trim();
|
|
16
|
+
if (!expr) return null;
|
|
17
|
+
// Guard: must be purely numeric + arithmetic operators + parens + spaces.
|
|
18
|
+
if (!/^[0-9+\-*\/^%().\s]+$/.test(expr)) return null;
|
|
19
|
+
if (!/\d/.test(expr)) return null;
|
|
20
|
+
if (/[a-zA-Z]/.test(expr)) return null;
|
|
21
|
+
// Must contain at least one operator (otherwise "2" alone isn't a math ask).
|
|
22
|
+
if (!/[\+\-\*\/\^%]/.test(expr)) return null;
|
|
23
|
+
return expr;
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Tokenize an expression into numbers and operators.
|
|
27
|
+
function tokenize(expr) {
|
|
28
|
+
const tokens = [];
|
|
29
|
+
let i = 0;
|
|
30
|
+
const n = expr.length;
|
|
31
|
+
while (i < n) {
|
|
32
|
+
const c = expr[i];
|
|
33
|
+
if (/\s/.test(c)) { i++; continue; }
|
|
34
|
+
if ('+-*/^%()'.includes(c)) {
|
|
35
|
+
tokens.push(c);
|
|
36
|
+
i++;
|
|
37
|
+
continue;
|
|
38
|
+
}
|
|
39
|
+
// read a number (int, decimal, scientific)
|
|
40
|
+
const num = expr.slice(i).match(/^(\d+(?:\.\d+)?|\.\d+)([eE][+-]?\d+)?/);
|
|
41
|
+
if (!num) return null;
|
|
42
|
+
tokens.push(parseFloat(num[0]));
|
|
43
|
+
i += num[0].length;
|
|
44
|
+
}
|
|
45
|
+
return tokens;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
// Recursive-descent parser with correct precedence:
|
|
49
|
+
// expr := term (('+'|'-') term)*
|
|
50
|
+
// term := power (('*'|'/'|'%') power)*
|
|
51
|
+
// power := unary ('^' power)? right-associative
|
|
52
|
+
// unary := ('+'|'-') unary | factor
|
|
53
|
+
// factor := '(' expr ')' | number
|
|
54
|
+
class Parser {
|
|
55
|
+
constructor(tokens) {
|
|
56
|
+
this.tokens = tokens;
|
|
57
|
+
this.pos = 0;
|
|
58
|
+
}
|
|
59
|
+
peek() { return this.tokens[this.pos]; }
|
|
60
|
+
next() { return this.tokens[this.pos++]; }
|
|
61
|
+
expect(val) {
|
|
62
|
+
const t = this.next();
|
|
63
|
+
if (t !== val) throw new Error('syntax error');
|
|
64
|
+
}
|
|
65
|
+
parseExpr() { return this.parseTerm(); }
|
|
66
|
+
parseTerm() {
|
|
67
|
+
let left = this.parseMul();
|
|
68
|
+
while (['+', '-'].includes(this.peek())) {
|
|
69
|
+
const op = this.next();
|
|
70
|
+
const right = this.parseMul();
|
|
71
|
+
left = op === '+' ? left + right : left - right;
|
|
72
|
+
}
|
|
73
|
+
return left;
|
|
74
|
+
}
|
|
75
|
+
parseMul() {
|
|
76
|
+
let left = this.parsePower();
|
|
77
|
+
while (['*', '/', '%'].includes(this.peek())) {
|
|
78
|
+
const op = this.next();
|
|
79
|
+
const right = this.parsePower();
|
|
80
|
+
if (op === '*') left = left * right;
|
|
81
|
+
else if (op === '/') left = left / right;
|
|
82
|
+
else left = left % right;
|
|
83
|
+
}
|
|
84
|
+
return left;
|
|
85
|
+
}
|
|
86
|
+
parsePower() {
|
|
87
|
+
let left = this.parseUnary();
|
|
88
|
+
if (this.peek() === '^') {
|
|
89
|
+
this.next();
|
|
90
|
+
const right = this.parsePower();
|
|
91
|
+
left = Math.pow(left, right);
|
|
92
|
+
}
|
|
93
|
+
return left;
|
|
94
|
+
}
|
|
95
|
+
parseUnary() {
|
|
96
|
+
if (this.peek() === '+') { this.next(); return this.parseUnary(); }
|
|
97
|
+
if (this.peek() === '-') {
|
|
98
|
+
this.next();
|
|
99
|
+
return -this.parseUnary();
|
|
100
|
+
}
|
|
101
|
+
return this.parseFactor();
|
|
102
|
+
}
|
|
103
|
+
parseFactor() {
|
|
104
|
+
const t = this.peek();
|
|
105
|
+
if (t === '(') {
|
|
106
|
+
this.next();
|
|
107
|
+
const v = this.parseExpr();
|
|
108
|
+
this.expect(')');
|
|
109
|
+
return v;
|
|
110
|
+
}
|
|
111
|
+
if (typeof t === 'number') {
|
|
112
|
+
this.next();
|
|
113
|
+
return t;
|
|
114
|
+
}
|
|
115
|
+
throw new Error('syntax error');
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
export function evaluateMath(expr) {
|
|
120
|
+
try {
|
|
121
|
+
const tokens = tokenize(expr);
|
|
122
|
+
if (!tokens || !tokens.length) return null;
|
|
123
|
+
const p = new Parser(tokens);
|
|
124
|
+
const result = p.parseExpr();
|
|
125
|
+
if (p.pos !== tokens.length) return null; // leftover tokens = not a clean expr
|
|
126
|
+
if (!Number.isFinite(result)) return null;
|
|
127
|
+
// Round to avoid floating point noise like 0.30000000000000004
|
|
128
|
+
const rounded = Math.round((result + Number.EPSILON) * 1e10) / 1e10;
|
|
129
|
+
return Number(rounded.toPrecision(12));
|
|
130
|
+
} catch {
|
|
131
|
+
return null;
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// Front-door: given a chat line, return a clean math answer or null.
|
|
136
|
+
export function trySolveMath(text) {
|
|
137
|
+
const expr = extractMathExpression(text);
|
|
138
|
+
if (!expr) return null;
|
|
139
|
+
const value = evaluateMath(expr);
|
|
140
|
+
if (value === null) return null;
|
|
141
|
+
return `${expr} = ${value}`;
|
|
142
|
+
}
|
package/src/utils/webfetch.js
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
-
const UA =
|
|
1
|
+
const UA =
|
|
2
|
+
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 ' +
|
|
3
|
+
'(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36';
|
|
2
4
|
|
|
3
5
|
const MAX_BYTES = 200000; // 200KB cap
|
|
4
6
|
const MAX_TEXT = 8000; // ~8k chars of readable text
|
|
@@ -8,6 +10,40 @@ const BLOCK_TAGS = new Set([
|
|
|
8
10
|
'nav', 'footer', 'aside', 'iframe', 'form', 'button', 'noscript',
|
|
9
11
|
]);
|
|
10
12
|
|
|
13
|
+
// Body markers that indicate an anti-bot wall instead of real content.
|
|
14
|
+
const BLOCK_MARKERS = [
|
|
15
|
+
'just a moment',
|
|
16
|
+
'client challenge',
|
|
17
|
+
'checking your browser',
|
|
18
|
+
'enable javascript and cookies',
|
|
19
|
+
'cf-chl',
|
|
20
|
+
'challenge-platform',
|
|
21
|
+
'attention required!',
|
|
22
|
+
'powershell install',
|
|
23
|
+
];
|
|
24
|
+
|
|
25
|
+
// Browser-like headers so header-based bot checks see a real browser.
|
|
26
|
+
function browserHeaders(accept = 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8') {
|
|
27
|
+
return {
|
|
28
|
+
'User-Agent': UA,
|
|
29
|
+
Accept: accept,
|
|
30
|
+
'Accept-Language': 'en-US,en;q=0.9',
|
|
31
|
+
Connection: 'keep-alive',
|
|
32
|
+
'Upgrade-Insecure-Requests': '1',
|
|
33
|
+
'Sec-Fetch-Dest': accept.startsWith('application/json') ? 'empty' : 'document',
|
|
34
|
+
'Sec-Fetch-Mode': accept.startsWith('application/json') ? 'cors' : 'navigate',
|
|
35
|
+
'Sec-Fetch-Site': 'none',
|
|
36
|
+
'Sec-Fetch-User': accept.startsWith('application/json') ? '?0' : '?1',
|
|
37
|
+
'Cache-Control': 'no-cache',
|
|
38
|
+
Pragma: 'no-cache',
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
export function isBlockedBody(html) {
|
|
43
|
+
const s = String(html || '').toLowerCase().slice(0, 4000);
|
|
44
|
+
return BLOCK_MARKERS.some((m) => s.includes(m));
|
|
45
|
+
}
|
|
46
|
+
|
|
11
47
|
export class WebFetch {
|
|
12
48
|
static normalizeUrl(input) {
|
|
13
49
|
const trimmed = (input || '').trim();
|
|
@@ -34,10 +70,45 @@ export class WebFetch {
|
|
|
34
70
|
throw new Error('Invalid URL. Provide a valid http(s) address.');
|
|
35
71
|
}
|
|
36
72
|
|
|
73
|
+
let finalUrl = url;
|
|
74
|
+
let html = '';
|
|
75
|
+
let blocked = false;
|
|
76
|
+
try {
|
|
77
|
+
const res = await this._fetchHtml(url);
|
|
78
|
+
finalUrl = res.url;
|
|
79
|
+
html = res.html;
|
|
80
|
+
const status = res.status;
|
|
81
|
+
blocked = status !== 200 || isBlockedBody(html);
|
|
82
|
+
} catch (err) {
|
|
83
|
+
blocked = true;
|
|
84
|
+
// _fetchHtml errors are network/cert/PROTOCOL issues; a fallback may
|
|
85
|
+
// still succeed (e.g. registry API is reachable even when the HTML
|
|
86
|
+
// page's CDN is not).
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Bot-walled npm/pypi pages degrade to their open JSON API.
|
|
90
|
+
if (blocked) {
|
|
91
|
+
const registry = await this._registryFallback(finalUrl || url, maxText);
|
|
92
|
+
if (registry) return registry;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
if (blocked) {
|
|
96
|
+
const status = 'blocked';
|
|
97
|
+
throw new Error(`Request blocked (${status}) for ${finalUrl || url}`);
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
return {
|
|
101
|
+
url: finalUrl,
|
|
102
|
+
title: this._extractTitle(html),
|
|
103
|
+
text: this._extractText(html).slice(0, maxText),
|
|
104
|
+
};
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
static async _fetchHtml(url) {
|
|
37
108
|
let resp;
|
|
38
109
|
try {
|
|
39
110
|
resp = await fetch(url, {
|
|
40
|
-
headers:
|
|
111
|
+
headers: browserHeaders(),
|
|
41
112
|
redirect: 'follow',
|
|
42
113
|
signal: AbortSignal.timeout(15000),
|
|
43
114
|
});
|
|
@@ -45,11 +116,6 @@ export class WebFetch {
|
|
|
45
116
|
throw new Error(`Could not reach ${url}`);
|
|
46
117
|
}
|
|
47
118
|
|
|
48
|
-
if (!resp.ok) {
|
|
49
|
-
throw new Error(`Request failed (${resp.status}) for ${url}`);
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
// Refuse to read the whole body past the cap.
|
|
53
119
|
const reader = resp.body.getReader();
|
|
54
120
|
const decoder = new TextDecoder('utf-8', { fatal: false });
|
|
55
121
|
let html = '';
|
|
@@ -63,13 +129,95 @@ export class WebFetch {
|
|
|
63
129
|
reader.cancel();
|
|
64
130
|
html += decoder.decode();
|
|
65
131
|
|
|
132
|
+
return { url: resp.url || url, html, status: resp.status };
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
// --- Registry fallback (npm + PyPI) via their open JSON APIs ---------- //
|
|
136
|
+
|
|
137
|
+
static async _registryFallback(url, maxText) {
|
|
138
|
+
const npm = this._npmSpec(url);
|
|
139
|
+
if (npm) {
|
|
140
|
+
try {
|
|
141
|
+
return await this._npmFallback(npm, url, maxText);
|
|
142
|
+
} catch {}
|
|
143
|
+
}
|
|
144
|
+
const pypi = this._pypiSpec(url);
|
|
145
|
+
if (pypi) {
|
|
146
|
+
try {
|
|
147
|
+
return await this._pypiFallback(pypi, url, maxText);
|
|
148
|
+
} catch {}
|
|
149
|
+
}
|
|
150
|
+
return null;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
static _npmSpec(url) {
|
|
154
|
+
const m = /npmjs\.\w+\/package\/([@a-zA-Z0-9._~-]+(?:\/[@a-zA-Z0-9._~-]+)?)/i.exec(url);
|
|
155
|
+
return m ? m[1] : null;
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
static _pypiSpec(url) {
|
|
159
|
+
const m = /pypi\.org\/project\/([a-zA-Z0-9._-]+)/i.exec(url);
|
|
160
|
+
return m ? m[1] : null;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
static async _fetchJson(apiUrl) {
|
|
164
|
+
const resp = await fetch(apiUrl, {
|
|
165
|
+
headers: browserHeaders('application/json'),
|
|
166
|
+
redirect: 'follow',
|
|
167
|
+
signal: AbortSignal.timeout(15000),
|
|
168
|
+
});
|
|
169
|
+
if (!resp.ok) throw new Error(`API ${resp.status}`);
|
|
170
|
+
return resp.json();
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
static async _npmFallback(spec, pageUrl, maxText) {
|
|
174
|
+
const pkg = encodeURIComponent(spec); // '@vierratale/ai' -> %40vierratale%2Fai
|
|
175
|
+
const data = await this._fetchJson(`https://registry.npmjs.org/${pkg}`);
|
|
176
|
+
const version =
|
|
177
|
+
(data['dist-tags'] && (data['dist-tags'].latest || data['dist-tags'].beta)) ||
|
|
178
|
+
Object.keys(data.versions || {})[0];
|
|
179
|
+
const meta = (data.versions && data.versions[version]) || {};
|
|
180
|
+
const deps = meta.dependencies || data.dependencies || {};
|
|
181
|
+
const lines = [
|
|
182
|
+
`${data.name} (npm package)`,
|
|
183
|
+
`Latest version: ${version}`,
|
|
184
|
+
meta.description || data.description || '',
|
|
185
|
+
`License: ${meta.license || data.license || 'N/A'}`,
|
|
186
|
+
`Dependencies: ${Object.keys(deps).length}`,
|
|
187
|
+
'',
|
|
188
|
+
];
|
|
189
|
+
const readme = String(data.readme || '').trim();
|
|
190
|
+
if (readme) lines.push(readme.slice(0, maxText));
|
|
66
191
|
return {
|
|
67
|
-
url:
|
|
68
|
-
title:
|
|
69
|
-
text:
|
|
192
|
+
url: pageUrl,
|
|
193
|
+
title: `${data.name} · npm`,
|
|
194
|
+
text: lines.filter(Boolean).join('\n').slice(0, maxText),
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
static async _pypiFallback(spec, pageUrl, maxText) {
|
|
199
|
+
const data = await this._fetchJson(`https://pypi.org/pypi/${encodeURIComponent(spec)}/json`);
|
|
200
|
+
const info = data.info || {};
|
|
201
|
+
const lines = [
|
|
202
|
+
`${info.name} (PyPI package)`,
|
|
203
|
+
`Version: ${info.version}`,
|
|
204
|
+
info.summary || '',
|
|
205
|
+
`Requires Python: ${info.requires_python || 'N/A'}`,
|
|
206
|
+
`License: ${info.license || 'N/A'}`,
|
|
207
|
+
`Home: ${info.home_page || info.project_url || 'N/A'}`,
|
|
208
|
+
'',
|
|
209
|
+
];
|
|
210
|
+
const desc = String(info.description || '').trim();
|
|
211
|
+
if (desc) lines.push(desc.slice(0, maxText));
|
|
212
|
+
return {
|
|
213
|
+
url: pageUrl,
|
|
214
|
+
title: `${info.name} · PyPI`,
|
|
215
|
+
text: lines.filter(Boolean).join('\n').slice(0, maxText),
|
|
70
216
|
};
|
|
71
217
|
}
|
|
72
218
|
|
|
219
|
+
// --- extraction -------------------------------------------------------- //
|
|
220
|
+
|
|
73
221
|
static _extractTitle(html) {
|
|
74
222
|
const m = /<title[^>]*>([^<]*)<\/title>/i.exec(html);
|
|
75
223
|
return m ? this._stripEntities(m[1].trim()) : '';
|
|
@@ -112,4 +260,4 @@ export class WebFetch {
|
|
|
112
260
|
.replace(/\s+/g, ' ')
|
|
113
261
|
.trim();
|
|
114
262
|
}
|
|
115
|
-
}
|
|
263
|
+
}
|
package/build.sh
DELETED
|
@@ -1,17 +0,0 @@
|
|
|
1
|
-
#!/bin/bash
|
|
2
|
-
# Build VierrataleAI Node.js version for packaging
|
|
3
|
-
set -e
|
|
4
|
-
|
|
5
|
-
cd "$(dirname "$0")"
|
|
6
|
-
|
|
7
|
-
echo "Installing dependencies..."
|
|
8
|
-
npm install
|
|
9
|
-
|
|
10
|
-
echo "Testing..."
|
|
11
|
-
node bin/vierrataleai.js --version
|
|
12
|
-
|
|
13
|
-
echo "Packaging..."
|
|
14
|
-
npm pack
|
|
15
|
-
|
|
16
|
-
echo "Build complete!"
|
|
17
|
-
echo "To publish: npm publish --tag beta"
|