@vierratale/ai 0.1.0-beta.10 → 0.1.0-beta.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/cli.js +52 -1
- package/src/cmd/agent.js +117 -29
- package/src/cmd/todos.js +64 -0
- package/src/cmd/tools.js +34 -0
- package/src/ui/branding.js +1 -1
- package/src/utils/webfetch.js +380 -45
package/package.json
CHANGED
package/src/cli.js
CHANGED
|
@@ -18,6 +18,7 @@ import { Session } from './session.js';
|
|
|
18
18
|
import { ChatUI, FrameThrottle } from './ui/chatbox.js';
|
|
19
19
|
import { logger } from './utils/logger.js';
|
|
20
20
|
import { runWithTools, TOOL_RESULTS_PROMPT, looksLikeOperationRequest } from './cmd/agent.js';
|
|
21
|
+
import { loadTodos, addTodo, updateTodo, clearTodos, saveTodos, formatTodos, todosFile } from './cmd/todos.js';
|
|
21
22
|
|
|
22
23
|
function parseArgs(args) {
|
|
23
24
|
const parsed = { provider: null, model: null, clear: false, version: false, help: false, install: false };
|
|
@@ -54,6 +55,7 @@ Commands (in chat):
|
|
|
54
55
|
/search <query> Search the web and summarize with AI
|
|
55
56
|
/fetch <url> Open a link and summarize its content
|
|
56
57
|
/download <url> Download a file, folder listing or archive
|
|
58
|
+
/todos List the todo list (add <text> | done/undone <n> | clear)
|
|
57
59
|
/log [n|path] Show the last n log entries (default 30), or the log file path
|
|
58
60
|
/clear Clear screen
|
|
59
61
|
/new Start a new empty session (keeps old ones)
|
|
@@ -79,6 +81,7 @@ export const SLASH_COMMANDS = [
|
|
|
79
81
|
{ name: 'fetch', args: '<url>', desc: 'Open a link and summarize' },
|
|
80
82
|
{ name: 'download', args: '<url>', desc: 'Download a file / folder / archive' },
|
|
81
83
|
{ name: 'system', desc: 'Show or set the system prompt' },
|
|
84
|
+
{ name: 'todos', args: '[add <text> | done <n> | undone <n> | clear]', desc: 'Manage the todo list' },
|
|
82
85
|
{ name: 'sessions', desc: 'List saved sessions' },
|
|
83
86
|
{ name: 'session', args: 'new | open | delete | backup | restore', desc: 'Manage sessions' },
|
|
84
87
|
{ name: 'log', args: '[n | path]', desc: 'Show recent log entries' },
|
|
@@ -656,7 +659,7 @@ async function chat(provider, systemPrompt) {
|
|
|
656
659
|
if (cmd === '/help') {
|
|
657
660
|
chatUI.clearNotices();
|
|
658
661
|
chatUI.notify(
|
|
659
|
-
`Commands:\n /search <query> · /fetch <url> · /model [name] · /provider [name]\n /models · /log [n|path] · /new · /sessions · /session new|open|delete|backup|restore\n /clear · /help · /quit`
|
|
662
|
+
`Commands:\n /search <query> · /fetch <url> · /model [name] · /provider [name]\n /models · /todos · /log [n|path] · /new · /sessions · /session new|open|delete|backup|restore\n /clear · /help · /quit`
|
|
660
663
|
);
|
|
661
664
|
chatUI.render(messages);
|
|
662
665
|
continue;
|
|
@@ -977,6 +980,11 @@ async function chat(provider, systemPrompt) {
|
|
|
977
980
|
continue;
|
|
978
981
|
}
|
|
979
982
|
|
|
983
|
+
if (cmd === '/todos' || cmd.startsWith('/todos ')) {
|
|
984
|
+
handleTodosCommand(trimmed, chatUI, messages, process.cwd());
|
|
985
|
+
continue;
|
|
986
|
+
}
|
|
987
|
+
|
|
980
988
|
const intent = detectIntent(trimmed);
|
|
981
989
|
if (intent.type === 'self') {
|
|
982
990
|
const model = Config.getEffectiveModel(provider.name);
|
|
@@ -1125,6 +1133,49 @@ async function ensureRawLocalModel(realModel, notify) {
|
|
|
1125
1133
|
export { looksLikeFileRequest, requestedFileName, extractFallbackFile, storyFileName, responseWantsFile, responseIsSoloCode, languageFileName, fallbackHasUsefulCode, validateWrittenCode, looksLikeJunkCode, warnIfJunkCode, writeFilesFromResponse };
|
|
1126
1134
|
// Command/tool layer: asks the model to plan filesystem/shell operations,
|
|
1127
1135
|
// executes them with the sandboxed executor, then streams the final answer.
|
|
1136
|
+
function handleTodosCommand(trimmed, chatUI, messages, cwd) {
|
|
1137
|
+
chatUI.clearNotices();
|
|
1138
|
+
const rest = trimmed.slice('/todos'.length).trim();
|
|
1139
|
+
let out;
|
|
1140
|
+
if (!rest) {
|
|
1141
|
+
out = `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1142
|
+
} else if (/^add\s+/i.test(rest)) {
|
|
1143
|
+
const err = addTodo(cwd, rest.replace(/^add\s+/i, '').trim());
|
|
1144
|
+
out = err && err.error ? err.error : `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1145
|
+
} else if (/^done\s+\d+$/i.test(rest)) {
|
|
1146
|
+
const err = updateTodo(cwd, parseInt(rest.replace(/^done\s+/i, ''), 10), { done: true });
|
|
1147
|
+
out = err && err.error ? err.error : `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1148
|
+
} else if (/^undone\s+\d+$/i.test(rest)) {
|
|
1149
|
+
const err = updateTodo(cwd, parseInt(rest.replace(/^undone\s+/i, ''), 10), { done: false });
|
|
1150
|
+
out = err && err.error ? err.error : `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1151
|
+
} else if (/^remove\s+\d+$/i.test(rest)) {
|
|
1152
|
+
const n = parseInt(rest.replace(/^remove\s+/i, ''), 10);
|
|
1153
|
+
const todos = loadTodos(cwd);
|
|
1154
|
+
if (!Number.isInteger(n - 1) || n - 1 < 0 || n - 1 >= todos.length) {
|
|
1155
|
+
out = `No todo #${n}.`;
|
|
1156
|
+
} else {
|
|
1157
|
+
const next = todos.filter((_, i) => i !== n - 1);
|
|
1158
|
+
saveTodos(cwd, next);
|
|
1159
|
+
out = `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1160
|
+
}
|
|
1161
|
+
} else if (/^clear$/i.test(rest)) {
|
|
1162
|
+
clearTodos(cwd);
|
|
1163
|
+
out = `Todos cleared (${todosFile(cwd)}). (no todos yet)`;
|
|
1164
|
+
} else if (/^\d+$/.test(rest)) {
|
|
1165
|
+
const n = parseInt(rest, 10);
|
|
1166
|
+
const todos = loadTodos(cwd);
|
|
1167
|
+
const err = (Number.isInteger(n - 1) && n - 1 >= 0 && n - 1 < todos.length)
|
|
1168
|
+
? updateTodo(cwd, n, { done: !todos[n - 1].done })
|
|
1169
|
+
: { error: `No todo #${n}.` };
|
|
1170
|
+
out = err && err.error ? err.error : `Todos (${todosFile(cwd)}):\n${formatTodos(loadTodos(cwd))}`;
|
|
1171
|
+
} else {
|
|
1172
|
+
out = 'Usage: /todos [add <text> | done <n> | undone <n> | remove <n> | <n> (toggle) | clear]';
|
|
1173
|
+
}
|
|
1174
|
+
chatUI.notify(out);
|
|
1175
|
+
logger.log('CMD', `todos: ${rest || '(list)'}`);
|
|
1176
|
+
chatUI.render(messages);
|
|
1177
|
+
}
|
|
1178
|
+
|
|
1128
1179
|
async function answerWithTools(messages, provider, requestText, systemPrompt, confirm, chatUI, cwd, timeoutMs) {
|
|
1129
1180
|
let results = [];
|
|
1130
1181
|
try {
|
package/src/cmd/agent.js
CHANGED
|
@@ -19,6 +19,10 @@ Reply with EXACTLY ONE JSON array of the tool calls needed to do the job, in ord
|
|
|
19
19
|
{"name":"list_directory","path":"."}
|
|
20
20
|
{"name":"delete_file","path":"old.txt"}
|
|
21
21
|
{"name":"download_url","url":"https://example.com/file.zip","dir":"downloads"}
|
|
22
|
+
{"name":"todo_add","text":"install dependencies"}
|
|
23
|
+
{"name":"todo_list"}
|
|
24
|
+
{"name":"todo_update","index":1,"done":true}
|
|
25
|
+
{"name":"todo_clear"}
|
|
22
26
|
|
|
23
27
|
Rules:
|
|
24
28
|
- run_command runs WITHOUT a shell. Allowed commands: mkdir touch ls pwd cat cp mv rm npm npx node python python3 git echo printf.
|
|
@@ -27,6 +31,7 @@ Rules:
|
|
|
27
31
|
- Prefer create_file over redirecting output.
|
|
28
32
|
- If you need to create a folder first, plan create_directory before create_file inside it.
|
|
29
33
|
- To download a file, web asset, archive or folder listing from the web, use download_url with the full URL. dir is optional (defaults to ~/Downloads).
|
|
34
|
+
- For multi-step work, track your progress with todo_add/todo_list/todo_update (e.g. todo_add for each remaining subtask, todo_update to mark steps done).
|
|
30
35
|
- Code you create must be runnable. For Python servers bind to 127.0.0.1, read PORT from the environment (default 8000), and fall back to a free port (port 0) when the default is taken. HTML/CSS/JS must be self-contained and reference each other by relative filename.
|
|
31
36
|
- If NO tools are needed, reply exactly: NONE
|
|
32
37
|
- Output the JSON array only — no prose, no markdown fences unless the array is inside them.`;
|
|
@@ -37,6 +42,20 @@ ${resultsJson}
|
|
|
37
42
|
|
|
38
43
|
Give the user a short, natural reply (in the user's own language) telling them what happened. Keep the whole reply under 40 words. Do NOT tell them to run the commands yourself. If they asked for a file containing code and files were created with content, you may also emit FILE: blocks if more files are still required.`;
|
|
39
44
|
|
|
45
|
+
// Auto error fixing: shown to the model when a tool call fails so it can emit
|
|
46
|
+
// a corrected plan instead of giving up. Bounded by MAX_AUTO_FIXES per run.
|
|
47
|
+
const MAX_AUTO_FIXES = 3;
|
|
48
|
+
|
|
49
|
+
const TOOL_FIX_PROMPT = (call, result) => `A tool call made while working in the workspace FAILED. Fix the problem so the original request can complete.
|
|
50
|
+
|
|
51
|
+
Failed tool call:
|
|
52
|
+
${JSON.stringify(call, null, 2)}
|
|
53
|
+
|
|
54
|
+
Failure:
|
|
55
|
+
${JSON.stringify({ success: false, exitCode: result.exitCode, stdout: result.stdout, stderr: result.stderr }, null, 2)}
|
|
56
|
+
|
|
57
|
+
Emit a NEW tool plan (a JSON array of tool calls) that corrects the failure — for example re-create the file with the error fixed, run a command to repair it, or download/read what is missing. If your previous plan skipped a required step (like creating a parent folder), include it. Use todo tools to track remaining subtasks if useful. Reply with exactly one JSON array, or the single word NONE if no tool call can fix this.`;
|
|
58
|
+
|
|
40
59
|
// ---------------------------------------------------------------------------
|
|
41
60
|
// Request detection
|
|
42
61
|
// ---------------------------------------------------------------------------
|
|
@@ -84,6 +103,18 @@ export function normalizeToolCall(raw) {
|
|
|
84
103
|
} else if (name === 'download_url') {
|
|
85
104
|
if (typeof raw.url !== 'string' || raw.url.trim() === '') return null;
|
|
86
105
|
params = { url: raw.url, dir: typeof raw.dir === 'string' && raw.dir.trim() ? raw.dir.trim() : undefined };
|
|
106
|
+
} else if (name === 'todo_add') {
|
|
107
|
+
if (typeof raw.text !== 'string' || raw.text.trim() === '') return null;
|
|
108
|
+
params = { text: raw.text };
|
|
109
|
+
} else if (name === 'todo_update') {
|
|
110
|
+
if (!Number.isInteger(raw.index) && !/^\d+$/.test(String(raw.index ?? ''))) return null;
|
|
111
|
+
params = {
|
|
112
|
+
index: Number(raw.index),
|
|
113
|
+
done: typeof raw.done === 'boolean' ? raw.done : undefined,
|
|
114
|
+
text: typeof raw.text === 'string' && raw.text.trim() ? raw.text : undefined,
|
|
115
|
+
};
|
|
116
|
+
} else if (name === 'todo_list' || name === 'todo_clear') {
|
|
117
|
+
params = {};
|
|
87
118
|
} else {
|
|
88
119
|
return null;
|
|
89
120
|
}
|
|
@@ -1631,6 +1662,75 @@ export async function planOnce(provider, messages, requestText, options = {}) {
|
|
|
1631
1662
|
return heuristic;
|
|
1632
1663
|
}
|
|
1633
1664
|
|
|
1665
|
+
async function executeToolCall({ executor, call, chatUI, messages, cwd }) {
|
|
1666
|
+
const label = call.name === 'run_command' ? call.params.command
|
|
1667
|
+
: call.name === 'download_url' ? `download_url(${call.params.url})`
|
|
1668
|
+
: (call.name === 'todo_add' || call.name === 'todo_update')
|
|
1669
|
+
? `${call.name}(${JSON.stringify(call.params ?? {})})`
|
|
1670
|
+
: `${call.name}(${call.params.path})`;
|
|
1671
|
+
if (chatUI) {
|
|
1672
|
+
chatUI.setStatus(`Running: ${label}`);
|
|
1673
|
+
chatUI.setThinking(true);
|
|
1674
|
+
chatUI.render(messages);
|
|
1675
|
+
}
|
|
1676
|
+
const result = await dispatchTool(executor, call);
|
|
1677
|
+
logger.log('EXEC', `${label} -> ${result.exitCode}${result.timedOut ? ' [timeout]' : ''}`);
|
|
1678
|
+
if (result.success) {
|
|
1679
|
+
if (chatUI) chatUI.notify(`\u2713 ${label}`);
|
|
1680
|
+
if (result.stdout.trim()) logger.log('EXEC', `${label} stdout: ${result.stdout.trim().slice(0, 500)}`);
|
|
1681
|
+
if (result.stderr.trim()) logger.log('WARN', `${label} stderr: ${result.stderr.trim().slice(0, 500)}`);
|
|
1682
|
+
if (call.name === 'create_file' && /\.py$/i.test(call.params.path)) {
|
|
1683
|
+
const check = spawnSync('python3', [
|
|
1684
|
+
'-c', 'import ast,sys; ast.parse(open(sys.argv[1], encoding="utf-8").read())',
|
|
1685
|
+
join(cwd || '.', call.params.path),
|
|
1686
|
+
], { encoding: 'utf-8', timeout: 30000 });
|
|
1687
|
+
if (check.status === 1) {
|
|
1688
|
+
const detail = (check.stderr || '').trim().split('\n').slice(-2).join(' ');
|
|
1689
|
+
if (chatUI) chatUI.notify(`\u2717 python check failed: ${call.params.path} (${detail})`);
|
|
1690
|
+
logger.log('WARN', `${label} failed python syntax check: ${detail}`);
|
|
1691
|
+
}
|
|
1692
|
+
}
|
|
1693
|
+
} else {
|
|
1694
|
+
if (chatUI) chatUI.notify(`\u2717 ${label}: ${(result.stderr || 'failed').trim().slice(0, 300)}`);
|
|
1695
|
+
logger.log('ERROR', `${label} failed: ${(result.stderr || result.stdout || 'exit ' + result.exitCode).trim().slice(0, 500)}`);
|
|
1696
|
+
}
|
|
1697
|
+
return { label, result };
|
|
1698
|
+
}
|
|
1699
|
+
|
|
1700
|
+
// Ask the model for a corrected tool plan after a failed call. Returns []
|
|
1701
|
+
// when the model emits nothing usable. Does NOT mutate the session messages.
|
|
1702
|
+
async function planAutoFix(provider, messages, requestText, call, result) {
|
|
1703
|
+
const history = (messages && messages.length ? messages.slice(-6) : [{ role: 'user', content: String(requestText || '') }]);
|
|
1704
|
+
const deadline = Config.get('planTimeoutMs');
|
|
1705
|
+
const controller = new AbortController();
|
|
1706
|
+
let text = '';
|
|
1707
|
+
let timer;
|
|
1708
|
+
try {
|
|
1709
|
+
text = await Promise.race([
|
|
1710
|
+
provider.complete([...history, { role: 'user', content: TOOL_FIX_PROMPT(call, result) }], {
|
|
1711
|
+
model: Config.getEffectiveModel(provider.name),
|
|
1712
|
+
system_prompt: TOOL_PLAN_PROMPT,
|
|
1713
|
+
maxTokens: 400,
|
|
1714
|
+
signal: controller.signal,
|
|
1715
|
+
}),
|
|
1716
|
+
new Promise((_resolve, reject) => {
|
|
1717
|
+
timer = setTimeout(() => {
|
|
1718
|
+
if (!controller.signal.aborted) controller.abort();
|
|
1719
|
+
reject(Object.assign(new Error(`auto-fix exceeded ${deadline}ms`), { name: 'DeadlineError' }));
|
|
1720
|
+
}, deadline);
|
|
1721
|
+
}),
|
|
1722
|
+
]);
|
|
1723
|
+
} catch (err) {
|
|
1724
|
+
logger.log('FIX', `auto-fix plan failed: ${err.message}`);
|
|
1725
|
+
return [];
|
|
1726
|
+
} finally {
|
|
1727
|
+
clearTimeout(timer);
|
|
1728
|
+
}
|
|
1729
|
+
const plan = parseToolPlan(text);
|
|
1730
|
+
if (plan.length) logger.log('FIX', `auto-fix plan: ${plan.map((p) => p.name).join(', ')}`);
|
|
1731
|
+
return plan;
|
|
1732
|
+
}
|
|
1733
|
+
|
|
1634
1734
|
export async function runWithTools({
|
|
1635
1735
|
provider,
|
|
1636
1736
|
messages,
|
|
@@ -1648,6 +1748,7 @@ export async function runWithTools({
|
|
|
1648
1748
|
const executor = makeExecutor({ cwd, timeoutMs });
|
|
1649
1749
|
const results = [];
|
|
1650
1750
|
const plannedByModel = [];
|
|
1751
|
+
let autoFixes = 0;
|
|
1651
1752
|
|
|
1652
1753
|
for (let round = 0; round < MAX_PLANNING_ROUNDS; round++) {
|
|
1653
1754
|
let plan;
|
|
@@ -1662,38 +1763,25 @@ export async function runWithTools({
|
|
|
1662
1763
|
const steps = Math.min(plan.length, MAX_STEPS - results.length);
|
|
1663
1764
|
let allOk = true;
|
|
1664
1765
|
for (const call of plan.slice(0, steps)) {
|
|
1665
|
-
const
|
|
1666
|
-
: call.name === 'download_url' ? `download_url(${call.params.url})`
|
|
1667
|
-
: `${call.name}(${call.params.path})`;
|
|
1668
|
-
if (chatUI) {
|
|
1669
|
-
chatUI.setStatus(`Running: ${label}`);
|
|
1670
|
-
chatUI.setThinking(true);
|
|
1671
|
-
chatUI.render(messages);
|
|
1672
|
-
}
|
|
1673
|
-
const result = await dispatchTool(executor, call);
|
|
1766
|
+
const { result } = await executeToolCall({ executor, call, chatUI, messages, cwd });
|
|
1674
1767
|
results.push({ name: call.name, params: call.params, result });
|
|
1675
|
-
|
|
1676
|
-
if (result.success)
|
|
1677
|
-
|
|
1678
|
-
|
|
1679
|
-
|
|
1680
|
-
if (
|
|
1681
|
-
|
|
1682
|
-
|
|
1683
|
-
|
|
1684
|
-
|
|
1685
|
-
|
|
1686
|
-
|
|
1687
|
-
|
|
1688
|
-
logger.log('WARN', `${label} failed python syntax check: ${detail}`);
|
|
1689
|
-
}
|
|
1768
|
+
if (!result.success) allOk = false;
|
|
1769
|
+
if (result.timedOut && !result.success) break;
|
|
1770
|
+
|
|
1771
|
+
if (!result.success && autoFixes < MAX_AUTO_FIXES && results.length < MAX_STEPS) {
|
|
1772
|
+
autoFixes++;
|
|
1773
|
+
if (chatUI) chatUI.setStatus('Auto-fixing failed step…');
|
|
1774
|
+
const fixPlan = await planAutoFix(provider, messages, requestText, call, result);
|
|
1775
|
+
const fixSteps = Math.min(fixPlan.length, MAX_STEPS - results.length);
|
|
1776
|
+
let fixedOk = fixPlan.length > 0;
|
|
1777
|
+
for (const fcall of fixPlan.slice(0, fixSteps)) {
|
|
1778
|
+
const f = await executeToolCall({ executor, call: fcall, chatUI, messages, cwd });
|
|
1779
|
+
results.push({ name: fcall.name, params: fcall.params, result: f.result });
|
|
1780
|
+
if (!f.result.success) { fixedOk = false; break; }
|
|
1690
1781
|
}
|
|
1691
|
-
|
|
1692
|
-
allOk =
|
|
1693
|
-
if (chatUI) chatUI.notify(`\u2717 ${label}: ${(result.stderr || 'failed').trim().slice(0, 300)}`);
|
|
1694
|
-
logger.log('ERROR', `${label} failed: ${(result.stderr || result.stdout || 'exit ' + result.exitCode).trim().slice(0, 500)}`);
|
|
1782
|
+
if (fixedOk && chatUI) chatUI.notify('Auto-fixed after failure.');
|
|
1783
|
+
if (fixedOk) allOk = true;
|
|
1695
1784
|
}
|
|
1696
|
-
if (result.timedOut && !result.success) break;
|
|
1697
1785
|
}
|
|
1698
1786
|
if (allOk || !results.length || round >= MAX_PLANNING_ROUNDS - 1) break;
|
|
1699
1787
|
}
|
package/src/cmd/todos.js
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import { readFileSync, writeFileSync } from 'fs';
|
|
2
|
+
import { join } from 'path';
|
|
3
|
+
|
|
4
|
+
const FILE_NAME = '.todos.json';
|
|
5
|
+
|
|
6
|
+
export function todosFile(cwd) {
|
|
7
|
+
return join(cwd || '.', FILE_NAME);
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export function loadTodos(cwd) {
|
|
11
|
+
try {
|
|
12
|
+
const data = JSON.parse(readFileSync(todosFile(cwd), 'utf-8'));
|
|
13
|
+
if (!Array.isArray(data)) return [];
|
|
14
|
+
return data.map((t, i) => {
|
|
15
|
+
if (t && typeof t === 'object') {
|
|
16
|
+
return {
|
|
17
|
+
text: String(t.text ?? '').trim(),
|
|
18
|
+
done: Boolean(t.done),
|
|
19
|
+
createdAt: t.createdAt ?? null,
|
|
20
|
+
updatedAt: t.updatedAt ?? null,
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
return { text: String(t ?? '').trim(), done: false, createdAt: null, updatedAt: null };
|
|
24
|
+
}).filter((t) => t.text.length > 0);
|
|
25
|
+
} catch {
|
|
26
|
+
return [];
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export function saveTodos(cwd, todos) {
|
|
31
|
+
writeFileSync(todosFile(cwd), JSON.stringify(todos, null, 2) + '\n', 'utf-8');
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function formatTodos(todos) {
|
|
35
|
+
if (!todos || !todos.length) return '(no todos yet)';
|
|
36
|
+
return todos.map((t, i) => `${i + 1}. ${t.done ? '[x]' : '[ ]'} ${t.text}`).join('\n');
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export function addTodo(cwd, text) {
|
|
40
|
+
const clean = String(text ?? '').trim();
|
|
41
|
+
if (!clean) return { error: 'Missing todo text.' };
|
|
42
|
+
const todos = loadTodos(cwd);
|
|
43
|
+
todos.push({ text: clean, done: false, createdAt: new Date().toISOString(), updatedAt: null });
|
|
44
|
+
saveTodos(cwd, todos);
|
|
45
|
+
return { okay: true };
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function updateTodo(cwd, index, { done, text } = {}) {
|
|
49
|
+
const todos = loadTodos(cwd);
|
|
50
|
+
const i = Number(index) - 1;
|
|
51
|
+
if (!Number.isInteger(i) || i < 0 || i >= todos.length) {
|
|
52
|
+
return { error: `No todo #${index}.` };
|
|
53
|
+
}
|
|
54
|
+
if (typeof done === 'boolean') todos[i].done = done;
|
|
55
|
+
if (typeof text === 'string' && text.trim()) todos[i].text = text.trim();
|
|
56
|
+
todos[i].updatedAt = new Date().toISOString();
|
|
57
|
+
saveTodos(cwd, todos);
|
|
58
|
+
return { okay: true };
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function clearTodos(cwd) {
|
|
62
|
+
saveTodos(cwd, []);
|
|
63
|
+
return { okay: true };
|
|
64
|
+
}
|
package/src/cmd/tools.js
CHANGED
|
@@ -3,6 +3,7 @@ import { resolve, sep } from 'path';
|
|
|
3
3
|
import { toolResult, SafeCommandExecutor } from './executor.js';
|
|
4
4
|
import { Downloader } from '../utils/downloader.js';
|
|
5
5
|
import { Config } from '../config.js';
|
|
6
|
+
import { loadTodos, addTodo, updateTodo, clearTodos, formatTodos } from './todos.js';
|
|
6
7
|
|
|
7
8
|
// Resolve a path relative to the workspace and require it to stay inside.
|
|
8
9
|
// Throws if the path escapes the workspace (or hits the workspace root for
|
|
@@ -124,6 +125,35 @@ export async function downloadUrl(cwd, url, dir) {
|
|
|
124
125
|
}
|
|
125
126
|
}
|
|
126
127
|
|
|
128
|
+
export function todoAdd(cwd, text) {
|
|
129
|
+
const err = addTodo(cwd, text);
|
|
130
|
+
if (err && err.error) return fail({ command: `todo_add("${text}")`, stderr: err.error, cwd });
|
|
131
|
+
return ok({ command: `todo_add("${text}")`, stdout: formatTodos(loadTodos(cwd)), cwd });
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
export function todoList(cwd) {
|
|
135
|
+
try {
|
|
136
|
+
return ok({ command: 'todo_list', stdout: formatTodos(loadTodos(cwd)), cwd });
|
|
137
|
+
} catch (err) {
|
|
138
|
+
return fail({ command: 'todo_list', stderr: err.message, cwd });
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
export function todoUpdate(cwd, index, done, text) {
|
|
143
|
+
const err = updateTodo(cwd, index, { done, text });
|
|
144
|
+
if (err && err.error) return fail({ command: `todo_update(${index})`, stderr: err.error, cwd });
|
|
145
|
+
return ok({ command: `todo_update(${index})`, stdout: formatTodos(loadTodos(cwd)), cwd });
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
export function todoClear(cwd) {
|
|
149
|
+
try {
|
|
150
|
+
clearTodos(cwd);
|
|
151
|
+
return ok({ command: 'todo_clear', stdout: '(no todos yet)', cwd });
|
|
152
|
+
} catch (err) {
|
|
153
|
+
return fail({ command: 'todo_clear', stderr: err.message, cwd });
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
127
157
|
// Structured tool registry shared with the agent loop.
|
|
128
158
|
export const TOOL_EXECUTORS = {
|
|
129
159
|
run_command: (executor, params) => executor.runCommand(params.command),
|
|
@@ -133,6 +163,10 @@ export const TOOL_EXECUTORS = {
|
|
|
133
163
|
list_directory: (executor, params) => listDirectory(executor.cwd, params.path || '.'),
|
|
134
164
|
delete_file: (executor, params) => deleteFile(executor.cwd, params.path),
|
|
135
165
|
download_url: (executor, params) => downloadUrl(executor.cwd, params.url, params.dir),
|
|
166
|
+
todo_add: (executor, params) => todoAdd(executor.cwd, params ? params.text : undefined),
|
|
167
|
+
todo_list: (executor, params) => todoList(executor.cwd),
|
|
168
|
+
todo_update: (executor, params) => todoUpdate(executor.cwd, params ? params.index : undefined, params ? params.done : undefined, params ? params.text : undefined),
|
|
169
|
+
todo_clear: (executor, params) => todoClear(executor.cwd),
|
|
136
170
|
};
|
|
137
171
|
|
|
138
172
|
export function makeExecutor({ cwd, timeoutMs } = {}) {
|
package/src/ui/branding.js
CHANGED
package/src/utils/webfetch.js
CHANGED
|
@@ -2,6 +2,10 @@ export const UA =
|
|
|
2
2
|
'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 ' +
|
|
3
3
|
'(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36';
|
|
4
4
|
|
|
5
|
+
const ALT_UA =
|
|
6
|
+
'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 ' +
|
|
7
|
+
'(KHTML, like Gecko) Version/17.4 Safari/605.1.15';
|
|
8
|
+
|
|
5
9
|
const MAX_BYTES = 200000; // 200KB cap
|
|
6
10
|
const MAX_TEXT = 8000; // ~8k chars of readable text
|
|
7
11
|
|
|
@@ -16,10 +20,18 @@ const BLOCK_MARKERS = [
|
|
|
16
20
|
'client challenge',
|
|
17
21
|
'checking your browser',
|
|
18
22
|
'enable javascript and cookies',
|
|
19
|
-
'
|
|
23
|
+
'verify you are human',
|
|
20
24
|
'challenge-platform',
|
|
21
25
|
'attention required!',
|
|
22
|
-
'
|
|
26
|
+
'unusual traffic',
|
|
27
|
+
'access denied',
|
|
28
|
+
'too many requests',
|
|
29
|
+
'cf-error-details',
|
|
30
|
+
'atomic challenge',
|
|
31
|
+
'incapsula',
|
|
32
|
+
'rh-captcha',
|
|
33
|
+
'privacy pass',
|
|
34
|
+
'error code: 1020',
|
|
23
35
|
];
|
|
24
36
|
|
|
25
37
|
// Browser-like headers so header-based bot checks see a real browser.
|
|
@@ -39,6 +51,12 @@ export function browserHeaders(accept = 'text/html,application/xhtml+xml,applica
|
|
|
39
51
|
};
|
|
40
52
|
}
|
|
41
53
|
|
|
54
|
+
function altHeaders() {
|
|
55
|
+
const h = browserHeaders();
|
|
56
|
+
h['User-Agent'] = ALT_UA;
|
|
57
|
+
return h;
|
|
58
|
+
}
|
|
59
|
+
|
|
42
60
|
// File types the AI can download off a fetched page.
|
|
43
61
|
const ASSET_TYPES = {
|
|
44
62
|
pdf: 'pdf', zip: 'zip', tar: 'tar', gz: 'gz', '7z': '7z', rar: 'rar',
|
|
@@ -61,6 +79,9 @@ export function isBlockedBody(html) {
|
|
|
61
79
|
return BLOCK_MARKERS.some((m) => s.includes(m));
|
|
62
80
|
}
|
|
63
81
|
|
|
82
|
+
// Optional namespace prefix for XML tags (e.g. <sm:loc>).
|
|
83
|
+
const NSTAG = (name) => `(?:[a-z][\\w.-]*:)?${name}`;
|
|
84
|
+
|
|
64
85
|
export class WebFetch {
|
|
65
86
|
static normalizeUrl(input) {
|
|
66
87
|
const trimmed = (input || '').trim();
|
|
@@ -81,52 +102,63 @@ export class WebFetch {
|
|
|
81
102
|
}
|
|
82
103
|
}
|
|
83
104
|
|
|
105
|
+
// Fetch any URL and return readable text regardless of content type.
|
|
84
106
|
static async fetch(rawUrl, maxText = MAX_TEXT) {
|
|
85
107
|
const url = this.normalizeUrl(rawUrl);
|
|
86
108
|
if (!url) {
|
|
87
109
|
throw new Error('Invalid URL. Provide a valid http(s) address.');
|
|
88
110
|
}
|
|
89
111
|
|
|
90
|
-
let
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
const
|
|
95
|
-
|
|
96
|
-
html = res.html;
|
|
97
|
-
const status = res.status;
|
|
98
|
-
blocked = status !== 200 || isBlockedBody(html);
|
|
99
|
-
} catch (err) {
|
|
100
|
-
blocked = true;
|
|
101
|
-
// _fetchHtml errors are network/cert/PROTOCOL issues; a fallback may
|
|
102
|
-
// still succeed (e.g. registry API is reachable even when the HTML
|
|
103
|
-
// page's CDN is not).
|
|
112
|
+
let res = await this._get(url, browserHeaders());
|
|
113
|
+
|
|
114
|
+
// Transport-level failure (TLS, no https) → retry plain http.
|
|
115
|
+
if (!res.ok) {
|
|
116
|
+
const alt = url.replace(/^https:/i, 'http:');
|
|
117
|
+
if (alt !== url) res = await this._get(alt, browserHeaders());
|
|
104
118
|
}
|
|
105
119
|
|
|
106
|
-
//
|
|
107
|
-
if (
|
|
108
|
-
const
|
|
120
|
+
// Hard wall with a challenge challenge → one retry with a different UA.
|
|
121
|
+
if (res.ok && (res.status === 403 || res.status === 429) && isBlockedBody(res.text)) {
|
|
122
|
+
const retried = await this._get(url, altHeaders());
|
|
123
|
+
if (retried.ok && retried.status === 200 && !isBlockedBody(retried.text)) res = retried;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
if (!res.ok) {
|
|
127
|
+
const registry = await this._registryFallback(res.url || url, maxText);
|
|
109
128
|
if (registry) return registry;
|
|
129
|
+
throw new Error(`Could not reach ${res.url || url}`);
|
|
110
130
|
}
|
|
111
131
|
|
|
132
|
+
const status = res.status;
|
|
133
|
+
const bodyText = res.text;
|
|
134
|
+
const blocked = status >= 400 || isBlockedBody(bodyText);
|
|
135
|
+
if (blocked) {
|
|
136
|
+
const registry = await this._registryFallback(res.url || url, maxText);
|
|
137
|
+
if (registry) return registry;
|
|
138
|
+
}
|
|
112
139
|
if (blocked) {
|
|
113
|
-
|
|
114
|
-
throw new Error(`Request blocked (${status}) for ${finalUrl || url}`);
|
|
140
|
+
throw new Error(`Request blocked (HTTP ${status}) for ${res.url || url}`);
|
|
115
141
|
}
|
|
116
142
|
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
143
|
+
const kind = this._classify(res.url, res.contentType, bodyText);
|
|
144
|
+
return this._render(kind, res, bodyText, maxText);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
static async _get(url, headers) {
|
|
148
|
+
try {
|
|
149
|
+
const r = await this._fetchBytes(url, headers);
|
|
150
|
+
const text = this._decodeText(r.body, r.contentType);
|
|
151
|
+
return { ok: true, url: r.url, status: r.status, contentType: r.contentType, body: r.body, text };
|
|
152
|
+
} catch {
|
|
153
|
+
return { ok: false, url };
|
|
154
|
+
}
|
|
123
155
|
}
|
|
124
156
|
|
|
125
|
-
static async
|
|
157
|
+
static async _fetchBytes(url, headers) {
|
|
126
158
|
let resp;
|
|
127
159
|
try {
|
|
128
160
|
resp = await fetch(url, {
|
|
129
|
-
headers
|
|
161
|
+
headers,
|
|
130
162
|
redirect: 'follow',
|
|
131
163
|
signal: AbortSignal.timeout(15000),
|
|
132
164
|
});
|
|
@@ -134,20 +166,316 @@ export class WebFetch {
|
|
|
134
166
|
throw new Error(`Could not reach ${url}`);
|
|
135
167
|
}
|
|
136
168
|
|
|
137
|
-
const
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
169
|
+
const chunks = [];
|
|
170
|
+
let total = 0;
|
|
171
|
+
const reader = resp.body && resp.body.getReader();
|
|
172
|
+
if (reader) {
|
|
173
|
+
while (total < MAX_BYTES) {
|
|
174
|
+
const { done, value } = await reader.read();
|
|
175
|
+
if (done) break;
|
|
176
|
+
total += value.byteLength;
|
|
177
|
+
chunks.push(Buffer.from(value));
|
|
178
|
+
}
|
|
179
|
+
reader.cancel();
|
|
146
180
|
}
|
|
147
|
-
|
|
148
|
-
|
|
181
|
+
return {
|
|
182
|
+
url: resp.url || url,
|
|
183
|
+
status: resp.status,
|
|
184
|
+
contentType: String(resp.headers.get('content-type') || '').toLowerCase(),
|
|
185
|
+
body: Buffer.concat(chunks, total),
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
// --- charset-aware decoding ------------------------------------------- //
|
|
149
190
|
|
|
150
|
-
|
|
191
|
+
static _sniffCharset(contentType, bytes) {
|
|
192
|
+
const m = /charset=(["']?)([a-z0-9._-]+)\1/i.exec(contentType || '');
|
|
193
|
+
if (m) return m[2].toLowerCase();
|
|
194
|
+
const head = bytes.slice(0, 2048).toString('latin1').toLowerCase();
|
|
195
|
+
const meta = /<meta[^>]+charset=["']?\s*([a-z0-9._-]+)/i.exec(head);
|
|
196
|
+
if (meta) return meta[1].toLowerCase();
|
|
197
|
+
const httpEquiv = /<meta[^>]+http-equiv=["']content-type["'][^>]+content=["']text\/html; charset=([a-z0-9._-]+)/i.exec(head);
|
|
198
|
+
return httpEquiv ? httpEquiv[1].toLowerCase() : '';
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
static _decodeText(bytes, contentType) {
|
|
202
|
+
const charset = this._sniffCharset(contentType, bytes);
|
|
203
|
+
if (charset) {
|
|
204
|
+
try {
|
|
205
|
+
return new TextDecoder(charset, { fatal: false }).decode(bytes);
|
|
206
|
+
} catch {}
|
|
207
|
+
}
|
|
208
|
+
let text = new TextDecoder('utf-8', { fatal: false }).decode(bytes);
|
|
209
|
+
if (this._badDecode(text, bytes)) {
|
|
210
|
+
try {
|
|
211
|
+
text = new TextDecoder('windows-1252', { fatal: false }).decode(bytes);
|
|
212
|
+
} catch {}
|
|
213
|
+
}
|
|
214
|
+
return text;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
static _badDecode(text, bytes) {
|
|
218
|
+
if (!text || !bytes || bytes.length === 0) return false;
|
|
219
|
+
let bad = 0;
|
|
220
|
+
for (let i = 0; i < text.length; i++) if (text.charCodeAt(i) === 0xfffd) bad++;
|
|
221
|
+
return bad / bytes.length > 0.001;
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
// --- content-type classification --------------------------------------- //
|
|
225
|
+
|
|
226
|
+
static _classify(url, contentType, text) {
|
|
227
|
+
const meta = ((contentType || '').split(';')[0] || '').trim().toLowerCase();
|
|
228
|
+
const head = (text || '').slice(0, 240).trimStart();
|
|
229
|
+
if (/\bjson\b/.test(meta) || /\+\w*json\b/.test(meta)) return 'json';
|
|
230
|
+
if (/\b(?:rss|atom|xml)\b/.test(meta) || meta === 'application/sitemap+xml') return 'xml';
|
|
231
|
+
if (/\/xhtml\+html$/.test(meta) || /html/.test(meta)) return 'html';
|
|
232
|
+
if (/markdown/.test(meta)) return 'markdown';
|
|
233
|
+
if (meta === 'application/pdf') return 'pdf';
|
|
234
|
+
if (/^application\/(?:json|x-ndjson|xml|rss|atom)/.test(meta)) return 'json';
|
|
235
|
+
if (/^text\//.test(meta) && !/html/.test(meta)) return 'text';
|
|
236
|
+
if (head.startsWith('{') || head.startsWith('[')) return 'json';
|
|
237
|
+
if (head.startsWith('<?xml') || /^<!DOCTYPE\s+(?!html)/i.test(head)) return 'xml';
|
|
238
|
+
if (/^<!doctype html/i.test(head) || /<html\b/i.test(head.slice(0, 400))) return 'html';
|
|
239
|
+
if (/^(image|video|audio|font)\//.test(meta)) return 'binary';
|
|
240
|
+
if (/^application\//.test(meta)) return 'binary';
|
|
241
|
+
if (/\/pdf$/.test(url.split('?')[0].toLowerCase())) return 'pdf';
|
|
242
|
+
return 'text';
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
static _render(kind, res, bodyText, maxText) {
|
|
246
|
+
const url = res.url;
|
|
247
|
+
|
|
248
|
+
if (kind === 'json') {
|
|
249
|
+
let flat = bodyText.trim();
|
|
250
|
+
try {
|
|
251
|
+
flat = this._flattenJson(JSON.parse(bodyText));
|
|
252
|
+
} catch {}
|
|
253
|
+
return { url, title: this._titleFromUrl(url), text: flat.slice(0, maxText), assets: [] };
|
|
254
|
+
}
|
|
255
|
+
|
|
256
|
+
if (kind === 'xml') {
|
|
257
|
+
return {
|
|
258
|
+
url,
|
|
259
|
+
title: this._titleFromUrl(url),
|
|
260
|
+
text: this._xmlToText(bodyText).slice(0, maxText),
|
|
261
|
+
assets: this._extractAssets(bodyText, url),
|
|
262
|
+
};
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
if (kind === 'text' || kind === 'markdown') {
|
|
266
|
+
return { url, title: this._titleFromUrl(url), text: bodyText.trim().slice(0, maxText), assets: [] };
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
if (kind === 'pdf' || kind === 'binary') {
|
|
270
|
+
let parsed;
|
|
271
|
+
try {
|
|
272
|
+
parsed = new URL(url);
|
|
273
|
+
} catch {
|
|
274
|
+
parsed = null;
|
|
275
|
+
}
|
|
276
|
+
let type = detectAssetType(url) || 'file';
|
|
277
|
+
if (kind === 'pdf') type = 'pdf';
|
|
278
|
+
return {
|
|
279
|
+
url,
|
|
280
|
+
title: this._titleFromUrl(url),
|
|
281
|
+
text: this._binaryNote(res, kind, type),
|
|
282
|
+
assets: [{ url, name: this._assetName(parsed, type), type }],
|
|
283
|
+
};
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
// HTML
|
|
287
|
+
const desc = this._extractDescription(bodyText);
|
|
288
|
+
let text = this._extractText(bodyText);
|
|
289
|
+
if (text.length < 2400 && text.length < maxText) {
|
|
290
|
+
const embedded = this._extractEmbeddedJson(bodyText);
|
|
291
|
+
if (embedded !== null && embedded !== undefined) {
|
|
292
|
+
const flat = this._flattenJson(embedded).trim();
|
|
293
|
+
if (flat) text = `${text}\n\n[embedded page data]\n${flat}`.trim();
|
|
294
|
+
}
|
|
295
|
+
const links = this._extractLinks(bodyText, url);
|
|
296
|
+
if (links.length >= 3 && text.length < 2400) {
|
|
297
|
+
text = `${text}\n\n[Links on page]\n${links.slice(0, 25).map((l) => `- ${l.text}: ${l.url}`).join('\n')}`.trim();
|
|
298
|
+
}
|
|
299
|
+
}
|
|
300
|
+
let out = text.slice(0, maxText);
|
|
301
|
+
if (desc && out.length < maxText - 600 && !out.includes(desc.slice(0, 40))) {
|
|
302
|
+
out = `${desc.slice(0, 400)}\n\n${out}`;
|
|
303
|
+
}
|
|
304
|
+
let title = this._extractTitle(bodyText);
|
|
305
|
+
if (!title) title = this._titleFromUrl(url);
|
|
306
|
+
return { url, title, text: out, assets: this._extractAssets(bodyText, url) };
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
static _binaryNote(res, kind, type) {
|
|
310
|
+
const ct = res.contentType || 'unknown content type';
|
|
311
|
+
const size = res.body ? `~${(res.body.length / 1024).toFixed(1)} KB` : '?';
|
|
312
|
+
return [
|
|
313
|
+
`This resource is a ${kind === 'pdf' ? 'PDF' : 'binary/non-text file'} (${ct}, ${size}) and cannot be read as text.`,
|
|
314
|
+
`Use the download_url tool to save it locally: ${res.url}${type !== 'file' ? ` (type: ${type})` : ''}`,
|
|
315
|
+
].join('\n');
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
static _titleFromUrl(url) {
|
|
319
|
+
try {
|
|
320
|
+
const u = new URL(url);
|
|
321
|
+
const seg = decodeURIComponent(u.pathname.split('/').filter(Boolean).pop() || '');
|
|
322
|
+
const base = u.hostname.replace(/^www\./i, '');
|
|
323
|
+
if (seg && !/\.(html?|php|aspx?)$/i.test(seg)) return base;
|
|
324
|
+
const cleaned = seg.replace(/\.(html?|php|aspx?)$/i, '').replace(/[-_]+/g, ' ').trim();
|
|
325
|
+
return cleaned || base;
|
|
326
|
+
} catch {
|
|
327
|
+
return url;
|
|
328
|
+
}
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
static _flattenJson(data, maxLines = 600) {
|
|
332
|
+
const out = [];
|
|
333
|
+
const walk = (v, label) => {
|
|
334
|
+
if (out.length >= maxLines) return;
|
|
335
|
+
if (v === null || v === undefined) {
|
|
336
|
+
if (label) out.push(`${label}: null`);
|
|
337
|
+
return;
|
|
338
|
+
}
|
|
339
|
+
if (Array.isArray(v)) {
|
|
340
|
+
if (label) out.push(`${label} (${v.length})`);
|
|
341
|
+
for (let i = 0; i < v.length; i++) walk(v[i], label ? `${label}[${i}]` : `[${i}]`);
|
|
342
|
+
} else if (typeof v === 'object') {
|
|
343
|
+
if (label) out.push(`${label} {${Object.keys(v).length}}`);
|
|
344
|
+
for (const k of Object.keys(v)) walk(v[k], label ? `${label}.${k}` : k);
|
|
345
|
+
} else {
|
|
346
|
+
const s = typeof v === 'string' ? v.replace(/\s+/g, ' ').trim() : String(v);
|
|
347
|
+
out.push(`${label ? label + ': ' : ''}${s.slice(0, 400)}`);
|
|
348
|
+
}
|
|
349
|
+
};
|
|
350
|
+
walk(data, '');
|
|
351
|
+
return out.join('\n');
|
|
352
|
+
}
|
|
353
|
+
|
|
354
|
+
// --- XML / feeds / sitemaps -------------------------------------------- //
|
|
355
|
+
|
|
356
|
+
static _xmlToText(xml) {
|
|
357
|
+
const doc = (xml || '').trimStart();
|
|
358
|
+
if (/<sitemapindex\b/i.test(doc) || /<urlset\b/i.test(doc)) {
|
|
359
|
+
const locs = [];
|
|
360
|
+
const re = new RegExp(`<${NSTAG('loc')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('loc')}>`, 'gi');
|
|
361
|
+
let m;
|
|
362
|
+
while ((m = re.exec(xml)) !== null) {
|
|
363
|
+
const u = this._stripEntities(m[1]).replace(/\s+/g, ' ').trim();
|
|
364
|
+
if (u && /^https?:\/\//i.test(u) && !locs.includes(u)) locs.push(u);
|
|
365
|
+
if (locs.length >= 100) break;
|
|
366
|
+
}
|
|
367
|
+
if (locs.length) {
|
|
368
|
+
return `Sitemap (${locs.length} URLs)\n${locs.slice(0, 60).map((u) => `* ${u}`).join('\n')}`;
|
|
369
|
+
}
|
|
370
|
+
}
|
|
371
|
+
if (/<(?:rss|feed)\b/i.test(doc)) return this._feedToText(xml);
|
|
372
|
+
return this._extractText(xml).slice(0, MAX_TEXT);
|
|
373
|
+
}
|
|
374
|
+
|
|
375
|
+
static _feedToText(xml) {
|
|
376
|
+
const lines = [];
|
|
377
|
+
const top = new RegExp(`<${NSTAG('channel')}[\\s\\S]*?<\\/${NSTAG('channel')}>|<${NSTAG('feed')}[\\s\\S]*?<\\/${NSTAG('feed')}>`, 'i');
|
|
378
|
+
const topMatch = top.exec(xml);
|
|
379
|
+
let title = '';
|
|
380
|
+
if (topMatch) {
|
|
381
|
+
const t = new RegExp(`<${NSTAG('title')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('title')}>`, 'i').exec(topMatch[0]);
|
|
382
|
+
if (t) title = this._stripTags(t[1]);
|
|
383
|
+
}
|
|
384
|
+
if (title) lines.push(title);
|
|
385
|
+
|
|
386
|
+
const itemRe = new RegExp(`<${NSTAG('item')}[\\s\\S]*?<\\/${NSTAG('item')}>|<${NSTAG('entry')}[\\s\\S]*?<\\/${NSTAG('entry')}>`, 'gi');
|
|
387
|
+
let m;
|
|
388
|
+
let count = 0;
|
|
389
|
+
while ((m = itemRe.exec(xml)) !== null && count < 40) {
|
|
390
|
+
const b = m[0];
|
|
391
|
+
count++;
|
|
392
|
+
const it = new RegExp(`<${NSTAG('title')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('title')}>`, 'i').exec(b);
|
|
393
|
+
const il = new RegExp(`<${NSTAG('link')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('link')}>`, 'i').exec(b) ||
|
|
394
|
+
new RegExp(`<${NSTAG('link')}(?:[^>]*)\\bhref=["']([^"']+)["']`, 'i').exec(b);
|
|
395
|
+
const di = /<(?:[a-z][\w.-]*:)?(?:description|summary|subtitle)[^>]*>([\s\S]*?)<\/(?:[a-z][\w.-]*:)?(?:description|summary|subtitle)>/i.exec(b);
|
|
396
|
+
const item = [];
|
|
397
|
+
if (it) item.push(this._stripTags(it[1]));
|
|
398
|
+
if (il) item.push(` ${this._stripTags(il[1]).replace(/\s+/g, ' ').trim()}`);
|
|
399
|
+
if (di && di[1].trim()) {
|
|
400
|
+
const d = this._stripTags(di[1]);
|
|
401
|
+
if (d) item.push(` ${d}`);
|
|
402
|
+
}
|
|
403
|
+
if (item.length) lines.push(item.join('\n'));
|
|
404
|
+
}
|
|
405
|
+
if (lines.length === 0) return this._extractText(xml).slice(0, MAX_TEXT);
|
|
406
|
+
return lines.join('\n\n');
|
|
407
|
+
}
|
|
408
|
+
|
|
409
|
+
// --- SPA / embedded JSON ------------------------------------------------ //
|
|
410
|
+
|
|
411
|
+
static _extractEmbeddedJson(html) {
|
|
412
|
+
const next = /<script[^>]+id=["']__NEXT_DATA__["'][^>]*>([\s\S]*?)<\/script>/i.exec(html);
|
|
413
|
+
if (next && next[1].trim()) {
|
|
414
|
+
try {
|
|
415
|
+
return JSON.parse(next[1].trim());
|
|
416
|
+
} catch {}
|
|
417
|
+
}
|
|
418
|
+
const ld = /<script[^>]+type=["']application\/ld\+json["'][^>]*>([\s\S]*?)<\/script>/gi;
|
|
419
|
+
let c;
|
|
420
|
+
while ((c = ld.exec(html)) !== null) {
|
|
421
|
+
if (!c[1].trim()) continue;
|
|
422
|
+
try {
|
|
423
|
+
return JSON.parse(c[1].trim());
|
|
424
|
+
} catch {}
|
|
425
|
+
}
|
|
426
|
+
const st = /window\.__PRELOADED_STATE__\s*=\s*/i.exec(html);
|
|
427
|
+
if (st) {
|
|
428
|
+
let i = st.index + st[0].length;
|
|
429
|
+
let depth = 0;
|
|
430
|
+
const start = i;
|
|
431
|
+
while (i < html.length) {
|
|
432
|
+
const ch = html[i];
|
|
433
|
+
if (ch === '{') depth++;
|
|
434
|
+
else if (ch === '}') {
|
|
435
|
+
depth--;
|
|
436
|
+
if (depth === 0) break;
|
|
437
|
+
}
|
|
438
|
+
i++;
|
|
439
|
+
}
|
|
440
|
+
if (depth === 0) {
|
|
441
|
+
try {
|
|
442
|
+
return JSON.parse(html.slice(start, i + 1).replace(/;?\s*$/, '').trim());
|
|
443
|
+
} catch {}
|
|
444
|
+
}
|
|
445
|
+
}
|
|
446
|
+
return null;
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
static _extractDescription(html) {
|
|
450
|
+
const m1 = /<meta[^>]+name=["']description["'][^>]+content=["']([^"']*)["'][^>]*>/i.exec(html);
|
|
451
|
+
if (m1) return this._stripTags(m1[1]);
|
|
452
|
+
const m2 = /<meta[^>]+content=["']([^"']*)["'][^>]+name=["']description["'][^>]*>/i.exec(html);
|
|
453
|
+
return m2 ? this._stripTags(m2[1]) : '';
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
static _extractLinks(html, baseUrl) {
|
|
457
|
+
const links = [];
|
|
458
|
+
const seen = new Set();
|
|
459
|
+
const re = /<a\b[^>]*href=["']([^"']+)["'][^>]*>((?:[^<]|<(?!\/a\b)[^>]+>)*?)<\/a>/gi;
|
|
460
|
+
let m;
|
|
461
|
+
while ((m = re.exec(html)) !== null) {
|
|
462
|
+
const href = m[1].trim().split('#')[0];
|
|
463
|
+
if (!href || /^(javascript|mailto|tel|data):/i.test(href)) continue;
|
|
464
|
+
let resolved;
|
|
465
|
+
try {
|
|
466
|
+
const u = new URL(href, baseUrl);
|
|
467
|
+
if (u.protocol !== 'http:' && u.protocol !== 'https:') continue;
|
|
468
|
+
resolved = u.toString();
|
|
469
|
+
} catch {
|
|
470
|
+
continue;
|
|
471
|
+
}
|
|
472
|
+
if (seen.has(resolved)) continue;
|
|
473
|
+
seen.add(resolved);
|
|
474
|
+
const text = this._stripTags(m[2]).slice(0, 80);
|
|
475
|
+
links.push({ text: text || resolved, url: resolved });
|
|
476
|
+
if (links.length >= 30) break;
|
|
477
|
+
}
|
|
478
|
+
return links;
|
|
151
479
|
}
|
|
152
480
|
|
|
153
481
|
// --- Registry fallback (npm + PyPI) via their open JSON APIs ---------- //
|
|
@@ -280,6 +608,7 @@ export class WebFetch {
|
|
|
280
608
|
}
|
|
281
609
|
|
|
282
610
|
static _assetName(url, type) {
|
|
611
|
+
if (!url) return `download.${type}`;
|
|
283
612
|
const path = url.pathname.split('/').filter(Boolean);
|
|
284
613
|
let name = path.length ? decodeURIComponent(path[path.length - 1]) : '';
|
|
285
614
|
if (!name || type === 'unknown') name = `${url.hostname.replace(/[^a-z0-9.-]/gi, '_')}.${type}`;
|
|
@@ -293,17 +622,19 @@ export class WebFetch {
|
|
|
293
622
|
}
|
|
294
623
|
|
|
295
624
|
static _extractText(html) {
|
|
296
|
-
//
|
|
297
|
-
let s = html
|
|
625
|
+
// Drop comments and tag-blocks that add no readable value.
|
|
626
|
+
let s = String(html || '')
|
|
627
|
+
.replace(/<!--[\s\S]*?-->/g, ' ')
|
|
628
|
+
.replace(/<(script|style|noscript|svg|head|iframe|form|nav|footer|aside|template|object)[^>]*>[\s\S]*?<\/\1>/gi, ' ');
|
|
298
629
|
|
|
299
630
|
// Force spacing around block-level elements so words don't merge.
|
|
300
|
-
s = s.replace(/<\/(p|div|h[1-6]|li|tr|br|section|article)>/gi, '\n');
|
|
631
|
+
s = s.replace(/<\/(p|div|h[1-6]|li|tr|br|section|article|blockquote|pre|table)>/gi, '\n');
|
|
301
632
|
s = s.replace(/<(br|li|tr)[^>]*>/gi, '\n');
|
|
302
633
|
|
|
303
634
|
// Remove remaining tags.
|
|
304
635
|
s = s.replace(/<[^>]+>/g, ' ');
|
|
305
636
|
|
|
306
|
-
//
|
|
637
|
+
// Collapse entities.
|
|
307
638
|
s = this._stripEntities(s);
|
|
308
639
|
|
|
309
640
|
// Collapse whitespace and trim lines.
|
|
@@ -316,6 +647,10 @@ export class WebFetch {
|
|
|
316
647
|
.trim();
|
|
317
648
|
}
|
|
318
649
|
|
|
650
|
+
static _stripTags(text) {
|
|
651
|
+
return this._stripEntities(text);
|
|
652
|
+
}
|
|
653
|
+
|
|
319
654
|
static _stripEntities(text) {
|
|
320
655
|
const map = {
|
|
321
656
|
'&': '&', '<': '<', '>': '>', '"': '"',
|