ft-scout 8.0.1 → 8.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@ import { saveAgentSession, getRecentSessionsSummary } from '../utils/session.js'
9
9
  import { safeNote, renderMarkdown } from '../utils/markdown.js';
10
10
  import { openai, callOpenAIWithRetry, isQuotaExceededError, sanitizeMessage, sanitizeMessages } from './llm.js';
11
11
  import { checkFileSyntax, verifyAndSelfHealFiles, executeSmartCommand, extractErrorDiagnostics } from './verifier.js';
12
- import { openApp, executeInApp } from './appControl.js';
12
+ import { openApp, executeInApp, cleanupScreenshots } from './appControl.js';
13
13
  import { speakText, listenSpeechToText } from './voiceEngine.js';
14
14
  const execAsync = promisify(exec);
15
15
  export function robustSnippetReplace(origContent, target, replacement) {
@@ -284,13 +284,13 @@ export const AGENT_TOOLS = [
284
284
  },
285
285
  {
286
286
  name: 'app_action',
287
- description: 'Execute interactive desktop/browser scratchpad automation actions, screen takeover, mouse cursor positioning & click takeover, keyboard keystrokes, hotkeys, screen seeing/screenshot capture, visual screen analysis, focus locks, or command sequences without shell commands (e.g. action: "takeover", "capture_screen", "see_screen", "analyze_screen", "get_screen_info", "click_app", "move_mouse", "type_text", "send_keys", "key_combo", "lock_app", "fetch_page", "navigate", "search", "send_dm", "exec_command", "open_file").',
287
+ description: 'Execute interactive desktop/browser scratchpad automation actions, screen seeing & analysis, UI element clicking, screen takeover, mouse cursor positioning & click takeover, mouse scrolling, window management, keyboard keystrokes, hotkeys, focus locks, or command sequences without shell commands (e.g. action: "see_screen", "analyze_screen", "capture_screen", "click_element", "click_app", "scroll", "list_windows", "focus_window", "takeover", "move_mouse", "type_text", "send_keys", "key_combo", "lock_app", "fetch_page", "navigate", "search", "send_dm", "exec_command", "open_file"). All temporary screenshot files are automatically deleted upon task completion.',
288
288
  parameters: {
289
289
  type: 'object',
290
290
  properties: {
291
291
  app: { type: 'string', description: 'Target application name, window title, screen, or category ("browser", "terminal", "editor", "notepad", "chrome", "desktop")' },
292
- action: { type: 'string', description: 'Action type ("takeover", "capture_screen", "see_screen", "analyze_screen", "get_screen_info", "type_text", "click_app", "move_mouse", "send_keys", "key_combo", "lock_app", "fetch_page", "search", "send_dm", "exec_command", "open_file")' },
293
- payload: { description: 'Action details/payload object or string (e.g. text string, { text: "...", enter: true }, { x: 100, y: 200, button: "left" }, { keyCombo: "ctrl+v" }, { duration: 3000 }, URL, search query, command, or file path)' },
292
+ action: { type: 'string', description: 'Action type ("see_screen", "analyze_screen", "capture_screen", "click_element", "click_app", "scroll", "list_windows", "focus_window", "takeover", "send_mail", "type_text", "move_mouse", "send_keys", "key_combo", "lock_app", "fetch_page", "search", "send_dm", "exec_command", "open_file")' },
293
+ payload: { description: 'Action details/payload object or string (e.g. element name/description to click { element: "Search Google" }, coordinates { x: 100, y: 200 }, scroll { direction: "down", amount: 4 }, text string, { text: "...", enter: true }, { keyCombo: "ctrl+v" }, { duration: 3000 }, URL, search query, command, or file path)' },
294
294
  },
295
295
  required: ['app', 'action'],
296
296
  },
@@ -399,12 +399,26 @@ export class AgentExecutionLoop {
399
399
  readFilesHistory = new Set();
400
400
  consecutiveReads = 0;
401
401
  consecutiveCommands = 0;
402
+ consecutiveScreenChecks = 0;
403
+ toolCallHistory = [];
402
404
  scratchpadState = { plan: [], completedSteps: [], notes: '' };
403
405
  constructor(options) {
404
406
  this.cwd = process.cwd();
405
407
  this.autoApprove = Boolean(options?.autoApprove);
406
408
  this.maxSteps = options?.maxSteps || 30;
407
409
  this.projectName = options?.projectName || path.basename(this.cwd);
410
+ // Ensure temporary screenshot cleanup on early exit or interrupt
411
+ const onExitOrInterrupt = () => {
412
+ try {
413
+ cleanupScreenshots();
414
+ }
415
+ catch { }
416
+ };
417
+ process.once('exit', onExitOrInterrupt);
418
+ process.once('SIGINT', () => {
419
+ onExitOrInterrupt();
420
+ process.exit(0);
421
+ });
408
422
  }
409
423
  getModifiedFiles() {
410
424
  return Array.from(this.modifiedFiles);
@@ -430,6 +444,8 @@ export class AgentExecutionLoop {
430
444
  this.currentStep = 0;
431
445
  this.consecutiveReads = 0;
432
446
  this.consecutiveCommands = 0;
447
+ this.consecutiveScreenChecks = 0;
448
+ this.toolCallHistory = [];
433
449
  this.scratchpadState = { plan: [], completedSteps: [], notes: '' };
434
450
  }
435
451
  updateScratchpad(plan, completedSteps, notes) {
@@ -472,7 +488,104 @@ export class AgentExecutionLoop {
472
488
  }
473
489
  return summary;
474
490
  }
475
- handleToolConsecutiveTracking(fnName, step) {
491
+ autoSyncScratchpadFromText(text) {
492
+ try {
493
+ const planLines = text.match(/[-*]\s*\[([ xX✓/])\]\s*(.*)/g);
494
+ if (planLines && planLines.length > 0) {
495
+ const plan = [];
496
+ const completed = [];
497
+ planLines.forEach((l, idx) => {
498
+ const isDone = l.includes('[x]') || l.includes('[X]') || l.includes('[✓]');
499
+ const cleanText = l.replace(/^[-*]\s*\[[ xX✓/]\]\s*/, '').trim();
500
+ plan.push(cleanText);
501
+ if (isDone)
502
+ completed.push(idx);
503
+ });
504
+ this.scratchpadState.plan = plan;
505
+ this.scratchpadState.completedSteps = completed;
506
+ }
507
+ const obsMatch = text.match(/Observations?:\s*([^\n]+(?:\n[^\n]+)*)/i);
508
+ const thoughtMatch = text.match(/Thought:\s*([^\n]+(?:\n[^\n]+)*)/i);
509
+ const notes = [
510
+ thoughtMatch ? `Thought: ${thoughtMatch[1]?.trim()}` : '',
511
+ obsMatch ? `Observations: ${obsMatch[1]?.trim()}` : '',
512
+ ].filter(Boolean).join('\n\n');
513
+ if (notes) {
514
+ this.scratchpadState.notes = notes;
515
+ }
516
+ const ftDir = path.join(this.cwd, '.ft');
517
+ if (!fs.existsSync(ftDir))
518
+ fs.mkdirSync(ftDir, { recursive: true });
519
+ const scratchpadPath = path.join(ftDir, 'scratchpad.md');
520
+ let mdContent = `# Scout Agent Scratchpad & Working Memory\n\n`;
521
+ if (this.scratchpadState.plan.length > 0) {
522
+ mdContent += `## Plan Checklist\n`;
523
+ this.scratchpadState.plan.forEach((item, idx) => {
524
+ const isDone = this.scratchpadState.completedSteps.includes(idx);
525
+ mdContent += `- [${isDone ? 'x' : ' '}] Step ${idx + 1}: ${item}\n`;
526
+ });
527
+ mdContent += `\n`;
528
+ }
529
+ if (this.scratchpadState.notes) {
530
+ mdContent += `## Working Memory & Observations\n${this.scratchpadState.notes}\n`;
531
+ }
532
+ fs.writeFileSync(scratchpadPath, mdContent, 'utf-8');
533
+ }
534
+ catch { }
535
+ }
536
+ handleToolConsecutiveTracking(fnName, args, step) {
537
+ const argsKey = JSON.stringify(args || {});
538
+ this.toolCallHistory.push({ fnName, argsKey });
539
+ if (this.toolCallHistory.length > 10) {
540
+ this.toolCallHistory.shift();
541
+ }
542
+ // 1. Direct duplicate call check: Same tool + exact same arguments executed consecutively
543
+ const len = this.toolCallHistory.length;
544
+ if (len >= 2) {
545
+ const prev = this.toolCallHistory[len - 2];
546
+ const curr = this.toolCallHistory[len - 1];
547
+ if (prev && curr && prev.fnName === curr.fnName && prev.argsKey === curr.argsKey && !['task_completed', 'ask_user'].includes(fnName)) {
548
+ this.historyMessages.push({
549
+ role: 'user',
550
+ content: `CRITICAL ACTION LOOP DETECTED (Step ${step}): You called "${fnName}" with the EXACT same arguments as the previous step! Repeating this action will not change the outcome. DO NOT call this tool with these arguments again. Change your strategy immediately: search with different terms, navigate to the correct page, take a concrete input action, or call "task_completed" if done.`,
551
+ });
552
+ }
553
+ }
554
+ // 2. Ping-Pong / Oscillation loop check: A -> B -> A -> B
555
+ if (len >= 4) {
556
+ const a1 = this.toolCallHistory[len - 4];
557
+ const b1 = this.toolCallHistory[len - 3];
558
+ const a2 = this.toolCallHistory[len - 2];
559
+ const b2 = this.toolCallHistory[len - 1];
560
+ if (a1 && b1 && a2 && b2 &&
561
+ a1.fnName === a2.fnName && a1.argsKey === a2.argsKey &&
562
+ b1.fnName === b2.fnName && b1.argsKey === b2.argsKey &&
563
+ (a1.fnName !== b1.fnName || a1.argsKey !== b1.argsKey)) {
564
+ this.historyMessages.push({
565
+ role: 'user',
566
+ content: `CRITICAL OSCILLATION LOOP DETECTED (Step ${step}): You are stuck alternating between "${a1.fnName}" and "${b1.fnName}" without progressing! STOP repeating these actions immediately. If the page or app is not responding as expected, adopt an alternative path, use web search, launch the desktop app directly, or conclude with task_completed.`,
567
+ });
568
+ }
569
+ }
570
+ // 3. Screen inspection stagnation tracking
571
+ if (fnName === 'app_action') {
572
+ const act = String(args?.action || '').toLowerCase();
573
+ if (['see_screen', 'analyze_screen', 'capture_screen'].includes(act)) {
574
+ this.consecutiveScreenChecks++;
575
+ if (this.consecutiveScreenChecks >= 2) {
576
+ this.historyMessages.push({
577
+ role: 'user',
578
+ content: `URGENT ACTION MANDATE (Step ${step}): You have inspected the screen ${this.consecutiveScreenChecks} times consecutively without taking any physical action! Do not keep calling see_screen. Take a concrete action now: click an element ("click_element"), enter text ("type_text"), open/focus a window, or complete the task.`,
579
+ });
580
+ }
581
+ }
582
+ else {
583
+ this.consecutiveScreenChecks = 0;
584
+ }
585
+ }
586
+ else {
587
+ this.consecutiveScreenChecks = 0;
588
+ }
476
589
  if (['read_file', 'list_dir', 'glob_search', 'grep_search', 'tree_view', 'file_info'].includes(fnName)) {
477
590
  this.consecutiveReads++;
478
591
  this.consecutiveCommands = 0;
@@ -541,17 +654,45 @@ Core Directives & Behavioral Guidelines:
541
654
  19. EXTERNAL APP TAKEOVER & CONTROL PROTOCOL: When requested by user prompt to open, take over, or work inside external applications (browser, terminal, VS Code, Notepad, social apps like Instagram, WhatsApp, Twitter/X, Telegram, or custom apps):
542
655
  a. Launch App: Use \`open_app\` to launch or focus the target application with optional URL, file path, or initial script.
543
656
  b. Social DM & Messaging Automation: For Instagram, WhatsApp, Twitter/X, or Telegram messaging requests (e.g. "open instagram and send message to @user"), immediately invoke \`app_action\` with action "send_dm" or "open_dm" (or \`open_app\`) specifying the target username/phone and message text so the agent automatically opens the direct messaging link in the browser!
544
- c. Work Inside App: Use \`app_action\` or \`run_command\` to execute actions inside the app context (e.g. fetching browser page content, running commands inside terminal, searching web, opening files in editor).
545
- 20. RECKLESS SCRATCHPAD DESKTOP CONTROL, SCREEN TAKEOVER & VISUAL ANALYSIS (ZERO SHELL COMMANDS):
546
- a. Mouse & Keyboard Screen Takeover: During scratchpad execution, you act as a full desktop automation agent that takes over the mouse cursor, screen focus, and keyboard typing on the specified screen/app using \`app_action\` (\`action: "takeover"\`, \`action: "click_app"\`, \`action: "move_mouse"\`, \`action: "type_text"\`, \`action: "send_keys"\`, \`action: "key_combo"\`).
547
- b. Native Screen Seeing & Visual Analysis (No Shell Commands): You can view, capture, and visually analyze the specified screen/window directly using \`app_action\` (\`action: "capture_screen"\`, \`action: "see_screen"\`, \`action: "analyze_screen"\`, \`action: "get_screen_info"\`). DO NOT run shell commands (\`run_command\`) to capture or inspect screens—\`app_action\` handles native screen capture, visual analysis, mouse takeover, and key input internally!
548
- c. CLI Credential Input Prompt: If an application requires login credentials, passwords, 2FA codes, or secret tokens to proceed, call \`ask_user\` tool with a clear prompt. This presents a secure, interactive input bar directly in the user's running terminal CLI. Once the user enters the secret, take the received input, inject it into the target application window via \`app_action\` (\`action: "type_text"\`).
549
- 21. MANDATORY TAKEOVER TOOL INVOCATION MANDATE: Whenever the user goal requests to open, launch, take over, click, type, or interact with an external app or desktop screen (e.g. 'take over browser and open website', 'open notepad and type', 'take over desktop'), YOU MUST CALL \`open_app\` AND \`app_action\` TOOLS (with \`action: "takeover"\`, \`action: "click_app"\`, \`action: "type_text"\`, \`action: "send_keys"\`, \`action: "capture_screen"\`) IN YOUR VERY FIRST STEP! You MUST NOT return a plain text answer or claim you did it in text without calling the tool! The tool calls ARE MANDATORY for executing the physical takeover.
657
+ c. Email & Gmail Automation: When requested to send or compose an email (e.g. "send mail to frontterrain@gmail.com saying...", "takeover mail.google.com and send mail"):
658
+ - Immediately invoke \`app_action\` with \`action: "send_mail"\` (or \`action: "send_dm"\`) specifying the target recipient email and body text. The agent automatically constructs the direct Gmail compose URL (\`https://mail.google.com/mail/?view=cm&fs=1&to=<recipient>&su=<subject>&body=<body>\`) which pre-populates the compose window and dispatches the email via Ctrl+Enter!
659
+ - Alternatively, navigate directly to \`https://mail.google.com/mail/?view=cm&fs=1&to=<recipient>&su=...&body=...\` and trigger hotkey \`app_action(action: "key_combo", payload: { keyCombo: "ctrl+enter" })\`. NEVER click blind coordinates like (20, 20) in a web browser!
660
+ d. Work Inside App: Use \`app_action\` or \`run_command\` to execute actions inside the app context (e.g. fetching browser page content, running commands inside terminal, searching web, opening files in editor).
661
+ e. Universal Gaming & External Application Protocol:
662
+ - Desktop Game/App Launch: To open or play ANY installed game or app on the PC (Steam games, Epic Games, Minecraft, Roblox, Discord, Spotify, etc.), call open_app("<game_name>"). The universal OS launcher automatically resolves installed Windows/macOS applications and store packages via system registry and launches them directly!
663
+ - Web Game Play: If the game is web-based (e.g. Chess.com, Slither.io, 2048, Poki) or not installed locally, NEVER guess speculative URL subpaths! Use app_action(app: "browser", action: "search", payload: { query: "<game_name> play online official" }) or navigate to the official domain homepage to find the verified play URL.
664
+ - Autonomous Gameplay Progression Loop:
665
+ 1. Window Focus: Use app_action(action: "focus_window", payload: { app: "<game_name>" }) to bring the window front-and-center.
666
+ 2. Screen Observation: Use app_action(action: "see_screen") to inspect the UI, loading screen, or active state.
667
+ 3. Menu Traversal: Click menu buttons ("Play", "Start Game", "New Game", "Continue") using app_action(action: "click_element", payload: { element: "Play" }) .
668
+ 4. Interactive Controls: Send gameplay controls using app_action(action: "send_keys" / "type_text" / "key_combo") with standard gaming keys (WASD, Arrow keys, Space, Enter, Escape, mouse clicks/drags).
669
+ 5. Conclude task with task_completed when the requested gameplay actions or objectives are achieved.
670
+ 20. DESKTOP APP TAKEOVER & AUTOMATION PROTOCOL:
671
+ a. STEP 1 SCREEN TAKEOVER: Whenever the user goal asks to take over the screen, control the desktop, or automate an external app (e.g. 'take over screen', 'open maps and click', 'take over desktop', 'open notepad'):
672
+ - Your VERY FIRST tool call in Step 1 MUST be \`app_action\` with \`action: "takeover"\` (e.g. \`app: "desktop"\` or the target app).
673
+ - Calling \`app_action\` with \`action: "takeover"\` immediately activates the full-screen sky-blue aura HUD, displays the warning banner "⚡ Scout is on the screen.", blocks external input interruptions, and takes over the mouse cursor!
674
+ - Then immediately proceed to launch/navigate with \`open_app\` or \`app_action\` (\`action: "click_element"\` / \`action: "click_app"\` / \`action: "type_text"\` / \`action: "send_keys"\`).
675
+ b. SCREEN CAPTURE & VISION DIRECTIVE: You CAN take screenshots and visually inspect or analyze the screen using \`app_action\` with \`action: "see_screen"\`, \`"analyze_screen"\`, or \`"capture_screen"\` whenever needed to verify UI state, check screen contents, or inspect open windows. All temporary screenshot files are automatically and securely deleted upon task completion for privacy and storage cleanliness.
676
+ c. VISUAL ELEMENT GROUNDING (click_element): Instead of guessing blind coordinates (x, y), click buttons, inputs, or menus by descriptive label using \`app_action(action: "click_element", payload: { element: "Search" })\`. Vision AI and native OS UI automation will locate the element and click it accurately.
677
+ d. WINDOW & SCROLL CONTROLS: Use \`app_action(action: "scroll", payload: { direction: "down", amount: 4 })\` to scroll pages. Use \`app_action(action: "list_windows")\` to see all open windows, and \`app_action(action: "focus_window", payload: { app: "chrome" })\` to bring a window front-and-center.
678
+ e. CLI Credential Input Prompt: If an application requires login credentials, passwords, 2FA codes, or secret tokens to proceed, call \`ask_user\` tool with a clear prompt. This presents a secure, interactive input bar directly in the user's running terminal CLI. Once the user enters the secret, take the received input, inject it into the target application window via \`app_action\` (\`action: "type_text"\`).
679
+ 21. MANDATORY TAKEOVER TOOL INVOCATION MANDATE: Whenever the user goal requests to open, launch, take over, click, type, or interact with an external app or desktop screen (e.g. 'take over browser and open website', 'open notepad and type', 'take over desktop', 'take over screen'):
680
+ YOU MUST CALL \`app_action(action: "takeover")\` IN STEP 1. Then call \`open_app\` and \`app_action\` (\`click_element\`, \`click_app\`, \`type_text\`, \`send_keys\`). DO NOT call \`read_file\` or \`write_file\` for workspace code files when asked to take over external desktop apps! The takeover tool call is MANDATORY for executing the physical takeover.
550
681
  22. BROWSER DIRECT URL NAVIGATION MANDATE: When asked to open or navigate to a specific website or web app (e.g. Apple Maps, GitHub, YouTube, etc.), NEVER call Google Search or issue repeated \`app_action: search\` calls with text queries! IMMEDIATELY pass the exact URL (e.g. "https://maps.apple.com") to \`open_app(app: "browser", target: "https://maps.apple.com")\` or \`app_action(app: "browser", action: "navigate", payload: { url: "https://maps.apple.com" })\`. Direct URL navigation must always target the exact site URL directly without putting queries into Google Search!
551
- 23. WORKING MEMORY & SCRATCHPAD PROTOCOL: For multi-step complex goals, maintain an active working memory scratchpad using the \`agent_scratchpad\` tool.
552
- a. Plan Checklist: Call \`agent_scratchpad(action: "update", plan: ["step 1", "step 2", ...])\` at the start of complex multi-step tasks.
553
- b. Reasoning Block: In every step, write a clear 1-2 sentence \`<scratchpad>\` reasoning block in your response before invoking tool calls to explain current state, working notes, and step selection.
554
- c. Real-Time Tracking: Update completed steps using \`agent_scratchpad(action: "update", completedSteps: [...])\` as you finish tasks so state is saved in \`.ft/scratchpad.md\`.
682
+ 23. WORKING MEMORY & STRUCTURED SCRATCHPAD PROTOCOL:
683
+ Like Antigravity and leading autonomous agents, maintain disciplined working memory. In EVERY step, start your response with a structured <scratchpad> reasoning block before returning tool calls:
684
+ \`\`\`markdown
685
+ <scratchpad>
686
+ Thought: [1-2 sentences on what you are doing on this turn and why]
687
+ Plan:
688
+ [x] 1. [Completed step]
689
+ [/] 2. [In-progress step]
690
+ [ ] 3. [Next upcoming step]
691
+ Observations: [What you learned from the last tool result or screen capture]
692
+ Next Action: [The exact tool you are invoking now]
693
+ </scratchpad>
694
+ \`\`\`
695
+ This keeps your reasoning crystal-clear, ensures plan progression, and syncs automatically with .ft/scratchpad.md.
555
696
 
556
697
  ${this.getScratchpadPromptContext()}
557
698
 
@@ -695,192 +836,229 @@ When returning tool calls, use standard OpenAI function calling format or JSON t
695
836
  let step = 0;
696
837
  let finalSummary = '';
697
838
  const stepDurations = [];
698
- while (step < this.maxSteps) {
699
- step++;
700
- this.currentStep = step;
701
- const stepStartTime = Date.now();
702
- const avgStepMs = stepDurations.length > 0
703
- ? stepDurations.reduce((a, b) => a + b, 0) / stepDurations.length
704
- : 12000;
705
- const remainingSteps = (this.maxSteps - step + 1);
706
- const estSecsLeft = Math.max(5, Math.round((remainingSteps * avgStepMs) / 1000));
707
- const estLeftStr = estSecsLeft >= 60
708
- ? `${Math.floor(estSecsLeft / 60)}m ${estSecsLeft % 60}s`
709
- : `${estSecsLeft}s`;
710
- const stepSpinner = spinner();
711
- stepSpinner.start(chalk.cyan(`Scout is Working (Step ${step}/${this.maxSteps} • Est. completion: ~${estLeftStr} left)`));
712
- try {
713
- let response;
839
+ try {
840
+ while (step < this.maxSteps) {
841
+ step++;
842
+ this.currentStep = step;
843
+ const stepStartTime = Date.now();
844
+ const avgStepMs = stepDurations.length > 0
845
+ ? stepDurations.reduce((a, b) => a + b, 0) / stepDurations.length
846
+ : 12000;
847
+ const remainingSteps = (this.maxSteps - step + 1);
848
+ const estSecsLeft = Math.max(5, Math.round((remainingSteps * avgStepMs) / 1000));
849
+ const estLeftStr = estSecsLeft >= 60
850
+ ? `${Math.floor(estSecsLeft / 60)}m ${estSecsLeft % 60}s`
851
+ : `${estSecsLeft}s`;
852
+ const stepSpinner = spinner();
853
+ stepSpinner.start(chalk.cyan(`Scout is Working (Step ${step}/${this.maxSteps} • Est. completion: ~${estLeftStr} left)`));
714
854
  try {
715
- response = await callOpenAIWithRetry(async (model) => {
716
- return await openai.chat.completions.create({
717
- model,
718
- messages: sanitizeMessages(this.historyMessages),
719
- tools: AGENT_TOOLS.map((t) => ({ type: 'function', function: t })),
720
- tool_choice: 'auto',
721
- temperature: 0.1,
722
- });
723
- });
724
- }
725
- catch (err) {
726
- const errStr = String(err?.message || err?.error || err || '').toLowerCase();
727
- if (errStr.includes('tool') || errStr.includes('400') || errStr.includes('not supported') || errStr.includes('reasoning')) {
855
+ let response;
856
+ try {
728
857
  response = await callOpenAIWithRetry(async (model) => {
729
858
  return await openai.chat.completions.create({
730
859
  model,
731
860
  messages: sanitizeMessages(this.historyMessages),
861
+ tools: AGENT_TOOLS.map((t) => ({ type: 'function', function: t })),
862
+ tool_choice: 'auto',
732
863
  temperature: 0.1,
864
+ presence_penalty: 0.1,
865
+ frequency_penalty: 0.1,
733
866
  });
734
867
  });
735
868
  }
736
- else {
737
- throw err;
738
- }
739
- }
740
- const choice = response.choices[0];
741
- if (!choice) {
742
- stepSpinner.stop(chalk.yellow('No response from AI model. Retrying step...'));
743
- continue;
744
- }
745
- const msg = sanitizeMessage(choice.message);
746
- this.historyMessages.push(msg);
747
- // Render assistant's thought/reasoning text if provided on this turn
748
- const textContent = msg.content || '';
749
- if (textContent.trim()) {
750
- safeNote(renderMarkdown(textContent), `Scout Thought (Step ${step})`);
751
- }
752
- // Check native tool calls
753
- if (msg.tool_calls && msg.tool_calls.length > 0) {
754
- stepSpinner.stop(chalk.green(`Step ${step}: Scout issued ${msg.tool_calls.length} tool call(s).`));
755
- for (const tc of msg.tool_calls) {
756
- if (tc.type === 'function' && tc.function) {
757
- const fnName = tc.function.name;
758
- let args = {};
759
- try {
760
- args = JSON.parse(tc.function.arguments || '{}');
761
- }
762
- catch { }
763
- const toolResult = await this.dispatchToolCall(fnName, args);
764
- this.historyMessages.push({
869
+ catch (err) {
870
+ const errStr = String(err?.message || err?.error || err || '').toLowerCase();
871
+ if (errStr.includes('tool') || errStr.includes('400') || errStr.includes('not supported') || errStr.includes('reasoning')) {
872
+ // Model doesn't support native function calling — inject tool-call formatting hint
873
+ // so extractJsonToolCall can parse the response as a structured tool invocation
874
+ const toolHintMsg = {
765
875
  role: 'user',
766
- tool_call_id: tc.id,
767
- content: `Tool Execution Result (${fnName}):\n${toolResult.result}`,
876
+ content: `IMPORTANT: This model does not support native function/tool calling. You MUST format your tool invocations as a JSON code block in your response like this:
877
+ \`\`\`json
878
+ { "tool": "tool_name", "args": { ... } }
879
+ \`\`\`
880
+ Available tools: read_file, write_file, edit_file, run_command, list_dir, grep_search, glob_search, tree_view, file_info, multi_edit_file, fetch_url, git_diff, open_app, app_action, agent_scratchpad, ask_user, speak_text, task_completed.
881
+ For app takeover/control use: { "tool": "app_action", "args": { "app": "browser", "action": "takeover" } }
882
+ For opening apps use: { "tool": "open_app", "args": { "app": "chrome", "target": "https://..." } }
883
+ For clicking use: { "tool": "app_action", "args": { "app": "desktop", "action": "click_app", "payload": { "x": 500, "y": 300 } } }
884
+ For typing text use: { "tool": "app_action", "args": { "app": "browser", "action": "type_text", "payload": { "text": "...", "enter": true } } }
885
+ You MUST output exactly ONE JSON code block per tool call. Do NOT describe what you would do in plain text—output the JSON tool call directly!`,
886
+ };
887
+ const fallbackMessages = [...this.historyMessages, toolHintMsg];
888
+ response = await callOpenAIWithRetry(async (model) => {
889
+ return await openai.chat.completions.create({
890
+ model,
891
+ messages: sanitizeMessages(fallbackMessages),
892
+ temperature: 0.1,
893
+ presence_penalty: 0.1,
894
+ frequency_penalty: 0.1,
895
+ });
768
896
  });
769
- this.handleToolConsecutiveTracking(fnName, step);
770
- if (fnName === 'task_completed') {
771
- finalSummary = args.summary || toolResult.result;
772
- if (this.modifiedFiles.size > 0) {
773
- const filesList = getDirectoryFiles(this.cwd);
774
- const healRes = await verifyAndSelfHealFiles(Array.from(this.modifiedFiles), this.cwd, this.projectName, filesList, { maxRetries: 3 });
775
- if (healRes.verifiedFiles.length > 0) {
776
- safeNote(chalk.green(` Self-Healing Verification Confirmed: ${healRes.verifiedFiles.length} file(s) syntax & build clean!`), ' Code Verification Clean');
777
- }
778
- if (healRes.remainingErrors.length > 0) {
779
- safeNote(chalk.yellow(`️ Remaining verification issues:\n${healRes.remainingErrors.join('\n')}`), '️ Verification Warning');
780
- }
897
+ }
898
+ else {
899
+ throw err;
900
+ }
901
+ }
902
+ const choice = response.choices[0];
903
+ if (!choice) {
904
+ stepSpinner.stop(chalk.yellow('No response from Scout. Retrying step...'));
905
+ continue;
906
+ }
907
+ const msg = sanitizeMessage(choice.message);
908
+ this.historyMessages.push(msg);
909
+ // Render assistant's thought/scratchpad reasoning text if provided on this turn
910
+ const textContent = msg.content || '';
911
+ if (textContent.trim()) {
912
+ const scratchMatch = textContent.match(/<scratchpad>([\s\S]*?)<\/scratchpad>/i);
913
+ if (scratchMatch && scratchMatch[1]) {
914
+ const scratchText = scratchMatch[1].trim();
915
+ safeNote(renderMarkdown(scratchText), chalk.cyan.bold(`Scout Scratchpad & Working Memory (Step ${step})`));
916
+ this.autoSyncScratchpadFromText(scratchText);
917
+ const remainingText = textContent.replace(/<scratchpad>[\s\S]*?<\/scratchpad>/i, '').trim();
918
+ if (remainingText) {
919
+ safeNote(renderMarkdown(remainingText), `Scout Thought (Step ${step})`);
920
+ }
921
+ }
922
+ else {
923
+ safeNote(renderMarkdown(textContent), `Scout Thought (Step ${step})`);
924
+ }
925
+ }
926
+ // Check native tool calls
927
+ if (msg.tool_calls && msg.tool_calls.length > 0) {
928
+ stepSpinner.stop(chalk.green(`Step ${step}: Scout issued ${msg.tool_calls.length} tool call(s).`));
929
+ for (const tc of msg.tool_calls) {
930
+ if (tc.type === 'function' && tc.function) {
931
+ const fnName = tc.function.name;
932
+ let args = {};
933
+ try {
934
+ args = JSON.parse(tc.function.arguments || '{}');
781
935
  }
782
- saveAgentSession({
783
- goal: userGoal,
784
- summary: finalSummary,
785
- modifiedFiles: Array.from(this.modifiedFiles),
936
+ catch { }
937
+ const toolResult = await this.dispatchToolCall(fnName, args);
938
+ this.historyMessages.push({
939
+ role: 'user',
940
+ tool_call_id: tc.id,
941
+ content: `Tool Execution Result (${fnName}):\n${toolResult.result}`,
786
942
  });
787
- return { success: true, summary: finalSummary };
943
+ this.handleToolConsecutiveTracking(fnName, args, step);
944
+ if (fnName === 'task_completed') {
945
+ finalSummary = args.summary || toolResult.result;
946
+ if (this.modifiedFiles.size > 0) {
947
+ const filesList = getDirectoryFiles(this.cwd);
948
+ const healRes = await verifyAndSelfHealFiles(Array.from(this.modifiedFiles), this.cwd, this.projectName, filesList, { maxRetries: 3 });
949
+ if (healRes.verifiedFiles.length > 0) {
950
+ safeNote(chalk.green(` Self-Healing Verification Confirmed: ${healRes.verifiedFiles.length} file(s) syntax & build clean!`), ' Code Verification Clean');
951
+ }
952
+ if (healRes.remainingErrors.length > 0) {
953
+ safeNote(chalk.yellow(`️ Remaining verification issues:\n${healRes.remainingErrors.join('\n')}`), '️ Verification Warning');
954
+ }
955
+ }
956
+ saveAgentSession({
957
+ goal: userGoal,
958
+ summary: finalSummary,
959
+ modifiedFiles: Array.from(this.modifiedFiles),
960
+ });
961
+ return { success: true, summary: finalSummary };
962
+ }
788
963
  }
789
964
  }
965
+ continue;
790
966
  }
791
- continue;
792
- }
793
- // Check text response or fallback JSON tool call
794
- stepSpinner.stop(chalk.blue(`Step ${step} thinking complete.`));
795
- const parsedJsonTool = this.extractJsonToolCall(textContent);
796
- if (parsedJsonTool) {
797
- const toolResult = await this.dispatchToolCall(parsedJsonTool.tool, parsedJsonTool.args);
967
+ // Check text response or fallback JSON tool call
968
+ stepSpinner.stop(chalk.blue(`Step ${step} thinking complete.`));
969
+ const parsedJsonTool = this.extractJsonToolCall(textContent);
970
+ if (parsedJsonTool) {
971
+ const toolResult = await this.dispatchToolCall(parsedJsonTool.tool, parsedJsonTool.args);
972
+ this.historyMessages.push({
973
+ role: 'user',
974
+ content: `Tool Execution Result (${parsedJsonTool.tool}):\n${toolResult.result}`,
975
+ });
976
+ this.handleToolConsecutiveTracking(parsedJsonTool.tool, parsedJsonTool.args, step);
977
+ if (parsedJsonTool.tool === 'task_completed') {
978
+ finalSummary = parsedJsonTool.args?.summary || toolResult.result;
979
+ saveAgentSession({
980
+ goal: userGoal,
981
+ summary: finalSummary,
982
+ modifiedFiles: Array.from(this.modifiedFiles),
983
+ });
984
+ return { success: true, summary: finalSummary };
985
+ }
986
+ continue;
987
+ }
988
+ if (textContent.trim()) {
989
+ let cleanedThought = textContent
990
+ .replace(/[\u0600-\u06FF\u0750-\u077F\uAC00-\uD7AF\u3040-\u30FF\u4E00-\u9FFF\u0D80-\u0DFF]+/g, '')
991
+ .trim();
992
+ if (!cleanedThought || cleanedThought.length < 5) {
993
+ cleanedThought = 'Analyzing codebase files and executing next tool operation...';
994
+ }
995
+ safeNote(renderMarkdown(cleanedThought), ` Scout Agent Thought (Step ${step})`);
996
+ const lowerText = textContent.toLowerCase();
997
+ const isExplicitCompletion = lowerText.includes('task is complete') ||
998
+ lowerText.includes('task complete') ||
999
+ lowerText.includes('goal completed') ||
1000
+ lowerText.includes('goal is completed') ||
1001
+ lowerText.includes('all tasks completed') ||
1002
+ lowerText.includes('i have completed') ||
1003
+ lowerText.includes('no further changes needed') ||
1004
+ lowerText.includes('the fix is complete') ||
1005
+ lowerText.includes('has been created') ||
1006
+ lowerText.includes('successfully created') ||
1007
+ lowerText.includes('created the file') ||
1008
+ lowerText.includes('file created') ||
1009
+ lowerText.includes('implementation complete') ||
1010
+ lowerText.includes('work is complete');
1011
+ // If explicit completion phrase found, OR files have already been modified and assistant returned a final summary without calling tools
1012
+ if (isExplicitCompletion || (this.modifiedFiles.size > 0 && !lowerText.includes('?') && textContent.length > 50)) {
1013
+ saveAgentSession({
1014
+ goal: userGoal,
1015
+ summary: textContent,
1016
+ modifiedFiles: Array.from(this.modifiedFiles),
1017
+ });
1018
+ return { success: true, summary: textContent };
1019
+ }
1020
+ }
1021
+ // If assistant responded with text without calling tools, prompt it to execute tools to complete the goal
798
1022
  this.historyMessages.push({
799
1023
  role: 'user',
800
- content: `Tool Execution Result (${parsedJsonTool.tool}):\n${toolResult.result}`,
1024
+ content: 'You provided a text response but have not called any tools (write_file, edit_file, run_command, task_completed). Please execute necessary tool calls to complete the user goal, or invoke task_completed if finished.',
801
1025
  });
802
- this.handleToolConsecutiveTracking(parsedJsonTool.tool, step);
803
- if (parsedJsonTool.tool === 'task_completed') {
804
- finalSummary = parsedJsonTool.args?.summary || toolResult.result;
805
- saveAgentSession({
806
- goal: userGoal,
807
- summary: finalSummary,
808
- modifiedFiles: Array.from(this.modifiedFiles),
809
- });
810
- return { success: true, summary: finalSummary };
811
- }
812
- continue;
1026
+ stepDurations.push(Date.now() - stepStartTime);
813
1027
  }
814
- if (textContent.trim()) {
815
- let cleanedThought = textContent
816
- .replace(/[\u0600-\u06FF\u0750-\u077F\uAC00-\uD7AF\u3040-\u30FF\u4E00-\u9FFF\u0D80-\u0DFF]+/g, '')
817
- .trim();
818
- if (!cleanedThought || cleanedThought.length < 5) {
819
- cleanedThought = 'Analyzing codebase files and executing next tool operation...';
820
- }
821
- safeNote(renderMarkdown(cleanedThought), ` Scout Agent Thought (Step ${step})`);
822
- const lowerText = textContent.toLowerCase();
823
- const isExplicitCompletion = lowerText.includes('task is complete') ||
824
- lowerText.includes('task complete') ||
825
- lowerText.includes('goal completed') ||
826
- lowerText.includes('goal is completed') ||
827
- lowerText.includes('all tasks completed') ||
828
- lowerText.includes('i have completed') ||
829
- lowerText.includes('no further changes needed') ||
830
- lowerText.includes('the fix is complete') ||
831
- lowerText.includes('has been created') ||
832
- lowerText.includes('successfully created') ||
833
- lowerText.includes('created the file') ||
834
- lowerText.includes('file created') ||
835
- lowerText.includes('implementation complete') ||
836
- lowerText.includes('work is complete');
837
- // If explicit completion phrase found, OR files have already been modified and assistant returned a final summary without calling tools
838
- if (isExplicitCompletion || (this.modifiedFiles.size > 0 && !lowerText.includes('?') && textContent.length > 50)) {
839
- saveAgentSession({
840
- goal: userGoal,
841
- summary: textContent,
842
- modifiedFiles: Array.from(this.modifiedFiles),
843
- });
844
- return { success: true, summary: textContent };
1028
+ catch (err) {
1029
+ stepDurations.push(Date.now() - stepStartTime);
1030
+ if (isQuotaExceededError(err)) {
1031
+ stepSpinner.stop(chalk.red(`Oops! it\'s not you, it\'s us`));
1032
+ safeNote(`${chalk.bold.red(`Something went wrong in (Step ${step}), please try again in a moment`)}\n\n` +
1033
+ `${chalk.yellow('The configured AI provider has reached its API usage limit or rate cap.')}\n` +
1034
+ `${chalk.dim('This is separate from your Scout credit balance shown by `scout quota`.')}\n\n` +
1035
+ `${chalk.bold.cyan(' Please come back and try again in a few hours (or check back later today).')}\n\n` +
1036
+ `${chalk.dim('Scout Agent session has ended gracefully to protect remaining workflow.')}`, '️ API Quota Limit Reached');
1037
+ return {
1038
+ success: false,
1039
+ summary: 'Something went wrong, please try again in a moment.',
1040
+ };
845
1041
  }
1042
+ stepSpinner.stop(chalk.red(`Step ${step} execution error: ${err?.message || String(err)}`));
1043
+ this.historyMessages.push({
1044
+ role: 'user',
1045
+ content: `Error in previous turn: ${err?.message || String(err)}. Please try alternative steps or call tools.`,
1046
+ });
846
1047
  }
847
- // If assistant responded with text without calling tools, prompt it to execute tools to complete the goal
848
- this.historyMessages.push({
849
- role: 'user',
850
- content: 'You provided a text response but have not called any tools (write_file, edit_file, run_command, task_completed). Please execute necessary tool calls to complete the user goal, or invoke task_completed if finished.',
851
- });
852
- stepDurations.push(Date.now() - stepStartTime);
853
- }
854
- catch (err) {
855
- stepDurations.push(Date.now() - stepStartTime);
856
- if (isQuotaExceededError(err)) {
857
- stepSpinner.stop(chalk.red(`Oops! it\'s not you, it\'s us`));
858
- safeNote(`${chalk.bold.red(`Something went wrong in (Step ${step}), please try again in a moment`)}\n\n` +
859
- `${chalk.yellow('The configured AI provider has reached its API usage limit or rate cap.')}\n` +
860
- `${chalk.dim('This is separate from your Scout credit balance shown by `scout quota`.')}\n\n` +
861
- `${chalk.bold.cyan(' Please come back and try again in a few hours (or check back later today).')}\n\n` +
862
- `${chalk.dim('Scout Agent session has ended gracefully to protect remaining workflow.')}`, '️ API Quota Limit Reached');
863
- return {
864
- success: false,
865
- summary: 'Something went wrong, please try again in a moment.',
866
- };
1048
+ // If current max steps limit is reached while task is still in progress, auto-extend by +15 extra steps (up to 60 max threshold)
1049
+ if (step >= this.maxSteps && this.maxSteps < 60) {
1050
+ this.maxSteps += 15;
1051
+ safeNote(chalk.bold.yellow(`⚡ Step limit reached while task is in progress. Automatically extending execution by +15 extra steps (New Max Limit: ${this.maxSteps})...`), ' Auto Extra Steps Extension');
867
1052
  }
868
- stepSpinner.stop(chalk.red(`Step ${step} execution error: ${err?.message || String(err)}`));
869
- this.historyMessages.push({
870
- role: 'user',
871
- content: `Error in previous turn: ${err?.message || String(err)}. Please try alternative steps or call tools.`,
872
- });
873
- }
874
- // If current max steps limit is reached while task is still in progress, auto-extend by +15 extra steps (up to 60 max threshold)
875
- if (step >= this.maxSteps && this.maxSteps < 60) {
876
- this.maxSteps += 15;
877
- safeNote(chalk.bold.yellow(`⚡ Step limit reached while task is in progress. Automatically extending execution by +15 extra steps (New Max Limit: ${this.maxSteps})...`), ' Auto Extra Steps Extension');
878
1053
  }
1054
+ return {
1055
+ success: false,
1056
+ summary: `Reached max iteration steps limit (${this.maxSteps}). Modified files: ${Array.from(this.modifiedFiles).join(', ')}`,
1057
+ };
1058
+ }
1059
+ finally {
1060
+ cleanupScreenshots();
879
1061
  }
880
- return {
881
- success: false,
882
- summary: `Reached max iteration steps limit (${this.maxSteps}). Modified files: ${Array.from(this.modifiedFiles).join(', ')}`,
883
- };
884
1062
  }
885
1063
  extractJsonToolCall(content) {
886
1064
  if (!content)
@@ -900,7 +1078,10 @@ When returning tool calls, use standard OpenAI function calling format or JSON t
900
1078
  const knownTools = [
901
1079
  'read_file', 'write_file', 'edit_file', 'run_command', 'list_dir',
902
1080
  'grep_search', 'glob_search', 'tree_view', 'file_info', 'multi_edit_file',
903
- 'fetch_url', 'git_diff', 'open_app', 'app_action', 'ask_user', 'task_completed'
1081
+ 'fetch_url', 'git_diff', 'open_app', 'app_action', 'agent_scratchpad',
1082
+ 'ask_user', 'speak_text', 'task_completed',
1083
+ 'desktop_action', 'type_text', 'click_app', 'send_keys', 'takeover',
1084
+ 'capture_screen', 'see_screen', 'analyze_screen', 'scratchpad',
904
1085
  ];
905
1086
  for (const toolName of knownTools) {
906
1087
  const tagRegex = new RegExp(`<${toolName}>([\\s\\S]*?)<\\/${toolName}>`, 'i');
@@ -1374,7 +1555,7 @@ When returning tool calls, use standard OpenAI function calling format or JSON t
1374
1555
  case 'app_action': {
1375
1556
  const appName = String(args.app || args.appName || 'desktop').trim();
1376
1557
  const action = String(args.action || (name !== 'app_action' ? name : 'type_text')).trim();
1377
- const payload = args.payload !== undefined ? args.payload : (args.text || args.content || args.target || args.url || args.command || args);
1558
+ const payload = args.payload !== undefined ? args.payload : (args.text || args.content || args.target || args.query || args.element || args.url || args.command || args);
1378
1559
  const actionRes = await executeInApp(appName, action, payload);
1379
1560
  return { result: actionRes.output };
1380
1561
  }
@@ -1407,6 +1588,7 @@ When returning tool calls, use standard OpenAI function calling format or JSON t
1407
1588
  case 'task_completed': {
1408
1589
  const summaryStr = String(args.summary || 'Task completed successfully.');
1409
1590
  speakText(summaryStr, { async: true });
1591
+ cleanupScreenshots();
1410
1592
  return { result: summaryStr };
1411
1593
  }
1412
1594
  default: