diff --git a/app/build.gradle.kts b/app/build.gradle.kts index 0b226f21..c2dc3273 100644 --- a/app/build.gradle.kts +++ b/app/build.gradle.kts @@ -15,8 +15,8 @@ android { applicationId = "com.autonion.automationcompanion" minSdk = 24 targetSdk = 36 - versionCode = 9 - versionName = "1.1.0" + versionCode = 10 + versionName = "1.1.1" testInstrumentationRunner = "androidx.test.runner.AndroidJUnitRunner" diff --git a/app/src/main/assets/knowledge/android_app_guide.md b/app/src/main/assets/knowledge/android_app_guide.md index 56f2966d..25ff84d8 100644 --- a/app/src/main/assets/knowledge/android_app_guide.md +++ b/app/src/main/assets/knowledge/android_app_guide.md @@ -9,7 +9,7 @@ Autonion Automation Companion is an Android application that enables intelligent Autonion has 8 core features. Here is a complete list: 1. Omni-Chat (Unified Chatbot Interface) - The main interaction point. Accessible via the floating action button (FAB). Routes commands to the appropriate engine using on-device NLU. -2. Semantic Automation (AI-Powered Agent) - Autonomous multi-step task execution powered by LLM. Uses an agentic loop to interact with device UI. Supports Server LLM (Ollama), On-Device SLM (GGUF), and Cloud API inference modes. +2. Semantic Automation - Smart multi-step task execution powered by LLM. Uses a step-by-step automation loop to interact with device UI. Supports Server LLM (Ollama), On-Device SLM (GGUF), and Cloud API inference modes. 3. Cross-Device Automation - Send commands from Android to a desktop computer with secure OTP-based pairing. Includes remote Desktop Flow triggering, clipboard sync, and rule-based automation. Requires the Autonion Desktop Agent (github.com/Autonion/Autonion-Agent). 4. Gesture Recording and Playback - Record touch interactions (taps, swipes, long presses, drags) and replay them automatically. Coordinate-based replay. 5. Flow Builder - Visual drag-and-drop automation workflow builder. Create flows with triggers, actions, conditions, and connections. Includes Screen Understanding nodes with YOLO+UI attribute, UI attribute-only, and OCR modes. @@ -43,7 +43,7 @@ Omni-Chat modes and what they mean: - Desktop (link icon): Command sent to connected desktop computer. Example: "on my laptop open chrome". - Timer (clock icon): Scheduled or recurring task with stop control. Example: "click next every 1 minute". - FAQ (lightbulb icon): Instant answer from the built-in FAQ database, no LLM required. Example: "how do I connect devices?". -- Knowledge (book icon): Answer synthesized from documentation using RAG (Retrieval-Augmented Generation). Example: "explain how the agentic loop works". +- Knowledge (book icon): Answer synthesized from documentation using RAG (Retrieval-Augmented Generation). Example: "explain how the automation loop works". - Chat (speech bubble icon): General conversational LLM response. - System (gear icon): Error or system notification messages. @@ -68,23 +68,23 @@ LLM settings inside Omni-Chat: - You can toggle between Server LLM and Local SLM inference modes. - The connection auto-reconnects when the chat is reopened. -### Semantic Automation (AI-Powered Agent) -The AI-powered automation engine that understands natural language commands and executes complex multi-step tasks on your device autonomously. +### Semantic Automation +The AI-powered automation engine that understands natural language commands and helps you execute complex multi-step tasks on your device. How the Semantic Automation Engine works in detail: 1. Goal Parsing: Your natural language command (e.g., "search for shoes under 2000 on Flipkart") is sent to the LLM. The LLM extracts structured data: the task type (search, open, enable, etc.), the target app (Flipkart), the search query (shoes under 2000), and the domain (flipkart.com). 2. Pre-Actions: The engine launches the target app or opens system settings. If the target app is not installed, a dialog asks you to choose between Play Store, Browser fallback, or Cancel. -3. Screen Loop (the Agentic Loop): This is the core automation cycle that repeats up to 50 iterations: +3. Screen Loop (the Automation Loop): This is the core automation cycle that repeats up to 50 iterations: - Step A: Capture the current screen via screenshot. - Step B: Build a ScreenUIState from the accessibility tree. Elements include buttons, text fields, labels with their names, types, and bounding boxes. When a browser with the extension is active, DOM elements are used instead. - Step C: Compare with the previous UI state for post-action verification. If the screen has not changed after an action, it counts as a failure. - - Step D: Predict the next action using a multi-tier fallback system (see Action Prediction below). - - Step E: Execute the predicted action (click, type, scroll, press key, finish). + - Step D: Determine the next step using a multi-tier fallback system (see Action Matching below). + - Step E: Perform the selected action (click, type, scroll, press key, finish). - Step F: Wait 2.5 seconds for the screen to settle, then repeat. -4. Completion: The loop stops when the LLM predicts a FINISH action, when the user cancels, or when max iterations (50) are reached. +4. Completion: The loop stops when a FINISH action is reached, when the user cancels, or when max iterations (50) are reached. -Action prediction (multi-tier fallback): -The engine tries these prediction tiers in order until one succeeds: +Action matching (multi-tier fallback): +The engine tries these matching tiers in order until one succeeds: - Tier 0: Deterministic Task Planner - handles standard flows (search, play) without the LLM. Reserved for common patterns. - Tier 1 (Server LLM mode): Ollama server via REST API. Sends a system prompt, user prompt with screen elements, and step history. Receives structured JSON with the action type, target element index, and input text. - Tier 1 (Cloud API mode): Cloud LLM via OpenAI-compatible API. Supports OpenAI (GPT-4o), Google Gemini, Groq, DeepSeek, Mistral, Together AI, OpenRouter, Ollama Cloud, or any custom OpenAI-compatible endpoint. Same prompt format as Server LLM. @@ -125,7 +125,7 @@ Intent types: - DIRECT_KEY_ACTION: Press a specific key or type text. Examples: "press enter", "type hello world" - DIRECT_TOGGLE: Turn a system setting on or off. Examples: "turn off wifi", "enable bluetooth" - SCHEDULED_ACTION: Repeat an action on a timer. Examples: "click next every 1 minute", "press space 5 times" -- DEVICE_AUTOMATION: Complex on-device task requiring the AI agent. Examples: "search shoes on Flipkart", "open settings and enable dark mode" +- DEVICE_AUTOMATION: Complex on-device task using Semantic Automation. Examples: "search shoes on Flipkart", "open settings and enable dark mode" - CROSS_DEVICE: Command targeted at the desktop computer. Examples: "on my laptop open chrome", "on desktop open notepad" - FAQ: Question that matches the FAQ database. Detected via semantic similarity. - Q_AND_A: General knowledge question answered via RAG or LLM. Examples: "what features does this app have?", "how to set up Ollama?" @@ -332,9 +332,9 @@ How it works on Android: 3. When the Semantic Engine detects a browser task, it launches the browser with the target URL. 4. The extension captures the webpage DOM (interactive elements, text, links, buttons). 5. The DOM elements are sent back to the Android app and presented to the LLM as numbered elements. -6. The LLM predicts actions (click element, type text, scroll) using element IDs from the DOM snapshot. +6. The AI model maps the request to the next step (click element, type text, scroll) using element IDs from the DOM snapshot. 7. The action commands are sent back to the extension which executes them in the webpage. -8. After each action, the extension automatically captures a fresh DOM snapshot, enabling the agentic loop: DOM Snapshot → LLM Decision → Action Command → DOM Snapshot → repeat. +8. After each action, the extension automatically captures a fresh DOM snapshot, enabling the automation loop: DOM Snapshot → LLM Decision → Action Command → DOM Snapshot → repeat. Supported browsers with extensions on Android: Kiwi Browser (recommended), Lemur Browser, Firefox Nightly. @@ -388,7 +388,7 @@ Source code: github.com/Autonion/Autonion-Android-Extension #### Difference between the two extensions - Autonion Extension (Desktop): Runs in Chrome/Edge on your PC. Connects to the Autonion Desktop Agent. Used for cross-device browser automation where you control the desktop browser from your phone. -- Autonion Android Extension (Mobile): Runs in Kiwi/Lemur/Firefox Nightly on your phone. Connects directly to the Autonion Android app. Used for on-device browser automation where the AI agent controls the mobile browser. +- Autonion Android Extension (Mobile): Runs in Kiwi/Lemur/Firefox Nightly on your phone. Connects directly to the Autonion Android app. Used for on-device browser automation to help navigate web content in the mobile browser. ### Automation Debugger View detailed, categorized logs of all automation activities. @@ -396,7 +396,7 @@ View detailed, categorized logs of all automation activities. Log categories: - Accessibility Events: Raw accessibility service events (node clicks, text changes, window transitions) - Cross-Device Sync: WebSocket communication logs (messages sent, received, connection status) -- Semantic Actions: AI agent action predictions and executions (LLM prompts, predicted actions, execution results, step history) +- Semantic Actions: Automation step execution history (LLM prompts, selected actions, execution results, step history) - System Events: App-level events and errors How to use: @@ -537,7 +537,7 @@ This is usually caused by the LLM making poor predictions: - Break complex commands into simpler steps. Instead of "open WhatsApp, find Mom's chat, and send her a birthday wish", try "open WhatsApp" first, then "search for Mom" separately. - Close unnecessary apps to reduce screen UI noise. The fewer elements on screen, the easier it is for the LLM to identify the right target. - Check the Automation Debugger > Semantic Actions to see exactly what the LLM predicted. This helps identify whether the issue is with element detection or action selection. -- Use Omni-Chat for simple commands (like "press next" or "turn off wifi") which bypass the AI agent entirely and execute instantly. +- Use Omni-Chat for simple commands (like "press next" or "turn off wifi") which execute instantly via direct actions. ### Cannot connect to desktop - Make sure the Autonion Desktop Agent (from github.com/Autonion/Autonion-Agent/releases) is installed and running on your PC. diff --git a/app/src/main/assets/knowledge/desktop_agent_guide.md b/app/src/main/assets/knowledge/desktop_agent_guide.md index 4a5a996b..4e3f96c9 100644 --- a/app/src/main/assets/knowledge/desktop_agent_guide.md +++ b/app/src/main/assets/knowledge/desktop_agent_guide.md @@ -188,7 +188,7 @@ Use cases: For web automation tasks, the Autonion Extension can be installed in Chrome or Edge on the desktop: - The extension captures the webpage DOM (links, buttons, text fields, forms) - DOM snapshots are sent to the Desktop Agent, which forwards them to the Android app -- The AI agent can then predict actions on web content using element IDs from the DOM +- The engine can then perform requested actions on web content using element IDs from the DOM - This is significantly more reliable than using the Windows accessibility tree for web content Installation: diff --git a/app/src/main/assets/knowledge/desktop_agent_guide.md.bak b/app/src/main/assets/knowledge/desktop_agent_guide.md.bak deleted file mode 100644 index b8dd8c30..00000000 --- a/app/src/main/assets/knowledge/desktop_agent_guide.md.bak +++ /dev/null @@ -1,343 +0,0 @@ -# Autonion Desktop Agent Guide - -## Overview - -The Autonion Desktop Agent is a Flutter-based companion application that runs on Windows (with potential macOS and Linux support). It receives commands from the Autonion Android app over your local WiFi network and executes them on your desktop. Think of it as a remote assistant: you speak to your phone, and your computer acts. All communication stays on your local network with no cloud dependencies. - -## Architecture - -### Core Components - -1. WebSocket Server (websocket_service.dart) -The central communication hub. It listens on port 4545 (or a dynamic fallback if 4545 is busy) for incoming connections from the Android app and the browser extension. -- Manages multiple simultaneous WebSocket clients -- Tracks whether the browser extension is connected (separate from Android clients) -- Broadcasts events to all clients or sends targeted messages to just the extension -- Handles connection acknowledgment, ping/pong heartbeats, and graceful disconnection - -2. mDNS Discovery (discovery_service.dart) -Broadcasts the Desktop Agent's presence on your local network using Bonjour/mDNS (service type: _autonion._tcp). This allows the Android app to discover the desktop automatically without manual IP entry. - -3. Connection Provider (connection_provider.dart) -The central orchestrator that wires all services together and routes incoming WebSocket commands to the right handler. It is the "brain" of the Desktop Agent that decides what to do with each incoming message. - -4. Desktop Agent Service (desktop_agent_service.dart) -The agentic loop for desktop automation. When a desktop-related task arrives, this service takes over and autonomously interacts with the desktop UI using an LLM for decision-making. - -5. Python Bridge Service (python_bridge_service.dart) -Manages a Python subprocess that handles system-level interactions. The Python bridge uses the uiautomation and pyautogui libraries to read the Windows accessibility tree, simulate mouse clicks, keyboard input, and capture screenshots. - -6. Browser Launcher Service (browser_launcher_service.dart) -Detects installed Chromium-based browsers on the system (Chrome, Edge, Brave) and launches them when browser automation is needed. Auto-selects the first detected browser. - -7. Clipboard Sync Service (clipboard_sync_service.dart) -Bidirectional clipboard synchronization between Android and desktop. Polls the desktop clipboard every second and sends text changes to connected Android devices. Receives clipboard text from Android and writes it to the desktop clipboard. - -8. Trigger Rule Service (trigger_rule_service.dart) -Manages event-based automation rules registered by the Android app. Rules are stored in memory and forwarded to the browser extension. When the extension detects a rule condition is met, the event is relayed back through the Desktop Agent to Android. - -9. AI Provider System (ai/ directory) -Abstracted AI service layer supporting multiple LLM providers: -- Ollama Service: Connects to a local Ollama instance at http://localhost:11434/api/chat. Supports structured JSON output via the format parameter, vision models (base64 images in messages), and model listing/selection. -- API Key Service: For commercial API providers with API key authentication. -- Web-Based AI Service: For web-based AI platforms (ChatGPT, Gemini) through the browser extension content scripts. -The active provider is managed by AiProviderNotifier and can be switched at runtime. - -10. System Services -- System Tray Service: Puts the app in the Windows system tray so it runs in the background. -- Window Manager Service: Controls the Flutter window visibility, size, and position. -- Startup Service: Handles auto-start on system boot. - -### How the Python Bridge Works - -The Python bridge is the layer that performs actual system-level actions on the desktop: - -1. Setup process: - - On first use, the bridge locates a Python 3 installation on the system. - - It creates a virtual environment (venv) in the AppData directory. - - It installs required packages: uiautomation, pyautogui, mss, Pillow. - - It spawns the desktop_agent.py script as a subprocess. - -2. Communication protocol: - - The Flutter app writes JSON commands to the Python process's stdin. - - The Python process writes JSON responses back on stdout. - - Each command has a unique ID for request/response matching. - - Timeout: 30 seconds per command. - - Python stderr is captured as debug logs. - -3. Available Python bridge commands: - - ping: Health check. Returns "pong". - - get_screen_state: Reads the Windows accessibility tree of the foreground window and optionally captures a screenshot. Returns a list of UI elements with their names, roles, bounding boxes, and interactability. - - execute_action: Performs a UI action (click, type, scroll, hotkey, wait). - -4. Accessibility tree reading: - - Uses the Windows UI Automation framework (via Python uiautomation library). - - Reads the foreground window's element tree up to depth 10. - - Extracts actionable elements: buttons, menu items, tabs, hyperlinks, list items, checkboxes, radio buttons, edit controls, combo boxes. - - Each element gets a node_ID for targeting. Elements cached for click execution. - - Names are truncated to 100 characters for manageable prompt sizes. - -5. Action execution: - - click: Moves the mouse to the center of the target element's bounding box and clicks. - - type: Optionally clicks a target element first, then types text character by character. - - scroll: Scrolls up or down using pyautogui. - - hotkey: Presses key combinations (e.g., Ctrl+C, Win+R, Alt+Tab). - - wait: Pauses for 1 second. - -Python installation requirement: Python 3.8 or newer must be installed and accessible in PATH. The bridge will search for "python", "py", or "python3" commands automatically. - -## Communication Flow - -### When Android Sends a Command - -Step-by-step flow of what happens when the Android app sends a command to the Desktop Agent: - -1. The Android app sends a JSON message over the WebSocket connection. The message contains either a natural language prompt, a structured key_press command, a schedule command, or other action types. - -2. The ConnectionProvider's _executeCommand method receives the message and determines its type: - - If it contains a "prompt" field (natural language): - a. The agent first classifies whether the prompt is browser-related or desktop-related. - b. Classification method: An LLM prompt asks the AI to classify the prompt as "browser" or "desktop" with one word. If the LLM is unavailable, keyword matching is used as fallback (looking for words like "youtube", "website", "amazon", "google", ".com", etc.). - c. Browser-related prompts go to the browser extension via the agentic DOM-aware loop. - d. Desktop-related prompts go to the Desktop Agent Service for autonomous execution. - - If type is "key_press" (structured command): - - The key is sent directly to the Python bridge for execution via hotkey. No LLM needed. - - If type is "schedule": - - A periodic timer starts that executes the specified key press at the specified interval. - - Runs until the repeat count is reached or a schedule_cancel command is received. - - If type is "clipboard.text_copied": - - The text is written to the desktop clipboard via ClipboardSyncService. - - If type is "register_triggers": - - Trigger rules are stored and forwarded to the browser extension. - - If type is "open_url": - - The URL is launched in the system default browser. - -3. Status responses are sent back to Android at each stage: started, in_progress, completed, or failed. - -### Desktop Automation (Agentic Loop) - -When a desktop task is identified, the DesktopAgentService runs its agentic loop: - -1. Observe: The Python bridge reads the foreground window's accessibility tree and returns a list of UI elements. -2. Build Prompt: The screen state, user goal, and action history are formatted into a prompt sent to the LLM. -3. Predict: The LLM returns a structured JSON response with a "thought" (reasoning) and "action" (what to do). -4. Execute: The predicted action is sent to the Python bridge for execution. -5. Wait: 500ms delay for the UI to settle. -6. Repeat: Steps 1-5 repeat until the LLM outputs "done" or max steps (15) are reached. - -Available desktop actions: -- click: Click a UI element by its index in the accessibility tree -- type: Type text into a specific element or the active focus -- scroll: Scroll the view up or down -- hotkey: Press key combinations (e.g., Win+R for Run dialog, Ctrl+C for copy) -- wait: Wait 1 second for UI to settle -- done: Task is complete, stop the loop - -Safety rules for desktop automation: -- Maximum 15 steps per task to prevent infinite loops -- Action history is tracked so the LLM can avoid repeating failed actions -- Win+R is used for file and folder operations (avoids File Explorer search bugs) -- Win key is used for launching applications (reliable across Windows versions) -- User can stop the task at any time via the Android app - -### Browser Automation (Agentic DOM-Aware Loop) - -When a browser task is identified, the ConnectionProvider runs a DOM-aware browser loop: - -1. Initial Planning: The LLM is asked for the first action (usually open_url to navigate to the target website). -2. Extension Execution: The action is sent to the browser extension which executes it in the active tab. -3. DOM Snapshot: After execution, the extension captures the current page DOM and sends it back. The snapshot includes interactive elements with their IDs, tag types, text content, ARIA labels, placeholders, roles, and URLs. -4. Next Step Planning: The DOM snapshot, action history, and user goal are sent to the LLM. It decides the next action. -5. Repeat: This cycle continues for up to 8 steps. - -Available browser actions: -- open_url: Navigate to a URL -- click_element: Click a DOM element by its ID (e.g., "el_5") -- type_into: Type text into an input field identified by ID, with optional Enter press -- press_key: Press keyboard keys (Enter, Tab, Escape, ArrowDown) -- wait: Wait for a specified number of milliseconds -- scroll_to: Scroll to a specific element by ID - -The extension must be connected to the Desktop Agent via WebSocket. If no extension is connected when a browser task arrives, the Desktop Agent will attempt to launch a browser automatically and wait up to 15 seconds for the extension to connect. - -## Browser Extension - -The Autonion browser extension (Manifest V3) is a Chromium-based extension that enables web page automation. - -Extension components: -- Background Service Worker (background.js): Maintains the WebSocket connection to the Desktop Agent. Handles command routing and DOM snapshot requests. -- Content Scripts: Injected into web pages to capture DOM elements and execute actions. - - semantic-dom.js: Extracts interactive elements from the page (links, buttons, inputs, textareas) and assigns IDs. - - chatgpt.js: Content script for ChatGPT, enabling web-based AI interaction. - - gemini.js: Content script for Google Gemini, enabling web-based AI interaction. -- Popup (popup.html): Configuration UI for setting the WebSocket server URL and viewing connection status. - -Extension connection flow: -1. User configures the Desktop Agent's IP and port in the extension popup. -2. The extension connects via WebSocket to ws://:/automation. -3. The extension identifies itself by including "source: extension" in its messages. -4. The Desktop Agent tracks the extension as a separate client for targeted messaging. - -## Setup - -### Requirements -- Windows 10 or 11 (primary support). macOS and Linux may work but are not fully tested. -- Flutter SDK installed (for development or building from source) -- Python 3.8 or newer installed and in system PATH (for the automation bridge) -- Ollama installed locally (for AI-powered automation features) -- A Chromium-based browser installed (Chrome, Edge, Brave) for web automation - -### Installation Steps -1. Clone or download the Desktop Agent source code. -2. Open a terminal in the project directory. -3. Run "flutter pub get" to install Dart/Flutter dependencies. -4. Run "flutter run -d windows" to start the Desktop Agent in development mode. For a release build: "flutter build windows". -5. The agent window will display: the WebSocket server port, your local IP addresses, and the connection status. - -### Python Bridge Setup -The Python bridge is set up automatically on first use: -1. Make sure Python 3.8+ is installed on your system and accessible from the command line. -2. The bridge creates a virtual environment in your AppData directory (autonion_venv). -3. Required Python packages (uiautomation, pyautogui, mss, Pillow) are installed automatically via pip. -4. The bridge spawns the python/desktop_agent.py script as a subprocess. -5. A ping/pong health check runs on startup to verify the bridge is working. - -If Python is not installed, the Desktop Agent will show an error "Python 3 is not installed or not in PATH" when you try to use desktop automation features. - -### Ollama Setup -1. Download Ollama from https://ollama.ai and install it. -2. Open a terminal and run: ollama pull qwen2.5:3b (or any model you prefer). -3. Ollama starts automatically at http://localhost:11434. -4. In the Desktop Agent settings, configure the Ollama URL and model. The default URL is http://localhost:11434. -5. The agent uses Ollama for both desktop automation (predicting UI actions) and browser automation (planning browser steps). - -### Browser Extension Setup -1. Open your Chromium browser (Chrome, Edge, Brave). -2. Go to the extensions page: chrome://extensions or edge://extensions. -3. Enable "Developer mode" toggle. -4. Click "Load unpacked" and select the Autonion-Extension folder. -5. The extension icon will appear in the browser toolbar. -6. Click the extension icon and enter your Desktop Agent's IP and port (e.g., 192.168.1.100:4545). -7. Click "Connect". The status should show "Connected". - -## WebSocket Protocol Reference - -### Connection Details -- Default port: 4545 (automatically falls back to a random available port if 4545 is in use) -- Endpoint path: /automation -- Full URL: ws://:4545/automation -- Auto-discovery: mDNS service type _autonion._tcp -- Upon connection, the server sends a connection_ack message with agent info and timestamp - -### Incoming Message Types (from Android or Extension) - -prompt: Natural language command to execute -- Fields: prompt (string), transactionId (string), timestamp (number), sourceDeviceId (string) -- The agent classifies the prompt as browser or desktop and routes accordingly. -- A started response is sent immediately as acknowledgment. - -key_press: Execute a key press directly without LLM -- Fields: type ("key_press"), keyName (string), transactionId (string) -- The key is sent to the Python bridge for immediate execution. -- Much faster than the LLM path. Used for structured commands from the Android NLU. - -schedule: Start a recurring action -- Fields: type ("schedule"), action (object with keyName), intervalMs (number), repeatCount (number or null), transactionId (string) -- Creates a periodic timer that executes the key press at the specified interval. -- If repeatCount is null, runs indefinitely until cancelled. - -schedule_cancel: Stop a recurring action -- Fields: type ("schedule_cancel"), transactionId (string) -- Cancels the timer identified by transactionId. - -open_url: Launch a URL in the default browser -- Fields: type ("open_url"), url (string) or payload.url (string) - -clipboard.text_copied: Clipboard sync event from Android -- Fields: type ("clipboard.text_copied"), payload.text (string) -- Writes the received text to the desktop system clipboard. - -register_triggers: Register automation rules from Android -- Fields: type ("register_triggers"), payload.rules (array of rule objects) -- Rules are forwarded to the browser extension. - -Extension messages: Messages with source "extension" are routed to the extension handler. -- execution_status: Step-level progress updates from extension -- execution_result: Final result of an extension automation task -- dom_snapshot: DOM element snapshot from a webpage -- step_result: Result of a single agentic step with DOM snapshot for next planning cycle -- kill_switch_ack: Extension acknowledges a stop command -- rule_triggered: A trigger rule condition was met - -### Outgoing Message Types (to Android or Extension) - -connection_ack: Sent on new connection. Contains status, agent name, timestamp, and server info (port, client count). - -prompt_response: Command execution status update sent to Android. -- Fields: type ("prompt_response"), transactionId (string), status (string), message (string), timestamp (string) -- Status values: started, in_progress, completed, failed, scheduled, cancelled - -pong: Heartbeat response to a ping message. Contains timestamp. - -clipboard.text_copied: Clipboard sync event from desktop to Android. Sent when a new text is copied on the desktop clipboard. - -## Troubleshooting - -### Desktop Agent will not start -- Make sure Flutter SDK is properly installed. Run "flutter doctor" to check for issues. -- Try running with the verbose flag: flutter run -d windows --verbose -- Check if port 4545 is already in use by another application. The agent will fall back to a random port, but check the logs. -- If the window appears briefly and closes, check the terminal/console for error messages. - -### Android cannot find the Desktop -- Both devices must be on the same WiFi network. Verify this first. -- Check the Desktop Agent window. It should show "Server listening on 0.0.0.0:". -- Look for the "Reachable at: ws://..." lines in the agent window. These show your actual IP addresses. -- Check Windows Firewall: Allow the autonion_agent.exe (or flutter_windows.exe during development) through the firewall. -- If mDNS discovery fails, use manual IP entry on the Android app with the IP and port shown. -- Some corporate or guest WiFi networks block mDNS. Try a home or personal WiFi network. - -### Desktop automation commands fail -- For AI-powered automation: Make sure Ollama is running and has at least one model loaded. Check by visiting http://localhost:11434 in your browser. -- For direct key presses: Make sure the Python bridge is initialized. Check the Desktop Agent logs for "PythonBridge" messages. If the bridge failed, check that Python 3 is installed and in PATH. -- Python bridge troubleshooting: If the virtual environment creation fails, manually delete the autonion_venv folder in your AppData/Roaming directory and restart the agent. -- Check the Automation Debugger on the Android app for error details about what went wrong. - -### Browser extension does not connect -- The extension only works with Chromium-based browsers: Chrome, Edge, Brave. -- Make sure the extension is loaded and enabled in the browser's extension page. -- In the extension popup, verify the WebSocket URL matches the Desktop Agent's IP and port. -- Try reloading the extension: go to chrome://extensions, find Autonion, click the refresh icon. -- If the browser was launched by the Desktop Agent automatically, the extension may need a few seconds to establish the WebSocket connection. - -### Clipboard sync is not working -- Both devices must be connected via WebSocket. Check the Desktop Agent shows at least 1 connected client. -- Only text content is synced. Images, files, and rich content are not supported. -- There is a 1-second polling interval for detecting clipboard changes on the desktop. -- If you copy text on Android and it does not appear on the desktop, check the WebSocket connection status on both sides. - -### Python bridge errors -- "Python 3 is not installed or not in PATH": Install Python 3.8+ and make sure it is accessible from the command line. Run "python --version" in a terminal to verify. -- "Failed to create venv": Delete the existing venv folder at AppData/Roaming/autonion_venv and restart the agent. -- "desktop_agent.py not found": Make sure the python/ folder with desktop_agent.py exists in the project directory. -- "Command timed out": The Python bridge has a 30-second timeout per command. Complex accessibility trees can take longer. Try closing unnecessary windows on the desktop. - -### Agent reaches max steps without completing -- The desktop agentic loop has a maximum of 15 steps. If the task is complex, try breaking it into simpler sub-tasks. -- Check the action history in the logs. If the agent is clicking the wrong elements or getting stuck, the accessibility tree may not have the expected elements. -- Some Windows applications have complex or non-standard UI that the accessibility tree cannot read properly. Try using hotkey-based shortcuts instead. - -## Privacy and Security - -- All communication happens over your local WiFi network. No data is sent to cloud servers or external APIs. -- Ollama runs entirely locally on your machine. LLM inference happens on your hardware. -- No telemetry, analytics, or usage tracking of any kind. -- No account or login required. -- The Python bridge has system-level access (mouse, keyboard, screen). Only run the Desktop Agent on trusted machines. -- The WebSocket server accepts connections from any device on the local network. Keep your WiFi network secure. diff --git a/app/src/main/assets/knowledge/faq_database.json b/app/src/main/assets/knowledge/faq_database.json index 4636d823..5176cf44 100644 --- a/app/src/main/assets/knowledge/faq_database.json +++ b/app/src/main/assets/knowledge/faq_database.json @@ -11,7 +11,7 @@ }, { "question": "What is Semantic Automation?", - "answer": "Semantic Automation is Autonion's AI-powered feature that understands natural language commands and executes them on your device. For example, you can say 'search for shoes on Flipkart' and the AI agent will open Flipkart, find the search bar, type your query, and submit it \u2014 all autonomously.\n\nIt uses a combination of:\n- On-device accessibility tree reading to understand screen elements\n- LLM (Large Language Model) via local Ollama for action prediction\n- Rule-based heuristics for common tasks like search, toggle, and navigation", + "answer": "Semantic Automation is Autonion's AI-powered feature that understands natural language commands and executes them on your device. For example, you can say 'search for shoes on Flipkart' and the AI will open Flipkart, find the search bar, type your query, and submit it \u2014 all based on your request.\n\nIt uses a combination of:\n- On-device accessibility tree reading to understand screen elements\n- LLM (Large Language Model) via local Ollama for action prediction\n- Rule-based heuristics for common tasks like search, toggle, and navigation", "tags": [ "semantic", "automation", @@ -80,7 +80,7 @@ }, { "question": "How does the browser extension work?", - "answer": "The Autonion browser extension enables web automation:\n1. Install the extension in your browser (Lemur/Chrome-based browsers work best).\n2. The extension connects to the Desktop Agent via WebSocket.\n3. When you send a browser-related command, the extension captures the webpage DOM.\n4. The AI agent reads the page structure and interacts with elements (clicks, types, scrolls).\n5. Results are sent back to your Android app.\n\nThe extension is needed because the Accessibility Service can only see native Android UI elements, not web page content.", + "answer": "The Autonion browser extension enables web automation:\n1. Install the extension in your browser (Lemur/Chrome-based browsers work best).\n2. The extension connects to the Desktop Agent via WebSocket.\n3. When you send a browser-related command, the extension captures the webpage DOM.\n4. The engine reads the page structure and carries out requested interactions with elements (clicks, types, scrolls).\n5. Results are sent back to your Android app.\n\nThe extension is needed because the Accessibility Service can only see native Android UI elements, not web page content.", "tags": [ "browser", "extension", @@ -353,7 +353,7 @@ }, { "question": "How does the Python Bridge work on Desktop?", - "answer": "The Desktop Agent uses a background Python process (the 'bridge') to interact with the OS. It uses libraries like `uiautomation` and `pyautogui` to scan the UI elements, read the screen, and execute mouse/keyboard events autonomously.", + "answer": "The Desktop Agent uses a background Python process (the 'bridge') to interact with the OS. It uses libraries like `uiautomation` and `pyautogui` to scan the UI elements, read the screen, and execute mouse/keyboard events based on your configured tasks.", "tags": [ "python bridge", "desktop agent", @@ -437,7 +437,7 @@ }, { "question": "What is the difference between Direct Action and Semantic Automation?", - "answer": "A Direct Action is a hardcoded response to a command (like turning bluetooth on/off). Semantic Automation involves the AI Agent analyzing the screen, planning steps, and actually simulating taps/swipes to accomplish a complex goal like 'order an uber'.", + "answer": "A Direct Action is a hardcoded response to a command (like turning bluetooth on/off). Semantic Automation uses your connected AI model to interpret your request and assist with step-by-step UI interactions.", "tags": [ "direct action", "semantic automation", @@ -534,8 +534,8 @@ ] }, { - "question": "How do I make the agent type text?", - "answer": "You don't need to do anything special. Just describe the task (e.g., 'send a message to John saying Hello'). The AI agent will detect the input field, focus it, and automatically invoke the type action.", + "question": "How do I make the app type text?", + "answer": "You don't need to do anything special. Just describe the task (e.g., 'send a message to John saying Hello'). Autonion will locate the input field, focus it, and perform the requested text entry.", "tags": [ "type", "text", @@ -555,7 +555,7 @@ }, { "question": "How do I chain multiple actions?", - "answer": "Autonion's agentic loop is designed to automatically chain actions. If you say 'Find a recipe for lasagna and save it', the agent will search, wait for the page to load, click the result, and then try to find a save button over multiple steps.", + "answer": "Autonion's automation loop is designed to chain actions step by step. If you say 'Find a recipe for lasagna and save it', the engine will search, wait for the page to load, click the result, and then try to find a save button over multiple steps.", "tags": [ "chain", "multiple", @@ -564,8 +564,8 @@ ] }, { - "question": "What is the maximum number of steps an agent will take?", - "answer": "To prevent runaway loops, the Semantic Automation agent has a hard limit of 15 consecutive steps. If it hasn't achieved the goal by then, it will stop and notify you.", + "question": "What is the maximum number of steps an automation will take?", + "answer": "To prevent runaway loops, Semantic Automation has a hard limit of 15 consecutive steps. If it hasn't achieved the goal by then, it will stop and notify you.", "tags": [ "limit", "steps", @@ -574,8 +574,8 @@ ] }, { - "question": "How does the agent know when it's done?", - "answer": "After every action, the agent re-evaluates the screen. If it detects UI elements confirming success (like a 'Sent' message or the target app's home screen) or if its reasoning output declares the goal achieved, it emits a 'terminate' action.", + "question": "How does automation know when it's done?", + "answer": "After every action, the screen is re-evaluated. If it detects UI elements confirming success (like a 'Sent' message or the target app's home screen) or if the model declares the goal achieved, it emits a 'terminate' action.", "tags": [ "done", "terminate", @@ -767,17 +767,17 @@ ] }, { - "question": "How does the Agentic Loop recover from errors?", - "answer": "If an action fails (e.g., clicking a button that just vanished), the loop captures a fresh screen snapshot and replans. It adjusts its strategy based on the new UI state.", + "question": "How does the Automation Loop recover from errors?", + "answer": "If an action fails (e.g., clicking a button that just vanished), the system checks the updated screen and retries the current step when needed.", "tags": [ - "agentic loop", + "automation loop", "error recovery", - "replanning" + "retry" ] }, { "question": "What is a 'Scroll' action?", - "answer": "If the agent determines that the target element might be off-screen, it emits a 'scroll' action. The engine then simulates a swipe up/down and sends the newly revealed UI state back to the model.", + "answer": "If the model determines that the target element might be off-screen, it emits a 'scroll' action. The engine then simulates a swipe up/down and sends the newly revealed UI state back to the model.", "tags": [ "scroll", "swipe", @@ -1209,7 +1209,7 @@ ] }, { - "question": "Can I use Autonion to buy things autonomously?", + "question": "Can I use Autonion to buy things automatically?", "answer": "Yes, but this is dangerous because AI hallucinations might lead to incorrect purchases or quantities. It is highly recommended to ONLY use Autonion for read-only or low-risk tasks, or ensure manual confirmation steps exist in your flows.", "tags": [ "purchase", @@ -1544,7 +1544,7 @@ }, { "question": "How does the browser extension connect to the app?", - "answer": "The Autonion Android Extension connects via WebSocket:\n1. The Autonion app runs a WebSocket server on port 54321 (ExtensionBridgeServer).\n2. When you open a supported browser with the extension installed, it automatically connects to ws://localhost:54321.\n3. The extension captures the webpage DOM and sends it to the app.\n4. The AI agent reads the DOM elements and sends action commands back through the extension.\n\nThis all happens locally on your device — no internet required.", + "answer": "The Autonion Android Extension connects via WebSocket:\n1. The Autonion app runs a WebSocket server on port 54321 (ExtensionBridgeServer).\n2. When you open a supported browser with the extension installed, it automatically connects to ws://localhost:54321.\n3. The extension captures the webpage DOM and sends it to the app.\n4. The engine reads the DOM elements and performs the requested actions through the extension.\n\nThis all happens locally on your device — no internet required.", "tags": [ "extension", "websocket", @@ -1566,7 +1566,7 @@ }, { "question": "What is the difference between Autonion Extension and Autonion Android Extension?", - "answer": "They are two separate extensions for different platforms:\n\n- Autonion Extension (Desktop): Runs in Chrome/Edge on your PC. Connects to the Autonion Desktop Agent. Used for cross-device browser automation — controlling the desktop browser from your phone.\n\n- Autonion Android Extension: Runs in Kiwi/Lemur/Firefox Nightly on your phone. Connects directly to the Autonion Android app. Used for on-device browser automation — the AI agent controls the mobile browser.\n\nBoth provide DOM access for web page interaction, but they connect to different apps.", + "answer": "They are two separate extensions for different platforms:\n\n- Autonion Extension (Desktop): Runs in Chrome/Edge on your PC. Connects to the Autonion Desktop Agent. Used for cross-device browser automation — controlling the desktop browser from your phone.\n\n- Autonion Android Extension: Runs in Kiwi/Lemur/Firefox Nightly on your phone. Connects directly to the Autonion Android app. Used for on-device browser automation — helps navigate web content on your mobile browser.\n\nBoth provide DOM access for web page interaction, but they connect to different apps.", "tags": [ "extension", "difference", @@ -1577,7 +1577,7 @@ }, { "question": "Do I need the extension for web automation?", - "answer": "Yes, the browser extension is essential for effective web automation. Without it, the AI agent can only use the Android Accessibility Service, which sees browser UI elements (address bar, tabs) but cannot read or interact with actual web page content like links, buttons, and text fields inside the page. The extension captures the full DOM (Document Object Model) of the webpage, giving the agent precise control over page elements.", + "answer": "Yes, the browser extension is essential for effective web automation. Without it, Autonion can only use the Android Accessibility Service, which sees browser UI elements (address bar, tabs) but cannot read or interact with actual web page content like links, buttons, and text fields inside the page. The extension captures the full DOM (Document Object Model) of the webpage, giving the engine precise access to page elements.", "tags": [ "extension", "web automation", diff --git a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/OmniChatbotViewModel.kt b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/OmniChatbotViewModel.kt index 6c9adae8..38c02b4a 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/OmniChatbotViewModel.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/OmniChatbotViewModel.kt @@ -348,6 +348,7 @@ class OmniChatbotViewModel( isUser = false, mode = ResponseMode.FAQ )) + recordChatTurn(faq.question, faq.answer) } // ─── LLM Connection Management ────────────────────────── @@ -748,6 +749,17 @@ class OmniChatbotViewModel( } private fun handleQAndA(result: IntentResult) { + faqRepository.findExactQuestionMatch(result.rawPrompt)?.let { faq -> + addMessage(OmniChatMessage( + text = faq.answer, + isUser = false, + mode = ResponseMode.FAQ, + suggestedWalkthroughId = getWalkthroughFeatureForPrompt(result.rawPrompt) + )) + recordChatTurn(result.rawPrompt, faq.answer) + return + } + addMessage(OmniChatMessage( text = "💬 Let me think about that...", isUser = false, @@ -756,6 +768,23 @@ class OmniChatbotViewModel( )) viewModelScope.launch { + if (!faqRepository.isLoaded) { + val waitStart = System.currentTimeMillis() + while (!faqRepository.isLoaded && System.currentTimeMillis() - waitStart < 1_500) { + delay(100) + } + } + + faqRepository.findExactQuestionMatch(result.rawPrompt)?.let { faq -> + updateLastBotMessage( + faq.answer, + ResponseMode.FAQ, + suggestedWalkthroughId = getWalkthroughFeatureForPrompt(result.rawPrompt) + ) + recordChatTurn(result.rawPrompt, faq.answer) + return@launch + } + // Wait for the knowledge store to finish loading (handles the race // condition where the user sends a question before background init // completes — previously this returned "I don't have information"). @@ -776,16 +805,16 @@ class OmniChatbotViewModel( Log.d(TAG, "Q&A: Knowledge store ready after ${System.currentTimeMillis() - waitStart}ms") } - // Retrieve top 3 relevant chunks (filtered by min similarity 0.35) - val chunks = knowledgeStore.search(result.rawPrompt, topK = 3) + // Retrieve top 3 relevant chunks. For short follow-ups like + // "what modes does it support?", include recent chat as search context. + val chunks = knowledgeStore.search(buildKnowledgeSearchQuery(result.rawPrompt), topK = 3) // Case A: No knowledge chunks found at all if (chunks.isEmpty()) { - updateLastBotMessage( - "I don't have information about that in my knowledge base. " + - "Try asking about app features, automation, or troubleshooting!", - ResponseMode.KNOWLEDGE - ) + val fallback = "I don't have information about that in my knowledge base. " + + "Try asking about app features, automation, or troubleshooting!" + updateLastBotMessage(fallback, ResponseMode.KNOWLEDGE) + recordChatTurn(result.rawPrompt, fallback) return@launch } @@ -824,6 +853,7 @@ class OmniChatbotViewModel( } } updateLastBotMessage(fallback, ResponseMode.KNOWLEDGE) + recordChatTurn(result.rawPrompt, fallback) return@launch } @@ -841,21 +871,22 @@ class OmniChatbotViewModel( append("/no_think\n") append("You are Autonion, an AI assistant built into an Android automation app.\n\n") append("STRICT RULES:\n") - append("1. Answer ONLY using the reference knowledge provided for each question. Do NOT add information that is not in the knowledge.\n") - append("2. If the reference knowledge does NOT contain information to answer the question, say exactly: \"I don't have specific information about that in my knowledge base.\"\n") - append("3. Do NOT make up features, capabilities, or instructions that are not explicitly described in the knowledge.\n") - append("4. Be concise and direct. Use bullet points where appropriate.\n") - append("5. Do NOT use tags or internal reasoning. Answer immediately.\n") - append("6. If your answer is primarily about one of these features, append the tag on a new line at the very end of your response: [WALKTHROUGH:feature_id]\n") + append("1. Answer using the reference knowledge provided for the current question. Use prior chat only to resolve follow-up references like \"it\", \"that\", or \"same\".\n") + append("2. Do NOT add factual details that are not supported by the reference knowledge or already-stated chat context.\n") + append("3. If the reference knowledge and already-stated chat context do NOT contain information to answer the question, say exactly: \"I don't have specific information about that in my knowledge base.\"\n") + append("4. Do NOT make up features, capabilities, or instructions that are not explicitly described in the knowledge.\n") + append("5. Be concise and direct. Use bullet points where appropriate.\n") + append("6. Do NOT use tags or internal reasoning. Answer immediately.\n") + append("7. If your answer is primarily about one of these features, append the tag on a new line at the very end of your response: [WALKTHROUGH:feature_id]\n") append(" Available features: flow_builder, gesture_recording, semantic_automation, cross_device, visual_trigger, screen_ml, system_context, debugger\n") append(" IMPORTANT: Do NOT append a WALKTHROUGH tag for browser extension, extension installation, or extension setup topics. Those have no walkthrough.\n") - append("7. IMPORTANT: There are TWO different extensions. The 'Autonion Extension' is for Desktop PC browsers. The 'Autonion Android Extension' is for Mobile phone browsers. If the user asks about an 'extension' without specifying PC or Mobile, explicitly mention both, explain the difference, and you MUST provide the exact github.com download URLs for BOTH extensions exactly as they appear in the knowledge below.\n") + append("8. IMPORTANT: There are TWO different extensions. The 'Autonion Extension' is for Desktop PC browsers. The 'Autonion Android Extension' is for Mobile phone browsers. If the user asks about an 'extension' without specifying PC or Mobile, explicitly mention both, explain the difference, and you MUST provide the exact github.com download URLs for BOTH extensions exactly as they appear in the knowledge below.\n") } // ── Per-query knowledge — scoped to current question only ── val knowledgeContext = buildString { - append("REFERENCE KNOWLEDGE FOR THE FOLLOWING QUESTION ONLY:\n") - append("(Use this knowledge to answer the user's next message. ") + append("REFERENCE KNOWLEDGE FOR THE CURRENT QUESTION ONLY:\n") + append("(Use this knowledge to answer the current question below. ") append("Do NOT apply it to previous conversation topics.)\n\n") append(contextText) } @@ -866,12 +897,20 @@ class OmniChatbotViewModel( withContext(Dispatchers.IO) { val slm = com.autonion.automationcompanion.features.semantic_automation.ml.PredictorCache.getSLMEngine(context, modelStorageManager) if (slm != null) { + val recentConversation = buildRecentConversationContext() val slmPrompt = buildString { append("You are Autonion, an AI assistant for Android automation.\n") - append("Answer the question concisely using ONLY the reference knowledge.\n\n") - append("Knowledge:\n") + append("Answer concisely using ONLY the reference knowledge for factual details.\n") + append("Use recent conversation only to resolve follow-up references like \"it\", \"that\", or \"same\".\n") + append("If the reference knowledge does not answer the current question, say exactly: \"I don't have specific information about that in my knowledge base.\"\n\n") + if (recentConversation.isNotBlank()) { + append("Recent conversation:\n") + append(recentConversation) + append("\n\n") + } + append("Reference knowledge for the current question:\n") append(contextText.take(1500)) - append("\n\nQuestion: ") + append("\n\nCurrent question: ") append(result.rawPrompt) append("\nAnswer:") } @@ -894,12 +933,14 @@ class OmniChatbotViewModel( withContext(Dispatchers.IO) { val model = getLangchainModel() if (model != null) { - val userMsg = UserMessage(result.rawPrompt) + val userMsg = UserMessage(buildRagUserPrompt(result.rawPrompt, knowledgeContext)) val allMessages = mutableListOf() + // Keep exactly one system message. Put per-question RAG + // context in the current user turn so chat templates don't + // drop it and old history can't outrank it. allMessages.add(SystemMessage(baseSystemPrompt)) allMessages.addAll(chatMemory.messages()) - allMessages.add(SystemMessage(knowledgeContext)) allMessages.add(userMsg) val response = model.generate(allMessages) @@ -913,8 +954,6 @@ class OmniChatbotViewModel( if (content.isNotBlank()) { rawAnswer = content - chatMemory.add(userMsg) - chatMemory.add(AiMessage(rawAnswer)) } } } @@ -938,14 +977,11 @@ class OmniChatbotViewModel( ?.trim() // Priority: prompt-based match → LLM tag → null - val walkthroughFeature = if (FeatureMatcher.isExcludedFromWalkthrough(result.rawPrompt)) { - null - } else { - FeatureMatcher.matchFeature(result.rawPrompt) ?: llmSuggestedFeature - } + val walkthroughFeature = getWalkthroughFeatureForPrompt(result.rawPrompt, llmSuggestedFeature) if (!answer.isNullOrBlank()) { updateLastBotMessage(answer, ResponseMode.KNOWLEDGE, suggestedWalkthroughId = walkthroughFeature) + recordChatTurn(result.rawPrompt, answer) } else { // LLM failed — show clean chunk fallback val fallback = cleanKnowledgeChunk(chunks.first().text.take(1000), maxLength = 600) @@ -960,6 +996,7 @@ class OmniChatbotViewModel( ResponseMode.KNOWLEDGE, suggestedWalkthroughId = walkthroughFeature ) + recordChatTurn(result.rawPrompt, fallbackNote) } } } @@ -995,6 +1032,91 @@ class OmniChatbotViewModel( } } + private fun buildRagUserPrompt(question: String, knowledgeContext: String): String { + return buildString { + append(knowledgeContext) + append("\n\nCURRENT QUESTION:\n") + append(question.trim()) + append("\n\nAnswer directly using the reference knowledge.") + } + } + + private fun buildKnowledgeSearchQuery(question: String): String { + if (!isContextualFollowUp(question)) return question + + val recentConversation = buildRecentConversationContext( + maxMessages = 6, + maxCharsPerMessage = 220 + ) + + return if (recentConversation.isBlank()) { + question + } else { + "$recentConversation\nCurrent question: $question" + } + } + + private fun isContextualFollowUp(question: String): Boolean { + val lower = question.lowercase().trim() + val tokens = lower + .replace(Regex("[^a-z0-9\\s]"), " ") + .split(Regex("\\s+")) + .filter { it.isNotBlank() } + + val contextualTerms = setOf( + "it", "its", "that", "this", "they", "them", "those", "these", + "same", "above", "previous", "earlier", "also", "too" + ) + + return tokens.any { it in contextualTerms } || + lower.startsWith("what about") || + lower.startsWith("how about") || + lower.startsWith("and ") || + lower.startsWith("also ") + } + + private fun buildRecentConversationContext( + maxMessages: Int = 6, + maxCharsPerMessage: Int = 280 + ): String { + return chatMemory.messages() + .takeLast(maxMessages) + .mapNotNull { msg -> + when (msg) { + is UserMessage -> { + val text = msg.singleText().trim() + if (text.isBlank()) null else "User: ${text.take(maxCharsPerMessage)}" + } + is AiMessage -> { + val text = msg.text().trim() + if (text.isBlank()) null else "Assistant: ${text.take(maxCharsPerMessage)}" + } + else -> null + } + } + .joinToString("\n") + } + + private fun getWalkthroughFeatureForPrompt( + prompt: String, + llmSuggestedFeature: String? = null + ): String? { + return if (FeatureMatcher.isExcludedFromWalkthrough(prompt)) { + null + } else { + FeatureMatcher.matchFeature(prompt) ?: llmSuggestedFeature + } + } + + private fun recordChatTurn(userText: String, assistantText: String) { + val cleanedUserText = userText.trim() + val cleanedAssistantText = stripMarkdown(assistantText).trim() + if (cleanedUserText.isBlank() || cleanedAssistantText.isBlank()) return + + chatMemory.add(UserMessage(cleanedUserText.take(800))) + chatMemory.add(AiMessage(cleanedAssistantText.take(1200))) + } + // ═══════════════════════════════════════════════════════════ // TWO-WAY COMMUNICATION // ═══════════════════════════════════════════════════════════ diff --git a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/FeatureMatcher.kt b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/FeatureMatcher.kt index d91f6bf1..f0f19211 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/FeatureMatcher.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/FeatureMatcher.kt @@ -24,8 +24,8 @@ object FeatureMatcher { "gesture replay", "tap recording", "swipe recording", "record actions" ), "semantic_automation" to listOf( - "semantic", "ai agent", "semantic automation", "semantic ai", - "smart automation", "ai automation", "agentic" + "semantic", "semantic automation", "semantic ai", + "smart automation", "ai automation" ), "cross_device" to listOf( "cross device", "cross-device", "desktop agent", "desktop automation", diff --git a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/WalkthroughRegistry.kt b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/WalkthroughRegistry.kt index 0cc3a705..6e4f3812 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/WalkthroughRegistry.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/companion/WalkthroughRegistry.kt @@ -190,21 +190,21 @@ object WalkthroughRegistry { ), // ─────────────────────────────────────────────── - // Semantic AI Agent + // Semantic Automation // ─────────────────────────────────────────────── "semantic_automation" to WalkthroughScript( featureId = "semantic_automation", - featureName = "Semantic AI Agent", - description = "An AI-powered agent that understands and executes complex on-device tasks.", + featureName = "Semantic Automation", + description = "Natural language automation that carries out multi-step tasks on your device.", steps = listOf( WalkthroughStep( - instruction = "Taking you to the Semantic AI Agent screen…", + instruction = "Taking you to the Semantic Automation screen…", targetRoute = AutomationRoutes.SEMANTIC_AUTOMATION, stepType = StepType.NAVIGATE ), WalkthroughStep( - instruction = "This is the AI Agent chat interface. Here you describe a task in natural language, " + - "and the AI will autonomously navigate your phone to complete it.", + instruction = "This is the Semantic Automation interface. Here you describe a task in natural language, " + + "and Autonion will carry out the steps to complete your task.", stepType = StepType.OBSERVE ), WalkthroughStep( @@ -223,19 +223,19 @@ object WalkthroughRegistry { ), WalkthroughStep( instruction = "After you send a command, watch the Live Status card — it shows " + - "what the agent is doing in real time: parsing your goal, capturing the screen, " + - "deciding the next action, and executing it. You can toggle this card with the 👁 icon.", + "the process in real time: parsing your goal, inspecting the screen, " + + "and performing the requested steps. You can toggle this card with the 👁 icon.", stepType = StepType.OBSERVE ), WalkthroughStep( - instruction = "If the agent needs clarification, it will ask you a question with " + - "clickable options. You can also stop the agent at any time using the ■ Stop button " + + instruction = "If clarification is needed, a prompt will appear with " + + "clickable options. You can also stop the process at any time using the ■ Stop button " + "in the status card.", stepType = StepType.OBSERVE ), WalkthroughStep( - instruction = "That's the Semantic AI Agent! It's the most powerful automation tool — " + - "it can handle multi-step, cross-app tasks using AI-driven screen understanding. 🤖", + instruction = "That's Semantic Automation! It's a powerful natural language tool — " + + "helping you complete multi-step tasks using screen understanding. 💡", stepType = StepType.OBSERVE ) ) @@ -436,7 +436,7 @@ object WalkthroughRegistry { ), WalkthroughStep( instruction = "The Debugger shows logs from all your automation modules — " + - "Gesture, Visual Trigger, Flow Builder, AI Agent, and more.", + "Gesture, Visual Trigger, Flow Builder, Semantic Automation, and more.", stepType = StepType.OBSERVE ), WalkthroughStep( diff --git a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/knowledge/FAQRepository.kt b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/knowledge/FAQRepository.kt index 0105bd4e..3aba8635 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/knowledge/FAQRepository.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/knowledge/FAQRepository.kt @@ -27,6 +27,7 @@ class FAQRepository { ) private val _allFAQs = mutableListOf() + private val faqLock = Any() @Volatile var isLoaded = false @@ -44,6 +45,7 @@ class FAQRepository { val jsonArray = JSONArray(jsonStr) val start = System.currentTimeMillis() + val loadedFaqs = mutableListOf() for (i in 0 until jsonArray.length()) { val obj = jsonArray.getJSONObject(i) @@ -58,12 +60,17 @@ class FAQRepository { } } - _allFAQs.add(FAQ(question, answer, tags)) + loadedFaqs.add(FAQ(question, answer, tags)) + } + + synchronized(faqLock) { + _allFAQs.clear() + _allFAQs.addAll(loadedFaqs) + isLoaded = true } val elapsed = System.currentTimeMillis() - start - isLoaded = true - Log.d(TAG, "Loaded ${_allFAQs.size} Static FAQs in ${elapsed}ms") + Log.d(TAG, "Loaded ${loadedFaqs.size} Static FAQs in ${elapsed}ms") } catch (e: Exception) { Log.e(TAG, "Failed to load static FAQs", e) @@ -73,5 +80,32 @@ class FAQRepository { /** * Get all loaded FAQs for the UI Browser. */ - fun getAllFAQs(): List = _allFAQs.toList() + fun getAllFAQs(): List = synchronized(faqLock) { + _allFAQs.toList() + } + + /** + * Returns a static FAQ when the user's prompt exactly matches a known FAQ + * question, ignoring case and punctuation. + */ + fun findExactQuestionMatch(prompt: String): FAQ? { + val normalizedPrompt = normalizeQuestion(prompt) + if (normalizedPrompt.isBlank()) return null + + val faqs = synchronized(faqLock) { + _allFAQs.toList() + } + + return faqs.firstOrNull { faq -> + normalizeQuestion(faq.question) == normalizedPrompt + } + } + + private fun normalizeQuestion(value: String): String { + return value + .lowercase() + .replace(Regex("[^a-z0-9\\s]"), " ") + .replace(Regex("\\s+"), " ") + .trim() + } } diff --git a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/model/OmniChatMessage.kt b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/model/OmniChatMessage.kt index 15a02f2c..a4e599c9 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/model/OmniChatMessage.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/omni_chatbot/model/OmniChatMessage.kt @@ -88,11 +88,11 @@ object ContextualFAQs { ) private val semanticChips = listOf( - FAQChip("How does the AI agent work?", "How AI works"), + FAQChip("How does Semantic Automation work?", "How AI works"), FAQChip("Why is automation doing random things?", "Random actions fix"), FAQChip("How to change the AI model?", "Change model"), FAQChip("What prompts work best?", "Best prompts"), - FAQChip("How to stop the agent?", "Stop agent") + FAQChip("How to stop automation?", "Stop automation") ) private val crossDeviceChips = listOf( diff --git a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/ExtensionBridgeServer.kt b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/ExtensionBridgeServer.kt index a760696a..af0a8e64 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/ExtensionBridgeServer.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/ExtensionBridgeServer.kt @@ -28,7 +28,7 @@ import java.util.concurrent.ConcurrentHashMap * passive relay: * * 1. **DOM Snapshots**: The extension captures the browser's interactive DOM - * elements and sends them here. The SemanticAutomationAgent consumes these + * elements and sends them here. The SemanticAutomationEngine consumes these * snapshots alongside accessibility UI tree data to build richer prompts * for the Local LLM (Ollama). * @@ -36,7 +36,7 @@ import java.util.concurrent.ConcurrentHashMap * scroll, etc.) to the extension, which dispatches them to the content * script for execution in the browser DOM. * - * 3. **Agentic Loop**: After each action, the extension automatically + * 3. **Automation Loop**: After each action, the extension automatically * captures a fresh DOM snapshot and sends it back, enabling the * LLM-driven automation loop: * diff --git a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/SemanticAutomationEngine.kt b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/SemanticAutomationEngine.kt index 61e7c5c7..6091ac55 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/SemanticAutomationEngine.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/core/SemanticAutomationEngine.kt @@ -757,7 +757,7 @@ class SemanticAutomationEngine(private val context: Context) { val rawPrompt = rGoal.rawCommand ?: goal.rawCommand val taskType = rGoal.task ?: goal.task - // Always launch the base domain. The agentic loop will wait for the extension + // Always launch the base domain. The automation loop will wait for the extension // to connect and provide the DOM to click the website's search box and interact. val urlToLaunch = baseDomain diff --git a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/ui/SemanticAutomationScreen.kt b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/ui/SemanticAutomationScreen.kt index 31128fd5..b50c1d5f 100644 --- a/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/ui/SemanticAutomationScreen.kt +++ b/app/src/main/java/com/autonion/automationcompanion/features/semantic_automation/ui/SemanticAutomationScreen.kt @@ -101,11 +101,11 @@ fun SemanticAutomationScreen( if (showTip) { FeatureTipSheet( - title = "Semantic AI Agent", + title = "Semantic Automation", tips = listOf( "Connect an **AI model** first via ⚙ Settings in the top bar", "Type a natural language command like **'open YouTube and play music'**", - "The AI agent will **take control** of your screen to execute the task" + "Autonion will **help perform** the requested steps on your screen" ), icon = Icons.Default.AutoAwesome, iconColor = Color(0xFFFF9800), @@ -239,7 +239,7 @@ fun SemanticAutomationScreen( Scaffold( topBar = { TopAppBar( - title = { Text("Semantic AI Agent", color = headerTextColor, fontWeight = FontWeight.Bold) }, + title = { Text("Semantic Automation", color = headerTextColor, fontWeight = FontWeight.Bold) }, navigationIcon = { IconButton(onClick = onBack) { Icon(Icons.AutoMirrored.Filled.ArrowBack, contentDescription = "Back", tint = headerTextColor) diff --git a/app/src/main/java/com/autonion/automationcompanion/ui/HomeScreen.kt b/app/src/main/java/com/autonion/automationcompanion/ui/HomeScreen.kt index 47f038f1..418633e7 100644 --- a/app/src/main/java/com/autonion/automationcompanion/ui/HomeScreen.kt +++ b/app/src/main/java/com/autonion/automationcompanion/ui/HomeScreen.kt @@ -274,8 +274,8 @@ private fun CompactHomeLayout( item { StaggeredEntry(index = 6) { ListCard( - title = "Semantic AI Agent", - description = "Natural language automation — describe a task and let the AI agent execute it.", + title = "Semantic Automation", + description = "Natural language automation — describe a task and run the requested steps.", icon = Icons.Default.AutoAwesome, iconColor = Color.White, iconContainerColor = AccentOrange, @@ -396,7 +396,7 @@ private fun MediumHomeLayout( } Text( - text = "Record gestures, build visual flows, and let AI agents automate tasks — all on-device with optional cloud power.", + text = "Record gestures, build visual flows, and automate tasks with natural language — all on-device with optional cloud power.", style = MaterialTheme.typography.bodyLarge.copy(color = MaterialTheme.colorScheme.onSurfaceVariant), modifier = Modifier.padding(bottom = 16.dp, start = 8.dp, end = 8.dp) ) @@ -467,8 +467,8 @@ private fun MediumHomeLayout( item { StaggeredEntry(index = 4) { GridCard( - title = "Semantic AI Agent", - description = "Natural language automation — describe a task and let AI execute it.", + title = "Semantic Automation", + description = "Natural language automation — describe a task and run the requested steps.", icon = Icons.Default.AutoAwesome, iconColor = Color.White, iconContainerColor = AccentOrange, @@ -667,8 +667,8 @@ private fun ExpandedHomeLayout( item { StaggeredEntry(index = 4) { GridCard( - title = "Semantic AI Agent", - description = "Describe a task and let the AI agent execute it.", + title = "Semantic Automation", + description = "Describe a task and run the requested steps.", icon = Icons.Default.AutoAwesome, iconColor = Color.White, iconContainerColor = AccentOrange, @@ -739,7 +739,7 @@ private fun HomeFooter() { verticalAlignment = Alignment.CenterVertically ) { Text( - text = "v1.1.0", + text = "v1.1.1", style = MaterialTheme.typography.bodySmall.copy( color = MaterialTheme.colorScheme.onSurfaceVariant.copy(alpha = 0.6f), fontSize = 12.sp diff --git a/app/src/main/java/com/autonion/automationcompanion/ui/components/DashboardComponents.kt b/app/src/main/java/com/autonion/automationcompanion/ui/components/DashboardComponents.kt index 7bed65a1..561bdd1d 100644 --- a/app/src/main/java/com/autonion/automationcompanion/ui/components/DashboardComponents.kt +++ b/app/src/main/java/com/autonion/automationcompanion/ui/components/DashboardComponents.kt @@ -761,7 +761,7 @@ fun TabletBrandingPanel( ) Spacer(modifier = Modifier.height(24.dp)) Text( - text = "Record gestures, build visual flows, and let AI agents automate tasks — all on-device with optional cloud power.", + text = "Record gestures, build visual flows, and automate tasks with natural language — all on-device with optional cloud power.", style = MaterialTheme.typography.bodyLarge.copy( color = MaterialTheme.colorScheme.onSurfaceVariant, lineHeight = 24.sp diff --git a/app/src/main/java/com/autonion/automationcompanion/ui/components/WhatsNewData.kt b/app/src/main/java/com/autonion/automationcompanion/ui/components/WhatsNewData.kt index ab6260ca..c30253d7 100644 --- a/app/src/main/java/com/autonion/automationcompanion/ui/components/WhatsNewData.kt +++ b/app/src/main/java/com/autonion/automationcompanion/ui/components/WhatsNewData.kt @@ -24,8 +24,8 @@ data class WhatsNewItem( ) data class WhatsNewRelease( - val versionName: String = "1.1.0", - val versionCode: Int = 9, + val versionName: String = "1.1.1", + val versionCode: Int = 10, val releaseDate: String = "August 2026", val headline: String = "Cross-Device Flows, SLM Inference & Enhanced Nodes", val description: String = "Experience faster local AI inference, seamless desktop pairing, remote desktop unlocks, and powerful new node modes in the Flow Builder.", @@ -34,10 +34,10 @@ data class WhatsNewRelease( ) object WhatsNewRepository { - fun getCurrentRelease(versionName: String = "1.1.0"): WhatsNewRelease { + fun getCurrentRelease(versionName: String = "1.1.1"): WhatsNewRelease { return WhatsNewRelease( versionName = versionName, - versionCode = 9, + versionCode = 10, releaseDate = "August 2026", headline = "Cross-Device Flows, SLM Inference & Enhanced Nodes", description = "Experience faster local AI inference, seamless desktop pairing, remote desktop unlocks, and powerful new node modes in the Flow Builder.", diff --git a/app/src/main/java/com/autonion/automationcompanion/ui/onboarding/OnboardingScreen.kt b/app/src/main/java/com/autonion/automationcompanion/ui/onboarding/OnboardingScreen.kt index e1e13e4f..327526f0 100644 --- a/app/src/main/java/com/autonion/automationcompanion/ui/onboarding/OnboardingScreen.kt +++ b/app/src/main/java/com/autonion/automationcompanion/ui/onboarding/OnboardingScreen.kt @@ -262,7 +262,7 @@ private fun WelcomePage() { // Benefits BenefitRow(Icons.Default.TouchApp, "Record & replay gestures across any app") Spacer(Modifier.height(16.dp)) - BenefitRow(Icons.Default.SmartToy, "AI agent that automates tasks with natural language") + BenefitRow(Icons.Default.SmartToy, "Smart automation that assists with natural language tasks") Spacer(Modifier.height(16.dp)) BenefitRow(Icons.Default.Devices, "Control your desktop from your phone") } @@ -333,7 +333,7 @@ private fun AISetupPage() { Spacer(Modifier.height(8.dp)) Text( - "Connecting to an AI model unlocks Omni-Chat,\nSemantic AI Agent, and smarter automation.", + "Connecting to an AI model unlocks Omni-Chat,\nSemantic Automation, and smarter assistance.", color = Color.White.copy(alpha = 0.5f), fontSize = 14.sp, textAlign = TextAlign.Center, @@ -462,7 +462,7 @@ private fun QuickStartPage( Spacer(Modifier.height(12.dp)) QuickStartCard( emoji = "🤖", - title = "Try AI Agent", + title = "Try Semantic Automation", subtitle = "Automate with natural language commands", color = Color(0xFFFF9800), onClick = { onPickFeature("feature/semantic_automation") } diff --git a/app/src/main/res/values/strings.xml b/app/src/main/res/values/strings.xml index 61ab6365..34787bd8 100644 --- a/app/src/main/res/values/strings.xml +++ b/app/src/main/res/values/strings.xml @@ -1,6 +1,6 @@ Autonion - Accessibility Service for Automation Companion. + Autonion uses the Accessibility Service only to carry out gesture recordings and automation flows that you explicitly create or trigger. It reads screen content solely to perform the specific action you request. Required for touch automation Overlay permission is required to show controls Accessibility permission is required for automation