diff --git a/.dockerignore b/.dockerignore index 14aab39..222b03b 100644 --- a/.dockerignore +++ b/.dockerignore @@ -3,10 +3,9 @@ __pycache__/ *.py[cod] # Virtual environments -venv/ -.venv/ +agentic_seek_env/ +.agentic_seek_env/ -# Environment variables (secrets) .env # Git metadata diff --git a/.env.example b/.env.example index 069f23c..7cf366d 100644 --- a/.env.example +++ b/.env.example @@ -1,4 +1,12 @@ SEARXNG_BASE_URL="http://127.0.0.1:8080" +REDIS_BASE_URL="redis://redis:6379/0" +WORK_DIR="/Users/username/Documents/workspace_with_my_files" +OLLAMA_PORT="11434" +LM_STUDIO_PORT="1234" +CUSTOM_ADDITIONAL_LLM_PORT="11435" OPENAI_API_KEY='xxxxx' DEEPSEEK_API_KEY='xxxxx' -OPENROUTER_API_KEY='xxxxx' \ No newline at end of file +OPENROUTER_API_KEY='xxxxx' +TOGETHER_API_KEY='xxxxx' +GOOGLE_API_KEY='xxxxx' +ANTHROPIC_API_KEY='xxxxx' \ No newline at end of file diff --git a/.gitignore b/.gitignore index 0e82296..d1e9b6e 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,8 @@ *.egg-info cookies.json test_agent.py +searxng/uwsgi.ini.new +searxng/settings.yml.new config.ini .voices/ experimental/ @@ -19,6 +21,7 @@ agentic_seek_env/* .env */.env dsk/ +chrome136/ ### react ### .DS_* diff --git a/Dockerfile.backend b/Dockerfile.backend index 1cb8149..f6dec0f 100644 --- a/Dockerfile.backend +++ b/Dockerfile.backend @@ -1,46 +1,102 @@ -FROM ubuntu:22.04 -# Warning: doesn't work yet, backend is run on host machine for now + +FROM --platform=linux/amd64 python:3.11-slim +ENV DEBIAN_FRONTEND=noninteractive + +# Install essential packages and Chrome dependencies +RUN apt-get update -y && apt-get install -y \ + wget \ + gnupg2 \ + ca-certificates \ + unzip \ + xvfb \ + libxss1 \ + libappindicator1 \ + fonts-liberation \ + libnss3 \ + libatk1.0-0 \ + libatk-bridge2.0-0 \ + libcups2 \ + libdrm2 \ + libxcomposite1 \ + libxdamage1 \ + libxrandr2 \ + xdg-utils \ + dbus \ + && rm -rf /var/lib/apt/lists/* + +RUN apt-get update -y && \ + apt-get install -y \ + gcc \ + g++ \ + gfortran \ + libportaudio2 \ + portaudio19-dev \ + ffmpeg \ + libavcodec-dev \ + libavformat-dev \ + libavutil-dev \ + gnupg2 \ + wget \ + unzip \ + python3 \ + python3-pip \ + libasound2 \ + libatk-bridge2.0-0 \ + libgtk-4-1 \ + libnss3 \ + xdg-utils \ + wget \ + && rm -rf /var/lib/apt/lists/* + + +RUN apt-get update -y && \ +apt-get install -y \ + alsa-utils \ +&& rm -rf /var/lib/apt/lists/* + +ENV CHROME_TESTING_VERSION=134.0.6998.88 +ENV DISPLAY=:99 WORKDIR /app -RUN apt-get update -qq -y && \ -apt-get install -y \ - gcc \ - g++ \ - gfortran \ - libportaudio2 \ - portaudio19-dev \ - ffmpeg \ - libavcodec-dev \ - libavformat-dev \ - libavutil-dev \ - gnupg2 \ - wget \ - unzip \ - python3 \ - python3-pip \ - libasound2 \ - libatk-bridge2.0-0 \ - libgtk-4-1 \ - libnss3 \ - xdg-utils \ - wget && \ +RUN set -eux; \ + wget -qO /tmp/chrome.zip \ + "https://storage.googleapis.com/chrome-for-testing-public/${CHROME_TESTING_VERSION}/linux64/chrome-linux64.zip"; \ + unzip -q /tmp/chrome.zip -d /opt; \ + rm /tmp/chrome.zip; \ + ln -s /opt/chrome-linux64/chrome /usr/local/bin/google-chrome; \ + ln -s /opt/chrome-linux64/chrome /usr/local/bin/chrome; \ + mkdir -p /opt/chrome; \ + ln -s /opt/chrome-linux64/chrome /opt/chrome/chrome; \ + google-chrome --version + +RUN set -eux; \ + wget -qO /tmp/chromedriver.zip \ + "https://storage.googleapis.com/chrome-for-testing-public/${CHROME_TESTING_VERSION}/linux64/chromedriver-linux64.zip"; \ + unzip -q /tmp/chromedriver.zip -d /tmp; \ + mv /tmp/chromedriver-linux64/chromedriver /usr/local/bin; \ + rm /tmp/chromedriver.zip; \ + chmod +x /usr/local/bin/chromedriver; \ + chromedriver --version RUN chmod +x /opt/chrome/chrome -# Install dependencies + +RUN pip3 install --upgrade pip setuptools wheel + COPY requirements.txt . RUN pip install --no-cache-dir -r requirements.txt +RUN mkdir -p /opt/workspace +RUN mkdir -p /tmp && chmod 1777 /tmp + # Copy application code COPY api.py . COPY sources/ ./sources/ COPY prompts/ ./prompts/ COPY crx/ crx/ COPY llm_router/ llm_router/ -COPY .env . COPY config.ini . -# Expose port EXPOSE 8000 # Run the application diff --git a/README.md b/README.md index f68d8b6..4be7c3d 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ * 📋 Plans & Executes Complex Tasks - From trip planning to complex projects — it can split big tasks into steps and get things done using multiple AI agents. -* 🎙️ Voice-Enabled - Clean, fast, futuristic voice and speech to text allowing you to talk to it like it's your personal AI from a sci-fi movie +* 🎙️ Voice-Enabled - Clean, fast, futuristic voice and speech to text allowing you to talk to it like it's your personal AI from a sci-fi movie. (In progress) ### **Demo** @@ -32,19 +32,17 @@ https://github.com/user-attachments/assets/b8ca60e9-7b3b-4533-840e-08f9ac426316 Disclaimer: This demo, including all the files that appear (e.g: CV_candidates.zip), are entirely fictional. We are not a corporation, we seek open-source contributors not candidates. -> 🛠⚠️️ **Active Work in Progress** – Please note that Code/Bash is not dockerized yet but will be soon (see docker_deployement branch) - Do not deploy over network or production. +> 🛠⚠️️ **Active Work in Progress** -> 🙏 This project started as a side-project with zero roadmap and zero funding. It's grown way beyond what I expected by ending in GitHub Trending. Contributions, feedback, and patience are deeply appreciated. +> 🙏 This project started as a side-project and has zero roadmap and zero funding. It's grown way beyond what I expected by ending in GitHub Trending. Contributions, feedback, and patience are deeply appreciated. -## Installation +## Prerequisites Make sure you have chrome driver, docker and python3.10 installed. -We highly advice you use exactly python3.10 for the setup. Dependencies error might happen otherwise. - For issues related to chrome driver, see the **Chromedriver** section. -### 1️⃣ **Clone the repository and setup** +### 1. **Clone the repository and setup** ```sh git clone https://github.com/Fosowl/agenticSeek.git @@ -52,82 +50,57 @@ cd agenticSeek mv .env.example .env ``` -## Step 2: Install UV Package Manager - -### For Linux/macOS: -```bash -curl -LsSf https://astral.sh/uv/install.sh | sh -``` - -### For Windows: -```powershell -powershell -c "irm https://astral.sh/uv/install.ps1 | iex" -``` - -## Step 3: Create Virtual Environment -```bash -uv venv -``` - -### 3️⃣ **Install package** - -Ensure Python, Docker and docker compose, and Google chrome are installed. - -We recommand Python 3.10.0. - -**Automatic Installation (Recommanded):** - -For Linux/Macos: -```sh -./install.sh -``` - -For windows: +### 2. Change the .env file content ```sh -./install.bat +SEARXNG_BASE_URL="http://127.0.0.1:8080" +REDIS_BASE_URL="redis://redis:6379/0" +WORK_DIR="/Users/mlg/Documents/workspace_for_ai" +OLLAMA_PORT="11434" +LM_STUDIO_PORT="1234" +CUSTOM_ADDITIONAL_LLM_PORT="11435" +OPENAI_API_KEY='optional' +DEEPSEEK_API_KEY='optional' +OPENROUTER_API_KEY='optional' +TOGETHER_API_KEY='optional' +GOOGLE_API_KEY='optional' +ANTHROPIC_API_KEY='optional' ``` -**Manually:** +**API Key are totally optional for user who choose to run LLM locally. Which is the primary purpose of this project. Leave empty if you have sufficient hardware** -**Note: For any OS, ensure the ChromeDriver you install matches your installed Chrome version. Run `google-chrome --version`. See known issues if you have chrome >135** +The following environment variables configure your application's connections and API keys. -- *Linux*: +Update the `.env` file with your own values as needed: -Update Package List: `sudo apt update` +- **SEARXNG_BASE_URL**: Leave unchanged +- **REDIS_BASE_URL**: Leave unchanged +- **WORK_DIR**: Path to your working directory on your local machine. AgenticSeek will be able to read and interact with these files. +- **OLLAMA_PORT**: Port number for the Ollama service. +- **LM_STUDIO_PORT**: Port number for the LM Studio service. +- **CUSTOM_ADDITIONAL_LLM_PORT**: Port for any additional custom LLM service. -Install Dependencies: `sudo apt install -y alsa-utils portaudio19-dev python3-pyaudio libgtk-3-dev libnotify-dev libgconf-2-4 libnss3 libxss1` +All API key environment variables below are **optional**. You only need to provide them if you plan to use external APIs instead of running LLMs locally. -Install ChromeDriver matching your Chrome browser version: -`sudo apt install -y chromium-chromedriver` +### 3. **Start Docker** -Install requirements: `pip3 install -r requirements.txt` +Make sure Docker is installed and running on your system. You can start Docker using the following commands: -- *Macos*: +- **On Linux/macOS:** + Open a terminal and run: + ```sh + sudo systemctl start docker + ``` + Or launch Docker Desktop from your applications menu if installed. -Update brew : `brew update` +- **On Windows:** + Start Docker Desktop from the Start menu. -Install chromedriver : `brew install --cask chromedriver` - -Install portaudio: `brew install portaudio` - -Upgrade pip : `python3 -m pip install --upgrade pip` - -Upgrade wheel : : `pip3 install --upgrade setuptools wheel` - -Install requirements: `pip3 install -r requirements.txt` - -- *Windows*: - -Install pyreadline3 `pip install pyreadline3` - -Install portaudio manually (e.g., via vcpkg or prebuilt binaries) and then run: `pip install pyaudio` - -Download and install chromedriver manually from: https://sites.google.com/chromium.org/driver/getting-started - -Place chromedriver in a directory included in your PATH. - -Install requirements: `pip3 install -r requirements.txt` +You can verify Docker is running by executing: +```sh +docker info +``` +If you see information about your Docker installation, it is running correctly. --- @@ -149,7 +122,7 @@ See below for a list of local supported provider. **Update the config.ini** -Change the config.ini file to set the provider_name to a supported provider and provider_model to a LLM supported by your provider. We recommand reasoning model such as *Qwen* or *Deepseek*. +Change the config.ini file to set the provider_name to a supported provider and provider_model to a LLM supported by your provider. We recommend reasoning model such as *Qwen* or *Deepseek*. See the **FAQ** at the end of the README for required hardware. @@ -162,19 +135,23 @@ provider_server_address = 127.0.0.1:11434 agent_name = Jarvis # name of your AI recover_last_session = True # whenever to recover the previous session save_session = True # whenever to remember the current session -speak = True # text to speech -listen = False # Speech to text, only for CLI -work_dir = /Users/mlg/Documents/workspace # The workspace for AgenticSeek. +speak = False # text to speech +listen = False # Speech to text, only for CLI, experimental jarvis_personality = False # Whenever to use a more "Jarvis" like personality (experimental) languages = en zh # The list of languages, Text to speech will default to the first language on the list [BROWSER] -headless_browser = True # Whenever to use headless browser, recommanded only if you use web interface. +headless_browser = True # leave unchanged unless using CLI on host. stealth_mode = True # Use undetected selenium to reduce browser detection ``` -Warning: Do *NOT* set provider_name to `openai` if using LM-studio for running LLMs. Set it to `lm-studio`. +**Warning**: -Note: Some provider (eg: lm-studio) require you to have `http://` in front of the IP. For example `http://127.0.0.1:1234` +- The `config.ini` file format does not support comments. +Do not copy and paste the example configuration directly, as comments will cause errors. Instead, manually modify the `config.ini` file with your desired settings, excluding any comments. + +- Do *NOT* set provider_name to `openai` if using LM-studio for running LLMs. Set it to `lm-studio`. + +- Some provider (eg: lm-studio) require you to have `http://` in front of the IP. For example `http://127.0.0.1:1234` **List of local providers** @@ -196,6 +173,8 @@ Next step: [Start services and run AgenticSeek](#Start-services-and-Run) ## Setup to run with an API +**Running with an API is optional, see above to run locally.** + Set the desired provider in the `config.ini`. See below for a list of API providers. ```sh @@ -221,9 +200,7 @@ Example: export `TOGETHER_API_KEY="xxxxx"` | togetherAI | No | Use together AI API (non-private) | | google | No | Use google gemini API (non-private) | -*We advice against using gpt-4o or other closedAI models*, performance are poor for web browsing and task planning. - -Please also note that coding/bash might fail with gemini, it seem to ignore our prompt for format to respect, which are optimized for deepseek r1. +Please note that coding/bash might fail with gemini, it seems to ignore our prompt for format to respect, which are optimized for deepseek r1. Model such are gpt-4o seem to perform poorly with our prompt as well. Next step: [Start services and run AgenticSeek](#Start-services-and-Run) @@ -235,44 +212,44 @@ Next step: [Start services and run AgenticSeek](#Start-services-and-Run) ## Start services and Run -Activate your python env if needed. -```sh -source .venv/bin/activate -``` - Start required services. This will start all services from the docker-compose.yml, including: - searxng - redis (required by searxng) - frontend + - backend (if using `full`) + +```sh +sudo ./start_services.sh full # MacOS +start ./start_services.cmd full # Window +``` + +**Warning:** This step will download and load all Docker images, which may take up to 30 minutes. After starting the services, please wait until the backend service is fully running (you should see backend: in the log) before sending any messages. The backend services may take longer to start than others. + +Go to `http://localhost:3000/` and you should see the web interface. + +**Optional:** Run with the CLI interface: + +To run with CLI interface you would have to install package on host: + +```sh +./install.sh +./install.bat # windows +``` + +Start services: ```sh sudo ./start_services.sh # MacOS start ./start_services.cmd # Window ``` -**Options 1:** Run with the CLI interface. - -```sh -python3 cli.py -``` - -We advice you set `headless_browser` to False in the config.ini for CLI mode. - -**Options 2:** Run with the Web interface. - -Start the backend. - -```sh -python3 api.py -``` - -Go to `http://localhost:3000/` and you should see the web interface. +Then run : `python3 cli.py` --- ## Usage -Make sure the services are up and running with `./start_services.sh` and run the AgenticSeek with `python3 cli.py` for CLI mode or `python3 api.py` then go to `localhost:3000` for web interface. +Make sure the services are up and running with `./start_services.sh full` and go to `localhost:3000` for web interface. You can also use speech to text by setting `listen = True` in the config. Only for CLI mode. @@ -368,6 +345,8 @@ Next step: [Start services and run AgenticSeek](#Start-services-and-Run) ## Speech to Text +Warning: speech to text only work in CLI mode at the moment. + Please note that currently speech to text only work in english. The speech-to-text functionality is disabled by default. To enable it, set the listen option to True in the config.ini file: @@ -407,7 +386,6 @@ recover_last_session = False save_session = False speak = False listen = False -work_dir = /Users/mlg/Documents/ai_folder jarvis_personality = False languages = en zh [BROWSER] @@ -435,8 +413,6 @@ stealth_mode = False - listen -> listen to voice input (True) or not (False). -- work_dir -> Folder the AI will have access to. eg: /Users/user/Documents/. - - jarvis_personality -> Uses a JARVIS-like personality (True) or not (False). This simply change the prompt file. - languages -> The list of supported language, needed for the llm router to work properly, avoid putting too many or too similar languages. @@ -550,7 +526,7 @@ Yes with Ollama, lm-studio or server providers, all speech to text, LLM and text **Q: Why should I use AgenticSeek when I have Manus?** This started as Side-Project we did out of interest about AI agents. What’s special about it is that we want to use local model and avoid APIs. -We draw inspiration from Jarvis and Friday (Iron man movies) to make it "cool" but for functionnality we take more inspiration from Manus, because that's what people want in the first place: a local manus alternative. +We draw inspiration from Jarvis and Friday (Iron man movies) to make it "cool" but for functionality we take more inspiration from Manus, because that's what people want in the first place: a local manus alternative. Unlike Manus, AgenticSeek prioritizes independence from external systems, giving you more control, privacy and avoid api cost. ## Contribute @@ -568,3 +544,8 @@ We’re looking for developers to improve AgenticSeek! Check out open issues or > [antoineVIVIES](https://github.com/antoineVIVIES) | Taipei Time > [steveh8758](https://github.com/steveh8758) | Taipei Time + +## Special Thanks: + + > [tcsenpai](https://github.com/tcsenpai) and [plitc](https://github.com/plitc) For helping with backend dockerization + diff --git a/api.py b/api.py index fa82896..9e70a4a 100755 --- a/api.py +++ b/api.py @@ -22,6 +22,10 @@ from sources.utility import pretty_print from sources.logger import Logger from sources.schemas import QueryRequest, QueryResponse +from dotenv import load_dotenv + +load_dotenv() + from celery import Celery @@ -34,7 +38,7 @@ config.read('config.ini') api.add_middleware( CORSMiddleware, - allow_origins=["http://localhost", "http://localhost:3000"], + allow_origins=["*"], allow_credentials=True, allow_methods=["*"], allow_headers=["*"], @@ -247,4 +251,9 @@ async def process_query(request: QueryRequest): interaction.save_session() if __name__ == "__main__": + envport = os.getenv("BACKEND_PORT") + if envport: + port = int(envport) + else: + port = 8000 uvicorn.run(api, host="0.0.0.0", port=8000) \ No newline at end of file diff --git a/config.ini b/config.ini index 19cb1e4..21bce48 100644 --- a/config.ini +++ b/config.ini @@ -3,12 +3,11 @@ is_local = True provider_name = ollama provider_model = deepseek-r1:14b provider_server_address = 127.0.0.1:11434 -agent_name = Name_of_your_AI +agent_name = Jarvis recover_last_session = False save_session = False speak = False listen = False -work_dir = /Users/mlg/Documents/workspace_for_agenticseek jarvis_personality = False languages = en [BROWSER] diff --git a/docker-compose.yml b/docker-compose.yml index 8dd20e2..5cd548d 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -3,6 +3,7 @@ version: '3' services: redis: container_name: redis + profiles: ["core", "full"] image: docker.io/valkey/valkey:8-alpine command: valkey-server --save 30 1 --loglevel warning restart: unless-stopped @@ -24,6 +25,7 @@ services: searxng: container_name: searxng + profiles: ["core", "full"] image: docker.io/searxng/searxng:latest restart: unless-stopped ports: @@ -31,8 +33,8 @@ services: volumes: - ./searxng:/etc/searxng:rw,z environment: - - SEARXNG_BASE_URL=http://localhost:8080/ - - SEARXNG_SECRET_KEY=$(openssl rand -hex 32) + - SEARXNG_BASE_URL=${SEARXNG_BASE_URL:-http://localhost:8080/} + - SEARXNG_SECRET_KEY=${SEARXNG_SECRET_KEY} - UWSGI_WORKERS=4 - UWSGI_THREADS=4 cap_add: @@ -51,6 +53,7 @@ services: frontend: container_name: frontend + profiles: ["core", "full"] build: context: ./frontend dockerfile: Dockerfile.frontend @@ -62,39 +65,38 @@ services: environment: - NODE_ENV=development - CHOKIDAR_USEPOLLING=true - - BACKEND_URL=http://backend:8000 + - REACT_APP_BACKEND_URL=http://0.0.0.0:${BACKEND_PORT:-8000} networks: - agentic-seek-net - - # NOTE: backend service is not working yet due to issue with chromedriver on docker. - # Therefore backend is run on host machine. - # Open to pull requests to fix this. - #backend: - # container_name: backend - # build: - # context: ./ - # dockerfile: Dockerfile.backend - # stdin_open: true - # tty: true - # shm_size: 8g - # ports: - # - "8000:8000" - # volumes: - # - ./:/app - # environment: - # - NODE_ENV=development - # - REDIS_URL=redis://redis:6379/0 - # - SEARXNG_URL=http://searxng:8080 - # - OLLAMA_URL=http://localhost:11434 - # - LM_STUDIO_URL=http://localhost:1234 - # extra_hosts: - # - "host.docker.internal:host-gateway" - # depends_on: - # - redis - # - searxng - # networks: - # - agentic-seek-net + backend: + container_name: backend + profiles: ["backend", "full"] + build: + context: . + dockerfile: Dockerfile.backend + ports: + - ${BACKEND_PORT:-7777}:${BACKEND_PORT:-7777} + - ${OLLAMA_PORT:-11434}:${OLLAMA_PORT:-11434} + - ${LM_STUDIO_PORT:-1234}:${LM_STUDIO_PORT:-1234} + - ${CUSTOM_ADDITIONAL_LLM_PORT:-8000}:${CUSTOM_ADDITIONAL_LLM_PORT:-8000} + volumes: + - ./:/app + - ${WORK_DIR:-.}:/opt/workspace + command: python3 api.py + environment: + - SEARXNG_URL=${SEARXNG_BASE_URL:-http://searxng:8080} + - REDIS_URL=${REDIS_BASE_URL:-redis://redis:6379/0} + - WORK_DIR=/opt/workspace + - OPENAI_API_KEY=${OPENAI_API_KEY} + - DEEPSEEK_API_KEY=${DEEPSEEK_API_KEY} + - OPENROUTER_API_KEY=${OPENROUTER_API_KEY} + - TOGETHER_API_KEY=${TOGETHER_API_KEY} + - GOOGLE_API_KEY=${GOOGLE_API_KEY} + - ANTHROPIC_API_KEY=${ANTHROPIC_API_KEY} + - HUGGINGFACE_API_KEY=${HUGGINGFACE_API_KEY} + - DSK_DEEPSEEK_API_KEY=${DSK_DEEPSEEK_API_KEY} + network_mode: "host" volumes: redis-data: diff --git a/frontend/agentic-seek-front/src/App.js b/frontend/agentic-seek-front/src/App.js index c1e6ba8..ed64d08 100644 --- a/frontend/agentic-seek-front/src/App.js +++ b/frontend/agentic-seek-front/src/App.js @@ -4,6 +4,8 @@ import axios from 'axios'; import './App.css'; import { colors } from './colors'; +const BACKEND_URL = process.env.BACKEND_PORT || 'http://0.0.0.0:8000'; + function App() { const [query, setQuery] = useState(''); const [messages, setMessages] = useState([]); @@ -27,7 +29,7 @@ function App() { const checkHealth = async () => { try { - await axios.get('http://127.0.0.1:8000/health'); + await axios.get(`${BACKEND_URL}/health`); setIsOnline(true); console.log('System is online'); } catch { @@ -39,7 +41,7 @@ function App() { const fetchScreenshot = async () => { try { const timestamp = new Date().getTime(); - const res = await axios.get(`http://127.0.0.1:8000/screenshots/updated_screen.png?timestamp=${timestamp}`, { + const res = await axios.get(`${BACKEND_URL}/screenshots/updated_screen.png?timestamp=${timestamp}`, { responseType: 'blob' }); console.log('Screenshot fetched successfully'); @@ -90,7 +92,7 @@ function App() { const fetchLatestAnswer = async () => { try { - const res = await axios.get('http://127.0.0.1:8000/latest_answer'); + const res = await axios.get(`${BACKEND_URL}/latest_answer`); const data = res.data; updateData(data); @@ -141,7 +143,7 @@ function App() { setIsLoading(false); setError(null); try { - const res = await axios.get('http://127.0.0.1:8000/stop'); + const res = await axios.get(`${BACKEND_URL}/stop`); setStatus("Requesting stop..."); } catch (err) { console.error('Error stopping the agent:', err); @@ -162,7 +164,7 @@ function App() { try { console.log('Sending query:', query); setQuery('waiting for response...'); - const res = await axios.post('http://127.0.0.1:8000/query', { + const res = await axios.post(`${BACKEND_URL}/query`, { query, tts_enabled: false }); diff --git a/requirements.txt b/requirements.txt index d92068d..58e09bb 100644 --- a/requirements.txt +++ b/requirements.txt @@ -17,7 +17,6 @@ playsound3>=1.0.0 soundfile>=0.13.1 transformers>=4.46.3 torch>=2.4.1 -python-dotenv>=1.0.0 ollama>=0.4.7 scipy>=1.9.3 soundfile>=0.13.1 @@ -41,6 +40,7 @@ fake_useragent>=2.1.0 selenium_stealth>=1.0.6 undetected-chromedriver>=3.5.5 sentencepiece>=0.2.0 +together>=1.5.0 tqdm>4 openai sniffio diff --git a/searxng/uwsgi.ini b/searxng/uwsgi.ini index 65fb79a..24b1972 100644 --- a/searxng/uwsgi.ini +++ b/searxng/uwsgi.ini @@ -5,12 +5,12 @@ gid = searxng # Number of workers (usually CPU count) # default value: %k (= number of CPU core, see Dockerfile) -workers = 1 +workers = 4 # Number of threads per worker # default value: 4 (see Dockerfile) -enable-threads = true -threads = 1 +enable-threads = 4 +threads = 4 # The right granted on the created socket chmod-socket = 666 diff --git a/sources/agents/browser_agent.py b/sources/agents/browser_agent.py index 3817fb3..3f5de5d 100644 --- a/sources/agents/browser_agent.py +++ b/sources/agents/browser_agent.py @@ -41,7 +41,7 @@ class BrowserAgent(Agent): self.memory = Memory(self.load_prompt(prompt_path), recover_last_session=False, # session recovery in handled by the interaction class memory_compression=False, - model_provider=provider.get_model_name()) + model_provider=provider.get_model_name() if provider else None) def get_today_date(self) -> str: """Get the date""" @@ -77,14 +77,14 @@ class BrowserAgent(Agent): def get_unvisited_links(self) -> List[str]: return "\n".join([f"[{i}] {link}" for i, link in enumerate(self.navigable_links) if link not in self.search_history]) - def make_newsearch_prompt(self, user_prompt: str, search_result: dict) -> str: + def make_newsearch_prompt(self, prompt: str, search_result: dict) -> str: search_choice = self.stringify_search_results(search_result) self.logger.info(f"Search results: {search_choice}") return f""" Based on the search result: {search_choice} Your goal is to find accurate and complete information to satisfy the user’s request. - User request: {user_prompt} + User request: {prompt} To proceed, choose a relevant link from the search results. Announce your choice by saying: "I will navigate to " Do not explain your choice. """ @@ -235,13 +235,17 @@ class BrowserAgent(Agent): return links def select_link(self, links: List[str]) -> str | None: + """ + Select the first unvisited link that is not the current page. + Preference is given to links not in search_history. + """ for lk in links: - if lk == self.current_page: - self.logger.info(f"Already visited {lk}. Skipping.") + if lk == self.current_page or lk in self.search_history: + self.logger.info(f"Skipping already visited or current link: {lk}") continue self.logger.info(f"Selected link: {lk}") return lk - self.logger.warning("No link selected.") + self.logger.warning("No suitable link selected.") return None def get_page_text(self, limit_to_model_ctx = False) -> str: @@ -396,7 +400,10 @@ class BrowserAgent(Agent): if (link == None and len(extracted_form) < 3) or Action.GO_BACK.value in answer or link in self.search_history: pretty_print(f"Going back to results. Still {len(unvisited)}", color="status") self.status_message = "Going back to search results..." - prompt = self.make_newsearch_prompt(user_prompt, unvisited) + request_prompt = user_prompt + if link is None: + request_prompt += f"\nYou previously choosen:\n{self.last_answer} but the website is unavailable. Consider other options." + prompt = self.make_newsearch_prompt(request_prompt, unvisited) self.search_history.append(link) self.current_page = link continue diff --git a/sources/browser.py b/sources/browser.py index 3604e9b..639eaec 100644 --- a/sources/browser.py +++ b/sources/browser.py @@ -19,6 +19,7 @@ import time import random import os import shutil +import uuid import tempfile import markdownify import sys @@ -42,7 +43,14 @@ def get_chrome_path() -> str: paths = ["/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", "/Applications/Google Chrome Beta.app/Contents/MacOS/Google Chrome Beta"] else: # Linux - paths = ["/usr/bin/google-chrome", "/usr/bin/chromium-browser", "/usr/bin/chromium", "/opt/chrome/chrome", "/usr/local/bin/chrome"] + paths = ["/usr/bin/google-chrome", + "/opt/chrome/chrome", + "/usr/bin/chromium-browser", + "/usr/bin/chromium", + "/usr/local/bin/chrome", + "/opt/google/chrome/chrome-headless-shell", + #"/app/chrome_bundle/chrome136/chrome-linux64" + ] for path in paths: if os.path.exists(path) and os.access(path, os.X_OK): @@ -75,6 +83,7 @@ def install_chromedriver() -> str: chromedriver_path = shutil.which("chromedriver") if not chromedriver_path: try: + print("ChromeDriver not found, attempting to install automatically...") chromedriver_path = chromedriver_autoinstaller.install() except Exception as e: raise FileNotFoundError( @@ -120,17 +129,28 @@ def create_driver(headless=False, stealth_mode=True, crx_path="./crx/nopecha.crx chrome_options.binary_location = chrome_path if headless: - chrome_options.add_argument("--headless") + #chrome_options.add_argument("--headless") + chrome_options.add_argument("--headless=new") chrome_options.add_argument("--disable-gpu") chrome_options.add_argument("--disable-webgl") user_data_dir = tempfile.mkdtemp() user_agent = get_random_user_agent() width, height = (1920, 1080) - chrome_options.add_argument(f"--user-data-dir={user_data_dir}") - chrome_options.add_argument(f"--accept-lang={lang}-{lang.upper()},{lang};q=0.9") - chrome_options.add_argument("--timezone=Europe/Paris") + user_data_dir = tempfile.mkdtemp(prefix="chrome_profile_") chrome_options.add_argument("--no-sandbox") - chrome_options.add_argument("--disable-dev-shm-usage") + chrome_options.add_argument('--disable-dev-shm-usage') + profile_dir = f"/tmp/chrome_profile_{uuid.uuid4().hex[:8]}" + chrome_options.add_argument(f'--user-data-dir={profile_dir}') + chrome_options.add_argument(f"--accept-lang={lang}-{lang.upper()},{lang};q=0.9") + chrome_options.add_argument("--disable-extensions") + chrome_options.add_argument("--disable-background-timer-throttling") + chrome_options.add_argument("--timezone=Europe/Paris") + chrome_options.add_argument('--remote-debugging-port=9222') + chrome_options.add_argument('--disable-background-timer-throttling') + chrome_options.add_argument('--disable-backgrounding-occluded-windows') + chrome_options.add_argument('--disable-renderer-backgrounding') + chrome_options.add_argument('--disable-features=TranslateUI') + chrome_options.add_argument('--disable-ipc-flooding-protection') chrome_options.add_argument("--mute-audio") chrome_options.add_argument("--disable-notifications") chrome_options.add_argument("--autoplay-policy=user-gesture-required") @@ -698,8 +718,6 @@ if __name__ == "__main__": input("press enter to continue") print("AntiCaptcha / Form Test") - browser.go_to("https://www.google.com/recaptcha/api2/demo") - time.sleep(50) browser.go_to("https://bot.sannysoft.com") time.sleep(5) #txt = browser.get_text() diff --git a/sources/language.py b/sources/language.py index b62c087..17e4c97 100644 --- a/sources/language.py +++ b/sources/language.py @@ -1,8 +1,6 @@ from typing import List, Tuple, Type, Dict import re import langid -import nltk -from nltk.sentiment.vader import SentimentIntensityAnalyzer from transformers import MarianMTModel, MarianTokenizer from sources.utility import pretty_print, animate_thinking @@ -16,7 +14,6 @@ class LanguageUtility: args: supported_language: list of languages for translation, determine which Helsinki-NLP model to load """ - self.sid = None self.translators_tokenizer = None self.translators_model = None self.logger = Logger("language.log") @@ -25,11 +22,6 @@ class LanguageUtility: def load_model(self) -> None: animate_thinking("Loading language utility...", color="status") - try: - nltk.data.find('vader_lexicon') - except LookupError: - nltk.download('vader_lexicon') - self.sid = SentimentIntensityAnalyzer() self.translators_tokenizer = {lang: MarianTokenizer.from_pretrained(f"Helsinki-NLP/opus-mt-{lang}-en") for lang in self.supported_language if lang != "en"} self.translators_model = {lang: MarianMTModel.from_pretrained(f"Helsinki-NLP/opus-mt-{lang}-en") for lang in self.supported_language if lang != "en"} @@ -65,49 +57,17 @@ class LanguageUtility: translation = model.generate(**inputs) return tokenizer.decode(translation[0], skip_special_tokens=True) - def detect_emotion(self, text: str) -> str: - """ - Detect the dominant emotion in the given text - Args: - text: string to analyze - Returns: string of the dominant emotion - """ - try: - scores = self.sid.polarity_scores(text) - emotions = { - 'Happy': max(scores['pos'], 0), - 'Angry': 0, - 'Sad': max(scores['neg'], 0), - 'Fear': 0, - 'Surprise': 0 - } - if scores['compound'] < -0.5: - emotions['Angry'] = abs(scores['compound']) * 0.5 - emotions['Fear'] = abs(scores['compound']) * 0.5 - elif scores['compound'] > 0.5: - emotions['Happy'] = scores['compound'] - emotions['Surprise'] = scores['compound'] * 0.5 - dominant_emotion = max(emotions, key=emotions.get) - if emotions[dominant_emotion] == 0: - return 'Neutral' - self.logger.info(f"Emotion: {dominant_emotion} for text: {text}") - return dominant_emotion - except Exception as e: - raise e - def analyze(self, text): """ Combined analysis of language and emotion Args: text: string to analyze - Returns: dictionary with language and emotion results + Returns: dictionary with language related information """ try: language = self.detect_language(text) - emotions = self.detect_emotion(text) return { - "language": language, - "emotions": emotions + "language": language } except Exception as e: raise e @@ -125,4 +85,4 @@ if __name__ == "__main__": pretty_print(f"Language: {detector.detect_language(text)}", color="status") result = detector.analyze(text) trans = detector.translate(text, result['language']) - pretty_print(f"Translation: {trans} - from: {result['language']} - Emotion: {result['emotions']}") \ No newline at end of file + pretty_print(f"Translation: {trans} - from: {result['language']}") \ No newline at end of file diff --git a/sources/speech_to_text.py b/sources/speech_to_text.py index a888326..9d6917b 100644 --- a/sources/speech_to_text.py +++ b/sources/speech_to_text.py @@ -3,11 +3,18 @@ from typing import List, Tuple, Type, Dict import queue import threading import numpy as np -import torch import time -from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline -import librosa -import pyaudio + +IMPORT_FOUND = True + +try: + import torch + import librosa + import pyaudio + from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline +except ImportError: + print(Fore.RED + "Speech To Text disabled." + Fore.RESET) + IMPORT_FOUND = False audio_queue = queue.Queue() done = False @@ -23,13 +30,18 @@ class AudioRecorder: self.chunk = chunk self.record_seconds = record_seconds self.verbose = verbose - self.audio = pyaudio.PyAudio() - self.thread = threading.Thread(target=self._record, daemon=True) + self.thread = None + self.audio = None + if IMPORT_FOUND: + self.audio = pyaudio.PyAudio() + self.thread = threading.Thread(target=self._record, daemon=True) def _record(self) -> None: """ Record audio from the microphone and add it to the audio queue. """ + if not IMPORT_FOUND: + return stream = self.audio.open(format=self.format, channels=self.channels, rate=self.rate, input=True, frames_per_buffer=self.chunk) if self.verbose: @@ -58,10 +70,14 @@ class AudioRecorder: def start(self) -> None: """Start the recording thread.""" + if not IMPORT_FOUND: + return self.thread.start() def join(self) -> None: """Wait for the recording thread to finish.""" + if not IMPORT_FOUND: + return self.thread.join() class Transcript: @@ -69,6 +85,9 @@ class Transcript: Transcript is a class that transcribes audio from the audio queue and adds it to the transcript. """ def __init__(self): + if not IMPORT_FOUND: + print(Fore.RED + "Transcript: Speech to Text is disabled." + Fore.RESET) + return self.last_read = None device = self.get_device() torch_dtype = torch.float16 if device == "cuda" else torch.float32 @@ -91,6 +110,8 @@ class Transcript: ) def get_device(self) -> str: + if not IMPORT_FOUND: + return "cpu" if torch.backends.mps.is_available(): return "mps" if torch.cuda.is_available(): @@ -108,6 +129,8 @@ class Transcript: def transcript_job(self, audio_data: np.ndarray, sample_rate: int = 16000) -> str: """Transcribe the audio data.""" + if not IMPORT_FOUND: + return "" if audio_data.dtype != np.float32: audio_data = audio_data.astype(np.float32) / np.iinfo(audio_data.dtype).max if len(audio_data.shape) > 1: @@ -122,6 +145,9 @@ class AudioTranscriber: AudioTranscriber is a class that transcribes audio from the audio queue and adds it to the transcript. """ def __init__(self, ai_name: str, verbose: bool = False): + if not IMPORT_FOUND: + print(Fore.RED + "AudioTranscriber: Speech to Text is disabled." + Fore.RESET) + return self.verbose = verbose self.ai_name = ai_name self.transcriptor = Transcript() @@ -152,6 +178,8 @@ class AudioTranscriber: """ Transcribe the audio data using AI stt model. """ + if not IMPORT_FOUND: + return global done if self.verbose: print(Fore.BLUE + "AudioTranscriber: Started processing..." + Fore.RESET) @@ -185,9 +213,13 @@ class AudioTranscriber: def start(self): """Start the transcription thread.""" + if not IMPORT_FOUND: + return self.thread.start() def join(self): + if not IMPORT_FOUND: + return """Wait for the transcription thread to finish.""" self.thread.join() diff --git a/sources/text_to_speech.py b/sources/text_to_speech.py index b57d782..310bfee 100644 --- a/sources/text_to_speech.py +++ b/sources/text_to_speech.py @@ -5,9 +5,14 @@ import subprocess from sys import modules from typing import List, Tuple, Type, Dict -from kokoro import KPipeline -from IPython.display import display, Audio -import soundfile as sf +IMPORT_FOUND = True +try: + from kokoro import KPipeline + from IPython.display import display, Audio + import soundfile as sf +except ImportError: + print("Speech synthesis disabled. Please install the kokoro package.") + IMPORT_FOUND = False if __name__ == "__main__": from utility import pretty_print, animate_thinking @@ -33,7 +38,7 @@ class Speech(): } self.pipeline = None self.language = language - if enable: + if enable and IMPORT_FOUND: self.pipeline = KPipeline(lang_code=self.lang_map[language]) self.voice = self.voice_map[language][voice_idx] self.speed = 1.2 @@ -57,7 +62,7 @@ class Speech(): sentence (str): The text to convert to speech. Will be pre-processed. voice_idx (int, optional): Index of the voice to use from the voice map. """ - if not self.pipeline: + if not self.pipeline or not IMPORT_FOUND: return if voice_idx >= len(self.voice_map[self.language]): pretty_print("Invalid voice number, using default voice", color="error") @@ -159,6 +164,7 @@ class Speech(): if __name__ == "__main__": # TODO add info message for cn2an, jieba chinese related import + IMPORT_FOUND = False sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) speech = Speech() tosay_en = """ diff --git a/sources/tools/tools.py b/sources/tools/tools.py index 555f30c..cb63869 100644 --- a/sources/tools/tools.py +++ b/sources/tools/tools.py @@ -49,20 +49,16 @@ class Tools(): def set_allow_language_exec_bash(value: bool) -> None: self.allow_language_exec_bash = value - - def check_config_dir_validity(self): - """Check if the config directory is valid.""" - path = self.config['MAIN']['work_dir'] - if path == "": - print("WARNING: Work directory not set in config.ini") - return False - if path.lower() == "none": - print("WARNING: Work directory set to none in config.ini") - return False - if not os.path.exists(path): - print(f"WARNING: Work directory {path} does not exist") - return False - return True + + def safe_get_work_dir_path(self): + path = None + path = os.getenv('WORK_DIR', path) + if path is None or path == "": + path = self.config['MAIN']['work_dir'] if 'MAIN' in self.config and 'work_dir' in self.config['MAIN'] else None + if path is None or path == "": + print("No work directory specified, using default.") + path = self.create_work_dir() + return path def config_exists(self): """Check if the config file exists.""" @@ -73,11 +69,10 @@ class Tools(): default_path = os.path.dirname(os.getcwd()) if self.config_exists(): self.config.read('./config.ini') - config_path = self.config['MAIN']['work_dir'] - dir_path = default_path if not self.check_config_dir_validity() else config_path + workdir_path = self.safe_get_work_dir_path() else: - dir_path = default_path - return dir_path + workdir_path = default_path + return workdir_path @abstractmethod def execute(self, blocks:[str], safety:bool) -> str: diff --git a/start_services.cmd b/start_services.cmd index 94e8649..ce278a8 100644 --- a/start_services.cmd +++ b/start_services.cmd @@ -1,10 +1,35 @@ @echo off -docker-compose up -if %ERRORLEVEL% neq 0 ( - echo Error: Failed to start containers. Check Docker logs with 'docker compose logs'. - echo Possible fixes: Ensure Docker Desktop is running or check if port 8080 is free. - exit /b 1 +if "%1"=="full" ( + echo Starting full deployment... +) else ( + echo Starting partial deployment... (backend run on host), use "full" to run all services in containers ) -timeout /t 10 /nobreak >nul \ No newline at end of file +where openssl >nul 2>&1 +if %ERRORLEVEL% == 0 ( + for /f %%i in ('openssl rand -hex 32') do set SEARXNG_SECRET_KEY=%%i +) else ( + where python3 >nul 2>&1 + if %ERRORLEVEL% == 0 ( + for /f %%i in ('python3 -c "import secrets; print(secrets.token_hex(32))"') do set SEARXNG_SECRET_KEY=%%i + ) else ( + echo Error: Neither openssl nor python is available to generate a secret key. + exit /b 1 + ) +) + +REM Stop all containers +echo Stopping containers... +docker stop $(docker ps -aq) >nul 2>&1 + +REM Generate secret key +for /f %%i in ('powershell -command "[System.Web.Security.Membership]::GeneratePassword(64,0)"') do set SEARXNG_SECRET_KEY=%%i + +if "%1"=="full" ( + docker compose up -d backend + timeout /t 5 /nobreak >nul + docker compose --profile full up +) else ( + docker compose --profile core up +) \ No newline at end of file diff --git a/start_services.sh b/start_services.sh index 995f6ff..79451ec 100755 --- a/start_services.sh +++ b/start_services.sh @@ -1,12 +1,35 @@ #!/bin/bash +source .env + command_exists() { command -v "$1" &> /dev/null } +if [ -z "$WORK_DIR" ]; then + echo "Error: WORK_DIR environment variable is not set. Please set it in your .env file." + exit 1 +fi -# -# Check if Docker is installed é running -# +if [[ "$OSTYPE" == "darwin"* ]]; then + dir_size_bytes=$(du -s -b "$WORK_DIR" 2>/dev/null | awk '{print $1}') +else + dir_size_bytes=$(du -s --bytes "$WORK_DIR" 2>/dev/null | awk '{print $1}') +fi + +max_size_bytes=$((2 * 1024 * 1024 * 1024)) + +echo "Mounting $WORK_DIR ($dir_size_bytes bytes) to docker." + +if [ "$dir_size_bytes" -gt "$max_size_bytes" ]; then + echo "Error: WORK_DIR ($WORK_DIR) contains more than 2GB of data ($(du -sh "$WORK_DIR" 2>/dev/null | awk '{print $1}'))." + exit 1 +fi + +if [ "$1" = "full" ]; then + echo "Starting full deployment with backend and all services..." +else + echo "Starting core deployment with frontend and search services only... use ./start_services.sh full to start backend as well" +fi if ! command_exists docker; then echo "Error: Docker is not installed. Please install Docker first." @@ -60,15 +83,57 @@ if [ ! -f "docker-compose.yml" ]; then exit 1 fi -# start docker compose for searxng, redis, frontend services +# Stop all running containers to ensure a clean state echo "Warning: stopping all docker containers (t-4 seconds)..." sleep 4 docker stop $(docker ps -a -q) echo "All containers stopped" -if ! $COMPOSE_CMD up; then - echo "Error: Failed to start containers. Check Docker logs with '$COMPOSE_CMD logs'." - echo "Possible fixes: Run with sudo or ensure port 8080 is free." - exit 1 +# export searxng secret key (cross-platform) +if command -v openssl &> /dev/null; then + export SEARXNG_SECRET_KEY=$(openssl rand -hex 32) +else + # Fallback: use Python if openssl is not available + if command -v python3 &> /dev/null; then + export SEARXNG_SECRET_KEY=$(python3 -c "import secrets; print(secrets.token_hex(32))") + else + echo "Error: Neither openssl nor python is available to generate a secret key." + exit 1 + fi +fi + +if [ "$1" = "full" ]; then + # First start backend and wait for it to be healthy + echo "Full docker deployement. Starting backend service..." + if ! $COMPOSE_CMD up -d backend; then + echo "Error: Failed to start backend container." + exit 1 + fi + # Wait for backend to be healthy (check if it's running and not restarting) + echo "Waiting for backend to be ready..." + for i in {1..30}; do + if [ "$(docker inspect -f '{{.State.Running}}' backend)" = "true" ] && \ + [ "$(docker inspect -f '{{.State.Restarting}}' backend)" = "false" ]; then + echo "backend is ready!" + break + fi + if [ $i -eq 30 ]; then + echo "Error: backend failed to start properly after 30 seconds" + $COMPOSE_CMD logs backend + exit 1 + fi + sleep 1 + done + if ! $COMPOSE_CMD --profile full up; then + echo "Error: Failed to start containers. Check Docker logs with '$COMPOSE_CMD logs'." + echo "Possible fixes: Run with sudo or ensure port 8080 is free." + exit 1 + fi +else + if ! $COMPOSE_CMD --profile core up; then + echo "Error: Failed to start containers. Check Docker logs with '$COMPOSE_CMD logs'." + echo "Possible fixes: Run with sudo or ensure port 8080 is free." + exit 1 + fi fi sleep 10 \ No newline at end of file diff --git a/tests/test_browser_agent_parsing.py b/tests/test_browser_agent_parsing.py index aac95bc..40d0eec 100644 --- a/tests/test_browser_agent_parsing.py +++ b/tests/test_browser_agent_parsing.py @@ -23,7 +23,7 @@ class TestBrowserAgentParsing(unittest.TestCase): "https://thriveonai.com/15-ai-startups-in-japan-to-take-note-of", "www.google.com", "https://test.org/about?page=1", - "https://weatherstack.com/documentation", + "https://weatherstack.com/documentation" ] result = self.agent.extract_links(test_text) self.assertEqual(result, expected)