services: ollama: image: ollama/ollama:latest container_name: ollama restart: unless-stopped runtime: nvidia ports: - "11434:11434" volumes: - ./ollama_data:/root/.ollama environment: - OLLAMA_HOST=0.0.0.0 - OLLAMA_ORIGINS=* - OLLAMA_NUM_PARALLEL=4 - OLLAMA_MAX_LOADED_MODELS=2 - OLLAMA_KEEP_ALIVE=10m - OLLAMA_FLASH_ATTENTION=1 - OLLAMA_KV_CACHE_TYPE=q4_0 deploy: resources: reservations: devices: - driver: nvidia count: all capabilities: [gpu] healthcheck: test: ["CMD", "ollama", "list"] interval: 30s timeout: 10s retries: 5 searxng: image: searxng/searxng:latest container_name: searxng restart: unless-stopped ports: - "11436:8080" volumes: - ./searxng:/etc/searxng environment: - BASE_URL=http://localhost:11436/ - INSTANCE_NAME=searxng - SEARXNG_BASE_URL=${SEARXNG_BASE_URL} - SEARXNG_SECRET=${SEARXNG_SECRET} - SEARXNG_LIMITER=false healthcheck: test: ["CMD-SHELL", "curl -fsS http://localhost:8080/healthz >/dev/null || exit 1"] interval: 30s timeout: 10s retries: 3 start_period: 30s open-webui: image: ghcr.io/open-webui/open-webui:main container_name: open-webui restart: unless-stopped ports: - "11435:8080" volumes: - ./open-webui_data:/app/backend/data environment: - OLLAMA_BASE_URL=http://ollama:11434 - ENABLE_RAG_WEB_SEARCH=true - RAG_WEB_SEARCH_ENGINE=searxng - SEARXNG_QUERY_URL=http://searxng:8080/search?q= depends_on: - ollama - searxng