services: whisper: image: fedirz/faster-whisper-server:latest-cpu container_name: andytranscribe-whisper ports: # Host 8090 avoids clashing with `php artisan serve` on 8000 - "${WHISPER_HOST_PORT:-8090}:8000" volumes: - whisper-huggingface-cache:/root/.cache/huggingface environment: # OpenAI-compatible API; model is also passed per request from Laravel WHISPER__MODEL: ${LOCAL_WHISPER_MODEL:-Systran/faster-whisper-base} restart: unless-stopped healthcheck: test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8000/health')"] interval: 15s timeout: 5s retries: 10 start_period: 30s # GPU variant (NVIDIA). Start with: # docker compose --profile gpu up -d whisper-gpu whisper-gpu: profiles: ["gpu"] image: fedirz/faster-whisper-server:latest-cuda container_name: andytranscribe-whisper-gpu ports: - "${WHISPER_HOST_PORT:-8090}:8000" volumes: - whisper-huggingface-cache:/root/.cache/huggingface environment: WHISPER__MODEL: ${LOCAL_WHISPER_MODEL:-Systran/faster-whisper-base} restart: unless-stopped deploy: resources: reservations: devices: - driver: nvidia count: 1 capabilities: [gpu] volumes: whisper-huggingface-cache: