Ref: https://github.com/jonesckevin/whisper-webapp

Whisper WebApp Dockerized version for running OpenAI’s Whisper speech-to-text model. It provides an easy-to-use web interface for transcribing audio/video files with support for multiple languages and models.
Deploying Whisper WebApp in a Docker container simplifies setup and ensures consistent environments for reliable transcription.
yaml 33 lines
---
services:
whisper-webapp:
image: 'jonesckevin/whisper-webapp:latest'
ports:
- "${WHISPERPORT:-8000}:5000"
volumes:
# Mapped directories for external file access via network share
- ./data/uploads:/data/uploads
- ./data/completed:/data/completed
# Whisper models cache - persists models outside container
- ./data/models:/root/.cache/whisper
environment:
- NVIDIA_VISIBLE_DEVICES=all
- CUDA_VISIBLE_DEVICES=0
- FLASK_ENV=production
- MAX_UPLOAD_SIZE_GB=5
- PRELOAD_WHISPER_MODELS=false # Set to 'true' to download all models on first run
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
capabilities: [gpu]
restart: unless-stopped
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:5000/api/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 60sDocker Run
GPU
bash 15 lines
# GPU Version
docker run -d \
--name whisper-webapp \
-p 8000:5000 \
-v "$(pwd)/data/uploads:/data/uploads" \
-v "$(pwd)/data:/data/db" \
-v "$(pwd)/data/models:/root/.cache/whisper" \
-e NVIDIA_VISIBLE_DEVICES=all \
-e CUDA_VISIBLE_DEVICES=0 \
-e FLASK_ENV=production \
-e MAX_UPLOAD_SIZE_GB=5 \
-e PRELOAD_WHISPER_MODELS=false \
--gpus '"device=0"' \
--restart unless-stopped \
jonesckevin/whisper-webapp:gpu