From da5b2da097103bbf922165d48999dfcd6e4e9c40 Mon Sep 17 00:00:00 2001 From: Daniel Cox Date: Tue, 11 Aug 2026 21:46:24 +0930 Subject: [PATCH] fix: add health check + clarify volume mounts - nginx: return 200 OK on GET / for load balancer health checks - Dockerfile: set HF_HOME=/app/models so Hub downloads persist in mounted volume - Dockerfile: add /app/data directory and volume declaration - docker-compose: explicit volume mounts for models, data, lora, output - README: document where to put training files and find output - .gitignore: exclude volume mount directories (models/, data/, lora/, output/) --- .gitignore | 6 +++ docker/Dockerfile | 5 ++- docker/README.md | 91 ++++++++++++++++++++++++++++++--------- docker/docker-compose.yml | 17 ++++++-- docker/nginx.conf | 6 +++ 5 files changed, 100 insertions(+), 25 deletions(-) diff --git a/.gitignore b/.gitignore index f7fa981..61bac30 100644 --- a/.gitignore +++ b/.gitignore @@ -5,3 +5,9 @@ voxcpm.egg-info .DS_Store ./pretrained_models/ app_local.py + +# Docker volume mount directories (large files, user-specific) +models/ +data/ +lora/ +output/ diff --git a/docker/Dockerfile b/docker/Dockerfile index e22273a..b37d20c 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -38,14 +38,15 @@ RUN pip install --no-cache-dir -e . COPY . /app/ # Create default directories and declare volumes -RUN mkdir -p /app/lora /app/models /app/output -VOLUME ["/app/models", "/app/lora", "/app/output"] +RUN mkdir -p /app/lora /app/models /app/output /app/data +VOLUME ["/app/models", "/app/lora", "/app/output", "/app/data"] EXPOSE 7860 # Environment variables for configuration ENV GRADIO_SERVER_PORT=7860 ENV GRADIO_ROOT_PATH="" +ENV HF_HOME=/app/models # Default: launch training WebUI CMD ["python", "lora_ft_webui.py"] diff --git a/docker/README.md b/docker/README.md index dcb4cf3..51609fb 100644 --- a/docker/README.md +++ b/docker/README.md @@ -21,6 +21,75 @@ This starts: Access the WebUI at **http://localhost/webui/**. +## Volume Mounts + +The compose file maps host directories to container paths. Create these directories at the project root before starting: + +``` +VoxCPM/ +├── docker/ +│ ├── docker-compose.yml +│ ├── Dockerfile +│ └── nginx.conf +├── models/ ← Pretrained model weights (or auto-downloaded via HF) +│ ├── openbmb__VoxCPM2/ +│ └── openbmb__VoxCPM1.5/ +├── data/ ← Training manifests + audio files +│ ├── train.jsonl +│ ├── val.jsonl (optional) +│ └── audio/ +│ ├── speaker1_001.wav +│ └── ... +├── lora/ ← LoRA training output (created automatically) +│ └── my-voice-2024/ +│ ├── checkpoints/ +│ ├── logs/ +│ └── train_config.yaml +└── output/ ← Additional training artifacts +``` + +### Mount Reference + +| Host Path | Container Path | Purpose | +|-----------|---------------|---------| +| `./models/` | `/app/models` | Pretrained model weights and HF cache (`HF_HOME`). Pre-populate with model dirs (e.g., `openbmb__VoxCPM2/`) or leave empty — models auto-download on first run and persist here. | +| `./data/` | `/app/data` | Training data. Put JSONL manifests and audio files here. In the WebUI, reference paths as `/app/data/train.jsonl`. | +| `./lora/` | `/app/lora` | LoRA checkpoint output. After training, find results in `lora//checkpoints/`. Also used to resume training from existing checkpoints. | +| `./output/` | `/app/output` | Miscellaneous training artifacts. | + +### Training Data Format + +The train manifest is a JSONL file where each line references an audio file: + +```json +{"audio_path": "/app/data/audio/speaker1_001.wav", "text": "Hello world", "speaker": "speaker1"} +``` + +Use absolute container paths (`/app/data/...`) in your manifest so the container can find the files. + +### Models + +If `models/openbmb__VoxCPM2/` exists on the host, the app loads directly from that path — no network access needed. If the directory is empty or missing, `from_pretrained` falls back to `snapshot_download` from HuggingFace Hub. + +The Dockerfile sets `HF_HOME=/app/models` so any Hub downloads land in the same mounted volume (matching the pattern in `deploy/Dockerfile.voxcpm-unified`). This means models persist across container restarts regardless of whether they were pre-populated or auto-downloaded. + +**Recommended:** Pre-populate to avoid first-run download delay: + +```bash +huggingface-cli download openbmb/VoxCPM2 --local-dir ./models/openbmb__VoxCPM2 +``` + +The Dockerfile creates empty `/app/models`, `/app/lora`, `/app/output` directories, but the volume mounts override them with your host directories. + +## Health Check + +The nginx proxy responds with `200 OK` on `GET /` for load balancer health checks (AWS ALB, etc.). This is separate from the WebUI at `/webui/`. + +```bash +curl http://localhost/ +# OK +``` + ## Direct Access (no proxy) If you want to bypass nginx and access Gradio directly: @@ -40,23 +109,12 @@ docker build -f docker/Dockerfile -t voxcpm-training . # Run with GPU access (no reverse proxy) docker run --gpus all -p 7860:7860 \ -v ./models:/app/models \ + -v ./data:/app/data \ -v ./lora:/app/lora \ -v ./output:/app/output \ voxcpm-training ``` -## Model Weights - -Models are **auto-downloaded** from HuggingFace Hub on first use. The `/app/models` volume persists them across container restarts so they don't need to be re-downloaded. - -To pre-populate (avoids download at startup): - -``` -models/ -├── openbmb__VoxCPM2/ # VoxCPM2 (preferred) -└── openbmb__VoxCPM1.5/ # VoxCPM1.5 (fallback) -``` - ## Environment Variables | Variable | Default | Description | @@ -88,17 +146,10 @@ Training subprocess output is streamed to stdout, visible via: docker compose -f docker/docker-compose.yml logs -f training-webui ``` -## Volumes - -| Mount Point | Purpose | -|-------------|---------| -| `/app/models` | Pre-trained model weights (read-only OK) | -| `/app/lora` | LoRA checkpoints — training output is saved here | -| `/app/output` | Additional training artifacts | - ## Troubleshooting - **"no NVIDIA GPU detected"**: Ensure the NVIDIA Container Toolkit is installed and `docker run --gpus all nvidia-smi` works. - **OOM errors**: Reduce batch size in the WebUI or use a GPU with more VRAM. - **WebUI not accessible**: Check that port 80 (nginx) or 7860 (direct) isn't blocked by a firewall. - **WebSocket errors behind proxy**: Ensure your proxy forwards `Upgrade` and `Connection` headers (the included nginx.conf handles this). +- **Health check failing**: Ensure nginx is running — `curl http://localhost/` should return `OK`. diff --git a/docker/docker-compose.yml b/docker/docker-compose.yml index 8dbf9b8..79621ab 100644 --- a/docker/docker-compose.yml +++ b/docker/docker-compose.yml @@ -8,9 +8,20 @@ services: ports: - "7860:7860" volumes: - - ../models:/app/models # Pre-downloaded model weights - - ../lora:/app/lora # LoRA checkpoints (input/output) - - ../output:/app/output # Training output artifacts + # Pretrained model weights + HF cache (HF_HOME=/app/models in Dockerfile). + # Pre-populate with model dirs, or leave empty — auto-downloads on first run. + - ../models:/app/models + + # Training data: JSONL manifests and audio files. + # Reference paths inside the container as /app/data/train.jsonl etc. + - ../data:/app/data + + # LoRA training output — checkpoints, configs, logs. + # Results appear in lora//checkpoints/ after training. + - ../lora:/app/lora + + # Additional training artifacts. + - ../output:/app/output deploy: resources: reservations: diff --git a/docker/nginx.conf b/docker/nginx.conf index 0ac18c6..5e5ea01 100644 --- a/docker/nginx.conf +++ b/docker/nginx.conf @@ -2,6 +2,12 @@ server { listen 80; server_name _; + # Health check for load balancers (AWS ALB, etc.) + location = / { + return 200 'OK\n'; + add_header Content-Type text/plain; + } + location /webui/ { proxy_pass http://training-webui:7860/; proxy_set_header Host $host;