Three P0 fixes bundled — foundation cleanup before v0.3.0 phase work. Closes release.yml drift (PR #51's tabs broke v0.3.0 tag releases), production bind exposure (Critic F1), and 9-endpoint LAN gap on /system/* (Critic F2+F3). 5 new tests; 243 full pass.
96 lines
3.6 KiB
YAML
96 lines
3.6 KiB
YAML
# ──────────────────────────────────────────────────────────────
|
|
# OmniVoice Studio — Docker Compose
|
|
#
|
|
# Quick start:
|
|
# docker compose -f deploy/docker-compose.yml --profile cpu up # CPU mode
|
|
# docker compose -f deploy/docker-compose.yml --profile gpu up # GPU mode
|
|
#
|
|
# Both services bind to port 3900, so they MUST be opt-in via profiles —
|
|
# otherwise `compose up` would race them and one would fail to bind.
|
|
#
|
|
# First run downloads ~4 GB of models. Progress is shown in logs.
|
|
# Open http://localhost:3900 once the health check passes.
|
|
#
|
|
# SECURITY: The port is bound to 127.0.0.1 by default — only this
|
|
# machine can reach the API. To expose OmniVoice on your LAN (or
|
|
# through a reverse proxy / tunnel), change the port mapping to
|
|
# "0.0.0.0:3900:3900" or "3900:3900". OmniVoice itself ships no
|
|
# authentication — if you expose it, put it behind a reverse proxy
|
|
# with auth (Caddy basic_auth, nginx + htpasswd, Tailscale, etc.).
|
|
# ──────────────────────────────────────────────────────────────
|
|
|
|
services:
|
|
# ── CPU mode — activate with: docker compose --profile cpu up
|
|
omnivoice:
|
|
image: ghcr.io/debpalash/omnivoice-studio:latest
|
|
# To build from source instead of pulling, comment out `image:` and
|
|
# uncomment the two lines below:
|
|
build:
|
|
context: ..
|
|
dockerfile: deploy/Dockerfile
|
|
container_name: omnivoice-studio
|
|
profiles: ["cpu"]
|
|
ports:
|
|
- "127.0.0.1:3900:3900"
|
|
volumes:
|
|
- omnivoice-data:/app/omnivoice_data
|
|
environment:
|
|
- HF_HOME=/app/omnivoice_data/huggingface
|
|
- HF_TOKEN=${HF_TOKEN:-}
|
|
- OMNIVOICE_DATA_DIR=/app/omnivoice_data
|
|
- PYTHONPATH=/app/backend
|
|
- PYTHONUNBUFFERED=1
|
|
# Bind uvicorn to 0.0.0.0 *inside* the container so the host-side port
|
|
# mapping above can forward traffic in. The 127.0.0.1 prefix on the
|
|
# `ports:` mapping is what enforces loopback-only on the host —
|
|
# OMNIVOICE_BIND_HOST=0.0.0.0 here only opens the container's own
|
|
# interface. The backend default is 127.0.0.1 (see backend/main.py).
|
|
- OMNIVOICE_BIND_HOST=0.0.0.0
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-sf", "http://localhost:3900/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 3
|
|
start_period: 120s
|
|
restart: unless-stopped
|
|
|
|
# ── GPU mode — activate with: docker compose --profile gpu up
|
|
omnivoice-gpu:
|
|
image: ghcr.io/debpalash/omnivoice-studio:latest
|
|
build:
|
|
context: ..
|
|
dockerfile: deploy/Dockerfile
|
|
container_name: omnivoice-studio-gpu
|
|
profiles: ["gpu"]
|
|
ports:
|
|
- "127.0.0.1:3900:3900"
|
|
volumes:
|
|
- omnivoice-data:/app/omnivoice_data
|
|
environment:
|
|
- HF_HOME=/app/omnivoice_data/huggingface
|
|
- HF_TOKEN=${HF_TOKEN:-}
|
|
- OMNIVOICE_DATA_DIR=/app/omnivoice_data
|
|
- PYTHONPATH=/app/backend
|
|
- PYTHONUNBUFFERED=1
|
|
# Bind uvicorn to 0.0.0.0 *inside* the container — same as the CPU
|
|
# service above. The host-side `127.0.0.1:3900:3900` mapping keeps
|
|
# LAN reachability off by default.
|
|
- OMNIVOICE_BIND_HOST=0.0.0.0
|
|
healthcheck:
|
|
test: ["CMD", "curl", "-sf", "http://localhost:3900/health"]
|
|
interval: 30s
|
|
timeout: 10s
|
|
retries: 3
|
|
start_period: 180s
|
|
deploy:
|
|
resources:
|
|
reservations:
|
|
devices:
|
|
- driver: nvidia
|
|
count: 1
|
|
capabilities: [gpu]
|
|
restart: unless-stopped
|
|
|
|
volumes:
|
|
omnivoice-data:
|