-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdocker-compose.yaml
More file actions
269 lines (255 loc) · 6.75 KB
/
Copy pathdocker-compose.yaml
File metadata and controls
269 lines (255 loc) · 6.75 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
# =============================================================================
# PRODUCTION DOCKER COMPOSE
# =============================================================================
# Usage: docker compose up -d
# Features:
# - Caddy reverse proxy with automatic HTTPS
# - Internal networking (no direct port exposure except Caddy)
# - Production-ready configuration
# =============================================================================
services:
redis:
image: redis:7-alpine
container_name: audio_redis
env_file: .env
expose:
- "6379"
volumes:
- redis_data:/data
command: redis-server --appendonly yes
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 10s
timeout: 5s
retries: 3
restart: unless-stopped
postgres:
image: postgres:15-alpine
container_name: audio_postgres
env_file: .env
expose:
- "5432"
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER:-user} -d ${POSTGRES_DB:-audio_db}"]
interval: 10s
timeout: 5s
retries: 3
restart: unless-stopped
minio:
image: minio/minio:latest
container_name: audio_minio
env_file: .env
expose:
- "9000"
- "9001"
ports:
- "127.0.0.1:9000:9000" # S3 API (localhost only)
- "127.0.0.1:9001:9001" # Web Console (localhost only)
volumes:
- minio_data:/data
command: server /data --console-address ":9001"
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:9000/minio/health/live"]
interval: 30s
timeout: 20s
retries: 3
restart: unless-stopped
app:
build:
context: .
dockerfile: Dockerfile
container_name: audio_app
env_file: .env
expose:
- "8000"
volumes:
- ./uploads:/app/uploads
- ./logs:/app/logs
runtime: nvidia
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
depends_on:
redis:
condition: service_healthy
postgres:
condition: service_healthy
minio:
condition: service_healthy
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 40s
restart: unless-stopped
# Model initialization service - runs once to download models
model_init:
build:
context: .
dockerfile: Dockerfile
command: python scripts/init_models.py
env_file: .env
volumes:
- models_cache:/models
runtime: nvidia
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: 1
capabilities: [gpu]
restart: "no" # Only runs once
rq_worker:
build:
context: .
dockerfile: Dockerfile
command: python -m src.rq_worker
env_file: .env
volumes:
- ./uploads:/app/uploads
- ./logs:/app/logs
- models_cache:/models:ro # Read-only mount for fast model access
runtime: nvidia
deploy:
replicas: ${MAX_WORKERS:-3}
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
depends_on:
redis:
condition: service_healthy
postgres:
condition: service_healthy
minio:
condition: service_healthy
app:
condition: service_healthy
model_init:
condition: service_completed_successfully
restart: unless-stopped
rq_dashboard:
image: eoranged/rq-dashboard:latest
container_name: audio_rq_dashboard
env_file: .env
expose:
- "9181"
depends_on:
redis:
condition: service_healthy
restart: unless-stopped
caddy:
image: caddy:2-alpine
container_name: audio_caddy
ports:
- "80:80"
- "443:443"
- "443:443/udp" # HTTP/3
volumes:
- ./Caddyfile:/etc/caddy/Caddyfile
- caddy_data:/data
- caddy_config:/config
depends_on:
app:
condition: service_healthy
restart: unless-stopped
prometheus:
image: prom/prometheus:latest
container_name: audio_prometheus
expose:
- "9090"
# Only expose on localhost for security
ports:
- "127.0.0.1:9090:9090"
volumes:
- ./prometheus.yml:/etc/prometheus/prometheus.yml
- prometheus_data:/prometheus
command:
- '--config.file=/etc/prometheus/prometheus.yml'
- '--storage.tsdb.path=/prometheus'
- '--web.console.libraries=/etc/prometheus/console_libraries'
- '--web.console.templates=/etc/prometheus/consoles'
depends_on:
- app
restart: unless-stopped
grafana:
image: grafana/grafana:latest
container_name: audio_grafana
env_file: .env
expose:
- "3000"
# Only expose on localhost for security
ports:
- "127.0.0.1:3000:3000"
volumes:
- grafana_data:/var/lib/grafana
- ./monitoring/grafana/dashboards:/etc/grafana/provisioning/dashboards
- ./monitoring/grafana/datasources:/etc/grafana/provisioning/datasources
depends_on:
prometheus:
condition: service_started
restart: unless-stopped
# NVIDIA GPU metrics exporter (official DCGM exporter)
nvidia-gpu-exporter:
image: nvcr.io/nvidia/k8s/dcgm-exporter:3.3.5-3.4.0-ubuntu22.04
container_name: audio_gpu_exporter
env_file: .env
expose:
- "9400" # DCGM exporter uses port 9400
runtime: nvidia
restart: unless-stopped
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
cap_add:
- SYS_ADMIN
# Container metrics (cAdvisor) - monitors all containers CPU/RAM/Network/Disk
cadvisor:
image: gcr.io/cadvisor/cadvisor:latest
container_name: audio_cadvisor
expose:
- "8080"
volumes:
- /:/rootfs:ro
- /var/run:/var/run:rw
- /sys:/sys:ro
- /var/lib/docker/:/var/lib/docker:ro
- /dev/disk/:/dev/disk:ro
privileged: true
restart: unless-stopped
# S3 Cleanup - runs daily to remove old files
s3_cleanup:
build:
context: .
dockerfile: Dockerfile
command: sh -c "while true; do python scripts/cleanup_s3_files.py && sleep 86400; done"
env_file: .env
depends_on:
minio:
condition: service_healthy
restart: unless-stopped
profiles:
- cleanup # Optional service - activate with: docker compose --profile cleanup up
volumes:
redis_data:
postgres_data:
minio_data:
caddy_data:
caddy_config:
prometheus_data:
grafana_data:
models_cache: # Shared volume for Whisper models - accessed by all workers