forked from ortus-boxlang/bx-ai
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathdocker-compose-ollama.yml
More file actions
98 lines (96 loc) · 3.25 KB
/
Copy pathdocker-compose-ollama.yml
File metadata and controls
98 lines (96 loc) · 3.25 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
################################################################
# Ollama LLM Server and Web UI - Production Configuration
################################################################
#
# PRODUCTION SETUP NOTES:
#
# 1. SECURITY - Update these settings before deploying:
# - Change WEBUI_ADMIN_USER and WEBUI_ADMIN_PASS to strong credentials
# - Consider using Docker secrets or environment files for sensitive data
# - Use a reverse proxy (nginx/traefik) with SSL/TLS in front
# - Restrict port access using firewall rules (only expose what's needed)
#
# 2. MODELS - Customize for your use case:
# - Update the preloaded model in command section (default: qwen3:0.6b)
# - Add multiple models: ollama pull model1 && ollama pull model2
# - For GPU support, add deploy.resources.reservations.devices config
#
# 3. RESOURCE LIMITS - Add resource constraints:
# - Set memory limits (e.g., mem_limit: 8g)
# - Set CPU limits (e.g., cpus: '4.0')
# - Adjust OLLAMA_NUM_PARALLEL and OLLAMA_MAX_LOADED_MODELS based on hardware
#
# 4. DATA PERSISTENCE:
# - All data stored in ./.ollama directory
# - Ensure ./.ollama is backed up regularly
# - Add .ollama/ to .gitignore to avoid committing model data
#
# 5. NETWORKING:
# - Consider using a custom network for service isolation
# - For production, bind to specific IPs instead of 0.0.0.0
# - Update port mappings if needed (default: 11434 for Ollama, 3000 for WebUI)
#
# 6. MONITORING:
# - Health checks are configured but consider adding logging drivers
# - Integrate with monitoring solutions (Prometheus, Grafana, etc.)
# - Set up alerting for service failures
#
# 7. UPDATES:
# - Pin specific image versions instead of :latest for stability
# - Example: ollama/ollama:0.1.26 instead of ollama/ollama:latest
# - Test updates in staging before deploying to production
#
################################################################
services:
################################################################
# Ollama LLM Server
###############################################################
ollama:
image: ollama/ollama:0.18.2
container_name: ollama
ports:
- "11434:11434"
volumes:
- ./.ollama/server:/root/.ollama
environment:
- OLLAMA_NUM_PARALLEL=4
- OLLAMA_MAX_LOADED_MODELS=3
- OLLAMA_HOST=0.0.0.0:11434
entrypoint: ["/bin/bash", "-c"]
command: |
"ollama serve &
sleep 10
# Preload models
ollama pull qwen3:0.6b
ollama pull gemma3
ollama pull nomic-embed-text
wait"
restart: always
healthcheck:
test: ["CMD", "ollama", "list"]
interval: 30s
timeout: 10s
retries: 3
start_period: 40s
webui:
image: ghcr.io/open-webui/open-webui:main
container_name: ollama-webui
ports:
- "3000:8080"
volumes:
- ./.ollama/webui:/app/backend/data
depends_on:
ollama:
condition: service_healthy
environment:
- OLLAMA_BASE_URL=http://ollama:11434
- WEBUI_AUTH=True
- WEBUI_EMAIL=admin@boxlang.io
- WEBUI_PASSWORD=rocks
restart: always
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8080/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 30s