-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathdocker-compose.yml
More file actions
143 lines (137 loc) · 7.02 KB
/
Copy pathdocker-compose.yml
File metadata and controls
143 lines (137 loc) · 7.02 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
# Codeborough - single-box stack for the DGX Spark (GB10, arm64).
#
# PRIVACY BOUNDARY (provable):
# core_net is `internal: true` -> vllm + gateway have NO route to the internet.
# edge_net is a normal bridge -> only egress-proxy + bridge touch it.
# The reasoning core (brain + civic-geo + memory) physically cannot phone home;
# only the voice bridge reaches ElevenLabs, through an allowlisted proxy.
#
# GPU: uses `gpus: all` (CDI) - NOT `runtime: nvidia` - requires Docker with NVIDIA GPU
# support; verify with `docker run --rm --gpus all nvidia/cuda:12.0-base nvidia-smi`.
#
# Brain weights are provisioned at SETUP time (make pull-model) into cb_models,
# so at demo time vllm runs on the internal network with no download / no egress.
name: codeborough
services:
# ---------- BRAIN: Nemotron NVFP4 via vLLM (GPU, internal-only) ----------
vllm:
image: vllm/vllm-openai:v0.22.1 # arm64; has the Nemotron-3 (norm_eps) model code
gpus: all
command:
- --model=${BRAIN_MODEL}
- --served-model-name=${BRAIN_SERVED_NAME}
- --port=8000
- --gpu-memory-utilization=${VLLM_GPU_FRAC}
- --max-model-len=${VLLM_MAX_MODEL_LEN}
- --trust-remote-code
# Tool calling + reasoning for Nemotron-3 Nano (so the model emits STRUCTURED tool calls
# instead of writing the tool-call as prose). The reasoning-parser plugin ships in the
# model snapshot. Per the NVIDIA/vLLM recipe for this model family.
- --enable-auto-tool-choice
- --tool-call-parser=qwen3_coder
- --reasoning-parser-plugin=/models/hub/models--nvidia--NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4/snapshots/ce1b118ae66ec705d02c241525192832eb045fd3/nano_v3_reasoning_parser.py
- --reasoning-parser=nano_v3
environment:
- HF_HOME=/models
- HF_HUB_OFFLINE=1 # read weights from volume at runtime, never fetch
volumes:
- cb_models:/models
networks: [core_net]
healthcheck:
test: ["CMD-SHELL", "python3 -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://localhost:8000/health').status==200 else 1)\""]
interval: 10s
timeout: 5s
retries: 60
start_period: 480s # 30B NVFP4 load + CUDA graph capture is slow; gate the gateway on it
restart: unless-stopped
# ---------- GATEWAY: OpenClaw + civic-geo plugin + memory (CPU, internal-only) ----------
gateway:
build:
context: .
dockerfile: deploy/Dockerfile.gateway
environment:
- OPENCLAW_GATEWAY_PORT=18789
- CIVIC_DATA_DIR=/data/datasets
# Brain = our NVFP4 Nemotron via vLLM. OpenClaw 2026.6.1 ships a bundled `vllm` plugin
# (enabled in deploy/openclaw.gateway.json -> plugins.entries.vllm). It reaches the vLLM
# endpoint over the OpenAI-compatible API; point it there. BRAIN_SERVED_NAME must match
# vLLM's --served-model-name so the agent default resolves to vllm/<this>.
- OPENAI_BASE_URL=http://vllm:8000/v1
- OPENAI_API_KEY=local
- VLLM_BASE_URL=http://vllm:8000/v1
- VLLM_API_KEY=local # the vllm provider needs a (placeholder) key; vLLM ignores the value
- BRAIN_SERVED_NAME=${BRAIN_SERVED_NAME}
# openclaw 2026.6.1 REFUSES a non-loopback bind without auth; supply a token (safe on the
# isolated internal core_net). The bridge must present the SAME token.
- OPENCLAW_GATEWAY_TOKEN=${OPENCLAW_GATEWAY_TOKEN:-codeborough-local}
volumes:
- ./datasets:/data/datasets:ro # London GeoJSON, read-only
- cb_memory:/root/.openclaw/agents # sessions/*.jsonl + MEMORY.md (long-session memory)
networks: [core_net] # NO edge_net -> no internet route. This is the privacy claim.
depends_on:
vllm: { condition: service_healthy }
# The gateway is a WebSocket server, so an HTTP GET is the wrong probe (and would 401 under
# token auth). Just check the port accepts TCP connections (Node is already in the image).
healthcheck:
test: ["CMD", "node", "-e", "require('net').connect(18789,require('os').hostname()).on('connect',()=>process.exit(0)).on('error',()=>process.exit(1))"]
interval: 10s
timeout: 5s
retries: 30
start_period: 40s
restart: unless-stopped
# ---------- EGRESS PROXY: the single controlled crossing point ----------
egress-proxy:
build:
context: ./deploy/egress-proxy
networks: [core_net, edge_net] # the ONLY dual-homed infra service
restart: unless-stopped
# ---------- BRIDGE: bridge.mjs - browser-facing voice + map (CPU, edge) ----------
bridge:
build:
context: .
dockerfile: deploy/Dockerfile.bridge
environment:
- BRIDGE_PORT=8091
- CIVIC_DATA_DIR=/data/datasets
- ELEVENLABS_API_KEY=${ELEVENLABS_API_KEY}
- ELEVENLABS_VOICE_ID=${ELEVENLABS_VOICE_ID}
- ELEVENLABS_MODEL=${ELEVENLABS_MODEL}
# bridge /ask spawns `openclaw agent` and runs it EMBEDDED (the gateway daemon's WS is
# unreliable across the version split), so it needs the brain provider env directly. It
# reaches vLLM on core_net; NO_PROXY (below) keeps that call off the egress proxy.
- OPENAI_BASE_URL=http://vllm:8000/v1
- OPENAI_API_KEY=local
- VLLM_BASE_URL=http://vllm:8000/v1
- VLLM_API_KEY=local
- BRAIN_SERVED_NAME=${BRAIN_SERVED_NAME}
# clamp bridge egress to ElevenLabs only (Tier-2 enforcement):
- HTTPS_PROXY=http://egress-proxy:8888
- HTTP_PROXY=http://egress-proxy:8888
- NO_PROXY=gateway,vllm,localhost,127.0.0.1
volumes:
- ./datasets:/data/datasets:ro
- cb_memory:/root/.openclaw/agents # the embedded agent's sessions + MEMORY.md (long-session persistence)
ports:
- "${UI_PORT}:8091" # browser reaches the UI here (publishing != container egress)
networks: [core_net, edge_net] # core_net to reach gateway; edge_net for ElevenLabs
depends_on:
gateway: { condition: service_started } # bridge serves the page + map even before the gateway's first reply
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://localhost:8091/health >/dev/null || exit 1"]
interval: 10s
timeout: 5s
retries: 20
start_period: 15s
restart: unless-stopped
networks:
core_net:
internal: true # <-- enforcement: Docker attaches no gateway/NAT; no internet route exists
edge_net:
driver: bridge # egress allowed; only egress-proxy + bridge are attached
volumes:
cb_models: # vLLM NVFP4 weights (HF cache) - provisioned by `make pull-model`
name: cb_models # pin the EXACT name (no `codeborough_` project prefix). `make pull-model`
# stages ~18GB into a bare `cb_models` volume; without this pin, compose
# mounted an empty `codeborough_cb_models` instead -> vLLM LocalEntryNotFound.
cb_memory: # OpenClaw sessions + MEMORY.md
name: cb_memory # pin for consistency so make targets and compose share one volume