Skip to content

Commit c955c7a

Browse files
committed
ci: add Windows CUDA build
1 parent d607fba commit c955c7a

2 files changed

Lines changed: 71 additions & 15 deletions

File tree

‎.github/workflows/ci.yml‎

Lines changed: 30 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -8,6 +8,7 @@ name: ci
88
# macos-arm64-metal — Apple Silicon, Metal backend
99
# macos-x64-metal — Intel Mac, Metal backend (cross-compiled on ARM runner)
1010
# windows-x64-vulkan — Windows, Vulkan backend
11+
# windows-x64-cuda — Windows Server 2022, CUDA backend (CUDA 12.6)
1112
#
1213
# On tag push, artifacts are bundled into a GitHub Release.
1314

@@ -97,34 +98,47 @@ jobs:
9798
name: linux-x64-vulkan
9899
backend: vulkan
99100
cmake_extra: "-DGAME_GGML_VULKAN=ON"
101+
build_jobs: 4
100102
pkg_ext: ""
101103
lib_glob: "libggml*.so*"
102104

103105
- os: ubuntu-22.04
104106
name: linux-x64-cuda
105107
backend: cuda
106-
cmake_extra: '-DGAME_GGML_CUDA=ON -DGGML_NATIVE=OFF -DCMAKE_CUDA_ARCHITECTURES=75\;80\;86\;89'
108+
cmake_extra: '-DGAME_GGML_CUDA=ON -DGGML_NATIVE=OFF -DCMAKE_CUDA_ARCHITECTURES=75-real\;80-real\;86-real\;89-real'
109+
build_jobs: 2
107110
pkg_ext: ""
108111
lib_glob: "libggml*.so*"
109112

110113
- os: macos-14
111114
name: macos-arm64-metal
112115
backend: metal
113116
cmake_extra: "-DGAME_GGML_METAL=ON -DCMAKE_OSX_ARCHITECTURES=arm64"
117+
build_jobs: 4
114118
pkg_ext: ""
115119
lib_glob: "libggml*.dylib"
116120

117121
- os: macos-14
118122
name: macos-x64-metal
119123
backend: metal
120124
cmake_extra: "-DGAME_GGML_METAL=ON -DCMAKE_OSX_ARCHITECTURES=x86_64"
125+
build_jobs: 4
121126
pkg_ext: ""
122127
lib_glob: "libggml*.dylib"
123128

124129
- os: windows-latest
125130
name: windows-x64-vulkan
126131
backend: vulkan
127132
cmake_extra: "-DGAME_GGML_VULKAN=ON"
133+
build_jobs: 4
134+
pkg_ext: ".exe"
135+
lib_glob: "ggml*.dll"
136+
137+
- os: windows-2022
138+
name: windows-x64-cuda
139+
backend: cuda
140+
cmake_extra: '-DGAME_GGML_CUDA=ON -DGGML_NATIVE=OFF -DCMAKE_CUDA_ARCHITECTURES=75-real\;80-real\;86-real\;89-real'
141+
build_jobs: 2
128142
pkg_ext: ".exe"
129143
lib_glob: "ggml*.dll"
130144

@@ -147,12 +161,21 @@ jobs:
147161
if: matrix.backend == 'vulkan'
148162
run: glslc --version
149163

150-
- name: Install CUDA Toolkit
151-
if: matrix.backend == 'cuda'
152-
uses: Jimver/cuda-toolkit@1a3c14e26833ccf292b268f9a790fb47dea7b2da
164+
- name: Install CUDA Toolkit (Linux)
165+
if: matrix.backend == 'cuda' && runner.os == 'Linux'
166+
uses: Jimver/cuda-toolkit@1a3c14e26833ccf292b268f9a790fb47dea7b2da # v0.2.28
167+
with:
168+
cuda: "12.6.3"
169+
method: network
170+
log-file-suffix: "${{ matrix.name }}.txt"
171+
172+
- name: Install CUDA Toolkit (Windows)
173+
if: matrix.backend == 'cuda' && runner.os == 'Windows'
174+
uses: Jimver/cuda-toolkit@b8bf9c6c28f8a92fbb04dcfcaee872e60c57462d # v0.2.36
153175
with:
154176
cuda: "12.6.3"
155177
method: network
178+
sub-packages: '["nvcc", "cudart", "cublas", "cublas_dev", "visual_studio_integration"]'
156179
log-file-suffix: "${{ matrix.name }}.txt"
157180

158181
- name: Verify CUDA compiler
@@ -181,7 +204,9 @@ jobs:
181204
${{ matrix.cmake_extra }}
182205
183206
- name: Build
184-
run: cmake --build build -j --config Release
207+
# ggml-cuda expands many template instances. Unbounded parallel builds can
208+
# exhaust the memory of GitHub-hosted runners and terminate nvcc.
209+
run: cmake --build build --parallel ${{ matrix.build_jobs }} --config Release
185210

186211
- name: Verify binary
187212
run: |

‎BUILDING.md‎

Lines changed: 41 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -20,7 +20,8 @@ sudo apt install libvulkan-dev vulkan-tools
2020
# sudo apt install glslc-tools # not available on all distros
2121

2222
# Optional: CUDA backend (NVIDIA GPUs)
23-
# Install a CUDA Toolkit supported by your compiler and driver. CI uses CUDA 12.6.
23+
# Install a CUDA Toolkit supported by your compiler and driver. CI uses CUDA 12.6.3.
24+
# The prebuilt CUDA packages target Turing (CC 7.5) and newer GPUs.
2425
# https://developer.nvidia.com/cuda-downloads
2526

2627
# Optional: ccache for faster rebuilds
@@ -53,6 +54,10 @@ brew install cmake ccache
5354

5455
# Vulkan SDK (optional, for Vulkan backend)
5556
# https://vulkan.lunarg.com/sdk/home
57+
#
58+
# CUDA Toolkit 12.6.x (optional, for CUDA backend)
59+
# https://developer.nvidia.com/cuda-downloads
60+
# CUDA 12.6 supports Visual Studio 2022 / MSVC 193x.
5661
```
5762

5863
## Quick start
@@ -100,7 +105,7 @@ cmake -B build -DCMAKE_BUILD_TYPE=Release \
100105
-DGAME_GGML_BUILD_CLI=ON \
101106
-DGAME_GGML_CUDA=ON \
102107
-DGGML_NATIVE=OFF \
103-
-DCMAKE_CUDA_ARCHITECTURES="75;80;86;89"
108+
-DCMAKE_CUDA_ARCHITECTURES="75-real;80-real;86-real;89-real"
104109

105110
# macOS Apple Silicon + Metal
106111
cmake -B build -DCMAKE_BUILD_TYPE=Release \
@@ -117,6 +122,13 @@ cmake -B build -DCMAKE_BUILD_TYPE=Release \
117122
cmake -B build -DCMAKE_BUILD_TYPE=Release `
118123
-DGAME_GGML_BUILD_CLI=ON `
119124
-DGAME_GGML_VULKAN=ON
125+
126+
# Windows + CUDA (from Visual Studio 2022 Developer PowerShell)
127+
cmake -B build -DCMAKE_BUILD_TYPE=Release `
128+
-DGAME_GGML_BUILD_CLI=ON `
129+
-DGAME_GGML_CUDA=ON `
130+
-DGGML_NATIVE=OFF `
131+
-DCMAKE_CUDA_ARCHITECTURES="75-real;80-real;86-real;89-real"
120132
```
121133

122134
## Converting a PyTorch checkpoint to GGUF
@@ -163,14 +175,33 @@ build/bin/game_ggml_cli serve game_medium.gguf
163175
# Then write binary request frames to stdin (see src/cli/main.cpp for protocol)
164176
```
165177

166-
## CI CUDA scope
167-
168-
The hosted CI builds and packages the Linux x64 CUDA backend with CUDA Toolkit
169-
12.6. It verifies Toolkit discovery, CUDA compilation, linking, and that the CLI
170-
starts with `--version`. GitHub-hosted runners do not provide an NVIDIA GPU, so
171-
actual CUDA inference must still be smoke-tested on an NVIDIA system. The
172-
packaged CUDA backend uses the CUDA runtime and cuBLAS libraries supplied by the
173-
installed NVIDIA CUDA runtime/toolkit.
178+
## CUDA compatibility and CI scope
179+
180+
The hosted CI builds Linux x64 and Windows x64 CUDA packages with CUDA Toolkit
181+
12.6.3 and Visual Studio 2022 on Windows. It verifies Toolkit discovery, CUDA
182+
compilation, linking, and that the CLI starts with `--version`. GitHub-hosted
183+
runners do not provide an NVIDIA GPU, so actual CUDA inference must still be
184+
smoke-tested on an NVIDIA system.
185+
186+
The release architecture list is `75-real;80-real;86-real;89-real`, covering:
187+
188+
- CC 7.5: Turing (for example, GeForce RTX 20 series)
189+
- CC 8.0/8.6: Ampere (A100 and GeForce RTX 30 series)
190+
- CC 8.9: Ada (GeForce RTX 40 series)
191+
192+
`-real` emits native SASS for each target and avoids requiring the display driver
193+
to JIT PTX generated by CUDA 12.6. Pascal and Volta are not included in the
194+
prebuilt package. Source builds that need these older GPUs can use CUDA 12.x and
195+
add `61-real` and/or `70-real`. CUDA 13.0 removed NVCC offline compilation for
196+
architectures older than CC 7.5; use CUDA 12.9 or earlier when maintaining such
197+
builds.
198+
199+
CUDA 12.x minor-version compatibility requires at least NVIDIA driver
200+
525.60.13 on Linux or 528.33 on Windows, subject to the limitations documented
201+
in NVIDIA's CUDA Compatibility Guide. Using the current production driver is
202+
recommended. The packages currently expect the CUDA 12 runtime and cuBLAS
203+
libraries to be installed on the target system; they do not bundle NVIDIA's
204+
runtime libraries.
174205

175206
## Troubleshooting
176207

0 commit comments

Comments
 (0)