diff --git a/README.md b/README.md index d63a6a1..ec6e695 100644 --- a/README.md +++ b/README.md @@ -1,11 +1,44 @@ **University of Pennsylvania, CIS 565: GPU Programming and Architecture, Project 1 - Flocking** -* (TODO) YOUR NAME HERE - * (TODO) [LinkedIn](), [personal website](), [twitter](), etc. -* Tested on: (TODO) Windows 22, i7-2222 @ 2.22GHz 22GB, GTX 222 222MB (Moore 2222 Lab) -### (TODO: Your README) +* Tom Donnelly + * [LinkedIn](https://www.linkedin.com/in/the-tom-donnelly/) +* Tested on: Windows 11, AMD Ryzen 9 5900X, NVIDIA GeForce RTX 3070 (Personal Desktop) + +--- + +![](images/boid_flock.gif) +### Analysis + Questions +*** +#### For each implementation, how does changing the number of boids affect performance? Why do you think this is? +![](images/graph_num_boid_vis.png) +![](images/graph_num_boid_novis.png) +*Taken at 128 Block Size* +As the number of boids increase the performance for all implementations decreases. This is expected as the number of threads available on the GPU to run at the same time is also limited. +As the number of boids increases more time will be needed to calculate the affect on other boids and performance will decrease. Interesting to note is that at a low number of boids the Naive implementation as +better performance. It's likely that checking every boid at a low boid count is faster than computing grid indices and sorting boids. + +#### For each implementation, how does changing the block count and block size affect performance? Why do you think this is? +![](images/graph_block_size.png) +*Taken with no visualization and 50,000 boids* +Block size did not have a large affect on performance, this is likely due to the number of blocks/threads used being a function of the number of boids. (numObjects + blockSize - 1) / blockSize. +This means the number of threads running is at maximum of the capability of the GPU or the number of boids being simulated at once. Changing the number of blocks to a smaller amount would break the simulation +for boids > blocks * blocksize. The lowered performance at a higher block size count may be due to inactive threads in a block. If the number of boids does not fit nicely into a block, the last block used will have a larger number +of inactive threads. +#### For the coherent uniform grid: did you experience any performance improvements with the more coherent uniform grid? Was this the outcome you expected? Why or why not? +There were small performance improvements at a high amount of boids (~100,000). I expected the improvement to be more noticeable and at all counts of boids so this result was unexpected. It may be the memory performance +improvements for a coherent grid are not noticeable until being done at a large scale. +#### Did changing cell width and checking 27 vs 8 neighboring cells affect performance? Why or why not? Be careful: it is insufficient (and possibly incorrect) to say that 27-cell is slower simply because there are more cells to check! +![](images/graph_cell_search.png) +*Taken with no visualization and 50,000 boids* +Changed the cell width and checking 27 neighboring cells decreased performance for both uniform and uniform coherent implementations. Searching the cells may have impacted performance however as the grid cells were only the +maximum size of the impacted rules, there were less boids per cell to search and likely less boids checked against flocking distance rules. What likely hurt performance more was the indexing of boids in the larger grid. +As the cell width went down the grid resolution had to increase along with the number of cells. This may have made locating start and end indices take longer. +### Blooper +![](images/blooper.gif) + + + + -Include screenshots, analysis, etc. (Remember, this is public, so don't put -anything here that you don't want to share with the world.) diff --git a/images/blooper.gif b/images/blooper.gif new file mode 100644 index 0000000..58f2715 Binary files /dev/null and b/images/blooper.gif differ diff --git a/images/boid_flock.gif b/images/boid_flock.gif new file mode 100644 index 0000000..7bf16a6 Binary files /dev/null and b/images/boid_flock.gif differ diff --git a/images/graph_block_size.png b/images/graph_block_size.png new file mode 100644 index 0000000..a6a7988 Binary files /dev/null and b/images/graph_block_size.png differ diff --git a/images/graph_cell_search.png b/images/graph_cell_search.png new file mode 100644 index 0000000..696c76e Binary files /dev/null and b/images/graph_cell_search.png differ diff --git a/images/graph_num_boid_novis.png b/images/graph_num_boid_novis.png new file mode 100644 index 0000000..b95aa38 Binary files /dev/null and b/images/graph_num_boid_novis.png differ diff --git a/images/graph_num_boid_vis.png b/images/graph_num_boid_vis.png new file mode 100644 index 0000000..5d45f34 Binary files /dev/null and b/images/graph_num_boid_vis.png differ diff --git a/src/kernel.cu b/src/kernel.cu index 74dffcb..d4ea4b2 100644 --- a/src/kernel.cu +++ b/src/kernel.cu @@ -16,19 +16,21 @@ #endif #define checkCUDAErrorWithLine(msg) checkCUDAError(msg, __LINE__) +#define SEARCH_27 1 +#define SEARCH_8 0 /** * Check for CUDA errors; print and exit if there was a problem. */ -void checkCUDAError(const char *msg, int line = -1) { - cudaError_t err = cudaGetLastError(); - if (cudaSuccess != err) { - if (line >= 0) { - fprintf(stderr, "Line %d: ", line); - } - fprintf(stderr, "Cuda error: %s: %s.\n", msg, cudaGetErrorString(err)); - exit(EXIT_FAILURE); - } +void checkCUDAError(const char* msg, int line = -1) { + cudaError_t err = cudaGetLastError(); + if (cudaSuccess != err) { + if (line >= 0) { + fprintf(stderr, "Line %d: ", line); + } + fprintf(stderr, "Cuda error: %s: %s.\n", msg, cudaGetErrorString(err)); + exit(EXIT_FAILURE); + } } @@ -66,25 +68,27 @@ dim3 threadsPerBlock(blockSize); // Consider why you would need two velocity buffers in a simulation where each // boid cares about its neighbors' velocities. // These are called ping-pong buffers. -glm::vec3 *dev_pos; -glm::vec3 *dev_vel1; -glm::vec3 *dev_vel2; +glm::vec3* dev_pos; +glm::vec3* dev_vel1; +glm::vec3* dev_vel2; // LOOK-2.1 - these are NOT allocated for you. You'll have to set up the thrust // pointers on your own too. // For efficient sorting and the uniform grid. These should always be parallel. -int *dev_particleArrayIndices; // What index in dev_pos and dev_velX represents this particle? -int *dev_particleGridIndices; // What grid cell is this particle in? +int* dev_particleArrayIndices; // What index in dev_pos and dev_velX represents this particle? +int* dev_particleGridIndices; // What grid cell is this particle in? // needed for use with thrust thrust::device_ptr dev_thrust_particleArrayIndices; thrust::device_ptr dev_thrust_particleGridIndices; -int *dev_gridCellStartIndices; // What part of dev_particleArrayIndices belongs -int *dev_gridCellEndIndices; // to this cell? +int* dev_gridCellStartIndices; // What part of dev_particleArrayIndices belongs +int* dev_gridCellEndIndices; // to this cell? // TODO-2.3 - consider what additional buffers you might need to reshuffle // the position and velocity data to be coherent within cells. +glm::vec3* dev_shuffledPos; +glm::vec3* dev_shuffledVel; // LOOK-2.1 - Grid parameters based on simulation parameters. // These are automatically computed for you in Boids::initSimulation @@ -99,13 +103,13 @@ glm::vec3 gridMinimum; ******************/ __host__ __device__ unsigned int hash(unsigned int a) { - a = (a + 0x7ed55d16) + (a << 12); - a = (a ^ 0xc761c23c) ^ (a >> 19); - a = (a + 0x165667b1) + (a << 5); - a = (a + 0xd3a2646c) ^ (a << 9); - a = (a + 0xfd7046c5) + (a << 3); - a = (a ^ 0xb55a4f09) ^ (a >> 16); - return a; + a = (a + 0x7ed55d16) + (a << 12); + a = (a ^ 0xc761c23c) ^ (a >> 19); + a = (a + 0x165667b1) + (a << 5); + a = (a + 0xd3a2646c) ^ (a << 9); + a = (a + 0xfd7046c5) + (a << 3); + a = (a ^ 0xb55a4f09) ^ (a >> 16); + return a; } /** @@ -113,63 +117,85 @@ __host__ __device__ unsigned int hash(unsigned int a) { * Function for generating a random vec3. */ __host__ __device__ glm::vec3 generateRandomVec3(float time, int index) { - thrust::default_random_engine rng(hash((int)(index * time))); - thrust::uniform_real_distribution unitDistrib(-1, 1); + thrust::default_random_engine rng(hash((int)(index * time))); + thrust::uniform_real_distribution unitDistrib(-1, 1); - return glm::vec3((float)unitDistrib(rng), (float)unitDistrib(rng), (float)unitDistrib(rng)); + return glm::vec3((float)unitDistrib(rng), (float)unitDistrib(rng), (float)unitDistrib(rng)); } /** * LOOK-1.2 - This is a basic CUDA kernel. * CUDA kernel for generating boids with a specified mass randomly around the star. */ -__global__ void kernGenerateRandomPosArray(int time, int N, glm::vec3 * arr, float scale) { - int index = (blockIdx.x * blockDim.x) + threadIdx.x; - if (index < N) { - glm::vec3 rand = generateRandomVec3(time, index); - arr[index].x = scale * rand.x; - arr[index].y = scale * rand.y; - arr[index].z = scale * rand.z; - } +__global__ void kernGenerateRandomPosArray(int time, int N, glm::vec3* arr, float scale) { + int index = (blockIdx.x * blockDim.x) + threadIdx.x; + if (index < N) { + glm::vec3 rand = generateRandomVec3(time, index); + arr[index].x = scale * rand.x; + arr[index].y = scale * rand.y; + arr[index].z = scale * rand.z; + } } /** * Initialize memory, update some globals */ void Boids::initSimulation(int N) { - numObjects = N; - dim3 fullBlocksPerGrid((N + blockSize - 1) / blockSize); - - // LOOK-1.2 - This is basic CUDA memory management and error checking. - // Don't forget to cudaFree in Boids::endSimulation. - cudaMalloc((void**)&dev_pos, N * sizeof(glm::vec3)); - checkCUDAErrorWithLine("cudaMalloc dev_pos failed!"); - - cudaMalloc((void**)&dev_vel1, N * sizeof(glm::vec3)); - checkCUDAErrorWithLine("cudaMalloc dev_vel1 failed!"); - - cudaMalloc((void**)&dev_vel2, N * sizeof(glm::vec3)); - checkCUDAErrorWithLine("cudaMalloc dev_vel2 failed!"); - - // LOOK-1.2 - This is a typical CUDA kernel invocation. - kernGenerateRandomPosArray<<>>(1, numObjects, - dev_pos, scene_scale); - checkCUDAErrorWithLine("kernGenerateRandomPosArray failed!"); - - // LOOK-2.1 computing grid params - gridCellWidth = 2.0f * std::max(std::max(rule1Distance, rule2Distance), rule3Distance); - int halfSideCount = (int)(scene_scale / gridCellWidth) + 1; - gridSideCount = 2 * halfSideCount; - - gridCellCount = gridSideCount * gridSideCount * gridSideCount; - gridInverseCellWidth = 1.0f / gridCellWidth; - float halfGridWidth = gridCellWidth * halfSideCount; - gridMinimum.x -= halfGridWidth; - gridMinimum.y -= halfGridWidth; - gridMinimum.z -= halfGridWidth; - - // TODO-2.1 TODO-2.3 - Allocate additional buffers here. - cudaDeviceSynchronize(); + numObjects = N; + dim3 fullBlocksPerGrid((N + blockSize - 1) / blockSize); + + // LOOK-1.2 - This is basic CUDA memory management and error checking. + // Don't forget to cudaFree in Boids::endSimulation. + cudaMalloc((void**)&dev_pos, N * sizeof(glm::vec3)); + checkCUDAErrorWithLine("cudaMalloc dev_pos failed!"); + + cudaMalloc((void**)&dev_vel1, N * sizeof(glm::vec3)); + checkCUDAErrorWithLine("cudaMalloc dev_vel1 failed!"); + + cudaMalloc((void**)&dev_vel2, N * sizeof(glm::vec3)); + checkCUDAErrorWithLine("cudaMalloc dev_vel2 failed!"); + + // LOOK-1.2 - This is a typical CUDA kernel invocation. + kernGenerateRandomPosArray << > > (1, numObjects, + dev_pos, scene_scale); + checkCUDAErrorWithLine("kernGenerateRandomPosArray failed!"); + + // LOOK-2.1 computing grid params + gridCellWidth = 2.0f * std::max(std::max(rule1Distance, rule2Distance), rule3Distance); +#if SEARCH_27 + gridCellWidth = std::max(std::max(rule1Distance, rule2Distance), rule3Distance); +#endif + int halfSideCount = (int)(scene_scale / gridCellWidth) + 1; + gridSideCount = 2 * halfSideCount; + + gridCellCount = gridSideCount * gridSideCount * gridSideCount; + gridInverseCellWidth = 1.0f / gridCellWidth; + float halfGridWidth = gridCellWidth * halfSideCount; + gridMinimum.x -= halfGridWidth; + gridMinimum.y -= halfGridWidth; + gridMinimum.z -= halfGridWidth; + + // TODO-2.1 TODO-2.3 - Allocate additional buffers here. + cudaMalloc((void**)&dev_particleArrayIndices, N * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_particleArrayIndices failed!"); + cudaMalloc((void**)&dev_particleGridIndices, N * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_particleGridIndices failed!"); + cudaMalloc((void**)&dev_gridCellStartIndices, gridCellCount * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_gridCellStartIndices failed!"); + cudaMalloc((void**)&dev_gridCellEndIndices, gridCellCount * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_gridCellEndIndices failed!"); + + cudaMalloc((void**)&dev_shuffledPos, N * sizeof(glm::vec3)); + checkCUDAErrorWithLine("cudaMalloc dev_shuffledPos failed!"); + + cudaMalloc((void**)&dev_shuffledVel, N * sizeof(glm::vec3)); + checkCUDAErrorWithLine("cudaMalloc dev_shuffledVel failed!"); + dev_thrust_particleArrayIndices = thrust::device_pointer_cast(dev_particleArrayIndices); + dev_thrust_particleGridIndices = thrust::device_pointer_cast(dev_particleGridIndices); + + + cudaDeviceSynchronize(); + } @@ -180,42 +206,42 @@ void Boids::initSimulation(int N) { /** * Copy the boid positions into the VBO so that they can be drawn by OpenGL. */ -__global__ void kernCopyPositionsToVBO(int N, glm::vec3 *pos, float *vbo, float s_scale) { - int index = threadIdx.x + (blockIdx.x * blockDim.x); +__global__ void kernCopyPositionsToVBO(int N, glm::vec3* pos, float* vbo, float s_scale) { + int index = threadIdx.x + (blockIdx.x * blockDim.x); - float c_scale = -1.0f / s_scale; + float c_scale = -1.0f / s_scale; - if (index < N) { - vbo[4 * index + 0] = pos[index].x * c_scale; - vbo[4 * index + 1] = pos[index].y * c_scale; - vbo[4 * index + 2] = pos[index].z * c_scale; - vbo[4 * index + 3] = 1.0f; - } + if (index < N) { + vbo[4 * index + 0] = pos[index].x * c_scale; + vbo[4 * index + 1] = pos[index].y * c_scale; + vbo[4 * index + 2] = pos[index].z * c_scale; + vbo[4 * index + 3] = 1.0f; + } } -__global__ void kernCopyVelocitiesToVBO(int N, glm::vec3 *vel, float *vbo, float s_scale) { - int index = threadIdx.x + (blockIdx.x * blockDim.x); +__global__ void kernCopyVelocitiesToVBO(int N, glm::vec3* vel, float* vbo, float s_scale) { + int index = threadIdx.x + (blockIdx.x * blockDim.x); - if (index < N) { - vbo[4 * index + 0] = vel[index].x + 0.3f; - vbo[4 * index + 1] = vel[index].y + 0.3f; - vbo[4 * index + 2] = vel[index].z + 0.3f; - vbo[4 * index + 3] = 1.0f; - } + if (index < N) { + vbo[4 * index + 0] = vel[index].x + 0.3f; + vbo[4 * index + 1] = vel[index].y + 0.3f; + vbo[4 * index + 2] = vel[index].z + 0.3f; + vbo[4 * index + 3] = 1.0f; + } } /** * Wrapper for call to the kernCopyboidsToVBO CUDA kernel. */ -void Boids::copyBoidsToVBO(float *vbodptr_positions, float *vbodptr_velocities) { - dim3 fullBlocksPerGrid((numObjects + blockSize - 1) / blockSize); +void Boids::copyBoidsToVBO(float* vbodptr_positions, float* vbodptr_velocities) { + dim3 fullBlocksPerGrid((numObjects + blockSize - 1) / blockSize); - kernCopyPositionsToVBO << > >(numObjects, dev_pos, vbodptr_positions, scene_scale); - kernCopyVelocitiesToVBO << > >(numObjects, dev_vel1, vbodptr_velocities, scene_scale); + kernCopyPositionsToVBO << > > (numObjects, dev_pos, vbodptr_positions, scene_scale); + kernCopyVelocitiesToVBO << > > (numObjects, dev_vel1, vbodptr_velocities, scene_scale); - checkCUDAErrorWithLine("copyBoidsToVBO failed!"); + checkCUDAErrorWithLine("copyBoidsToVBO failed!"); - cudaDeviceSynchronize(); + cudaDeviceSynchronize(); } @@ -229,47 +255,95 @@ void Boids::copyBoidsToVBO(float *vbodptr_positions, float *vbodptr_velocities) * Compute the new velocity on the body with index `iSelf` due to the `N` boids * in the `pos` and `vel` arrays. */ -__device__ glm::vec3 computeVelocityChange(int N, int iSelf, const glm::vec3 *pos, const glm::vec3 *vel) { - // Rule 1: boids fly towards their local perceived center of mass, which excludes themselves - // Rule 2: boids try to stay a distance d away from each other - // Rule 3: boids try to match the speed of surrounding boids - return glm::vec3(0.0f, 0.0f, 0.0f); +__device__ glm::vec3 computeVelocityChange(int N, int iSelf, const glm::vec3* pos, const glm::vec3* vel) { + // Rule 1: boids fly towards their local perceived center of mass, which excludes themselves + + glm::vec3 v1; + glm::vec3 v2; + glm::vec3 v3; + glm::vec3 posSelf = pos[iSelf]; + glm::vec3 perceived_center = glm::vec3(0); + int neighbors = 0; + for (int b = 0; b < N; b++) { + if (b != iSelf && glm::distance(pos[b], posSelf) < rule1Distance) { + perceived_center += pos[b]; + neighbors++; + } + } + if (neighbors > 0) + { + perceived_center /= neighbors; + v1 = (perceived_center - posSelf) * rule1Scale; + } + // Rule 2: boids try to stay a distance d away from each other + + glm::vec3 c = glm::vec3(0); + + for (int b = 0; b < N; b++) { + if (b != iSelf && glm::distance(pos[b], posSelf) < rule2Distance) { + c -= (pos[b] - posSelf); + } + } + v2 = c * rule2Scale; + + // Rule 3: boids try to match the speed of surrounding boids + glm::vec3 perceived_velocity = glm::vec3(0); + neighbors = 0; + for (int b = 0; b < N; b++) { + if (b != iSelf && glm::distance(pos[b], posSelf) < rule3Distance) { + perceived_velocity += vel[b]; + neighbors++; + } + } + if (neighbors > 0) + { + perceived_velocity /= neighbors; + v3 = perceived_velocity * rule3Scale; + } + + return v1 + v2 + v3; } /** * TODO-1.2 implement basic flocking * For each of the `N` bodies, update its position based on its current velocity. */ -__global__ void kernUpdateVelocityBruteForce(int N, glm::vec3 *pos, - glm::vec3 *vel1, glm::vec3 *vel2) { - // Compute a new velocity based on pos and vel1 - // Clamp the speed - // Record the new velocity into vel2. Question: why NOT vel1? +__global__ void kernUpdateVelocityBruteForce(int N, glm::vec3* pos, + glm::vec3* vel1, glm::vec3* vel2) { + // Compute a new velocity based on pos and vel1 + int index = threadIdx.x + (blockIdx.x * blockDim.x); + if (index >= N) { + return; + } + glm::vec3 new_vel = computeVelocityChange(N, index, pos, vel1); + // Clamp the speed + // Record the new velocity into vel2. Question: why NOT vel1? + vel2[index] = glm::clamp((new_vel + vel1[index]), -maxSpeed, maxSpeed); } /** * LOOK-1.2 Since this is pretty trivial, we implemented it for you. * For each of the `N` bodies, update its position based on its current velocity. */ -__global__ void kernUpdatePos(int N, float dt, glm::vec3 *pos, glm::vec3 *vel) { - // Update position by velocity - int index = threadIdx.x + (blockIdx.x * blockDim.x); - if (index >= N) { - return; - } - glm::vec3 thisPos = pos[index]; - thisPos += vel[index] * dt; - - // Wrap the boids around so we don't lose them - thisPos.x = thisPos.x < -scene_scale ? scene_scale : thisPos.x; - thisPos.y = thisPos.y < -scene_scale ? scene_scale : thisPos.y; - thisPos.z = thisPos.z < -scene_scale ? scene_scale : thisPos.z; - - thisPos.x = thisPos.x > scene_scale ? -scene_scale : thisPos.x; - thisPos.y = thisPos.y > scene_scale ? -scene_scale : thisPos.y; - thisPos.z = thisPos.z > scene_scale ? -scene_scale : thisPos.z; - - pos[index] = thisPos; +__global__ void kernUpdatePos(int N, float dt, glm::vec3* pos, glm::vec3* vel) { + // Update position by velocity + int index = threadIdx.x + (blockIdx.x * blockDim.x); + if (index >= N) { + return; + } + glm::vec3 thisPos = pos[index]; + thisPos += vel[index] * dt; + + // Wrap the boids around so we don't lose them + thisPos.x = thisPos.x < -scene_scale ? scene_scale : thisPos.x; + thisPos.y = thisPos.y < -scene_scale ? scene_scale : thisPos.y; + thisPos.z = thisPos.z < -scene_scale ? scene_scale : thisPos.z; + + thisPos.x = thisPos.x > scene_scale ? -scene_scale : thisPos.x; + thisPos.y = thisPos.y > scene_scale ? -scene_scale : thisPos.y; + thisPos.z = thisPos.z > scene_scale ? -scene_scale : thisPos.z; + + pos[index] = thisPos; } // LOOK-2.1 Consider this method of computing a 1D index from a 3D grid index. @@ -279,179 +353,561 @@ __global__ void kernUpdatePos(int N, float dt, glm::vec3 *pos, glm::vec3 *vel) { // for(y) // for(z)? Or some other order? __device__ int gridIndex3Dto1D(int x, int y, int z, int gridResolution) { - return x + y * gridResolution + z * gridResolution * gridResolution; + return x + y * gridResolution + z * gridResolution * gridResolution; } __global__ void kernComputeIndices(int N, int gridResolution, - glm::vec3 gridMin, float inverseCellWidth, - glm::vec3 *pos, int *indices, int *gridIndices) { - // TODO-2.1 - // - Label each boid with the index of its grid cell. - // - Set up a parallel array of integer indices as pointers to the actual - // boid data in pos and vel1/vel2 + glm::vec3 gridMin, float inverseCellWidth, + glm::vec3* pos, int* indices, int* gridIndices) { + // TODO-2.1 + // - Label each boid with the index of its grid cell. + // - Set up a parallel array of integer indices as pointers to the actual + // boid data in pos and vel1/vel2 + int index = (blockIdx.x * blockDim.x) + threadIdx.x; + if (index >= N) + { + return; + } + auto thisPos = pos[index]; + int iX = glm::floor((thisPos.x - gridMin.x) * inverseCellWidth); + int iY = glm::floor((thisPos.y - gridMin.y) * inverseCellWidth); + int iZ = glm::floor((thisPos.z - gridMin.z) * inverseCellWidth); + int gridIndex = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + gridIndices[index] = gridIndex; + indices[index] = index; + } // LOOK-2.1 Consider how this could be useful for indicating that a cell // does not enclose any boids -__global__ void kernResetIntBuffer(int N, int *intBuffer, int value) { - int index = (blockIdx.x * blockDim.x) + threadIdx.x; - if (index < N) { - intBuffer[index] = value; - } +__global__ void kernResetIntBuffer(int N, int* intBuffer, int value) { + int index = (blockIdx.x * blockDim.x) + threadIdx.x; + if (index < N) { + intBuffer[index] = value; + } } -__global__ void kernIdentifyCellStartEnd(int N, int *particleGridIndices, - int *gridCellStartIndices, int *gridCellEndIndices) { - // TODO-2.1 - // Identify the start point of each cell in the gridIndices array. - // This is basically a parallel unrolling of a loop that goes - // "this index doesn't match the one before it, must be a new cell!" +__global__ void kernIdentifyCellStartEnd(int N, int* particleGridIndices, + int* gridCellStartIndices, int* gridCellEndIndices) { + // TODO-2.1 + // Identify the start point of each cell in the gridIndices array. + // This is basically a parallel unrolling of a loop that goes + // "this index doesn't match the one before it, must be a new cell!" + int index = (blockIdx.x * blockDim.x) + threadIdx.x; + if (index >= N) + { + return; + } + auto thisGrid = particleGridIndices[index]; + if (index == 0) + { + gridCellStartIndices[thisGrid] = index; + return; + } + if (index == (N - 1)) + { + gridCellEndIndices[thisGrid] = index; + return; + } + auto previousGrid = particleGridIndices[index - 1]; + auto nextGrid = particleGridIndices[index + 1]; + if (thisGrid != previousGrid) + { + gridCellStartIndices[thisGrid] = index; + + } + if (thisGrid != nextGrid) + { + gridCellEndIndices[thisGrid] = index; + } + + //Otherwise still in the same cell, do nothing } __global__ void kernUpdateVelNeighborSearchScattered( - int N, int gridResolution, glm::vec3 gridMin, - float inverseCellWidth, float cellWidth, - int *gridCellStartIndices, int *gridCellEndIndices, - int *particleArrayIndices, - glm::vec3 *pos, glm::vec3 *vel1, glm::vec3 *vel2) { - // TODO-2.1 - Update a boid's velocity using the uniform grid to reduce - // the number of boids that need to be checked. - // - Identify the grid cell that this particle is in - // - Identify which cells may contain neighbors. This isn't always 8. - // - For each cell, read the start/end indices in the boid pointer array. - // - Access each boid in the cell and compute velocity change from - // the boids rules, if this boid is within the neighborhood distance. - // - Clamp the speed change before putting the new speed in vel2 + int N, int gridResolution, glm::vec3 gridMin, + float inverseCellWidth, float cellWidth, + int* gridCellStartIndices, int* gridCellEndIndices, + int* particleArrayIndices, + glm::vec3* pos, glm::vec3* vel1, glm::vec3* vel2) { + // TODO-2.1 - Update a boid's velocity using the uniform grid to reduce + // the number of boids that need to be checked. + // - Identify the grid cell that this particle is in + int indexP = (blockIdx.x * blockDim.x) + threadIdx.x; + if (indexP >= N) + { + return; + } + int index = particleArrayIndices[indexP]; + auto thisPos = pos[index]; + int iX = glm::floor((thisPos.x - gridMin.x) * inverseCellWidth); + int iY = glm::floor((thisPos.y - gridMin.y) * inverseCellWidth); + int iZ = glm::floor((thisPos.z - gridMin.z) * inverseCellWidth); + int thisGrid = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + + // - Identify which cells may contain neighbors. This isn't always 8. +#if SEARCH_8 + //Convert to local cell cordinate and test if larger than 1/2 cell width + float halfWidth = cellWidth / 2; + float localX = thisPos.x - gridMin.x - (iX * cellWidth) - halfWidth; + float localY = thisPos.y - gridMin.y - (iY * cellWidth) - halfWidth; + float localZ = thisPos.z - gridMin.z - (iZ * cellWidth) - halfWidth; + float xValues[2]; + float yValues[2]; + float zValues[2]; + xValues[0] = (localX < 0 ? -cellWidth : 0); + xValues[1] = (localX >= 0 ? cellWidth : 0); + yValues[0] = (localY < 0 ? -cellWidth : 0); + yValues[1] = (localY >= 0 ? cellWidth : 0); + zValues[0] = (localZ < 0 ? -cellWidth : 0); + zValues[1] = (localZ >= 0 ? cellWidth : 0); + int adjacentCells[8]; + int cellIter = 0; + for (auto k : zValues) { + for (auto j : yValues) { + for (auto i : xValues) + { + iX = glm::floor((thisPos.x + i - gridMin.x) * inverseCellWidth); + iY = glm::floor((thisPos.y + j - gridMin.y) * inverseCellWidth); + iZ = glm::floor((thisPos.z + k - gridMin.z) * inverseCellWidth); + int neighborCell = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + if ((neighborCell >= 0) && (neighborCell < gridResolution * gridResolution * gridResolution)) + { + adjacentCells[cellIter] = neighborCell; + cellIter++; + } + } + } + } + while (cellIter < 8) + { + adjacentCells[cellIter] = -1; + cellIter++; + } +#elif SEARCH_27 + int xValues[3] = { -cellWidth, 0, cellWidth }; + int yValues[3] = { -cellWidth, 0, cellWidth }; + int zValues[3] = { -cellWidth, 0, cellWidth }; + int adjacentCells[27]; + int cellIter = 0; + for (auto k : zValues) { + for (auto j : yValues) { + for (auto i : xValues) + { + iX = glm::floor((thisPos.x + i - gridMin.x) * inverseCellWidth); + iY = glm::floor((thisPos.y + j - gridMin.y) * inverseCellWidth); + iZ = glm::floor((thisPos.z + k - gridMin.z) * inverseCellWidth); + int neighborCell = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + if ((neighborCell >= 0) && (neighborCell < (gridResolution * gridResolution * gridResolution))) + { + adjacentCells[cellIter] = neighborCell; + cellIter++; + } + } + } + } + while (cellIter < 27) + { + adjacentCells[cellIter] = -1; + cellIter++; + } +#endif + + + // - For each cell, read the start/end indices in the boid pointer array. + // - Access each boid in the cell and compute velocity change from + // the boids rules, if this boid is within the neighborhood distance. + int neighborsRule1 = 0; + int neighborsRule3 = 0; + glm::vec3 v1(0.0); + glm::vec3 v2(0.0); + glm::vec3 v3(0.0); + glm::vec3 perceived_center = glm::vec3(0); + glm::vec3 c = glm::vec3(0); + glm::vec3 perceived_velocity = glm::vec3(0); + for (auto cell : adjacentCells) + { + if (cell == -1) + { + continue; + } + int startIndex = gridCellStartIndices[cell]; + int endIndex = gridCellEndIndices[cell]; + if (startIndex == -1 || endIndex == -1) + { + continue; + } + for (int bp = startIndex; bp <= endIndex; bp++) + { + int b = particleArrayIndices[bp]; + if (b != index && glm::distance(pos[b], thisPos) < rule1Distance) { + perceived_center += pos[b]; + neighborsRule1++; + } + // Rule 2: boids try to stay a distance d away from each other + + if (b != index && glm::distance(pos[b], thisPos) < rule2Distance) { + c -= (pos[b] - thisPos); + } + + // Rule 3: boids try to match the speed of surrounding boids + if (b != index && glm::distance(pos[b], thisPos) < rule3Distance) { + perceived_velocity += vel1[b]; + neighborsRule3++; + } + + } + } + + if (neighborsRule1 > 0) + { + perceived_center /= neighborsRule1; + v1 = (perceived_center - thisPos) * rule1Scale; + } + + v2 = c * rule2Scale; + + if (neighborsRule3 > 0) + { + perceived_velocity /= neighborsRule3; + v3 = perceived_velocity * rule3Scale; + } + + // - Clamp the speed change before putting the new speed in vel2 + vel2[index] = glm::clamp((v1 + v2 + v3 + vel1[index]), -maxSpeed, maxSpeed); +} +__global__ void kernReshuffleVelAndPos( + int N, int* particleArrayIndices, glm::vec3* pos, glm::vec3* shuffledPos, glm::vec3* vel, glm::vec3* shuffledVel) +{ + int indexP = (blockIdx.x * blockDim.x) + threadIdx.x; + if (indexP >= N) + { + return; + } + int index = particleArrayIndices[indexP]; + shuffledPos[indexP] = pos[index]; + shuffledVel[indexP] = vel[index]; } + __global__ void kernUpdateVelNeighborSearchCoherent( - int N, int gridResolution, glm::vec3 gridMin, - float inverseCellWidth, float cellWidth, - int *gridCellStartIndices, int *gridCellEndIndices, - glm::vec3 *pos, glm::vec3 *vel1, glm::vec3 *vel2) { - // TODO-2.3 - This should be very similar to kernUpdateVelNeighborSearchScattered, - // except with one less level of indirection. - // This should expect gridCellStartIndices and gridCellEndIndices to refer - // directly to pos and vel1. - // - Identify the grid cell that this particle is in - // - Identify which cells may contain neighbors. This isn't always 8. - // - For each cell, read the start/end indices in the boid pointer array. - // DIFFERENCE: For best results, consider what order the cells should be - // checked in to maximize the memory benefits of reordering the boids data. - // - Access each boid in the cell and compute velocity change from - // the boids rules, if this boid is within the neighborhood distance. - // - Clamp the speed change before putting the new speed in vel2 + int N, int gridResolution, glm::vec3 gridMin, + float inverseCellWidth, float cellWidth, + int* gridCellStartIndices, int* gridCellEndIndices, + glm::vec3* pos, glm::vec3* vel1, glm::vec3* vel2) { + // TODO-2.3 - This should be very similar to kernUpdateVelNeighborSearchScattered, + // except with one less level of indirection. + // This should expect gridCellStartIndices and gridCellEndIndices to refer + // directly to pos and vel1. + // - Identify the grid cell that this particle is in + // - Identify which cells may contain neighbors. This isn't always 8. + // - For each cell, read the start/end indices in the boid pointer array. + // DIFFERENCE: For best results, consider what order the cells should be + // checked in to maximize the memory benefits of reordering the boids data. + // - Access each boid in the cell and compute velocity change from + // the boids rules, if this boid is within the neighborhood distance. + // - Clamp the speed change before putting the new speed in vel2 + int index = (blockIdx.x * blockDim.x) + threadIdx.x; + if (index >= N) + { + return; + } + auto thisPos = pos[index]; + int iX = glm::floor((thisPos.x - gridMin.x) * inverseCellWidth); + int iY = glm::floor((thisPos.y - gridMin.y) * inverseCellWidth); + int iZ = glm::floor((thisPos.z - gridMin.z) * inverseCellWidth); + int thisGrid = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + + // - Identify which cells may contain neighbors. This isn't always 8. +#if SEARCH_8 + //Convert to local cell cordinate and test if larger than 1/2 cell width + float halfWidth = cellWidth / 2; + float localX = thisPos.x - gridMin.x - (iX * cellWidth) - halfWidth; + float localY = thisPos.y - gridMin.y - (iY * cellWidth) - halfWidth; + float localZ = thisPos.z - gridMin.z - (iZ * cellWidth) - halfWidth; + float xValues[2]; + float yValues[2]; + float zValues[2]; + xValues[0] = (localX < 0 ? -cellWidth : 0); + xValues[1] = (localX >= 0 ? cellWidth : 0); + yValues[0] = (localY < 0 ? -cellWidth : 0); + yValues[1] = (localY >= 0 ? cellWidth : 0); + zValues[0] = (localZ < 0 ? -cellWidth : 0); + zValues[1] = (localZ >= 0 ? cellWidth : 0); + int adjacentCells[8]; + int cellIter = 0; + for (auto k : zValues) { + for (auto j : yValues) { + for (auto i : xValues) + { + iX = glm::floor((thisPos.x + i - gridMin.x) * inverseCellWidth); + iY = glm::floor((thisPos.y + j - gridMin.y) * inverseCellWidth); + iZ = glm::floor((thisPos.z + k - gridMin.z) * inverseCellWidth); + int neighborCell = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + if ((neighborCell >= 0) && (neighborCell < gridResolution * gridResolution * gridResolution)) + { + adjacentCells[cellIter] = neighborCell; + cellIter++; + } + } + } + } + while (cellIter < 8) + { + adjacentCells[cellIter] = -1; + cellIter++; + } +#elif SEARCH_27 + int xValues[3] = { -cellWidth, 0, cellWidth }; + int yValues[3] = { -cellWidth, 0, cellWidth }; + int zValues[3] = { -cellWidth, 0, cellWidth }; + int adjacentCells[27]; + int cellIter = 0; + for (auto k : zValues) { + for (auto j : yValues) { + for (auto i : xValues) + { + iX = glm::floor((thisPos.x + i - gridMin.x) * inverseCellWidth); + iY = glm::floor((thisPos.y + j - gridMin.y) * inverseCellWidth); + iZ = glm::floor((thisPos.z + k - gridMin.z) * inverseCellWidth); + int neighborCell = gridIndex3Dto1D(iX, iY, iZ, gridResolution); + if ((neighborCell >= 0) && (neighborCell < (gridResolution * gridResolution * gridResolution))) + { + adjacentCells[cellIter] = neighborCell; + cellIter++; + } + } + } + } + while (cellIter < 27) + { + adjacentCells[cellIter] = -1; + cellIter++; + } +#endif + + + // - For each cell, read the start/end indices in the boid pointer array. + // - Access each boid in the cell and compute velocity change from + // the boids rules, if this boid is within the neighborhood distance. + int neighborsRule1 = 0; + int neighborsRule3 = 0; + glm::vec3 v1(0.0); + glm::vec3 v2(0.0); + glm::vec3 v3(0.0); + glm::vec3 perceived_center = glm::vec3(0); + glm::vec3 c = glm::vec3(0); + glm::vec3 perceived_velocity = glm::vec3(0); + for (auto cell : adjacentCells) + { + if (cell == -1) + { + continue; + } + int startIndex = gridCellStartIndices[cell]; + int endIndex = gridCellEndIndices[cell]; + if (startIndex == -1 || endIndex == -1) + { + continue; + } + for (int b = startIndex; b <= endIndex; b++) + { + if (b != index && glm::distance(pos[b], thisPos) < rule1Distance) { + perceived_center += pos[b]; + neighborsRule1++; + } + // Rule 2: boids try to stay a distance d away from each other + + if (b != index && glm::distance(pos[b], thisPos) < rule2Distance) { + c -= (pos[b] - thisPos); + } + + // Rule 3: boids try to match the speed of surrounding boids + if (b != index && glm::distance(pos[b], thisPos) < rule3Distance) { + perceived_velocity += vel1[b]; + neighborsRule3++; + } + + } + } + + if (neighborsRule1 > 0) + { + perceived_center /= neighborsRule1; + v1 = (perceived_center - thisPos) * rule1Scale; + } + + v2 = c * rule2Scale; + + if (neighborsRule3 > 0) + { + perceived_velocity /= neighborsRule3; + v3 = perceived_velocity * rule3Scale; + } + + // - Clamp the speed change before putting the new speed in vel2 + vel2[index] = glm::clamp((v1 + v2 + v3 + vel1[index]), -maxSpeed, maxSpeed); } /** * Step the entire N-body simulation by `dt` seconds. */ void Boids::stepSimulationNaive(float dt) { - // TODO-1.2 - use the kernels you wrote to step the simulation forward in time. - // TODO-1.2 ping-pong the velocity buffers + // TODO-1.2 - use the kernels you wrote to step the simulation forward in time. + dim3 fullBlocksPerGrid((numObjects + blockSize - 1) / blockSize); + kernUpdateVelocityBruteForce << > > (numObjects, dev_pos, dev_vel1, dev_vel2); + kernUpdatePos << < fullBlocksPerGrid, blockSize >> > (numObjects, dt, dev_pos, dev_vel2); + // TODO-1.2 ping-pong the velocity buffers + std::swap(dev_vel1, dev_vel2); } void Boids::stepSimulationScatteredGrid(float dt) { - // TODO-2.1 - // Uniform Grid Neighbor search using Thrust sort. - // In Parallel: - // - label each particle with its array index as well as its grid index. - // Use 2x width grids. - // - Unstable key sort using Thrust. A stable sort isn't necessary, but you - // are welcome to do a performance comparison. - // - Naively unroll the loop for finding the start and end indices of each - // cell's data pointers in the array of boid indices - // - Perform velocity updates using neighbor search - // - Update positions - // - Ping-pong buffers as needed + // TODO-2.1 + dim3 fullBlocksPerGrid((numObjects + blockSize - 1) / blockSize); + dim3 fullGrid((gridCellCount + blockSize - 1) / blockSize); + // Uniform Grid Neighbor search using Thrust sort. + // In Parallel: + // - label each particle with its array index as well as its grid index. + // Use 2x width grids. + auto gridResolution = gridSideCount; + kernComputeIndices << < fullBlocksPerGrid, blockSize >> > (numObjects, gridResolution, gridMinimum, gridInverseCellWidth, dev_pos, dev_particleArrayIndices, dev_particleGridIndices); + // - Unstable key sort using Thrust. A stable sort isn't necessary, but you + // are welcome to do a performance comparison. + thrust::sort_by_key(dev_thrust_particleGridIndices, dev_thrust_particleGridIndices + numObjects, dev_thrust_particleArrayIndices); + // // - Naively unroll the loop for finding the start and end indices of each + // cell's data pointers in the array of boid indices + kernResetIntBuffer << < fullGrid, blockSize >> > (gridCellCount, dev_gridCellStartIndices, -1); + kernResetIntBuffer << < fullGrid, blockSize >> > (gridCellCount, dev_gridCellEndIndices, -1); + kernIdentifyCellStartEnd << > > (numObjects, dev_particleGridIndices, dev_gridCellStartIndices, dev_gridCellEndIndices); + // - Perform velocity updates using neighbor search + kernUpdateVelNeighborSearchScattered << < fullBlocksPerGrid, blockSize >> > (numObjects, gridResolution, gridMinimum, gridInverseCellWidth, gridCellWidth, + dev_gridCellStartIndices, dev_gridCellEndIndices, dev_particleArrayIndices, dev_pos, dev_vel1, dev_vel2); + // - Update positions + kernUpdatePos << < fullBlocksPerGrid, blockSize >> > (numObjects, dt, dev_pos, dev_vel2); + // - Ping-pong buffers as needed + std::swap(dev_vel1, dev_vel2); } void Boids::stepSimulationCoherentGrid(float dt) { - // TODO-2.3 - start by copying Boids::stepSimulationNaiveGrid - // Uniform Grid Neighbor search using Thrust sort on cell-coherent data. - // In Parallel: - // - Label each particle with its array index as well as its grid index. - // Use 2x width grids - // - Unstable key sort using Thrust. A stable sort isn't necessary, but you - // are welcome to do a performance comparison. - // - Naively unroll the loop for finding the start and end indices of each - // cell's data pointers in the array of boid indices - // - BIG DIFFERENCE: use the rearranged array index buffer to reshuffle all - // the particle data in the simulation array. - // CONSIDER WHAT ADDITIONAL BUFFERS YOU NEED - // - Perform velocity updates using neighbor search - // - Update positions - // - Ping-pong buffers as needed. THIS MAY BE DIFFERENT FROM BEFORE. + // TODO-2.3 - start by copying Boids::stepSimulationNaiveGrid + // Uniform Grid Neighbor search using Thrust sort on cell-coherent data. + // In Parallel: + dim3 fullBlocksPerGrid((numObjects + blockSize - 1) / blockSize); + dim3 fullGrid((gridCellCount + blockSize - 1) / blockSize); + + // - Label each particle with its array index as well as its grid index. + // Use 2x width grids + auto gridResolution = gridSideCount; + kernComputeIndices << < fullBlocksPerGrid, blockSize >> > (numObjects, gridResolution, gridMinimum, gridInverseCellWidth, dev_pos, dev_particleArrayIndices, dev_particleGridIndices); + // - Unstable key sort using Thrust. A stable sort isn't necessary, but you + // are welcome to do a performance comparison. + + thrust::sort_by_key(dev_thrust_particleGridIndices, dev_thrust_particleGridIndices + numObjects, dev_thrust_particleArrayIndices); + // - Naively unroll the loop for finding the start and end indices of each + // cell's data pointers in the array of boid indices + kernResetIntBuffer << < fullGrid, blockSize >> > (gridCellCount, dev_gridCellStartIndices, -1); + kernResetIntBuffer << < fullGrid, blockSize >> > (gridCellCount, dev_gridCellEndIndices, -1); + kernIdentifyCellStartEnd << > > (numObjects, dev_particleGridIndices, dev_gridCellStartIndices, dev_gridCellEndIndices); + + // - BIG DIFFERENCE: use the rearranged array index buffer to reshuffle all + // the particle data in the simulation array. + // CONSIDER WHAT ADDITIONAL BUFFERS YOU NEED + kernReshuffleVelAndPos << < fullBlocksPerGrid, blockSize >> > (numObjects, dev_particleArrayIndices, dev_pos, dev_shuffledPos, dev_vel1, dev_shuffledVel); + // - Perform velocity updates using neighbor search + kernUpdateVelNeighborSearchCoherent << < fullBlocksPerGrid, blockSize >> > (numObjects, gridResolution, gridMinimum, + gridInverseCellWidth, gridCellWidth, dev_gridCellStartIndices, dev_gridCellEndIndices, dev_shuffledPos, + dev_shuffledVel, dev_vel2); + // - Update positions + kernUpdatePos << < fullBlocksPerGrid, blockSize >> > (numObjects, dt, dev_shuffledPos, dev_vel2); + //Ping - pong buffers as needed.THIS MAY BE DIFFERENT FROM BEFORE. + std::swap(dev_vel1, dev_vel2); + std::swap(dev_pos, dev_shuffledPos); + + } void Boids::endSimulation() { - cudaFree(dev_vel1); - cudaFree(dev_vel2); - cudaFree(dev_pos); - - // TODO-2.1 TODO-2.3 - Free any additional buffers here. + cudaFree(dev_vel1); + cudaFree(dev_vel2); + cudaFree(dev_pos); + + // TODO-2.1 TODO-2.3 - Free any additional buffers here. + cudaFree(dev_particleArrayIndices); + cudaFree(dev_particleGridIndices); + cudaFree(dev_gridCellStartIndices); + cudaFree(dev_gridCellStartIndices); + + cudaFree(dev_shuffledPos); + cudaFree(dev_shuffledVel); } void Boids::unitTest() { - // LOOK-1.2 Feel free to write additional tests here. - - // test unstable sort - int *dev_intKeys; - int *dev_intValues; - int N = 10; - - std::unique_ptrintKeys{ new int[N] }; - std::unique_ptrintValues{ new int[N] }; - - intKeys[0] = 0; intValues[0] = 0; - intKeys[1] = 1; intValues[1] = 1; - intKeys[2] = 0; intValues[2] = 2; - intKeys[3] = 3; intValues[3] = 3; - intKeys[4] = 0; intValues[4] = 4; - intKeys[5] = 2; intValues[5] = 5; - intKeys[6] = 2; intValues[6] = 6; - intKeys[7] = 0; intValues[7] = 7; - intKeys[8] = 5; intValues[8] = 8; - intKeys[9] = 6; intValues[9] = 9; - - cudaMalloc((void**)&dev_intKeys, N * sizeof(int)); - checkCUDAErrorWithLine("cudaMalloc dev_intKeys failed!"); - - cudaMalloc((void**)&dev_intValues, N * sizeof(int)); - checkCUDAErrorWithLine("cudaMalloc dev_intValues failed!"); - - dim3 fullBlocksPerGrid((N + blockSize - 1) / blockSize); - - std::cout << "before unstable sort: " << std::endl; - for (int i = 0; i < N; i++) { - std::cout << " key: " << intKeys[i]; - std::cout << " value: " << intValues[i] << std::endl; - } - - // How to copy data to the GPU - cudaMemcpy(dev_intKeys, intKeys.get(), sizeof(int) * N, cudaMemcpyHostToDevice); - cudaMemcpy(dev_intValues, intValues.get(), sizeof(int) * N, cudaMemcpyHostToDevice); - - // Wrap device vectors in thrust iterators for use with thrust. - thrust::device_ptr dev_thrust_keys(dev_intKeys); - thrust::device_ptr dev_thrust_values(dev_intValues); - // LOOK-2.1 Example for using thrust::sort_by_key - thrust::sort_by_key(dev_thrust_keys, dev_thrust_keys + N, dev_thrust_values); - - // How to copy data back to the CPU side from the GPU - cudaMemcpy(intKeys.get(), dev_intKeys, sizeof(int) * N, cudaMemcpyDeviceToHost); - cudaMemcpy(intValues.get(), dev_intValues, sizeof(int) * N, cudaMemcpyDeviceToHost); - checkCUDAErrorWithLine("memcpy back failed!"); - - std::cout << "after unstable sort: " << std::endl; - for (int i = 0; i < N; i++) { - std::cout << " key: " << intKeys[i]; - std::cout << " value: " << intValues[i] << std::endl; - } - - // cleanup - cudaFree(dev_intKeys); - cudaFree(dev_intValues); - checkCUDAErrorWithLine("cudaFree failed!"); - return; + // LOOK-1.2 Feel free to write additional tests here. + + // test unstable sort + int* dev_intKeys; + int* dev_intValues; + char* dev_charValues; + int N = 10; + + std::unique_ptrintKeys{ new int[N] }; + std::unique_ptrintValues{ new int[N] }; + + + intKeys[0] = 0; intValues[0] = 0; + intKeys[1] = 1; intValues[1] = 1; + intKeys[2] = 0; intValues[2] = 2; + intKeys[3] = 3; intValues[3] = 3; + intKeys[4] = 0; intValues[4] = 4; + intKeys[5] = 2; intValues[5] = 5; + intKeys[6] = 2; intValues[6] = 6; + intKeys[7] = 0; intValues[7] = 7; + intKeys[8] = 5; intValues[8] = 8; + intKeys[9] = 6; intValues[9] = 9; + + cudaMalloc((void**)&dev_intKeys, N * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_intKeys failed!"); + + cudaMalloc((void**)&dev_intValues, N * sizeof(int)); + checkCUDAErrorWithLine("cudaMalloc dev_intValues failed!"); + + + dim3 fullBlocksPerGrid((N + blockSize - 1) / blockSize); + + std::cout << "before unstable sort: " << std::endl; + for (int i = 0; i < N; i++) { + std::cout << " key: " << intKeys[i]; + std::cout << " value: " << intValues[i] << std::endl; + } + + // How to copy data to the GPU + cudaMemcpy(dev_intKeys, intKeys.get(), sizeof(int) * N, cudaMemcpyHostToDevice); + cudaMemcpy(dev_intValues, intValues.get(), sizeof(int) * N, cudaMemcpyHostToDevice); + + // Wrap device vectors in thrust iterators for use with thrust. + //thrust::device_ptr dev_thrust_keys(dev_intKeys); + //thrust::device_ptr dev_thrust_values(dev_intValues); + thrust::device_ptr dev_thrust_keys; + thrust::device_ptr dev_thrust_values; + dev_thrust_keys = thrust::device_pointer_cast(dev_intKeys); + dev_thrust_values = thrust::device_pointer_cast(dev_intValues); + // LOOK-2.1 Example for using thrust::sort_by_key + thrust::sort_by_key(dev_thrust_keys, dev_thrust_keys + N, dev_thrust_values); + + // How to copy data back to the CPU side from the GPU + cudaMemcpy(intKeys.get(), dev_intKeys, sizeof(int) * N, cudaMemcpyDeviceToHost); + cudaMemcpy(intValues.get(), dev_intValues, sizeof(int) * N, cudaMemcpyDeviceToHost); + checkCUDAErrorWithLine("memcpy back failed!"); + + std::cout << "after unstable sort: " << std::endl; + for (int i = 0; i < N; i++) { + std::cout << " key: " << intKeys[i]; + std::cout << " value: " << intValues[i] << std::endl; + } + + // cleanup + cudaFree(dev_intKeys); + cudaFree(dev_intValues); + checkCUDAErrorWithLine("cudaFree failed!"); + return; } diff --git a/src/main.cpp b/src/main.cpp index b82c8c6..c5f135f 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -14,11 +14,11 @@ // LOOK-2.1 LOOK-2.3 - toggles for UNIFORM_GRID and COHERENT_GRID #define VISUALIZE 1 -#define UNIFORM_GRID 0 -#define COHERENT_GRID 0 +#define UNIFORM_GRID 1 +#define COHERENT_GRID 1 // LOOK-1.2 - change this to adjust particle count in the simulation -const int N_FOR_VIS = 5000; +const int N_FOR_VIS = 50000; const float DT = 0.2f; /** @@ -216,6 +216,8 @@ void initShaders(GLuint * program) { double fps = 0; double timebase = 0; int frame = 0; + std::vector fps_avg; + double result = 0.0f; Boids::unitTest(); // LOOK-1.2 We run some basic example code to make sure // your CUDA development setup is ready to go. @@ -230,6 +232,17 @@ void initShaders(GLuint * program) { fps = frame / (time - timebase); timebase = time; frame = 0; + fps_avg.push_back(fps); + } + if (fps_avg.size() >= 10) + { + double temp = 0; + for (auto i : fps_avg) + { + temp += i; + } + result = temp / fps_avg.size(); + fps_avg.clear(); } runCUDA(); @@ -238,7 +251,8 @@ void initShaders(GLuint * program) { ss << "["; ss.precision(1); ss << std::fixed << fps; - ss << " fps] " << deviceName; + ss << " fps] "; + ss<