Files
rustytorch/crates/specialized/rtx-cfd/src/solvers/lbm/d3q19_gpu.rs
T
Omar Sobh 0de750bad1
CI / Python Bindings (maturin) (ubuntu-latest) (push) Blocked by required conditions
CI / Build (macos-latest) (push) Waiting to run
CI / Test (macos-latest) (push) Blocked by required conditions
CI / Test (ubuntu-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (macos-latest) (push) Blocked by required conditions
CI / WASM Build + Size Check (push) Blocked by required conditions
CI / Distributed Training Tests (push) Blocked by required conditions
CI / CI Success (push) Blocked by required conditions
CI / Build CPU-Only (Explicit) (push) Failing after 5s
Documentation / Build User Guide (push) Successful in 5s
Documentation / Build API Documentation (push) Failing after 5s
CI / Format Check (push) Failing after 10s
CI / Build (ubuntu-latest) (push) Failing after 1m22s
CI / Clippy Check (push) Failing after 1m38s
Performance Benchmarks / Run Benchmarks (push) Successful in 2m50s
rtx-cfd legacy GPU: kernel outputs were device-buffer CLONES (every write lost) — the buffers are borrowed mutably now; the Poisson kernels keep boundary Dirichlet values; tests: the advection pulse marched 50 steps, the Jacobi budget 20k, the reduction reference in f64
2026-09-16 07:28:42 -05:00

895 lines
28 KiB
Rust

//! GPU-accelerated D3Q19 Lattice Boltzmann Method implementation
//!
//! This module provides a CUDA-accelerated version of the D3Q19 LBM solver
//! for simulating incompressible fluid flows in 3D. It uses real CUDA kernels
//! for collision, streaming, and boundary condition operations.
use super::{D3Q19Parameters, D3Q19Solver, MacroscopicVariables3D};
use crate::kernels::*;
use crate::{CfdConfig, CfdError, CfdResult};
use cudarc::driver::{CudaModule, CudaSlice, LaunchConfig, PushKernelArg};
use nalgebra::Vector3;
use std::sync::Arc;
/// GPU-accelerated D3Q19 LBM solver
pub struct D3Q19GpuSolver {
/// Base CPU solver for reference and fallback
cpu_solver: D3Q19Solver,
/// GPU kernel manager
kernel_manager: CudaKernelManager,
/// LBM CUDA module
lbm_module: Option<Arc<CudaModule>>,
/// GPU memory buffers
gpu_buffers: Option<D3Q19GpuBuffers>,
/// Parameters
params: D3Q19Parameters,
}
/// GPU memory buffers for D3Q19 LBM
struct D3Q19GpuBuffers {
/// Grid dimensions
nx: usize,
ny: usize,
nz: usize,
/// Distribution functions f[x*y*z*19] flattened for GPU
f: CudaSlice<f32>,
/// Temporary storage for streaming step
f_temp: CudaSlice<f32>,
/// Equilibrium distributions
f_eq: CudaSlice<f32>,
/// Macroscopic variables
density: CudaSlice<f32>,
velocity_x: CudaSlice<f32>,
velocity_y: CudaSlice<f32>,
velocity_z: CudaSlice<f32>,
/// Temporary arrays
temp1: CudaSlice<f32>,
temp2: CudaSlice<f32>,
}
impl D3Q19GpuSolver {
/// Create new GPU-accelerated D3Q19 solver
pub fn new(
nx: usize,
ny: usize,
nz: usize,
params: D3Q19Parameters,
config: &CfdConfig,
) -> CfdResult<Self> {
// Create CPU solver for fallback
let cpu_solver = D3Q19Solver::new(nx, ny, nz, params.clone());
// Initialize GPU components
let kernel_manager = CudaKernelManager::new(config)?;
// Create solver instance
let mut solver = Self {
cpu_solver,
kernel_manager,
lbm_module: None,
gpu_buffers: None,
params,
};
// Compile LBM-specific kernels
solver.compile_lbm_kernels()?;
Ok(solver)
}
/// Compile D3Q19 LBM-specific CUDA kernels
fn compile_lbm_kernels(&mut self) -> CfdResult<()> {
// D3Q19 LBM kernels source code
let d3q19_kernels_src = r#"
extern "C" {
// D3Q19 lattice velocities and weights
__constant__ int d3q19_ex[19] = {0, 1, -1, 0, 0, 0, 0, 1, -1, 1, -1, 1, -1, 1, -1, 0, 0, 0, 0};
__constant__ int d3q19_ey[19] = {0, 0, 0, 1, -1, 0, 0, 1, -1, -1, 1, 0, 0, 0, 0, 1, -1, 1, -1};
__constant__ int d3q19_ez[19] = {0, 0, 0, 0, 0, 1, -1, 0, 0, 0, 0, 1, -1, -1, 1, 1, -1, -1, 1};
__constant__ float d3q19_w[19] = {
1.0f/3.0f, // rest particle
1.0f/18.0f, 1.0f/18.0f, 1.0f/18.0f, 1.0f/18.0f, 1.0f/18.0f, 1.0f/18.0f, // face neighbors
1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, // edge neighbors
1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f, 1.0f/36.0f
};
// Compute equilibrium distribution function
__global__ void d3q19_equilibrium(
float* f_eq,
const float* density,
const float* velocity_x,
const float* velocity_y,
const float* velocity_z,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
int k = blockIdx.z * blockDim.z + threadIdx.z;
if (i >= nx || j >= ny || k >= nz) return;
int idx = k * nx * ny + j * nx + i;
float rho = density[idx];
float ux = velocity_x[idx];
float uy = velocity_y[idx];
float uz = velocity_z[idx];
float u_sqr = ux * ux + uy * uy + uz * uz;
for (int q = 0; q < 19; q++) {
float ex = (float)d3q19_ex[q];
float ey = (float)d3q19_ey[q];
float ez = (float)d3q19_ez[q];
float e_dot_u = ex * ux + ey * uy + ez * uz;
float f_eq_val = d3q19_w[q] * rho * (
1.0f + 3.0f * e_dot_u + 4.5f * e_dot_u * e_dot_u - 1.5f * u_sqr
);
f_eq[idx * 19 + q] = f_eq_val;
}
}
// BGK collision operator
__global__ void d3q19_collision(
float* f,
const float* f_eq,
float omega,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
int k = blockIdx.z * blockDim.z + threadIdx.z;
if (i >= nx || j >= ny || k >= nz) return;
int idx = k * nx * ny + j * nx + i;
for (int q = 0; q < 19; q++) {
int f_idx = idx * 19 + q;
f[f_idx] = f[f_idx] - omega * (f[f_idx] - f_eq[f_idx]);
}
}
// Streaming step (propagation)
__global__ void d3q19_streaming(
float* f_new,
const float* f_old,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
int k = blockIdx.z * blockDim.z + threadIdx.z;
if (i >= nx || j >= ny || k >= nz) return;
int idx = k * nx * ny + j * nx + i;
for (int q = 0; q < 19; q++) {
// Source position for streaming
int i_src = i - d3q19_ex[q];
int j_src = j - d3q19_ey[q];
int k_src = k - d3q19_ez[q];
// Periodic boundary conditions
i_src = (i_src + nx) % nx;
j_src = (j_src + ny) % ny;
k_src = (k_src + nz) % nz;
int idx_src = k_src * nx * ny + j_src * nx + i_src;
f_new[idx * 19 + q] = f_old[idx_src * 19 + q];
}
}
// Extract macroscopic variables (density and velocity)
__global__ void d3q19_macroscopic_variables(
float* density,
float* velocity_x,
float* velocity_y,
float* velocity_z,
const float* f,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
int k = blockIdx.z * blockDim.z + threadIdx.z;
if (i >= nx || j >= ny || k >= nz) return;
int idx = k * nx * ny + j * nx + i;
// Compute density
float rho = 0.0f;
for (int q = 0; q < 19; q++) {
rho += f[idx * 19 + q];
}
density[idx] = rho;
// Compute momentum
float momentum_x = 0.0f;
float momentum_y = 0.0f;
float momentum_z = 0.0f;
for (int q = 0; q < 19; q++) {
momentum_x += f[idx * 19 + q] * d3q19_ex[q];
momentum_y += f[idx * 19 + q] * d3q19_ey[q];
momentum_z += f[idx * 19 + q] * d3q19_ez[q];
}
// Compute velocity
velocity_x[idx] = (rho > 1e-15f) ? momentum_x / rho : 0.0f;
velocity_y[idx] = (rho > 1e-15f) ? momentum_y / rho : 0.0f;
velocity_z[idx] = (rho > 1e-15f) ? momentum_z / rho : 0.0f;
}
// Initialize uniform flow field
__global__ void d3q19_initialize_uniform(
float* f,
float density,
float velocity_x,
float velocity_y,
float velocity_z,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
int k = blockIdx.z * blockDim.z + threadIdx.z;
if (i >= nx || j >= ny || k >= nz) return;
int idx = k * nx * ny + j * nx + i;
float u_sqr = velocity_x * velocity_x + velocity_y * velocity_y + velocity_z * velocity_z;
for (int q = 0; q < 19; q++) {
float ex = (float)d3q19_ex[q];
float ey = (float)d3q19_ey[q];
float ez = (float)d3q19_ez[q];
float e_dot_u = ex * velocity_x + ey * velocity_y + ez * velocity_z;
float f_eq_val = d3q19_w[q] * density * (
1.0f + 3.0f * e_dot_u + 4.5f * e_dot_u * e_dot_u - 1.5f * u_sqr
);
f[idx * 19 + q] = f_eq_val;
}
}
// Simple no-slip boundary conditions (bounce-back)
__global__ void d3q19_bounce_back_boundaries(
float* f,
int nx, int ny, int nz
) {
int i = blockIdx.x * blockDim.x + threadIdx.x;
int j = blockIdx.y * blockDim.y + threadIdx.y;
if (i >= nx || j >= ny) return;
// Bottom wall (k = 0)
int idx_bottom = 0 * nx * ny + j * nx + i;
// Top wall (k = nz-1)
int idx_top = (nz - 1) * nx * ny + j * nx + i;
// Bounce back velocities pointing into walls
// For D3Q19, we need to map opposing directions
// This is a simplified implementation
for (int q = 0; q < 19; q++) {
if (d3q19_ez[q] == 1) { // pointing up
// Find opposite direction (pointing down)
for (int opp = 0; opp < 19; opp++) {
if (d3q19_ex[opp] == -d3q19_ex[q] &&
d3q19_ey[opp] == -d3q19_ey[q] &&
d3q19_ez[opp] == -d3q19_ez[q]) {
float temp = f[idx_bottom * 19 + q];
f[idx_bottom * 19 + q] = f[idx_bottom * 19 + opp];
f[idx_bottom * 19 + opp] = temp;
break;
}
}
}
if (d3q19_ez[q] == -1) { // pointing down
// Find opposite direction (pointing up)
for (int opp = 0; opp < 19; opp++) {
if (d3q19_ex[opp] == -d3q19_ex[q] &&
d3q19_ey[opp] == -d3q19_ey[q] &&
d3q19_ez[opp] == -d3q19_ez[q]) {
float temp = f[idx_top * 19 + q];
f[idx_top * 19 + q] = f[idx_top * 19 + opp];
f[idx_top * 19 + opp] = temp;
break;
}
}
}
}
}
}
"#;
// Compile D3Q19 kernels
let ptx = cudarc::nvrtc::compile_ptx(d3q19_kernels_src)
.map_err(|e| CfdError::gpu_error(&format!("Failed to compile D3Q19 kernels: {}", e)))?;
// Load module into context
let module = self
.kernel_manager
.context()
.load_module(ptx)
.map_err(|e| CfdError::gpu_error(&format!("Failed to load D3Q19 module: {}", e)))?;
// Store module in solver
self.lbm_module = Some(module);
Ok(())
}
/// Initialize GPU buffers
fn initialize_gpu_buffers(&mut self, nx: usize, ny: usize, nz: usize) -> CfdResult<()> {
let total_f_size = nx * ny * nz * 19; // 19 distribution functions per cell
let grid_size = nx * ny * nz;
// Allocate GPU memory
let f = self.kernel_manager.allocate_f32(total_f_size)?;
let f_temp = self.kernel_manager.allocate_f32(total_f_size)?;
let f_eq = self.kernel_manager.allocate_f32(total_f_size)?;
let density = self.kernel_manager.allocate_f32(grid_size)?;
let velocity_x = self.kernel_manager.allocate_f32(grid_size)?;
let velocity_y = self.kernel_manager.allocate_f32(grid_size)?;
let velocity_z = self.kernel_manager.allocate_f32(grid_size)?;
let temp1 = self.kernel_manager.allocate_f32(grid_size)?;
let temp2 = self.kernel_manager.allocate_f32(grid_size)?;
self.gpu_buffers = Some(D3Q19GpuBuffers {
nx,
ny,
nz,
f,
f_temp,
f_eq,
density,
velocity_x,
velocity_y,
velocity_z,
temp1,
temp2,
});
Ok(())
}
/// GPU-accelerated BGK collision step
pub fn gpu_collision_step(&self) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let omega = 1.0 / self.params.tau;
// Step 1: Extract macroscopic variables
self.gpu_extract_macroscopic_variables()?;
// Step 2: Compute equilibrium distributions
self.gpu_compute_equilibrium()?;
// Step 3: Perform BGK collision
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module
.load_function("d3q19_collision")
.map_err(|e| CfdError::gpu_error(&format!("Failed to get collision kernel: {}", e)))?;
let grid_dim_x = (buffers.nx as u32 + 7) / 8;
let grid_dim_y = (buffers.ny as u32 + 7) / 8;
let grid_dim_z = (buffers.nz as u32 + 7) / 8;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, grid_dim_z),
block_dim: (8, 8, 8),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.f)
.arg(&buffers.f_eq)
.arg(&(omega as f32))
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!("Collision kernel launch failed: {}", e))
})?;
}
self.kernel_manager.synchronize()?;
Ok(())
}
/// GPU-accelerated streaming step
pub fn gpu_streaming_step(&self) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module
.load_function("d3q19_streaming")
.map_err(|e| CfdError::gpu_error(&format!("Failed to get streaming kernel: {}", e)))?;
let grid_dim_x = (buffers.nx as u32 + 7) / 8;
let grid_dim_y = (buffers.ny as u32 + 7) / 8;
let grid_dim_z = (buffers.nz as u32 + 7) / 8;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, grid_dim_z),
block_dim: (8, 8, 8),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.f_temp)
.arg(&buffers.f)
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!("Streaming kernel launch failed: {}", e))
})?;
}
// Swap buffers: f = f_temp
// In real implementation, would swap buffer pointers
self.kernel_manager.synchronize()?;
Ok(())
}
/// Extract macroscopic variables on GPU
fn gpu_extract_macroscopic_variables(&self) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module
.load_function("d3q19_macroscopic_variables")
.map_err(|e| {
CfdError::gpu_error(&format!(
"Failed to get macroscopic variables kernel: {}",
e
))
})?;
let grid_dim_x = (buffers.nx as u32 + 7) / 8;
let grid_dim_y = (buffers.ny as u32 + 7) / 8;
let grid_dim_z = (buffers.nz as u32 + 7) / 8;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, grid_dim_z),
block_dim: (8, 8, 8),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.density)
.arg(&mut buffers.velocity_x)
.arg(&mut buffers.velocity_y)
.arg(&mut buffers.velocity_z)
.arg(&buffers.f)
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!(
"Macroscopic variables kernel launch failed: {}",
e
))
})?;
}
self.kernel_manager.synchronize()?;
Ok(())
}
/// Compute equilibrium distributions on GPU
fn gpu_compute_equilibrium(&self) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module.load_function("d3q19_equilibrium").map_err(|e| {
CfdError::gpu_error(&format!("Failed to get equilibrium kernel: {}", e))
})?;
let grid_dim_x = (buffers.nx as u32 + 7) / 8;
let grid_dim_y = (buffers.ny as u32 + 7) / 8;
let grid_dim_z = (buffers.nz as u32 + 7) / 8;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, grid_dim_z),
block_dim: (8, 8, 8),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.f_eq)
.arg(&buffers.density)
.arg(&buffers.velocity_x)
.arg(&buffers.velocity_y)
.arg(&buffers.velocity_z)
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!("Equilibrium kernel launch failed: {}", e))
})?;
}
self.kernel_manager.synchronize()?;
Ok(())
}
/// GPU-accelerated bounce-back boundary conditions
pub fn gpu_apply_bounce_back_boundaries(&self) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module
.load_function("d3q19_bounce_back_boundaries")
.map_err(|e| CfdError::gpu_error(&format!("Failed to get boundary kernel: {}", e)))?;
let grid_dim_x = (buffers.nx as u32 + 15) / 16;
let grid_dim_y = (buffers.ny as u32 + 15) / 16;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, 1),
block_dim: (16, 16, 1),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.f)
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!("Boundary kernel launch failed: {}", e))
})?;
}
self.kernel_manager.synchronize()?;
Ok(())
}
/// Complete GPU LBM time step
pub fn gpu_step(&self) -> CfdResult<()> {
self.gpu_collision_step()?;
self.gpu_streaming_step()?;
self.gpu_apply_bounce_back_boundaries()?;
Ok(())
}
/// Initialize uniform flow field on GPU
pub fn gpu_initialize_uniform(
&mut self,
density: f64,
velocity: Vector3<f64>,
) -> CfdResult<()> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
let module = self
.lbm_module
.as_ref()
.ok_or_else(|| CfdError::gpu_error("LBM module not loaded"))?;
let func = module
.load_function("d3q19_initialize_uniform")
.map_err(|e| {
CfdError::gpu_error(&format!("Failed to get initialization kernel: {}", e))
})?;
let grid_dim_x = (buffers.nx as u32 + 7) / 8;
let grid_dim_y = (buffers.ny as u32 + 7) / 8;
let grid_dim_z = (buffers.nz as u32 + 7) / 8;
let config = LaunchConfig {
grid_dim: (grid_dim_x, grid_dim_y, grid_dim_z),
block_dim: (8, 8, 8),
shared_mem_bytes: 0,
};
unsafe {
self.kernel_manager
.stream()
.launch_builder(&func)
.arg(&mut buffers.f)
.arg(&(density as f32))
.arg(&(velocity.x as f32))
.arg(&(velocity.y as f32))
.arg(&(velocity.z as f32))
.arg(&(buffers.nx as i32))
.arg(&(buffers.ny as i32))
.arg(&(buffers.nz as i32))
.launch(config)
.map_err(|e| {
CfdError::gpu_error(&format!("Initialization kernel launch failed: {}", e))
})?;
}
self.kernel_manager.synchronize()?;
Ok(())
}
/// Get macroscopic variables at a specific point (copy from GPU)
pub fn gpu_macroscopic_variables_at(
&self,
x: usize,
y: usize,
z: usize,
) -> CfdResult<MacroscopicVariables3D> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
// Extract macroscopic variables first
self.gpu_extract_macroscopic_variables()?;
// Copy arrays from GPU
let density_host = self.kernel_manager.copy_from_device(&buffers.density)?;
let velocity_x_host = self.kernel_manager.copy_from_device(&buffers.velocity_x)?;
let velocity_y_host = self.kernel_manager.copy_from_device(&buffers.velocity_y)?;
let velocity_z_host = self.kernel_manager.copy_from_device(&buffers.velocity_z)?;
let idx = z * buffers.nx * buffers.ny + y * buffers.nx + x;
if idx >= density_host.len() {
return Err(CfdError::gpu_error("Index out of bounds"));
}
let density = density_host[idx] as f64;
let velocity = Vector3::new(
velocity_x_host[idx] as f64,
velocity_y_host[idx] as f64,
velocity_z_host[idx] as f64,
);
Ok(MacroscopicVariables3D::new(density, velocity))
}
/// Initialize GPU buffers and set up solver
pub fn initialize(&mut self, nx: usize, ny: usize, nz: usize) -> CfdResult<()> {
self.initialize_gpu_buffers(nx, ny, nz)?;
Ok(())
}
/// Get grid dimensions
pub fn dimensions(&self) -> (usize, usize, usize) {
if let Some(buffers) = &self.gpu_buffers {
(buffers.nx, buffers.ny, buffers.nz)
} else {
self.cpu_solver.dimensions()
}
}
/// Get solver parameters
pub fn parameters(&self) -> &D3Q19Parameters {
&self.params
}
/// Calculate total mass on GPU
pub fn gpu_total_mass(&self) -> CfdResult<f64> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
self.gpu_extract_macroscopic_variables()?;
let density_host = self.kernel_manager.copy_from_device(&buffers.density)?;
let total_mass: f32 = density_host.iter().sum();
Ok(total_mass as f64)
}
/// Calculate kinetic energy on GPU
pub fn gpu_kinetic_energy(&self) -> CfdResult<f64> {
let buffers = self
.gpu_buffers
.as_mut()
.ok_or_else(|| CfdError::gpu_error("GPU buffers not initialized"))?;
self.gpu_extract_macroscopic_variables()?;
let density_host = self.kernel_manager.copy_from_device(&buffers.density)?;
let velocity_x_host = self.kernel_manager.copy_from_device(&buffers.velocity_x)?;
let velocity_y_host = self.kernel_manager.copy_from_device(&buffers.velocity_y)?;
let velocity_z_host = self.kernel_manager.copy_from_device(&buffers.velocity_z)?;
let mut total_ke = 0.0f64;
for i in 0..density_host.len() {
let rho = density_host[i] as f64;
let ux = velocity_x_host[i] as f64;
let uy = velocity_y_host[i] as f64;
let uz = velocity_z_host[i] as f64;
total_ke += 0.5 * rho * (ux * ux + uy * uy + uz * uz);
}
Ok(total_ke)
}
/// Fallback to CPU solver
pub fn cpu_solver(&self) -> &D3Q19Solver {
&self.cpu_solver
}
/// Fallback to CPU solver (mutable)
pub fn cpu_solver_mut(&mut self) -> &mut D3Q19Solver {
&mut self.cpu_solver
}
}
#[cfg(test)]
mod tests {
use super::*;
use approx::assert_relative_eq;
#[test]
fn test_d3q19_gpu_solver_creation() -> CfdResult<()> {
let config = CfdConfig {
nx: 16,
ny: 16,
nz: 16,
lx: 1.0,
ly: 1.0,
lz: 1.0,
dt: 0.001,
viscosity: 0.01,
density: 1.0,
device_id: 0,
..CfdConfig::default()
};
let params = D3Q19Parameters::default();
// This will only work if CUDA is available
if let Ok(mut solver) = D3Q19GpuSolver::new(16, 16, 16, params, &config) {
solver.initialize(16, 16, 16)?;
println!("GPU D3Q19 solver created successfully");
} else {
println!("GPU not available, skipping GPU D3Q19 test");
}
Ok(())
}
#[test]
fn test_d3q19_gpu_initialization() -> CfdResult<()> {
let config = CfdConfig {
nx: 8,
ny: 8,
nz: 8,
lx: 1.0,
ly: 1.0,
lz: 1.0,
dt: 0.001,
viscosity: 0.01,
density: 1.0,
device_id: 0,
..CfdConfig::default()
};
let params = D3Q19Parameters::default();
if let Ok(mut solver) = D3Q19GpuSolver::new(8, 8, 8, params, &config) {
solver.initialize(8, 8, 8)?;
let density = 1.0;
let velocity = Vector3::new(0.1, 0.05, 0.02);
solver.gpu_initialize_uniform(density, velocity)?;
// Check a few points
let vars = solver.gpu_macroscopic_variables_at(4, 4, 4)?;
assert_relative_eq!(vars.density, density, epsilon = 1e-5);
assert_relative_eq!(vars.velocity.x, velocity.x, epsilon = 1e-5);
assert_relative_eq!(vars.velocity.y, velocity.y, epsilon = 1e-5);
assert_relative_eq!(vars.velocity.z, velocity.z, epsilon = 1e-5);
println!("GPU D3Q19 initialization test passed");
} else {
println!("GPU not available, skipping initialization test");
}
Ok(())
}
#[test]
fn test_d3q19_gpu_step() -> CfdResult<()> {
let config = CfdConfig {
nx: 8,
ny: 8,
nz: 8,
lx: 1.0,
ly: 1.0,
lz: 1.0,
dt: 0.001,
viscosity: 0.01,
density: 1.0,
device_id: 0,
..CfdConfig::default()
};
let params = D3Q19Parameters::default();
if let Ok(mut solver) = D3Q19GpuSolver::new(8, 8, 8, params, &config) {
solver.initialize(8, 8, 8)?;
// Initialize with simple flow
solver.gpu_initialize_uniform(1.0, Vector3::new(0.01, 0.0, 0.0))?;
// Perform one time step
solver.gpu_step()?;
// Check that simulation is stable
let total_mass = solver.gpu_total_mass()?;
assert!(total_mass > 0.0, "Total mass should be positive");
assert!(total_mass < 1000.0, "Total mass should be reasonable");
let kinetic_energy = solver.gpu_kinetic_energy()?;
assert!(
kinetic_energy >= 0.0,
"Kinetic energy should be non-negative"
);
println!("GPU D3Q19 time step test passed");
} else {
println!("GPU not available, skipping time step test");
}
Ok(())
}
}