//! Criterion benchmarks for PINN MRE Helmholtz solver //! //! Benchmarks: //! - Forward pass (network evaluation) //! - PDE residual computation //! - Single training step //! - Full training epoch //! //! Run with: cargo bench --bench pinn_benchmark use criterion::{criterion_group, criterion_main, Criterion, BenchmarkId, Throughput}; use pinn_mre_helmholtz::{Config, Mre1DPinnSolver, LffnUNet1D, ForwardWorkspace, GpuLossAccumulator, synthesize_displacement, calculate_k}; use rtx_tensor::{Tensor, Device}; /// Benchmark forward pass (network evaluation only) fn benchmark_forward_pass(c: &mut Criterion) { let mut group = c.benchmark_group("forward_pass"); group.sample_size(50); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; // Setup solver to get network and input tensor let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("lffn_mlp", n_points), &n_points, |b, _| { b.iter(|| { solver.u_net.forward(&solver.x_data).expect("Forward pass failed") }) }, ); } group.finish(); } /// Benchmark optimized forward pass with workspace (Phase 2 optimizations) fn benchmark_forward_pass_optimized(c: &mut Criterion) { let mut group = c.benchmark_group("forward_pass_optimized"); group.sample_size(50); // Include 100000 to test rayon parallelism (threshold is 100k) for n_points in [200, 1000, 10000, 100000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; #[cfg(feature = "cuda")] let device = Device::Cuda(0); #[cfg(not(feature = "cuda"))] let device = Device::Cpu; // Setup solver and workspace let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = ForwardWorkspace::new(n_points, &cfg, &device).expect("Failed to create workspace"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("lffn_mlp_workspace", n_points), &n_points, |b, _| { b.iter(|| { solver.u_net.forward_with_workspace(&solver.x_data, &mut ws) .expect("Forward pass failed") }) }, ); } group.finish(); } /// Benchmark PDE residual computation (analytical derivatives) fn benchmark_pde_residual_analytical(c: &mut Criterion) { let mut group = c.benchmark_group("pde_residual_analytical"); group.sample_size(50); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("helmholtz_cpu", n_points), &n_points, |b, _| { b.iter(|| { solver.compute_pde_residual_analytical() }) }, ); } group.finish(); } /// Benchmark PDE residual computation (tensor operations - GPU accelerable) fn benchmark_pde_residual_tensor(c: &mut Criterion) { let mut group = c.benchmark_group("pde_residual_tensor"); group.sample_size(50); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("helmholtz_tensor", n_points), &n_points, |b, _| { b.iter(|| { solver.compute_pde_residual_tensor().expect("PDE residual failed") }) }, ); } group.finish(); } /// Benchmark single training step (forward + loss + scheduler) fn benchmark_training_step(c: &mut Criterion) { let mut group = c.benchmark_group("training_step"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; // Need mutable solver for training step let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step().expect("Training step failed") }) }, ); } group.finish(); } /// Benchmark optimized training step with workspace (Phase 2 optimizations) fn benchmark_training_step_optimized(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_optimized"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; // Need mutable solver and workspace for training step let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = solver.create_workspace().expect("Failed to create workspace"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step_workspace", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step_with_workspace(&mut ws).expect("Training step failed") }) }, ); } group.finish(); } /// Benchmark data generation (synthetic displacement) fn benchmark_data_generation(c: &mut Criterion) { let mut group = c.benchmark_group("data_generation"); group.sample_size(100); for n_points in [200, 1000, 10000, 100000] { let cfg = Config { n_data: n_points, n_pde: n_points, ..Config::default() }; group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("synthesize_displacement", n_points), &n_points, |b, _| { b.iter(|| { synthesize_displacement(&cfg) }) }, ); } group.finish(); } /// Benchmark wave number calculation fn benchmark_wave_number(c: &mut Criterion) { let cfg = Config::default(); c.bench_function("calculate_k", |b| { b.iter(|| calculate_k(&cfg)) }); } /// Benchmark MSE loss computation fn benchmark_mse_loss(c: &mut Criterion) { let mut group = c.benchmark_group("mse_loss"); group.sample_size(50); #[cfg(feature = "cuda")] let device = Device::Cuda(0); #[cfg(not(feature = "cuda"))] let device = Device::Cpu; for n_points in [200, 1000, 10000] { // Create random tensors for MSE computation let data1: Vec = (0..n_points * 2).map(|_| fastrand::f32()).collect(); let data2: Vec = (0..n_points * 2).map(|_| fastrand::f32()).collect(); let tensor1 = Tensor::from_slice(&data1, &[n_points, 2], &device) .expect("Failed to create tensor"); let tensor2 = Tensor::from_slice(&data2, &[n_points, 2], &device) .expect("Failed to create tensor"); group.throughput(Throughput::Elements(n_points as u64 * 2)); group.bench_with_input( BenchmarkId::new("mse_computation", n_points), &n_points, |b, _| { b.iter(|| { Mre1DPinnSolver::mse_loss(&tensor1, &tensor2).expect("MSE failed") }) }, ); } group.finish(); } /// Benchmark full training (100 epochs) - UNOPTIMIZED baseline fn benchmark_training_100_epochs(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, // Disable printing ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train().expect("Training failed") }) }, ); } group.finish(); } /// Benchmark full training (100 epochs) - OPTIMIZED with cached PDE tensors /// This benchmark demonstrates the Phase 1 optimization that eliminates ~89% overhead /// by pre-computing PDE tensors at solver construction. fn benchmark_training_100_epochs_cached(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_cached"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, // Disable printing ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_cached", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_cached().expect("Training failed") // Use cached version! }) }, ); } group.finish(); } /// Benchmark single training step with CACHED PDE tensors fn benchmark_training_step_cached(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_cached"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step_cached", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step_cached().expect("Training step failed") }) }, ); } group.finish(); } /// Benchmark FULLY OPTIMIZED training step: workspace + cached PDE fn benchmark_training_step_fully_optimized(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_fully_optimized"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = solver.create_workspace().expect("Failed to create workspace"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step_fully_opt", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step_fully_optimized(&mut ws).expect("Training step failed") }) }, ); } group.finish(); } /// Benchmark FULLY OPTIMIZED 100 epochs: workspace + cached PDE fn benchmark_training_100_epochs_fully_optimized(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_fully_optimized"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_fully_opt", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_fully_optimized().expect("Training failed") }) }, ); } group.finish(); } /// Benchmark forward pass with CUDA graph acceleration #[cfg(feature = "cuda")] fn benchmark_forward_pass_cuda_graph(c: &mut Criterion) { use pinn_mre_helmholtz::CapturedForwardGraph; let mut group = c.benchmark_group("forward_pass_cuda_graph"); group.sample_size(50); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let device = Device::Cuda(0); let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = ForwardWorkspace::new(n_points, &cfg, &device).expect("Failed to create workspace"); let mut graph = CapturedForwardGraph::new(); // Warm up: capture the graph on first call solver.u_net.forward_with_graph(&solver.x_data, &mut ws, &mut graph) .expect("Initial graph capture failed"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("lffn_mlp_cuda_graph", n_points), &n_points, |b, _| { b.iter(|| { solver.u_net.forward_with_graph(&solver.x_data, &mut ws, &mut graph) .expect("Graph launch failed") }) }, ); } group.finish(); } /// Benchmark forward pass with uber-kernel (single launch, forward mode AD) #[cfg(feature = "cuda")] fn benchmark_forward_pass_uber_kernel(c: &mut Criterion) { use pinn_mre_helmholtz::uber_kernel::{forward_uber_value_only, forward_uber_with_gradients}; let mut group = c.benchmark_group("forward_pass_uber_kernel"); group.sample_size(100); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); // Extract weights using new helper method let weights = solver.u_net.to_uber_kernel_weights(&cfg); let scale = 2.0 * std::f64::consts::PI as f32; // Benchmark value-only (no gradients) group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("uber_value_only", n_points), &n_points, |b, _| { b.iter(|| { forward_uber_value_only(&solver.x_data, &weights, scale) .expect("Uber-kernel forward failed") }) }, ); // Benchmark with gradients (forward mode AD) group.bench_with_input( BenchmarkId::new("uber_with_gradients", n_points), &n_points, |b, _| { b.iter(|| { forward_uber_with_gradients(&solver.x_data, &weights, scale) .expect("Uber-kernel forward with gradients failed") }) }, ); } group.finish(); } /// Benchmark forward pass with MATMULS-ONLY CUDA graph /// /// Strategy: Pre/Post kernel execution with matmul graph /// 1. Fourier features: Safe API (before graph) /// 2. Matmuls (5x): CUDA Graph (captured and replayed) /// 3. Bias+activation: Safe API (after graph) #[cfg(feature = "cuda")] fn benchmark_forward_pass_matmuls_graph(c: &mut Criterion) { use pinn_mre_helmholtz::{CachedGpuPtrs, UnsafeGraph}; let mut group = c.benchmark_group("forward_pass_matmuls_graph"); group.sample_size(100); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; // Create solver with ALL tensors on capture stream let (solver, mut ws, ctx) = Mre1DPinnSolver::new_for_graph_capture(cfg.clone()) .expect("Failed to create graph-ready solver"); let stream = ctx.stream().clone(); // Force sync before capture ctx.device_synchronize().expect("Device sync failed"); ctx.join_with_default_stream().expect("Stream join failed"); // Pre-cache GPU pointers let ptrs = CachedGpuPtrs::from_forward_pass( &solver.x_data, solver.u_net.b_learnable(), solver.u_net.layers(), &ws, &stream, ).expect("Failed to cache pointers"); // Get raw cuBLAS handle let cublas_handle = *ctx.cublas().handle(); // Capture the matmuls graph ctx.synchronize().expect("Pre-capture sync failed"); let mut graph = UnsafeGraph::capture(stream.clone(), || { unsafe { ptrs.forward_matmuls_only(cublas_handle) } }).expect("Graph capture failed"); // Warm up for _ in 0..10 { ctx.fourier_features_out(&solver.x_data, solver.u_net.b_learnable(), 2.0 * std::f32::consts::PI, &mut ws.features).unwrap(); graph.launch().unwrap(); for i in 0..solver.u_net.layers().len() - 1 { if let Some(bias) = solver.u_net.layers()[i].bias() { ctx.bias_add_tanh_(&mut ws.hidden_layers[i], bias).unwrap(); } } let last_idx = solver.u_net.layers().len() - 1; if let Some(bias) = solver.u_net.layers()[last_idx].bias() { ctx.add_bias_(&mut ws.output, bias).unwrap(); } } ctx.synchronize().expect("Warm-up sync failed"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("hybrid_graph", n_points), &n_points, |b, _| { b.iter(|| { // Pre-graph: Fourier features ctx.fourier_features_out(&solver.x_data, solver.u_net.b_learnable(), 2.0 * std::f32::consts::PI, &mut ws.features).unwrap(); // Graph: 5 matmuls graph.launch().unwrap(); // Post-graph: bias+activation for i in 0..solver.u_net.layers().len() - 1 { if let Some(bias) = solver.u_net.layers()[i].bias() { ctx.bias_add_tanh_(&mut ws.hidden_layers[i], bias).unwrap(); } } let last_idx = solver.u_net.layers().len() - 1; if let Some(bias) = solver.u_net.layers()[last_idx].bias() { ctx.add_bias_(&mut ws.output, bias).unwrap(); } // Sync to measure actual completion ctx.synchronize().unwrap(); }) }, ); } group.finish(); } /// Benchmark forward pass with uber-kernel ZERO-ALLOCATION path /// This is the fastest possible inference: pre-allocated output, cached weights #[cfg(feature = "cuda")] fn benchmark_forward_pass_uber_kernel_zero_alloc(c: &mut Criterion) { use pinn_mre_helmholtz::uber_kernel::forward_uber_value_only_out; let mut group = c.benchmark_group("forward_pass_uber_kernel_zero_alloc"); group.sample_size(200); // More samples for µs-level timing for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); // Pre-compute weights ONCE (outside benchmark loop) let weights = solver.u_net.to_uber_kernel_weights(&cfg); let scale = 2.0 * std::f64::consts::PI as f32; // Pre-allocate output tensor ONCE let device = Device::Cuda(0); let mut output = Tensor::zeros([n_points, 2], &device) .expect("Failed to allocate output"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("uber_zero_alloc", n_points), &n_points, |b, _| { b.iter(|| { forward_uber_value_only_out(&solver.x_data, &weights, scale, &mut output) .expect("Uber-kernel failed") }) }, ); } group.finish(); } /// Benchmark FUSED-LITE kernel with shared memory weight caching /// This is the target: single-digit µs inference #[cfg(feature = "cuda")] fn benchmark_forward_pass_fused_lite(c: &mut Criterion) { use pinn_mre_helmholtz::uber_kernel::forward_fused_lite; let mut group = c.benchmark_group("forward_pass_fused_lite"); group.sample_size(200); for n_points in [200, 1000, 10000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); // Pre-compute weights ONCE let weights = solver.u_net.to_uber_kernel_weights(&cfg); let scale = 2.0 * std::f64::consts::PI as f32; // Pre-allocate output tensor let device = Device::Cuda(0); let mut output = Tensor::zeros([n_points, 2], &device) .expect("Failed to allocate output"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("fused_lite", n_points), &n_points, |b, _| { b.iter(|| { forward_fused_lite(&solver.x_data, &weights, scale, &mut output) .expect("Fused-lite kernel failed") }) }, ); } group.finish(); } /// Benchmark training step with CUDA graph acceleration #[cfg(feature = "cuda")] fn benchmark_training_step_cuda_graph(c: &mut Criterion) { use pinn_mre_helmholtz::CapturedForwardGraph; let mut group = c.benchmark_group("training_step_cuda_graph"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = solver.create_workspace().expect("Failed to create workspace"); let mut graph = CapturedForwardGraph::new(); // Warm up: capture the graph on first call solver.training_step_with_graph(&mut ws, &mut graph) .expect("Initial graph capture failed"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step_cuda_graph", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step_with_graph(&mut ws, &mut graph) .expect("Training step failed") }) }, ); } group.finish(); } /// Benchmark 100 epochs with CUDA graph acceleration #[cfg(feature = "cuda")] fn benchmark_training_100_epochs_cuda_graph(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_cuda_graph"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_cuda_graph", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_with_graph().expect("Training failed") }) }, ); } group.finish(); } // ============================================================================= // ZERO-SYNC TRAINING BENCHMARKS (Eliminates GPU-CPU sync overhead) // ============================================================================= /// Benchmark DEFERRED-LOSS training step: only computes loss when needed /// This should be faster than fully_optimized when loss is skipped fn benchmark_training_step_deferred_loss(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_deferred_loss"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = solver.create_workspace().expect("Failed to create workspace"); group.throughput(Throughput::Elements(n_points as u64)); // Benchmark step WITHOUT loss computation (most steps in training) group.bench_with_input( BenchmarkId::new("no_loss_step", n_points), &n_points, |b, _| { let mut step = 1; // Not divisible by 100, so no loss computed b.iter(|| { solver.training_step_deferred_loss(&mut ws, step, 100) .expect("Deferred loss step failed"); step += 1; if step % 100 == 0 { step += 1; } // Skip sync steps }) }, ); // Benchmark step WITH loss computation (every sync_interval steps) group.bench_with_input( BenchmarkId::new("with_loss_step", n_points), &n_points, |b, _| { b.iter(|| { // Step 100 will compute loss solver.training_step_deferred_loss(&mut ws, 100, 100) .expect("Deferred loss step failed") }) }, ); } group.finish(); } /// Benchmark ZERO-SYNC training step: no GPU-CPU sync in hot path /// NOTE: This was found to be SLOWER than fully_optimized due to accumulator overhead fn benchmark_training_step_zero_sync(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_zero_sync"); group.sample_size(30); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let device = Device::Cuda(0); let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); let mut ws = solver.create_workspace().expect("Failed to create workspace"); let mut accumulator = GpuLossAccumulator::new(&device, 100).expect("Failed to create accumulator"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("single_step_zero_sync", n_points), &n_points, |b, _| { b.iter(|| { solver.training_step_zero_sync(&mut ws, &mut accumulator) .expect("Zero-sync training step failed") }) }, ); } group.finish(); } /// Benchmark DEFERRED-LOSS 100 epochs: skips loss on 99% of steps fn benchmark_training_100_epochs_deferred_loss(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_deferred_loss"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_deferred_loss", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_deferred_loss(100).expect("Training failed") }) }, ); } group.finish(); } /// Benchmark ZERO-SYNC 100 epochs: GPU accumulation (slower due to accumulator overhead) fn benchmark_training_100_epochs_zero_sync(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_zero_sync"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_zero_sync", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_zero_sync(100).expect("Training failed") }) }, ); } group.finish(); } /// Benchmark FP16 Tensor Core matmul vs FP32 SGEMM /// Tests raw matmul performance for typical PINN layer sizes #[cfg(feature = "cuda")] fn benchmark_matmul_fp16_tensor_cores(c: &mut Criterion) { let mut group = c.benchmark_group("matmul_tensor_cores"); group.sample_size(100); let device = Device::Cuda(0); // Test typical PINN layer sizes for (batch, _m, k, n) in [ (200, 200, 128, 64), // Fourier features → first hidden (200, 200, 64, 64), // Hidden → hidden (1000, 1000, 128, 64), // Larger batch (1000, 1000, 64, 64), // Larger batch hidden (10000, 10000, 128, 64), // Very large batch ] { let label = format!("{}x{}x{}", batch, k, n); // Create FP32 matrices (randn creates F32 by default) let a_f32 = Tensor::randn(&[batch, k], &device) .expect("Failed to create tensor A"); let b_f32 = Tensor::randn(&[k, n], &device) .expect("Failed to create tensor B"); // FP32 benchmark (SGEMM) group.throughput(Throughput::Elements((2 * batch * k * n) as u64)); // FLOPs group.bench_with_input( BenchmarkId::new("fp32_sgemm", &label), &batch, |b, _| { b.iter(|| { a_f32.matmul(&b_f32).expect("FP32 matmul failed") }) }, ); // Create FP16 matrices let a_f16 = a_f32.to_half().expect("Failed to convert A to FP16"); let b_f16 = b_f32.to_half().expect("Failed to convert B to FP16"); // FP16 benchmark (HGEMM / Tensor Cores) group.bench_with_input( BenchmarkId::new("fp16_tensor_cores", &label), &batch, |b, _| { b.iter(|| { a_f16.matmul(&b_f16).expect("FP16 matmul failed") }) }, ); } group.finish(); } // ============================================================================= // PHASE 8: ANALYTICAL BACKPROP TRAINING BENCHMARKS // ============================================================================= /// Benchmark ANALYTICAL training step with proper GPU synchronization. /// /// This measures the TRUE latency of a complete training step: /// - Forward pass (cache activations) /// - Backward pass (analytical gradients via chain rule) /// - Optimizer step (GPU-native Adam) /// /// CRITICAL: Includes device synchronization to measure actual GPU execution time, /// not just CPU dispatch speed. #[cfg(feature = "cuda")] fn benchmark_training_step_analytical(c: &mut Criterion) { let mut group = c.benchmark_group("training_step_analytical"); group.sample_size(200); // More samples for µs-level timing group.measurement_time(std::time::Duration::from_secs(5)); for n_points in [200, 1000] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 1, ..Config::default() }; let mut solver = Mre1DPinnSolver::new(cfg.clone()).expect("Failed to create solver"); // Get CUDA context for synchronization let ctx = rtx_tensor::storage::cuda_manager::get_or_create_context(0) .expect("Failed to get CUDA context"); // Warm up for _ in 0..10 { solver.train_step_data_only().expect("Warm-up step failed"); } ctx.synchronize().expect("Warm-up sync failed"); group.throughput(Throughput::Elements(n_points as u64)); group.bench_with_input( BenchmarkId::new("analytical_backprop", n_points), &n_points, |b, _| { b.iter(|| { // 1. Dispatch the entire training step (Forward + Backward + Optimizer) solver.train_step_data_only().expect("Training step failed"); // 2. Block CPU until GPU finishes // This measures the TRUE latency of the step ctx.synchronize().expect("Sync failed"); }) }, ); } group.finish(); } /// Benchmark 100 epochs with analytical backprop. /// Uses the convenience train_analytical() method. #[cfg(feature = "cuda")] fn benchmark_training_100_epochs_analytical(c: &mut Criterion) { let mut group = c.benchmark_group("training_100_epochs_analytical"); group.sample_size(10); for n_points in [200] { let cfg = Config { n_data: n_points, n_pde: n_points, epochs: 100, print_every: 1000, // Disable printing ..Config::default() }; group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("train_100_analytical", n_points), &n_points, |b, _| { b.iter(|| { let mut solver = Mre1DPinnSolver::new(cfg.clone()) .expect("Failed to create solver"); solver.train_analytical(100, 1000).expect("Training failed") }) }, ); } group.finish(); } #[cfg(feature = "cuda")] criterion_group!( benches, benchmark_wave_number, benchmark_data_generation, benchmark_forward_pass, benchmark_forward_pass_optimized, // workspace-based forward benchmark_forward_pass_cuda_graph, // CUDA graph-accelerated forward (OLD - may fail) benchmark_forward_pass_matmuls_graph, // MATMULS-ONLY CUDA graph (NEW - hybrid approach) benchmark_forward_pass_uber_kernel, // UBER-KERNEL: single launch forward + AD benchmark_forward_pass_uber_kernel_zero_alloc, // UBER-KERNEL: ZERO-ALLOCATION path benchmark_forward_pass_fused_lite, // FUSED-LITE: shared memory optimized benchmark_matmul_fp16_tensor_cores, // FP16 Tensor Core vs FP32 comparison benchmark_pde_residual_analytical, benchmark_pde_residual_tensor, benchmark_mse_loss, benchmark_training_step, benchmark_training_step_optimized, // workspace-based training step benchmark_training_step_cached, // cached PDE tensors only benchmark_training_step_fully_optimized, // workspace + cached PDE benchmark_training_step_deferred_loss, // DEFERRED-LOSS training step (skips loss on most steps) benchmark_training_step_zero_sync, // ZERO-SYNC training step (GPU accumulation - slower) benchmark_training_step_cuda_graph, // CUDA graph-accelerated training step benchmark_training_100_epochs, benchmark_training_100_epochs_cached, // cached PDE only benchmark_training_100_epochs_fully_optimized, // workspace + cached PDE benchmark_training_100_epochs_deferred_loss, // DEFERRED-LOSS training (skips loss on 99% of steps) benchmark_training_100_epochs_zero_sync, // ZERO-SYNC training (GPU accumulation - slower) benchmark_training_100_epochs_cuda_graph, // CUDA graph-accelerated training benchmark_training_step_analytical, // PHASE 8: Analytical backprop training step benchmark_training_100_epochs_analytical, // PHASE 8: Analytical backprop 100 epochs ); #[cfg(not(feature = "cuda"))] criterion_group!( benches, benchmark_wave_number, benchmark_data_generation, benchmark_forward_pass, benchmark_forward_pass_optimized, // workspace-based forward benchmark_pde_residual_analytical, benchmark_pde_residual_tensor, benchmark_mse_loss, benchmark_training_step, benchmark_training_step_optimized, // workspace-based training step benchmark_training_step_cached, // cached PDE tensors only benchmark_training_step_fully_optimized, // workspace + cached PDE benchmark_training_100_epochs, benchmark_training_100_epochs_cached, // cached PDE only benchmark_training_100_epochs_fully_optimized, // workspace + cached PDE ); criterion_main!(benches);