/*! Memory Utilization Benchmarks Benchmarks measuring memory usage patterns, efficiency, and optimization across different scenarios. These benchmarks help identify memory bottlenecks and validate memory management strategies. ## Benchmark Categories 1. **Memory Allocation**: Allocation and deallocation patterns 2. **Memory Fragmentation**: Fragmentation impact on performance 3. **Cache Efficiency**: Memory access pattern optimization 4. **GPU Memory**: GPU memory allocation and transfer patterns 5. **Memory Pools**: Memory pool effectiveness 6. **Large Model Handling**: Memory usage with large models ## Usage ```bash # Run memory utilization benchmarks cargo bench --bench memory_utilization # Run with memory profiling RUST_LOG=debug cargo bench --bench memory_utilization # Generate memory usage report cargo bench --bench memory_utilization -- --memory-report ``` */ use criterion::{black_box, criterion_group, criterion_main, Criterion, BenchmarkId, Throughput}; use rustytorch_integration_tests::*; use std::time::Duration; use tokio::runtime::Runtime; fn benchmark_memory_allocation_patterns(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("memory_allocation"); group.measurement_time(Duration::from_secs(10)); let allocation_sizes = vec![1024, 4096, 16384, 65536, 262144]; // 1KB to 256KB for size in allocation_sizes { group.throughput(Throughput::Bytes(size as u64)); group.bench_with_input( BenchmarkId::new("allocate_deallocate", size), &size, |b, &size| { b.to_async(&rt).iter(|| async { benchmark_allocation_cycle(black_box(size)).await }); }, ); group.bench_with_input( BenchmarkId::new("bulk_allocation", size), &size, |b, &size| { b.to_async(&rt).iter(|| async { benchmark_bulk_allocation(black_box(size)).await }); }, ); } group.finish(); } fn benchmark_memory_fragmentation(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("memory_fragmentation"); group.measurement_time(Duration::from_secs(15)); let fragmentation_scenarios = vec![ ("sequential", false), ("fragmented", true), ]; for (scenario, fragmented) in fragmentation_scenarios { group.bench_with_input( BenchmarkId::new("allocation_pattern", scenario), &fragmented, |b, &fragmented| { b.to_async(&rt).iter(|| async { benchmark_fragmentation_pattern(black_box(fragmented)).await }); }, ); } group.finish(); } fn benchmark_cache_efficiency(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("cache_efficiency"); group.measurement_time(Duration::from_secs(12)); let access_patterns = vec![ ("sequential", AccessPattern::Sequential), ("random", AccessPattern::Random), ("strided", AccessPattern::Strided), ]; let data_sizes = vec![1024 * 1024, 10 * 1024 * 1024]; // 1MB, 10MB for (pattern_name, pattern) in access_patterns { for data_size in &data_sizes { group.throughput(Throughput::Bytes(*data_size as u64)); group.bench_with_input( BenchmarkId::new(pattern_name, format!("{}MB", data_size / (1024 * 1024))), &(*data_size, pattern), |b, &(data_size, pattern)| { b.to_async(&rt).iter(|| async { benchmark_memory_access_pattern(black_box(data_size), black_box(pattern)).await }); }, ); } } group.finish(); } fn benchmark_gpu_memory_operations(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("gpu_memory"); group.measurement_time(Duration::from_secs(20)); group.sample_size(20); // Fewer samples for GPU operations let transfer_sizes = vec![1024 * 1024, 16 * 1024 * 1024, 64 * 1024 * 1024]; // 1MB, 16MB, 64MB for size in transfer_sizes { group.throughput(Throughput::Bytes(size as u64)); group.bench_with_input( BenchmarkId::new("host_to_device", format!("{}MB", size / (1024 * 1024))), &size, |b, &size| { b.to_async(&rt).iter(|| async { benchmark_host_to_device_transfer(black_box(size)).await }); }, ); group.bench_with_input( BenchmarkId::new("device_to_host", format!("{}MB", size / (1024 * 1024))), &size, |b, &size| { b.to_async(&rt).iter(|| async { benchmark_device_to_host_transfer(black_box(size)).await }); }, ); group.bench_with_input( BenchmarkId::new("device_to_device", format!("{}MB", size / (1024 * 1024))), &size, |b, &size| { b.to_async(&rt).iter(|| async { benchmark_device_to_device_transfer(black_box(size)).await }); }, ); } group.finish(); } fn benchmark_memory_pools(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("memory_pools"); group.measurement_time(Duration::from_secs(15)); let pool_configurations = vec![ ("no_pool", PoolConfig::None), ("small_pool", PoolConfig::Small), ("large_pool", PoolConfig::Large), ("adaptive_pool", PoolConfig::Adaptive), ]; for (config_name, pool_config) in pool_configurations { group.bench_with_input( BenchmarkId::new("pooled_allocation", config_name), &pool_config, |b, &pool_config| { b.to_async(&rt).iter(|| async { benchmark_pooled_allocation(black_box(pool_config)).await }); }, ); } group.finish(); } fn benchmark_large_model_memory(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("large_model_memory"); group.measurement_time(Duration::from_secs(30)); group.sample_size(10); // Very few samples for large model tests let model_sizes = vec![ ("small", 100 * 1024 * 1024), // 100MB ("medium", 1024 * 1024 * 1024), // 1GB ("large", 4 * 1024 * 1024 * 1024), // 4GB ]; for (size_name, size_bytes) in model_sizes { group.throughput(Throughput::Bytes(size_bytes as u64)); group.bench_with_input( BenchmarkId::new("model_loading", size_name), &size_bytes, |b, &size_bytes| { b.to_async(&rt).iter(|| async { benchmark_large_model_loading(black_box(size_bytes)).await }); }, ); group.bench_with_input( BenchmarkId::new("model_inference", size_name), &size_bytes, |b, &size_bytes| { b.to_async(&rt).iter(|| async { benchmark_large_model_inference(black_box(size_bytes)).await }); }, ); } group.finish(); } fn benchmark_memory_pressure(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("memory_pressure"); group.measurement_time(Duration::from_secs(25)); let pressure_levels = vec![ ("low", 0.3), // 30% memory usage ("medium", 0.6), // 60% memory usage ("high", 0.9), // 90% memory usage ]; for (level_name, pressure_ratio) in pressure_levels { group.bench_with_input( BenchmarkId::new("under_pressure", level_name), &pressure_ratio, |b, &pressure_ratio| { b.to_async(&rt).iter(|| async { benchmark_memory_under_pressure(black_box(pressure_ratio)).await }); }, ); } group.finish(); } // Helper types and enums #[derive(Debug, Clone, Copy)] enum AccessPattern { Sequential, Random, Strided, } #[derive(Debug, Clone, Copy)] enum PoolConfig { None, Small, Large, Adaptive, } // Mock benchmark implementations async fn benchmark_allocation_cycle(size: usize) -> usize { // Simulate allocation and deallocation let alloc_time = Duration::from_nanos(100 + (size / 1024) as u64 * 10); tokio::time::sleep(alloc_time).await; let dealloc_time = Duration::from_nanos(50 + (size / 2048) as u64 * 5); tokio::time::sleep(dealloc_time).await; size } async fn benchmark_bulk_allocation(size: usize) -> usize { // Simulate bulk allocation of multiple blocks let num_blocks = 100; let total_size = size * num_blocks; let bulk_alloc_time = Duration::from_nanos(500 + (total_size / 1024) as u64 * 5); tokio::time::sleep(bulk_alloc_time).await; total_size } async fn benchmark_fragmentation_pattern(fragmented: bool) -> usize { let allocations = if fragmented { 200 } else { 50 }; let base_time = if fragmented { 2000 } else { 500 }; // Fragmented allocations are slower tokio::time::sleep(Duration::from_nanos(base_time + allocations * 10)).await; allocations } async fn benchmark_memory_access_pattern(data_size: usize, pattern: AccessPattern) -> usize { let access_time = match pattern { AccessPattern::Sequential => data_size / (1024 * 1024), // Fast sequential access AccessPattern::Random => data_size / (512 * 1024), // Slower random access AccessPattern::Strided => data_size / (256 * 1024), // Slowest strided access }; tokio::time::sleep(Duration::from_micros(access_time as u64)).await; data_size } async fn benchmark_host_to_device_transfer(size: usize) -> usize { // Simulate PCIe transfer bandwidth (~16 GB/s for PCIe 4.0 x16) let transfer_time_ns = (size as u64 * 1000) / 16; // nanoseconds tokio::time::sleep(Duration::from_nanos(transfer_time_ns)).await; size } async fn benchmark_device_to_host_transfer(size: usize) -> usize { // Similar to host-to-device but slightly slower let transfer_time_ns = (size as u64 * 1000) / 14; tokio::time::sleep(Duration::from_nanos(transfer_time_ns)).await; size } async fn benchmark_device_to_device_transfer(size: usize) -> usize { // Much faster GPU-to-GPU transfer let transfer_time_ns = (size as u64 * 1000) / 400; // ~400 GB/s for high-end GPUs tokio::time::sleep(Duration::from_nanos(transfer_time_ns)).await; size } async fn benchmark_pooled_allocation(pool_config: PoolConfig) -> usize { let (alloc_time, efficiency) = match pool_config { PoolConfig::None => (1000, 1.0), // No pooling, baseline PoolConfig::Small => (200, 0.8), // Small pool, fast but less efficient PoolConfig::Large => (150, 0.95), // Large pool, very efficient PoolConfig::Adaptive => (100, 0.98), // Adaptive pool, best of both }; tokio::time::sleep(Duration::from_nanos(alloc_time)).await; (1024.0 * efficiency) as usize // Effective allocation size } async fn benchmark_large_model_loading(size_bytes: usize) -> usize { // Simulate loading time based on disk I/O and memory bandwidth let loading_time_ms = (size_bytes / (1024 * 1024)) as u64 * 10; // ~100 MB/s loading tokio::time::sleep(Duration::from_millis(loading_time_ms)).await; size_bytes } async fn benchmark_large_model_inference(size_bytes: usize) -> usize { // Simulate inference time scaling with model size let inference_base_time = 100; // Base inference time in milliseconds let size_factor = (size_bytes / (100 * 1024 * 1024)) as u64; // Per 100MB let total_time = inference_base_time + size_factor * 50; tokio::time::sleep(Duration::from_millis(total_time)).await; size_bytes } async fn benchmark_memory_under_pressure(pressure_ratio: f64) -> usize { // Simulate performance degradation under memory pressure let base_time = 100; let pressure_multiplier = if pressure_ratio > 0.8 { 3.0 // Significant slowdown at high pressure } else if pressure_ratio > 0.6 { 1.5 // Moderate slowdown at medium pressure } else { 1.0 // No slowdown at low pressure }; let total_time = (base_time as f64 * pressure_multiplier) as u64; tokio::time::sleep(Duration::from_millis(total_time)).await; (1024.0 / pressure_multiplier) as usize // Effective processing amount } criterion_group!( benches, benchmark_memory_allocation_patterns, benchmark_memory_fragmentation, benchmark_cache_efficiency, benchmark_gpu_memory_operations, benchmark_memory_pools, benchmark_large_model_memory, benchmark_memory_pressure ); criterion_main!(benches);