/*! Throughput Scaling Benchmarks Benchmarks measuring throughput scaling characteristics across different load levels and resource configurations. These benchmarks validate performance scales appropriately with increased resources. ## Benchmark Categories 1. **Request Throughput**: Requests per second under various loads 2. **Concurrent Processing**: Throughput with concurrent requests 3. **Resource Scaling**: Performance scaling with additional resources 4. **Batch Throughput**: Throughput optimization with batching 5. **Memory Scaling**: Performance vs memory usage trade-offs 6. **GPU Scaling**: Multi-GPU throughput characteristics ## Usage ```bash # Run throughput scaling benchmarks cargo bench --bench throughput_scaling # Run with specific concurrency levels cargo bench --bench throughput_scaling -- --concurrency 16 # Generate detailed report cargo bench --bench throughput_scaling -- --verbose ``` */ use criterion::{black_box, criterion_group, criterion_main, Criterion, BenchmarkId, Throughput}; use rustytorch_integration_tests::*; use std::time::Duration; use tokio::runtime::Runtime; use futures::future::join_all; fn benchmark_request_throughput(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("request_throughput"); group.measurement_time(Duration::from_secs(20)); // Test different request loads let request_counts = vec![100, 500, 1000, 2000]; for request_count in request_counts { group.throughput(Throughput::Elements(request_count as u64)); group.bench_with_input( BenchmarkId::new("sequential_requests", request_count), &request_count, |b, &request_count| { b.to_async(&rt).iter(|| async { benchmark_sequential_requests(black_box(request_count)).await }); }, ); } group.finish(); } fn benchmark_concurrent_processing(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("concurrent_processing"); group.measurement_time(Duration::from_secs(15)); let concurrency_levels = vec![1, 2, 4, 8, 16, 32]; let requests_per_level = 100; for concurrency in concurrency_levels { group.throughput(Throughput::Elements(requests_per_level as u64)); group.bench_with_input( BenchmarkId::new("concurrent_requests", concurrency), &concurrency, |b, &concurrency| { b.to_async(&rt).iter(|| async { benchmark_concurrent_requests(black_box(concurrency), requests_per_level).await }); }, ); } group.finish(); } fn benchmark_resource_scaling(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("resource_scaling"); group.measurement_time(Duration::from_secs(25)); let resource_levels = vec![ ("1_cpu", 1), ("2_cpu", 2), ("4_cpu", 4), ("8_cpu", 8), ]; for (label, cpu_count) in resource_levels { group.throughput(Throughput::Elements(1000)); group.bench_with_input( BenchmarkId::new("cpu_scaling", label), &cpu_count, |b, &cpu_count| { b.to_async(&rt).iter(|| async { benchmark_cpu_scaling(black_box(cpu_count)).await }); }, ); } group.finish(); } fn benchmark_batch_throughput(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("batch_throughput"); group.measurement_time(Duration::from_secs(15)); let batch_configurations = vec![ (1, 1000), // No batching (8, 1000), // Small batches (32, 1000), // Medium batches (128, 1000), // Large batches ]; for (batch_size, total_items) in batch_configurations { group.throughput(Throughput::Elements(total_items as u64)); group.bench_with_input( BenchmarkId::new("batch_processing", format!("batch_{}", batch_size)), &(batch_size, total_items), |b, &(batch_size, total_items)| { b.to_async(&rt).iter(|| async { benchmark_batch_processing_throughput( black_box(batch_size), black_box(total_items) ).await }); }, ); } group.finish(); } fn benchmark_memory_scaling(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("memory_scaling"); group.measurement_time(Duration::from_secs(20)); let memory_sizes = vec![ (1024, "1KB"), (1024 * 1024, "1MB"), (10 * 1024 * 1024, "10MB"), (100 * 1024 * 1024, "100MB"), ]; for (size_bytes, label) in memory_sizes { group.throughput(Throughput::Bytes(size_bytes as u64)); group.bench_with_input( BenchmarkId::new("memory_processing", label), &size_bytes, |b, &size_bytes| { b.to_async(&rt).iter(|| async { benchmark_memory_processing(black_box(size_bytes)).await }); }, ); } group.finish(); } fn benchmark_gpu_scaling(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("gpu_scaling"); group.measurement_time(Duration::from_secs(30)); group.sample_size(10); // Fewer samples for GPU tests let gpu_configurations = vec![ (1, "single_gpu"), (2, "dual_gpu"), (4, "quad_gpu"), ]; for (gpu_count, label) in gpu_configurations { group.throughput(Throughput::Elements(100)); group.bench_with_input( BenchmarkId::new("gpu_processing", label), &gpu_count, |b, &gpu_count| { b.to_async(&rt).iter(|| async { benchmark_gpu_processing(black_box(gpu_count)).await }); }, ); } group.finish(); } fn benchmark_end_to_end_scaling(c: &mut Criterion) { let rt = Runtime::new().unwrap(); let mut group = c.benchmark_group("end_to_end_scaling"); group.measurement_time(Duration::from_secs(30)); let load_scenarios = vec![ (10, 1, "light_load"), (100, 4, "medium_load"), (500, 16, "heavy_load"), (1000, 32, "extreme_load"), ]; for (request_count, concurrency, label) in load_scenarios { group.throughput(Throughput::Elements(request_count as u64)); group.bench_with_input( BenchmarkId::new("full_pipeline", label), &(request_count, concurrency), |b, &(request_count, concurrency)| { b.to_async(&rt).iter(|| async { benchmark_full_pipeline_scaling( black_box(request_count), black_box(concurrency) ).await }); }, ); } group.finish(); } // Mock benchmark implementations async fn benchmark_sequential_requests(request_count: usize) -> usize { let mut processed = 0; for _ in 0..request_count { // Simulate request processing tokio::time::sleep(Duration::from_micros(100)).await; processed += 1; } processed } async fn benchmark_concurrent_requests(concurrency: usize, total_requests: usize) -> usize { let requests_per_task = total_requests / concurrency; let tasks: Vec<_> = (0..concurrency).map(|_| { tokio::spawn(async move { for _ in 0..requests_per_task { // Simulate request processing tokio::time::sleep(Duration::from_micros(50)).await; } requests_per_task }) }).collect(); let results = join_all(tasks).await; results.into_iter().map(|r| r.unwrap()).sum() } async fn benchmark_cpu_scaling(cpu_count: usize) -> usize { // Simulate CPU-intensive work scaling let work_per_cpu = 1000; let total_work = cpu_count * work_per_cpu; // Simulate parallel processing let sleep_time = Duration::from_micros(1000 / cpu_count as u64); tokio::time::sleep(sleep_time).await; total_work } async fn benchmark_batch_processing_throughput(batch_size: usize, total_items: usize) -> usize { let num_batches = (total_items + batch_size - 1) / batch_size; for _ in 0..num_batches { // Simulate batch processing overhead let batch_overhead = Duration::from_micros(10 + batch_size as u64 * 5); tokio::time::sleep(batch_overhead).await; } total_items } async fn benchmark_memory_processing(size_bytes: usize) -> usize { // Simulate memory-intensive processing let processing_time = Duration::from_micros((size_bytes / 1024) as u64); tokio::time::sleep(processing_time).await; size_bytes } async fn benchmark_gpu_processing(gpu_count: usize) -> usize { // Simulate GPU processing with scaling let base_work = 1000; let scaled_work = base_work * gpu_count; // GPU work scales better than linearly initially, then plateaus let efficiency = if gpu_count == 1 { 1.0 } else if gpu_count <= 4 { gpu_count as f64 * 0.9 } else { 4.0 * 0.9 + (gpu_count - 4) as f64 * 0.5 }; let processing_time = Duration::from_millis((1000.0 / efficiency) as u64); tokio::time::sleep(processing_time).await; scaled_work } async fn benchmark_full_pipeline_scaling(request_count: usize, concurrency: usize) -> usize { // Simulate full pipeline processing under load let requests_per_task = (request_count + concurrency - 1) / concurrency; let tasks: Vec<_> = (0..concurrency).map(|_| { tokio::spawn(async move { let mut processed = 0; for _ in 0..requests_per_task { // Simulate full pipeline: preprocessing, inference, postprocessing tokio::time::sleep(Duration::from_micros(200)).await; // preprocessing tokio::time::sleep(Duration::from_micros(500)).await; // inference tokio::time::sleep(Duration::from_micros(100)).await; // postprocessing processed += 1; } processed }) }).collect(); let results = join_all(tasks).await; results.into_iter().map(|r| r.unwrap()).sum() } criterion_group!( benches, benchmark_request_throughput, benchmark_concurrent_processing, benchmark_resource_scaling, benchmark_batch_throughput, benchmark_memory_scaling, benchmark_gpu_scaling, benchmark_end_to_end_scaling ); criterion_main!(benches);