Initial commit

This commit is contained in:
redclawsystems
2026-03-04 00:08:42 +00:00
commit 4d88dc0584
4449 changed files with 1556714 additions and 0 deletions
@@ -0,0 +1,301 @@
//! Legacy AutoTuner for backward compatibility.
use std::collections::HashMap;
use std::time::{Duration, Instant};
use crate::error::SynthesisResult;
use crate::hardware::HardwareProfile;
use crate::templates::{KernelOperation, KernelTemplate};
use super::bayesian_tuner::BayesianAutoTuner;
use super::config::TuningConfig;
use super::gaussian_process::CacheEntry;
use super::metrics::{PerformanceMetrics, TuningResult};
use super::parameters::ParameterRange;
use super::types::SearchStrategy;
/// Legacy AutoTuner for backward compatibility
#[derive(Debug, Clone)]
pub struct AutoTuner {
hardware_profile: Option<HardwareProfile>,
cache: HashMap<String, CacheEntry>,
rng_state: u64,
}
impl AutoTuner {
pub fn new() -> SynthesisResult<Self> {
Ok(Self {
hardware_profile: None,
cache: HashMap::new(),
rng_state: 12345,
})
}
pub fn set_hardware_profile(&mut self, profile: HardwareProfile) {
self.hardware_profile = Some(profile);
}
/// Create tuning configuration based on kernel operation and hardware
pub fn create_tuning_config(
&self,
operation: &KernelOperation,
) -> SynthesisResult<TuningConfig> {
let mut config = TuningConfig::default();
match operation {
KernelOperation::Gemm { m, n, k, .. } => {
config.parameter_ranges = vec![
ParameterRange::with_candidates("block_size_m", vec![16, 32, 64, 128]),
ParameterRange::with_candidates("block_size_n", vec![16, 32, 64, 128]),
ParameterRange::with_candidates("block_size_k", vec![8, 16, 32]),
ParameterRange::with_candidates("thread_tile_m", vec![4, 8, 16]),
ParameterRange::with_candidates("thread_tile_n", vec![4, 8, 16]),
ParameterRange::with_candidates("warp_tile_m", vec![16, 32, 64]),
ParameterRange::with_candidates("warp_tile_n", vec![16, 32, 64]),
ParameterRange::with_candidates("unroll_factor", vec![1, 2, 4, 8]),
];
config.search_strategy = SearchStrategy::BayesianOptimization { iterations: 50 };
let problem_size = (*m as u64) * (*n as u64) * (*k as u64);
if problem_size > 1_000_000 {
config.timeout = Duration::from_secs(600);
config.max_iterations = 100;
}
}
KernelOperation::Convolution { .. } => {
config.parameter_ranges = vec![
ParameterRange::with_candidates("block_size_x", vec![8, 16, 32]),
ParameterRange::with_candidates("block_size_y", vec![8, 16, 32]),
ParameterRange::with_candidates("tile_size", vec![1, 2, 4, 8]),
ParameterRange::with_candidates("unroll_factor", vec![1, 2, 4]),
];
config.search_strategy = SearchStrategy::BayesianOptimization { iterations: 30 };
}
KernelOperation::Attention {
sequence_length, ..
} => {
config.parameter_ranges = vec![
ParameterRange::with_candidates("block_size", vec![64, 128, 256, 512]),
ParameterRange::with_candidates("head_block_size", vec![1, 2, 4, 8]),
ParameterRange::with_candidates("sequence_tile", vec![32, 64, 128]),
];
config.search_strategy = SearchStrategy::BayesianOptimization { iterations: 100 };
if *sequence_length > 2048 {
config.timeout = Duration::from_secs(900);
}
}
KernelOperation::Reduction { .. } => {
config.parameter_ranges = vec![
ParameterRange::with_candidates("block_size", vec![128, 256, 512, 1024]),
ParameterRange::with_candidates("items_per_thread", vec![1, 2, 4, 8]),
ParameterRange::with_candidates("unroll_factor", vec![1, 2, 4]),
];
config.search_strategy = SearchStrategy::BayesianOptimization { iterations: 20 };
config.timeout = Duration::from_secs(120);
}
KernelOperation::Elementwise { .. } => {
config.parameter_ranges = vec![
ParameterRange::with_candidates("block_size", vec![256, 512, 1024]),
ParameterRange::with_candidates("elements_per_thread", vec![1, 2, 4, 8, 16]),
];
config.search_strategy = SearchStrategy::BayesianOptimization { iterations: 15 };
config.timeout = Duration::from_secs(60);
}
}
Ok(config)
}
/// Generate a cache key for the operation and hardware
fn generate_cache_key(&self, operation: &KernelOperation) -> String {
let hw_hash = self.hardware_profile.as_ref().map_or_else(
|| "unknown_hw".to_string(),
|p| {
format!(
"{}_{}.{}",
p.architecture, p.compute_capability.0, p.compute_capability.1
)
},
);
match operation {
KernelOperation::Gemm {
m,
n,
k,
transpose_a,
transpose_b,
} => {
format!("gemm_{m}x{n}x{k}_ta{transpose_a}_tb{transpose_b}_{hw_hash}")
}
KernelOperation::Convolution {
batch_size,
in_channels,
out_channels,
height,
width,
kernel_size,
} => {
format!(
"conv_{batch_size}x{in_channels}x{out_channels}x{height}x{width}x{kernel_size}_{hw_hash}"
)
}
KernelOperation::Attention {
sequence_length,
head_dim,
num_heads,
} => {
format!("attention_{sequence_length}x{head_dim}x{num_heads}_{hw_hash}")
}
KernelOperation::Reduction {
operation: op,
input_size,
axis,
} => {
format!("reduction_{op:?}_{input_size}x{axis}_{hw_hash}")
}
KernelOperation::Elementwise {
operation: op,
size,
} => {
format!("elementwise_{op:?}_{size}_{hw_hash}")
}
}
}
/// Simulate kernel execution and return performance metrics
fn profile_kernel_performance(
&mut self,
_template: &KernelTemplate,
parameters: &HashMap<String, i32>,
) -> SynthesisResult<PerformanceMetrics> {
let block_size = parameters
.get("block_size")
.or(parameters.get("block_size_m"))
.unwrap_or(&128);
let unroll_factor = parameters.get("unroll_factor").unwrap_or(&4);
let base_gflops = if let Some(ref hw) = self.hardware_profile {
hw.peak_flops.fp32 * 0.7
} else {
1000.0
};
let block_efficiency = match block_size {
64 => 0.85,
128 => 0.95,
256 => 0.90,
512 => 0.80,
_ => 0.75,
};
let unroll_efficiency = match unroll_factor {
1 => 0.70,
2 => 0.85,
4 => 0.95,
8 => 0.90,
_ => 0.80,
};
let gflops = base_gflops * block_efficiency * unroll_efficiency;
let execution_time = Duration::from_nanos((1_000_000.0 / gflops) as u64 * 1000);
let metrics = PerformanceMetrics {
execution_time,
gflops,
memory_bandwidth: gflops * 8.0,
occupancy: block_efficiency,
energy_efficiency: gflops / 300.0,
};
Ok(metrics)
}
/// Simple LCG for reproducible random numbers
fn next_random(&mut self) -> u32 {
self.rng_state = (self
.rng_state
.wrapping_mul(1_103_515_245)
.wrapping_add(12345))
& 0x7fffffff;
(self.rng_state >> 16) as u32
}
/// Main autotuning entry point (legacy interface)
pub async fn tune_kernel(
&mut self,
template: &KernelTemplate,
operation: &KernelOperation,
) -> SynthesisResult<TuningResult> {
let _start_time = Instant::now();
// Check cache first
let cache_key = self.generate_cache_key(operation);
if let Some(cached_entry) = self.cache.get(&cache_key) {
return Ok(TuningResult {
optimal_parameters: cached_entry.parameters.clone(),
best_metrics: cached_entry.metrics.clone(),
iterations_completed: 0,
total_time: Duration::from_nanos(1),
convergence_achieved: true,
pareto_frontier: None,
acquisition_history: None,
});
}
let config = self.create_tuning_config(operation)?;
// Use BayesianAutoTuner for actual optimization
let mut bayesian_tuner = BayesianAutoTuner::new()?;
if let Some(ref hw_profile) = self.hardware_profile {
bayesian_tuner.set_hardware_profile(hw_profile.clone())?;
}
let result = bayesian_tuner.tune_parameters(template, operation, &config)?;
// Cache the result if enabled
if config.cache_results {
let hw_hash = self.hardware_profile.as_ref().map_or(0, |p| {
use std::collections::hash_map::DefaultHasher;
use std::hash::{Hash, Hasher};
let mut hasher = DefaultHasher::new();
p.architecture.hash(&mut hasher);
hasher.finish()
});
let cache_entry = CacheEntry {
parameters: result.optimal_parameters.clone(),
metrics: result.best_metrics.clone(),
timestamp: Duration::from_nanos(1),
hardware_hash: hw_hash,
};
self.cache.insert(cache_key, cache_entry);
}
Ok(result)
}
/// Clear the autotuning cache
pub fn clear_cache(&mut self) {
self.cache.clear();
}
/// Get cache statistics
pub fn cache_stats(&self) -> (usize, usize) {
let total_entries = self.cache.len();
let expired_entries = self
.cache
.values()
.filter(|entry| entry.timestamp > Duration::from_secs(86400))
.count();
(total_entries, expired_entries)
}
}