Files
rustytorch/crates/specialized/rtx-cfd/tests/poisson_cache_exact.rs
T
Omar SobhandClaude Fable 5.1 79c18def32
CI / Build (macos-latest) (push) Waiting to run
CI / CI Success (push) Blocked by required conditions
CI / Test (macos-latest) (push) Blocked by required conditions
CI / Test (ubuntu-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (macos-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (ubuntu-latest) (push) Blocked by required conditions
CI / WASM Build + Size Check (push) Blocked by required conditions
CI / Distributed Training Tests (push) Blocked by required conditions
CI / Build CPU-Only (Explicit) (push) Failing after 6s
Documentation / Build API Documentation (push) Failing after 6s
Documentation / Build User Guide (push) Successful in 6s
CI / Format Check (push) Failing after 16s
CI / Build (ubuntu-latest) (push) Failing after 2m13s
CI / Clippy Check (push) Failing after 2m41s
Performance Benchmarks / Run Benchmarks (push) Successful in 3m7s
PERF-2 P1.1: the multigrid-PCG's prepared operator (hierarchy, f64 fine level, active cells, components) cached on the embedded solver and reused while the operator is bit-identical (an exact key over the coefficient bit patterns, the mask and the hierarchy parameters); solve_multigrid_pcg_cached gives the uncached solve's answer bit for bit (pin poisson_cache_exact.rs: five right-hand sides on one operator, a one-coefficient miss, both precisions); the CG driver split into Prepared::build + run_pcg
Co-Authored-By: Claude Fable 5.1 <[email protected]>
Claude-Session: https://claude.ai/code/session_01YJPeT6WA2e7YvAnS875AHL
2026-09-15 23:30:53 -05:00

118 lines
4.6 KiB
Rust

//! PERF-2 P1.1 (`docs/perf2_campaign.md`): the cached multigrid-PCG (the
//! operator's hierarchy, fine level and components kept across solves)
//! must give the uncached solve's answer BIT FOR BIT — a masked,
//! outlet-anchored problem like the overset background's, five right-hand
//! sides in a row on one operator, then a one-coefficient change that
//! must miss the cache and still match.
use rtx_cfd::solvers::incompressible::{
MgPrecision, MultigridParameters, PcgCache, PoissonProblem, solve_multigrid_pcg,
solve_multigrid_pcg_cached,
};
fn problem(nx: usize, ny: usize, seed: u64) -> PoissonProblem {
let mut p = PoissonProblem::new(nx, ny);
let (dx, dy, dt) = (1.0 / nx as f64, 0.41 / ny as f64, 1e-3);
let (ae, an) = (dt * dy / dx, dt * dx / dy);
let hole = |i: usize, j: usize| {
let (x, y) = ((i as f64 + 0.5) * dx, (j as f64 + 0.5) * dy);
(x - 0.2).powi(2) + (y - 0.2).powi(2) < 0.05 * 0.05
};
for j in 0..ny {
for i in 0..nx {
let idx = j * nx + i;
if hole(i, j) {
p.active[idx] = false;
continue;
}
if i + 1 < nx && !hole(i + 1, j) {
p.ae[idx] = ae;
}
if i > 0 && !hole(i - 1, j) {
p.aw[idx] = ae;
}
if j + 1 < ny && !hole(i, j + 1) {
p.an[idx] = an;
}
if j > 0 && !hole(i, j - 1) {
p.as_[idx] = an;
}
if i + 1 == nx {
p.extra_diag[idx] = 2.0 * ae; // the outlet
}
}
}
let mut state = seed | 1;
for idx in 0..nx * ny {
state ^= state << 13;
state ^= state >> 7;
state ^= state << 17;
p.rhs[idx] = if p.active[idx] {
1e-6 * ((state >> 11) as f64 / (1u64 << 53) as f64 - 0.5)
} else {
0.0
};
}
p
}
fn same(a: &[f64], b: &[f64]) -> bool {
a.len() == b.len() && a.iter().zip(b).all(|(x, y)| x.to_bits() == y.to_bits())
}
#[test]
fn cached_pcg_reproduces_the_uncached_solve_bit_for_bit() {
let (nx, ny) = (96, 40);
for precision in [MgPrecision::F64, MgPrecision::F32] {
let params = MultigridParameters {
precision,
..MultigridParameters::default()
};
let mut cache = PcgCache::default();
let base = problem(nx, ny, 7);
for k in 0..5u64 {
// Same operator, a new right-hand side each time.
let mut prob = base.clone();
prob.rhs = problem(nx, ny, 11 + k).rhs;
let (mut p0, mut p1) = (vec![0.0; nx * ny], vec![0.0; nx * ny]);
let s0 = solve_multigrid_pcg(&prob, &mut p0, &params, 1e-12, None);
let s1 = solve_multigrid_pcg_cached(&prob, &mut p1, &params, 1e-12, None, &mut cache);
assert_eq!(cache.len(), 1);
assert!(same(&p0, &p1), "{precision:?} rhs {k}: solutions differ");
assert_eq!(s0.iterations, s1.iterations);
assert_eq!(s0.residual.to_bits(), s1.residual.to_bits());
if k > 0 {
assert!(
s1.setup_ns < s0.setup_ns / 4,
"{precision:?} rhs {k}: cache hit should skip the setup ({} vs {} ns)",
s1.setup_ns,
s0.setup_ns
);
}
println!(
" {precision:?} rhs {k}: {} iterations, residual {:.3e}, setup {} vs {} ns",
s0.iterations, s0.residual, s0.setup_ns, s1.setup_ns
);
}
// One coefficient changes: a miss, still exact.
let mut changed = base.clone();
let idx = (ny / 2) * nx + nx / 3;
changed.ae[idx] *= 1.5;
changed.aw[idx + 1] *= 1.5;
let (mut p0, mut p1) = (vec![0.0; nx * ny], vec![0.0; nx * ny]);
let s0 = solve_multigrid_pcg(&changed, &mut p0, &params, 1e-12, None);
let s1 = solve_multigrid_pcg_cached(&changed, &mut p1, &params, 1e-12, None, &mut cache);
assert!(
same(&p0, &p1),
"{precision:?}: the changed operator's solutions differ"
);
assert_eq!(s0.iterations, s1.iterations);
// And back to the base operator: a miss again (the cache holds one), still exact.
let (mut p0, mut p1) = (vec![0.0; nx * ny], vec![0.0; nx * ny]);
let s0 = solve_multigrid_pcg(&base, &mut p0, &params, 1e-12, None);
let s1 = solve_multigrid_pcg_cached(&base, &mut p1, &params, 1e-12, None, &mut cache);
assert!(same(&p0, &p1));
assert_eq!(s0.iterations, s1.iterations);
}
}