embedded3 S2-2b: the moving path's host work 5.0 s → 0.6 s per step at 1.45 M cells, bit-identical — Problem::diagonals (the per-cell link diagonal was cells × links: 1.8 s twice per step), counting CSR + rayon in the device tables, rayon in the cut geometry builder; RTX_E3_MOVING_PROFILE laps
CI / Test (macos-latest) (push) Blocked by required conditions
CI / Test (ubuntu-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (macos-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (ubuntu-latest) (push) Blocked by required conditions
CI / WASM Build + Size Check (push) Blocked by required conditions
CI / Distributed Training Tests (push) Blocked by required conditions
CI / CI Success (push) Blocked by required conditions
CI / Build (macos-latest) (push) Waiting to run
CI / Build CPU-Only (Explicit) (push) Failing after 4s
CI / Clippy Check (push) Failing after 5s
Documentation / Build User Guide (push) Successful in 5s
CI / Format Check (push) Failing after 11s
Documentation / Build API Documentation (push) Failing after 1m34s
Performance Benchmarks / Run Benchmarks (push) Successful in 1m54s
CI / Build (ubuntu-latest) (push) Failing after 2m11s

Co-Authored-By: Claude Fable 5.1 <[email protected]>
This commit is contained in:
Omar Sobh
2026-09-18 12:48:45 -05:00
co-authored by Claude Fable 5.1
parent 99e4a7214b
commit cdd5ad0b66
7 changed files with 251 additions and 132 deletions
@@ -59,26 +59,32 @@ impl CutGeometry {
let g = grid; let g = grid;
let (nx, ny, nz, dx, dy, dz) = (g.nx, g.ny, g.nz, g.dx, g.dy, g.dz); let (nx, ny, nz, dx, dy, dz) = (g.nx, g.ny, g.nz, g.dx, g.dy, g.dz);
let n_nodes = (nx + 1) * (ny + 1) * (nz + 1); let n_nodes = (nx + 1) * (ny + 1) * (nz + 1);
// Every loop below is per-index with read-only inputs: rayon (S2-2b),
// the same arithmetic per entry.
use rayon::prelude::*;
let mut phi = vec![0.0; n_nodes]; let mut phi = vec![0.0; n_nodes];
let mut bound = vec![0.0; n_nodes]; let mut bound = vec![0.0; n_nodes];
for k in 0..=nz { phi.par_iter_mut()
for j in 0..=ny { .zip(bound.par_iter_mut())
for i in 0..=nx { .enumerate()
let n = Self::node(g, k, j, i); .for_each(|(n, (phi_n, bound_n))| {
let (k, j, i) = (
n / ((ny + 1) * (nx + 1)),
(n / (nx + 1)) % (ny + 1),
n % (nx + 1),
);
if let Some((p, band, motion)) = prev { if let Some((p, band, motion)) = prev {
let b = p.bound[n] - motion; let b = p.bound[n] - motion;
if b > band { if b > band {
phi[n] = p.phi[n]; *phi_n = p.phi[n];
bound[n] = b; *bound_n = b;
continue; return;
} }
} }
let v = body.phi(i as f64 * dx, j as f64 * dy, k as f64 * dz, t); let v = body.phi(i as f64 * dx, j as f64 * dy, k as f64 * dz, t);
phi[n] = v; *phi_n = v;
bound[n] = v.abs(); *bound_n = v.abs();
} });
}
}
let corner = |k: usize, j: usize, i: usize| phi[Self::node(g, k, j, i)]; let corner = |k: usize, j: usize, i: usize| phi[Self::node(g, k, j, i)];
// Face apertures: a face's two triangles along the diagonal from its // Face apertures: a face's two triangles along the diagonal from its
// (0, 0) to its (1, 1) corner in the face's own (a, b) order — the // (0, 0) to its (1, 1) corner in the face's own (a, b) order — the
@@ -113,51 +119,51 @@ impl CutGeometry {
let mut d_u = vec![0.0; (nx + 1) * ny * nz]; let mut d_u = vec![0.0; (nx + 1) * ny * nz];
let mut d_v = vec![0.0; nx * (ny + 1) * nz]; let mut d_v = vec![0.0; nx * (ny + 1) * nz];
let mut d_w = vec![0.0; nx * ny * (nz + 1)]; let mut d_w = vec![0.0; nx * ny * (nz + 1)];
for k in 0..nz { a_u.par_iter_mut()
for j in 0..ny { .zip(d_u.par_iter_mut())
for i in 0..=nx { .enumerate()
.for_each(|(f, (a, d))| {
// x-face at i: corners (j, k), (j+1, k), (j, k+1), (j+1, k+1) // x-face at i: corners (j, k), (j+1, k), (j, k+1), (j+1, k+1)
let (k, j, i) = (f / (ny * (nx + 1)), (f / (nx + 1)) % ny, f % (nx + 1));
let (q00, q10, q01, q11) = ( let (q00, q10, q01, q11) = (
corner(k, j, i), corner(k, j, i),
corner(k, j + 1, i), corner(k, j + 1, i),
corner(k + 1, j, i), corner(k + 1, j, i),
corner(k + 1, j + 1, i), corner(k + 1, j + 1, i),
); );
a_u[g.uface(k, j, i)] = quad_fraction(q00, q10, q01, q11); *a = quad_fraction(q00, q10, q01, q11);
d_u[g.uface(k, j, i)] = 0.25 * (q00 + q10 + q01 + q11); *d = 0.25 * (q00 + q10 + q01 + q11);
} });
} a_v.par_iter_mut()
} .zip(d_v.par_iter_mut())
for k in 0..nz { .enumerate()
for j in 0..=ny { .for_each(|(f, (a, d))| {
for i in 0..nx {
// y-face at j: corners (i, k), (i+1, k), (i, k+1), (i+1, k+1) // y-face at j: corners (i, k), (i+1, k), (i, k+1), (i+1, k+1)
let (k, j, i) = (f / ((ny + 1) * nx), (f / nx) % (ny + 1), f % nx);
let (q00, q10, q01, q11) = ( let (q00, q10, q01, q11) = (
corner(k, j, i), corner(k, j, i),
corner(k, j, i + 1), corner(k, j, i + 1),
corner(k + 1, j, i), corner(k + 1, j, i),
corner(k + 1, j, i + 1), corner(k + 1, j, i + 1),
); );
a_v[g.vface(k, j, i)] = quad_fraction(q00, q10, q01, q11); *a = quad_fraction(q00, q10, q01, q11);
d_v[g.vface(k, j, i)] = 0.25 * (q00 + q10 + q01 + q11); *d = 0.25 * (q00 + q10 + q01 + q11);
} });
} a_w.par_iter_mut()
} .zip(d_w.par_iter_mut())
for k in 0..=nz { .enumerate()
for j in 0..ny { .for_each(|(f, (a, d))| {
for i in 0..nx {
// z-face at k: corners (i, j), (i+1, j), (i, j+1), (i+1, j+1) // z-face at k: corners (i, j), (i+1, j), (i, j+1), (i+1, j+1)
let (k, j, i) = (f / (ny * nx), (f / nx) % ny, f % nx);
let (q00, q10, q01, q11) = ( let (q00, q10, q01, q11) = (
corner(k, j, i), corner(k, j, i),
corner(k, j, i + 1), corner(k, j, i + 1),
corner(k, j + 1, i), corner(k, j + 1, i),
corner(k, j + 1, i + 1), corner(k, j + 1, i + 1),
); );
a_w[g.wface(k, j, i)] = quad_fraction(q00, q10, q01, q11); *a = quad_fraction(q00, q10, q01, q11);
d_w[g.wface(k, j, i)] = 0.25 * (q00 + q10 + q01 + q11); *d = 0.25 * (q00 + q10 + q01 + q11);
} });
}
}
// Cell volumes by the Kuhn split: the six tetrahedra around the // Cell volumes by the Kuhn split: the six tetrahedra around the
// diagonal (0,0,0)(1,1,1) in unit-cube coordinates. // diagonal (0,0,0)(1,1,1) in unit-cube coordinates.
let mut vol = vec![0.0; g.cells()]; let mut vol = vec![0.0; g.cells()];
@@ -170,9 +176,7 @@ impl CutGeometry {
[[0, 0, 0], [0, 0, 1], [1, 0, 1], [1, 1, 1]], [[0, 0, 0], [0, 0, 1], [1, 0, 1], [1, 1, 1]],
[[0, 0, 0], [0, 0, 1], [0, 1, 1], [1, 1, 1]], [[0, 0, 0], [0, 0, 1], [0, 1, 1], [1, 1, 1]],
]; ];
for k in 0..nz { let cell_fluid = |k: usize, j: usize, i: usize| -> f64 {
for j in 0..ny {
for i in 0..nx {
let mut fluid = 0.0; let mut fluid = 0.0;
for tet in &KUHN { for tet in &KUHN {
let pts: Vec<[f64; 3]> = tet let pts: Vec<[f64; 3]> = tet
@@ -185,21 +189,23 @@ impl CutGeometry {
.collect(); .collect();
fluid += tet_fluid_volume(&pts, &vals); fluid += tet_fluid_volume(&pts, &vals);
} }
fluid
};
let (ax, ay, az) = (dy * dz, dx * dz, dx * dy);
vol.par_iter_mut()
.zip(wall.par_iter_mut())
.enumerate()
.for_each(|(idx, (v, w))| {
let (k, j, i) = (idx / (ny * nx), (idx / nx) % ny, idx % nx);
// The six tets fill the unit cube (volume 1). // The six tets fill the unit cube (volume 1).
let idx = g.cell(k, j, i); *v = cell_fluid(k, j, i);
vol[idx] = fluid;
let ax = dy * dz;
let ay = dx * dz;
let az = dx * dy;
// Outward normals of the cell's faces times their fluid area, // Outward normals of the cell's faces times their fluid area,
// summed; the wall closes the fluid part of the cell. // summed; the wall closes the fluid part of the cell.
let sx = (a_u[g.uface(k, j, i + 1)] - a_u[g.uface(k, j, i)]) * ax; let sx = (a_u[g.uface(k, j, i + 1)] - a_u[g.uface(k, j, i)]) * ax;
let sy = (a_v[g.vface(k, j + 1, i)] - a_v[g.vface(k, j, i)]) * ay; let sy = (a_v[g.vface(k, j + 1, i)] - a_v[g.vface(k, j, i)]) * ay;
let sz = (a_w[g.wface(k + 1, j, i)] - a_w[g.wface(k, j, i)]) * az; let sz = (a_w[g.wface(k + 1, j, i)] - a_w[g.wface(k, j, i)]) * az;
wall[idx] = [-sx, -sy, -sz]; *w = [-sx, -sy, -sz];
} });
}
}
Self { Self {
grid, grid,
phi, phi,
@@ -96,8 +96,12 @@ impl DeviceCg {
/// with a singular one among them. /// with a singular one among them.
pub fn new(problem: &Problem, params: &MultigridParameters) -> Self { pub fn new(problem: &Problem, params: &MultigridParameters) -> Self {
let rt = runtime(); let rt = runtime();
let profile = std::env::var("RTX_E3_MOVING_PROFILE").is_ok();
let lap = std::time::Instant::now();
let fine = Level::<f64>::new(problem.clone()); let fine = Level::<f64>::new(problem.clone());
let l_level = lap.elapsed();
let components = Components::find(problem, &fine.cells); let components = Components::find(problem, &fine.cells);
let l_components = lap.elapsed();
let singular_count = components.singular.iter().filter(|&&s| s).count(); let singular_count = components.singular.iter().filter(|&&s| s).count();
assert!( assert!(
singular_count == 0 || components.members.len() == 1, singular_count == 0 || components.members.len() == 1,
@@ -185,8 +189,12 @@ impl DeviceCg {
/// from the stale hierarchy get no correction rather than a stale one. /// from the stale hierarchy get no correction rather than a stale one.
pub fn refresh(&mut self, problem: &Problem, params: &MultigridParameters) { pub fn refresh(&mut self, problem: &Problem, params: &MultigridParameters) {
let rt = runtime(); let rt = runtime();
let profile = std::env::var("RTX_E3_MOVING_PROFILE").is_ok();
let lap = std::time::Instant::now();
let fine = Level::<f64>::new(problem.clone()); let fine = Level::<f64>::new(problem.clone());
let l_level = lap.elapsed();
let components = Components::find(problem, &fine.cells); let components = Components::find(problem, &fine.cells);
let l_components = lap.elapsed();
let singular_count = components.singular.iter().filter(|&&s| s).count(); let singular_count = components.singular.iter().filter(|&&s| s).count();
assert!( assert!(
singular_count == 0 || components.members.len() == 1, singular_count == 0 || components.members.len() == 1,
@@ -217,6 +225,7 @@ impl DeviceCg {
} }
link_ptr.push(link_idx.len() as u32); link_ptr.push(link_idx.len() as u32);
} }
let l_links = lap.elapsed();
self.n_cells = fine.cells.len(); self.n_cells = fine.cells.len();
self.n_blocks = self.n_cells.div_ceil(256).max(1); self.n_blocks = self.n_cells.div_ceil(256).max(1);
if self.partial.len() < self.n_blocks { if self.partial.len() < self.n_blocks {
@@ -241,11 +250,27 @@ impl DeviceCg {
}); });
self.singular = singular_count > 0; self.singular = singular_count > 0;
self.active_host = fine.active.clone(); self.active_host = fine.active.clone();
let l_upload = lap.elapsed();
self.key = OperatorKey::of(problem, params); self.key = OperatorKey::of(problem, params);
let l_key = lap.elapsed();
// The V-cycle's finest level follows the operator (its coarser // The V-cycle's finest level follows the operator (its coarser
// levels stay): the fine level alone, no hierarchy build. // levels stay): the fine level alone, no hierarchy build.
let fine_level = super::export::export_fine(problem); let fine_level = super::export::export_fine(problem);
let l_export = lap.elapsed();
self.vcycle.refresh_fine(&fine_level); self.vcycle.refresh_fine(&fine_level);
if profile {
let ms = |d: std::time::Duration| d.as_secs_f64() * 1e3;
eprintln!(
" refresh laps: level {:.0} ms, components {:.0} ms, links {:.0} ms, uploads {:.0} ms, key {:.0} ms, export_fine {:.0} ms, refresh_fine {:.0} ms",
ms(l_level),
ms(l_components - l_level),
ms(l_links - l_components),
ms(l_upload - l_links),
ms(l_key - l_upload),
ms(l_export - l_key),
ms(lap.elapsed() - l_export)
);
}
} }
pub fn n_cells(&self) -> usize { pub fn n_cells(&self) -> usize {
@@ -41,7 +41,7 @@ impl<T: MgScalar> Level<T> {
pub(crate) fn new(mut problem: Problem) -> Self { pub(crate) fn new(mut problem: Problem) -> Self {
let (nx, ny, nz) = (problem.nx, problem.ny, problem.nz); let (nx, ny, nz) = (problem.nx, problem.ny, problem.nz);
let n = nx * ny * nz; let n = nx * ny * nz;
let ap: Vec<f64> = (0..n).map(|idx| problem.diagonal(idx)).collect(); let ap: Vec<f64> = problem.diagonals();
let active: Vec<bool> = (0..n) let active: Vec<bool> = (0..n)
.map(|idx| problem.active[idx] && ap[idx] > 0.0) .map(|idx| problem.active[idx] && ap[idx] > 0.0)
.collect(); .collect();
@@ -128,6 +128,36 @@ impl Problem {
} }
} }
/// Every cell's diagonal in one pass over the links (the per-cell
/// [`Self::diagonal`] scans every link: `cells × links` on a merged
/// operator — 1.8 s per call at 1.45 M cells, S2-2b). Same sums in
/// the same order.
#[must_use]
pub fn diagonals(&self) -> Vec<f64> {
let n = self.nx * self.ny * self.nz;
let mut link_sum = vec![0.0; n];
for &(a, b, c) in &self.links {
link_sum[a] += c;
link_sum[b] += c;
}
(0..n)
.map(|idx| {
let stencil = self.ae[idx]
+ self.aw[idx]
+ self.an[idx]
+ self.as_[idx]
+ self.at[idx]
+ self.ab[idx]
+ self.extra_diag[idx];
if self.links.is_empty() {
stencil
} else {
stencil + link_sum[idx]
}
})
.collect()
}
/// Pure Neumann: no active cell has a Dirichlet contribution. /// Pure Neumann: no active cell has a Dirichlet contribution.
#[must_use] #[must_use]
pub fn is_singular(&self) -> bool { pub fn is_singular(&self) -> bool {
@@ -144,6 +174,7 @@ impl Problem {
let (nx, ny, nz) = (self.nx, self.ny, self.nz); let (nx, ny, nz) = (self.nx, self.ny, self.nz);
let links = self.link_lists(); let links = self.link_lists();
let mut sum = 0.0; let mut sum = 0.0;
let diagonals = self.diagonals();
for k in 0..nz { for k in 0..nz {
for j in 0..ny { for j in 0..ny {
for i in 0..nx { for i in 0..nx {
@@ -151,7 +182,7 @@ impl Problem {
if !self.active[idx] { if !self.active[idx] {
continue; continue;
} }
let ap = self.diagonal(idx); let ap = diagonals[idx];
if ap <= 0.0 { if ap <= 0.0 {
continue; continue;
} }
@@ -135,25 +135,30 @@ impl DeviceCut {
1 => (nx, ny + 1, nz), 1 => (nx, ny + 1, nz),
_ => (nx, ny, nz + 1), _ => (nx, ny, nz + 1),
}; };
// Per face, in parallel (S2-2b): the surface velocity within the
// band and the open flag.
use rayon::prelude::*;
let mut ubc = vec![0.0; counts[c]]; let mut ubc = vec![0.0; counts[c]];
let mut opc = vec![0i32; counts[c]]; let mut opc = vec![0i32; counts[c]];
for k in 0..nk { ubc.par_iter_mut()
for j in 0..nj { .zip(opc.par_iter_mut())
for i in 0..ni { .enumerate()
let idx = (k * nj + j) * ni + i; .for_each(|(idx, (ub_out, op_out))| {
let (k, j, i) = (idx / (nj * ni), (idx / ni) % nj, idx % ni);
let x = [ let x = [
(i as f64 + if c == 0 { 0.0 } else { 0.5 }) * h[0], (i as f64 + if c == 0 { 0.0 } else { 0.5 }) * h[0],
(j as f64 + if c == 1 { 0.0 } else { 0.5 }) * h[1], (j as f64 + if c == 1 { 0.0 } else { 0.5 }) * h[1],
(k as f64 + if c == 2 { 0.0 } else { 0.5 }) * h[2], (k as f64 + if c == 2 { 0.0 } else { 0.5 }) * h[2],
]; ];
ubc[idx] = match shared { debug_assert!(k < nk);
*ub_out = match shared {
Some(sh) => sh[c][idx], Some(sh) => sh[c][idx],
None if dists[c][idx].abs() <= band => { None if dists[c][idx].abs() <= band => {
mask.surface_velocity_at(body, x, c, t) mask.surface_velocity_at(body, x, c, t)
} }
None => 0.0, None => 0.0,
}; };
opc[idx] = i32::from(if projection { *op_out = i32::from(if projection {
match c { match c {
0 => mask.u_open(idx), 0 => mask.u_open(idx),
1 => mask.v_open(idx), 1 => mask.v_open(idx),
@@ -167,9 +172,7 @@ impl DeviceCut {
}; };
kind == FaceKind::Fluid kind == FaceKind::Fluid
}); });
} });
}
}
ub[c] = ubc; ub[c] = ubc;
open[c] = opc; open[c] = opc;
} }
@@ -201,20 +204,25 @@ impl DeviceCut {
let owner: Vec<u32> = (0..nc) let owner: Vec<u32> = (0..nc)
.map(|i| mask.master(i).unwrap_or(i) as u32) .map(|i| mask.master(i).unwrap_or(i) as u32)
.collect(); .collect();
let mut fold_ptr = Vec::with_capacity(nc + 1); // The slaves of every cell as a CSR by counting (no per-cell lists).
let mut fold_idx = Vec::new(); let mut fold_ptr = vec![0u32; nc + 1];
let mut slaves_of: Vec<Vec<u32>> = vec![Vec::new(); nc];
let mut merged = 0; let mut merged = 0;
for i in 0..nc { for i in 0..nc {
if let Some(m) = mask.master(i) { if let Some(m) = mask.master(i) {
slaves_of[m].push(i as u32); fold_ptr[m + 1] += 1;
merged += 1; merged += 1;
} }
} }
fold_ptr.push(0u32); for i in 0..nc {
for list in &slaves_of { fold_ptr[i + 1] += fold_ptr[i];
fold_idx.extend_from_slice(list); }
fold_ptr.push(fold_idx.len() as u32); let mut fold_idx = vec![0u32; merged];
let mut cursor: Vec<u32> = fold_ptr[..nc].to_vec();
for i in 0..nc {
if let Some(m) = mask.master(i) {
fold_idx[cursor[m] as usize] = i as u32;
cursor[m] += 1;
}
} }
let ub_host = ub; let ub_host = ub;
let ub_dev = [up_f(&ub_host[0]), up_f(&ub_host[1]), up_f(&ub_host[2])]; let ub_dev = [up_f(&ub_host[0]), up_f(&ub_host[1]), up_f(&ub_host[2])];
@@ -333,11 +341,27 @@ impl DeviceStep {
let t_rebuild = Instant::now(); let t_rebuild = Instant::now();
let mut fresh_cells = 0; let mut fresh_cells = 0;
if self.solver.is_moving() { if self.solver.is_moving() {
// `RTX_E3_MOVING_PROFILE=1`: the laps of the moving path (S2-2b).
let profile = std::env::var("RTX_E3_MOVING_PROFILE").is_ok();
let lap = Instant::now();
let mut field = Field::new(g); let mut field = Field::new(g);
self.download(&mut field); self.download(&mut field);
let l_down = lap.elapsed();
fresh_cells = self.solver.rebuild_moving_mask(&mut field, dt, t_new); fresh_cells = self.solver.rebuild_moving_mask(&mut field, dt, t_new);
let l_mask = lap.elapsed();
self.upload(&field); self.upload(&field);
let l_up = lap.elapsed();
self.cut = DeviceCut::build(&self.solver, g, Phase::Projection, t_new); self.cut = DeviceCut::build(&self.solver, g, Phase::Projection, t_new);
let l_tables = lap.elapsed();
if profile {
eprintln!(
" moving laps: download {:.0} ms, host mask {:.0} ms, upload {:.0} ms, projection tables {:.0} ms",
l_down.as_secs_f64() * 1e3,
(l_mask - l_down).as_secs_f64() * 1e3,
(l_up - l_mask).as_secs_f64() * 1e3,
(l_tables - l_up).as_secs_f64() * 1e3
);
}
// The operator: refreshed every step, the hierarchy every // The operator: refreshed every step, the hierarchy every
// `RTX_E3_PRECOND_REFRESH` steps (default 10; 1 = rebuild always). // `RTX_E3_PRECOND_REFRESH` steps (default 10; 1 = rebuild always).
let every: usize = std::env::var("RTX_E3_PRECOND_REFRESH") let every: usize = std::env::var("RTX_E3_PRECOND_REFRESH")
@@ -350,7 +374,14 @@ impl DeviceStep {
self.cg = None; self.cg = None;
self.steps_since_hierarchy = 0; self.steps_since_hierarchy = 0;
} else if let Some(cg) = self.cg.as_mut() { } else if let Some(cg) = self.cg.as_mut() {
let lap_op = Instant::now();
let problem = self.solver.poisson_operator(g, dt); let problem = self.solver.poisson_operator(g, dt);
if profile {
eprintln!(
" operator laps: poisson_operator {:.0} ms",
lap_op.elapsed().as_secs_f64() * 1e3
);
}
let params = MultigridParameters { let params = MultigridParameters {
precision: self.solver.params.poisson_precision, precision: self.solver.params.poisson_precision,
smoother: self.solver.params.poisson_smoother, smoother: self.solver.params.poisson_smoother,
@@ -380,6 +411,13 @@ impl DeviceStep {
self.cg_dt = dt; self.cg_dt = dt;
} }
let rebuild = t_rebuild.elapsed(); let rebuild = t_rebuild.elapsed();
if std::env::var("RTX_E3_MOVING_PROFILE").is_ok() {
eprintln!(
" moving laps: predictor block {:.0} ms, rebuild block (with the operator refresh) {:.0} ms",
t_pred.as_secs_f64() * 1e3,
rebuild.as_secs_f64() * 1e3
);
}
let anchor = self.solver.anchor_cell(g); let anchor = self.solver.anchor_cell(g);
let mut total = 0; let mut total = 0;
let mut final_residual = f64::INFINITY; let mut final_residual = f64::INFINITY;
@@ -272,6 +272,7 @@ impl Solver {
} }
} }
.map(|mut m| { .map(|mut m| {
let lap_closures = std::time::Instant::now();
m.scheme = self.params.convection_scheme; m.scheme = self.params.convection_scheme;
m.density = self.fluid.density; m.density = self.fluid.density;
m.wall_order = self.params.wall_order; m.wall_order = self.params.wall_order;
@@ -284,6 +285,12 @@ impl Solver {
if self.params.pressure_centroid { if self.params.pressure_centroid {
m.compute_gradient_weights(); m.compute_gradient_weights();
} }
if std::env::var("RTX_E3_MOVING_PROFILE").is_ok() {
eprintln!(
" mask laps: face shifts (within build_mask) {:.0} ms",
lap_closures.elapsed().as_secs_f64() * 1e3
);
}
m m
}) })
.expect("embedded mask") .expect("embedded mask")
@@ -18,7 +18,9 @@ impl Solver {
return 0; return 0;
}; };
let mut fresh_cells = 0; let mut fresh_cells = 0;
let lap = std::time::Instant::now();
let mut new_mask = self.build_mask(body, field.grid, t_new, dt); let mut new_mask = self.build_mask(body, field.grid, t_new, dt);
let l_build = lap.elapsed();
if let Some(old_mask) = &self.mask { if let Some(old_mask) = &self.mask {
fresh_cells = refill_fresh_cells(old_mask, &new_mask, field); fresh_cells = refill_fresh_cells(old_mask, &new_mask, field);
let n_in = self.params.aperture_substeps; let n_in = self.params.aperture_substeps;
@@ -48,6 +50,7 @@ impl Solver {
new_mask.set_step_apertures_with(old_mask, &refs); new_mask.set_step_apertures_with(old_mask, &refs);
} }
} }
let l_step = lap.elapsed();
new_mask.impose_from( new_mask.impose_from(
body, body,
&field.u_old, &field.u_old,
@@ -77,6 +80,15 @@ impl Solver {
}) })
.collect(); .collect();
} }
if std::env::var("RTX_E3_MOVING_PROFILE").is_ok() {
let ms = |d: std::time::Duration| d.as_secs_f64() * 1e3;
eprintln!(
" mask laps: build_mask {:.0} ms, refill + step apertures + merging {:.0} ms, impose + GCL + volumes {:.0} ms",
ms(l_build),
ms(l_step - l_build),
ms(lap.elapsed() - l_step)
);
}
self.mask = Some(new_mask); self.mask = Some(new_mask);
fresh_cells fresh_cells
} }