PERF-2 P2: the parallel maps run only on levels with ≥ 8,192 active cells and in tasks of ≥ 2,048 cells (a fork-join over a small level costs more than the map); values unchanged
CI / Build (macos-latest) (push) Waiting to run
CI / Test (macos-latest) (push) Blocked by required conditions
CI / Test (ubuntu-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (macos-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (ubuntu-latest) (push) Blocked by required conditions
CI / WASM Build + Size Check (push) Blocked by required conditions
CI / Distributed Training Tests (push) Blocked by required conditions
CI / CI Success (push) Blocked by required conditions
Performance Benchmarks / Run Benchmarks (push) Failing after 7s
CI / Format Check (push) Failing after 6s
Documentation / Build User Guide (push) Successful in 6s
CI / Build CPU-Only (Explicit) (push) Failing after 8s
CI / Clippy Check (push) Failing after 16s
CI / Build (ubuntu-latest) (push) Failing after 1m16s
Documentation / Build API Documentation (push) Failing after 1m18s
CI / Build (macos-latest) (push) Waiting to run
CI / Test (macos-latest) (push) Blocked by required conditions
CI / Test (ubuntu-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (macos-latest) (push) Blocked by required conditions
CI / Python Bindings (maturin) (ubuntu-latest) (push) Blocked by required conditions
CI / WASM Build + Size Check (push) Blocked by required conditions
CI / Distributed Training Tests (push) Blocked by required conditions
CI / CI Success (push) Blocked by required conditions
Performance Benchmarks / Run Benchmarks (push) Failing after 7s
CI / Format Check (push) Failing after 6s
Documentation / Build User Guide (push) Successful in 6s
CI / Build CPU-Only (Explicit) (push) Failing after 8s
CI / Clippy Check (push) Failing after 16s
CI / Build (ubuntu-latest) (push) Failing after 1m16s
Documentation / Build API Documentation (push) Failing after 1m18s
This commit is contained in:
@@ -604,6 +604,7 @@ impl<T: MgScalar> Level<T> {
|
||||
self.cells
|
||||
.par_iter()
|
||||
.zip(tmp[..n].par_iter_mut())
|
||||
.with_min_len(PAR_MIN_LEN)
|
||||
.for_each(|(&idx, t)| {
|
||||
*t = b[idx] - (self.ap[idx] * x[idx] - self.neighbour_sum(x, idx));
|
||||
});
|
||||
@@ -628,6 +629,7 @@ impl<T: MgScalar> Level<T> {
|
||||
colour
|
||||
.par_iter()
|
||||
.zip(tmp[..n].par_iter_mut())
|
||||
.with_min_len(PAR_MIN_LEN)
|
||||
.for_each(|(&idx, t)| {
|
||||
*t = (b[idx] + self.neighbour_sum(x_ro, idx)) / self.ap[idx];
|
||||
});
|
||||
@@ -802,10 +804,11 @@ impl<T: MgScalar> Hierarchy<T> {
|
||||
wf.x[idx] = T::ZERO;
|
||||
}
|
||||
let Work { b, x, r, tmp } = wf;
|
||||
let par_here = parallel && fine.cells.len() >= PAR_MIN_CELLS;
|
||||
for _ in 0..self.sweeps {
|
||||
fine.smooth(b, x, tmp, smoother, parallel);
|
||||
fine.smooth(b, x, tmp, smoother, par_here);
|
||||
}
|
||||
if parallel {
|
||||
if par_here {
|
||||
fine.residual_par(b, x, r, tmp);
|
||||
} else {
|
||||
fine.residual(b, x, r);
|
||||
@@ -844,8 +847,9 @@ impl<T: MgScalar> Hierarchy<T> {
|
||||
wf.x[idx] += T::from_f64(COARSE_CORRECTION) * wc.x[fine.coarse_of[idx]];
|
||||
}
|
||||
let Work { b, x, tmp, .. } = wf;
|
||||
let par_here = parallel && fine.cells.len() >= PAR_MIN_CELLS;
|
||||
for _ in 0..self.sweeps {
|
||||
fine.smooth(b, x, tmp, smoother, parallel);
|
||||
fine.smooth(b, x, tmp, smoother, par_here);
|
||||
}
|
||||
}
|
||||
for &idx in &levels[0].cells {
|
||||
@@ -976,6 +980,14 @@ impl<T: MgScalar> Prepared<T> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Threads pay only when a level is large: below this many active cells a
|
||||
/// level's maps run serially even with `threads > 1` (the fork-join of a
|
||||
/// map over a few thousand cells costs more than the map), and each task
|
||||
/// covers at least [`PAR_MIN_LEN`] cells. Both are pure scheduling: the
|
||||
/// values are the serial ones.
|
||||
const PAR_MIN_CELLS: usize = 8192;
|
||||
const PAR_MIN_LEN: usize = 2048;
|
||||
|
||||
/// Configure rayon's global thread pool for the multigrid's parallel maps
|
||||
/// (PERF-2 P2). Idempotent: a pool that already exists is kept (rayon
|
||||
/// refuses a second global pool) — the first caller decides.
|
||||
|
||||
Reference in New Issue
Block a user