From 9aaa4402a8f91389a8c27a4a3ceb2b867deb44ff Mon Sep 17 00:00:00 2001 From: Stephen Waits Date: Mon, 4 May 2026 19:36:38 -0600 Subject: [PATCH] feat: add optional `parallel` feature for population-evaluation parallelism MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds a `parallel` Cargo feature that pulls in rayon and parallelizes the only step that's actually expensive in practice — calls to `Problem::evaluate` — across the population. RNG-driven steps (parent and donor selection, variation, replacement decisions) stay serial, so seeded runs remain deterministic regardless of feature state, and the default and `--features parallel` builds produce bit-identical results. Wiring: - New `algorithms::parallel_eval::evaluate_batch` helper with two cfg-gated implementations (rayon's `into_par_iter` when the feature is on, plain `into_iter` otherwise). Both preserve input order, so pareto_front and crowding-distance decisions remain reproducible. - `RandomSearch`, `Nsga2`, and `DifferentialEvolution` now route population/offspring evaluation through the helper. NSGA-II's main loop is restructured into a serial selection-and-variation phase followed by a parallel-friendly batch evaluation phase. - DE's per-target loop is restructured into three phases (serial trial construction → batch evaluation → serial replacement). Side effect of the restructuring: DE is now the canonical synchronous DE/rand/1/bin rather than the asynchronous variant where target `i+1` sees `i`'s in-flight update. Synchronous is the textbook formulation, so this is a small correctness improvement on top of the parallelism enable. - PAES stays serial — its main loop has a sequential dependency on the current candidate and would gain nothing from rayon. Cost: algorithm impls now require `P: Sync` and `P::Decision: Send` unconditionally so a single impl serves both feature modes. This is a small bound tightening that any plain-data Problem already satisfies; in return the public `Problem` trait itself stays unchanged and the default build picks up no new dependencies. Verified: - `cargo test` and `cargo test --features parallel` both pass; the Nsga2 `deterministic_with_same_seed` test confirms reproducibility. - `cargo run --release --example benchmarks` and the same with `--features parallel` produce bit-identical ZDT1 / Rastrigin results. --- Cargo.toml | 2 + src/algorithms/differential_evolution.rs | 76 ++++++++++++------------ src/algorithms/mod.rs | 1 + src/algorithms/nsga2.rs | 32 +++++----- src/algorithms/parallel_eval.rs | 55 +++++++++++++++++ src/algorithms/random_search.rs | 11 ++-- 6 files changed, 116 insertions(+), 61 deletions(-) create mode 100644 src/algorithms/parallel_eval.rs diff --git a/Cargo.toml b/Cargo.toml index 04b4cf2..2a2d8d2 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -9,8 +9,10 @@ readme = "README.md" [features] default = [] serde = ["dep:serde"] +parallel = ["dep:rayon"] [dependencies] rand = "0.9" rand_distr = "0.5" +rayon = { version = "1", optional = true } serde = { version = "1", features = ["derive"], optional = true } diff --git a/src/algorithms/differential_evolution.rs b/src/algorithms/differential_evolution.rs index c2d5a29..1c93ef7 100644 --- a/src/algorithms/differential_evolution.rs +++ b/src/algorithms/differential_evolution.rs @@ -2,6 +2,7 @@ use rand::Rng as _; +use crate::algorithms::parallel_eval::evaluate_batch; use crate::core::candidate::Candidate; use crate::core::objective::Direction; use crate::core::population::Population; @@ -60,7 +61,7 @@ impl DifferentialEvolution { impl

Optimizer

for DifferentialEvolution where - P: Problem>, + P: Problem> + Sync, { fn run(&mut self, problem: &P) -> OptimizationResult { assert!( @@ -88,57 +89,56 @@ where use crate::traits::Initializer as _; self.bounds.initialize(n, &mut rng) }; - let mut evaluations = 0usize; - let mut evals: Vec = decisions - .iter() - .map(|d| { - let e = problem.evaluate(d); - evaluations += 1; - e.objectives[0] - }) - .collect(); + let initial_pop = evaluate_batch(problem, decisions.clone()); + let mut evaluations = initial_pop.len(); + let mut evals: Vec = + initial_pop.iter().map(|c| c.evaluation.objectives[0]).collect(); for _gen in 0..self.config.generations { - for i in 0..n { - let (r1, r2, r3) = pick_three_distinct(n, i, &mut rng); - let j_rand = rng.random_range(0..dim); - let mut trial = decisions[i].clone(); - for j in 0..dim { - let take_donor = - rng.random_bool(self.config.crossover_probability) || j == j_rand; - if take_donor { - let mutant = decisions[r1][j] - + self.config.differential_weight - * (decisions[r2][j] - decisions[r3][j]); - let (lo, hi) = self.bounds.bounds[j]; - trial[j] = mutant.clamp(lo, hi); + // Phase 1 (serial): construct one trial per target. RNG state is + // consumed in deterministic order so seeded runs reproduce + // exactly regardless of the `parallel` feature. + let trials: Vec> = (0..n) + .map(|i| { + let (r1, r2, r3) = pick_three_distinct(n, i, &mut rng); + let j_rand = rng.random_range(0..dim); + let mut trial = decisions[i].clone(); + for j in 0..dim { + let take_donor = + rng.random_bool(self.config.crossover_probability) || j == j_rand; + if take_donor { + let mutant = decisions[r1][j] + + self.config.differential_weight + * (decisions[r2][j] - decisions[r3][j]); + let (lo, hi) = self.bounds.bounds[j]; + trial[j] = mutant.clamp(lo, hi); + } } - } - let trial_obj = { - let e = problem.evaluate(&trial); - evaluations += 1; - e.objectives[0] - }; + trial + }) + .collect(); + + // Phase 2 (parallel-friendly): evaluate every trial. + let trial_cands = evaluate_batch(problem, trials); + evaluations += trial_cands.len(); + + // Phase 3 (serial): greedy replacement. + for (i, trial_cand) in trial_cands.into_iter().enumerate() { + let trial_obj = trial_cand.evaluation.objectives[0]; let target_obj = evals[i]; let trial_better = match direction { Direction::Minimize => trial_obj <= target_obj, Direction::Maximize => trial_obj >= target_obj, }; if trial_better { - decisions[i] = trial; + decisions[i] = trial_cand.decision; evals[i] = trial_obj; } } } - let final_pop: Vec>> = decisions - .into_iter() - .map(|d| { - let e = problem.evaluate(&d); - evaluations += 1; - Candidate::new(d, e) - }) - .collect(); + let final_pop: Vec>> = evaluate_batch(problem, decisions); + evaluations += final_pop.len(); let front = pareto_front(&final_pop, &objectives); let best = best_candidate(&final_pop, &objectives); OptimizationResult::new( diff --git a/src/algorithms/mod.rs b/src/algorithms/mod.rs index 3186af3..2952f7a 100644 --- a/src/algorithms/mod.rs +++ b/src/algorithms/mod.rs @@ -3,6 +3,7 @@ pub mod differential_evolution; pub mod nsga2; pub mod paes; +pub(crate) mod parallel_eval; pub mod random_search; pub use differential_evolution::*; diff --git a/src/algorithms/nsga2.rs b/src/algorithms/nsga2.rs index 2c62bbc..291ae16 100644 --- a/src/algorithms/nsga2.rs +++ b/src/algorithms/nsga2.rs @@ -2,6 +2,7 @@ use rand::Rng as _; +use crate::algorithms::parallel_eval::evaluate_batch; use crate::core::candidate::Candidate; use crate::core::population::Population; use crate::core::problem::Problem; @@ -56,7 +57,8 @@ struct Nsga2Entry { impl Optimizer

for Nsga2 where - P: Problem, + P: Problem + Sync, + P::Decision: Send, I: Initializer, V: Variation, { @@ -76,13 +78,8 @@ where n, "NSGA-II initializer must return exactly population_size decisions", ); - let mut population: Vec> = initial_decisions - .into_iter() - .map(|d| { - let e = problem.evaluate(&d); - Candidate::new(d, e) - }) - .collect(); + let population: Vec> = + evaluate_batch(problem, initial_decisions); let mut evaluations = population.len(); // Annotate the starting population with rank and crowding so the first @@ -90,9 +87,9 @@ where let mut annotated = annotate(population, &objectives); for _ in 0..self.config.generations { - // --- Parent selection + offspring generation --- - let mut offspring: Vec> = Vec::with_capacity(n); - while offspring.len() < n { + // --- Phase 1: serial parent selection + variation --- + let mut offspring_decisions: Vec = Vec::with_capacity(n); + while offspring_decisions.len() < n { let p1 = binary_tournament(&annotated, &mut rng); let p2 = binary_tournament(&annotated, &mut rng); let parents = vec![ @@ -105,14 +102,16 @@ where "NSGA-II variation returned no children", ); for child_decision in children { - if offspring.len() >= n { + if offspring_decisions.len() >= n { break; } - let eval = problem.evaluate(&child_decision); - evaluations += 1; - offspring.push(Candidate::new(child_decision, eval)); + offspring_decisions.push(child_decision); } } + // --- Phase 2: parallel-friendly batch evaluation --- + let offspring: Vec> = + evaluate_batch(problem, offspring_decisions); + evaluations += offspring.len(); // --- Combine + survival selection --- let mut combined: Vec> = Vec::with_capacity(2 * n); @@ -143,8 +142,7 @@ where break; } } - population = next; - annotated = annotate(population, &objectives); + annotated = annotate(next, &objectives); } // Return final state. diff --git a/src/algorithms/parallel_eval.rs b/src/algorithms/parallel_eval.rs new file mode 100644 index 0000000..741270c --- /dev/null +++ b/src/algorithms/parallel_eval.rs @@ -0,0 +1,55 @@ +//! Population-wide evaluation helper used by the population-based algorithms. +//! +//! Two cfg-gated implementations: +//! +//! - With the `parallel` feature: rayon's `into_par_iter` evaluates decisions +//! on the global thread pool. The result preserves input order so any +//! downstream sort/dominance/crowding decisions remain reproducible. +//! - Without the feature: plain serial `into_iter`. +//! +//! Both implementations require `P: Problem + Sync` and `P::Decision: Send`, +//! so each algorithm's `Optimizer

` impl carries the same bounds regardless +//! of feature state. This keeps the public `Problem` trait itself unchanged. + +use crate::core::candidate::Candidate; +use crate::core::problem::Problem; + +/// Evaluate every decision in `decisions` against `problem` and return the +/// resulting candidates in the same order. +#[cfg(feature = "parallel")] +pub(crate) fn evaluate_batch

( + problem: &P, + decisions: Vec, +) -> Vec> +where + P: Problem + Sync, + P::Decision: Send, +{ + use rayon::prelude::*; + decisions + .into_par_iter() + .map(|d| { + let e = problem.evaluate(&d); + Candidate::new(d, e) + }) + .collect() +} + +/// Serial fallback used when the `parallel` feature is disabled. +#[cfg(not(feature = "parallel"))] +pub(crate) fn evaluate_batch

( + problem: &P, + decisions: Vec, +) -> Vec> +where + P: Problem + Sync, + P::Decision: Send, +{ + decisions + .into_iter() + .map(|d| { + let e = problem.evaluate(&d); + Candidate::new(d, e) + }) + .collect() +} diff --git a/src/algorithms/random_search.rs b/src/algorithms/random_search.rs index 0b07627..a260863 100644 --- a/src/algorithms/random_search.rs +++ b/src/algorithms/random_search.rs @@ -3,6 +3,7 @@ //! This is the reference example for spec §2.4 / §12.1: read this file before //! writing your own optimizer. +use crate::algorithms::parallel_eval::evaluate_batch; use crate::core::candidate::Candidate; use crate::core::population::Population; use crate::core::problem::Problem; @@ -50,7 +51,8 @@ impl RandomSearch { impl Optimizer

for RandomSearch where - P: Problem, + P: Problem + Sync, + P::Decision: Send, I: Initializer, { fn run(&mut self, problem: &P) -> OptimizationResult { @@ -61,11 +63,8 @@ where for _ in 0..self.config.iterations { let decisions = self.initializer.initialize(self.config.batch_size, &mut rng); - for decision in decisions { - let eval = problem.evaluate(&decision); - evaluations += 1; - all.push(Candidate::new(decision, eval)); - } + evaluations += decisions.len(); + all.extend(evaluate_batch(problem, decisions)); } let front = pareto_front(&all, &objectives);