diff --git a/AGENTS.md b/AGENTS.md index 8c61512c..6b03a0b7 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -136,7 +136,7 @@ Stops: `Stop::target(score)` (at least as good), `generations(n)`, `evaluations( - Maximize is the default; use `.minimize()`, don't negate. - `None`, `Fitness::invalid()` and NaN are invalid: worse than everything. - **Constraints:** return `(score, violation)`, 0 when feasible, adding up `constraint::at_most(value, limit)`, `at_least`, `equal(value, target, tolerance)`. Deb's rules: feasible beats infeasible, then score or violation decides. Select with `Tournament` or `Rank`: roulette and SUS give infeasible solutions no weight. `Penalty::new(weight)?.fitness(objective, score, violation)` is a static penalty instead. -- **Test problems:** `problems::{Sphere, AxisParallelEllipsoid, Schwefel1_2, Rastrigin, Rosenbrock, Ackley, Griewank, Schwefel2_26, Levy, Zakharov, StyblinskiTang, Michalewicz, Schwefel2_21, Schwefel2_22, DixonPrice, Trid, Powell, SumOfDifferentPowers, Step, Quartic, Penalized1, Penalized2, HighConditionedElliptic, BentCigar, Discus, DifferentPowers, BucheRastrigin, NonContinuousRastrigin, Weierstrass, Katsuura, HappyCat, HgBat, SchafferF7, RotatedHyperEllipsoid}::new(n)` (`Powell` takes a multiple of 4; `Quartic::noisy(n)` adds noise drawn from the genome) and `problems::{Himmelblau, Branin, GoldsteinPrice, SixHumpCamel, Hartmann3, Hartmann6, Shekel5, Shekel7, Shekel10, Easom, Eggholder, SchafferF6, Beale, Booth, Matyas, Bohachevsky1, Bohachevsky2, Bohachevsky3, ThreeHumpCamel, Langermann, ShekelFoxholes, Kowalik}` are fitness functions for `Engine::new(algorithm, problem)`, all minimized. The `problems::Problem` trait gives `representation()` (the bounds), `optimum()` (`value()`, `solutions()`), `reference()`; `problems::all()` lists them as `Box`. `problems::Shifted::new(problem, seed)` and `problems::Rotated::new(problem, seed)` make CEC/BBOB-style instances of any of them (e.g. CEC 2005's F10: `Rotated::new(Shifted::new(Rastrigin::new(n), seed), seed)`), keeping the optimum's value and what the wrapped problem provides: its gradient (by the chain rule through the rotation), constraint values and their Jacobian. Constrained, with fitness `(score, violation)` and `constraints(&x)` (`g <= 0`, then `h = 0`): `problems::cec2006::{G01, …, G24}` (equalities met within `EQUALITY_TOLERANCE` = 1e-4; `with_tolerance(δ)` for the problems with equalities, e.g. `G03::with_tolerance(δ)`), and `problems::engineering::{WeldedBeam, WeldedBeamRagsdell, PressureVessel, TensionCompressionSpring, SpeedReducer, ThreeBarTruss, CantileverBeam, CarSideImpact}`. `PressureVessel` and `SpeedReducer` round their discrete genes when evaluated; `design(&x)` gives the rounded design. `engineering::GearTrain` has an `Integer` genome and isn't in `all()`. `Optimum::is_proven()` is false for a best known value. +- **Test problems:** `problems::{Sphere, AxisParallelEllipsoid, Schwefel1_2, Rastrigin, Rosenbrock, Ackley, Griewank, Schwefel2_26, Levy, Zakharov, StyblinskiTang, Michalewicz, Schwefel2_21, Schwefel2_22, DixonPrice, Trid, Powell, SumOfDifferentPowers, Step, Quartic, Penalized1, Penalized2, HighConditionedElliptic, BentCigar, Discus, DifferentPowers, BucheRastrigin, NonContinuousRastrigin, Weierstrass, Katsuura, HappyCat, HgBat, SchafferF7, RotatedHyperEllipsoid}::new(n)` (`Powell` takes a multiple of 4; `Quartic::noisy(n)` adds noise drawn from the genome) and `problems::{Himmelblau, Branin, GoldsteinPrice, SixHumpCamel, Hartmann3, Hartmann6, Shekel5, Shekel7, Shekel10, Easom, Eggholder, SchafferF6, Beale, Booth, Matyas, Bohachevsky1, Bohachevsky2, Bohachevsky3, ThreeHumpCamel, Langermann, ShekelFoxholes, Kowalik}` are fitness functions for `Engine::new(algorithm, problem)`, all minimized. The `problems::Problem` trait gives `representation()` (the bounds), `optimum()` (`value()`, `solutions()`), `reference()`; `problems::all()` lists them as `Box`. `problems::Shifted::new(problem, seed)` and `problems::Rotated::new(problem, seed)` make CEC/BBOB-style instances of any of them (e.g. CEC 2005's F10: `Rotated::new(Shifted::new(Rastrigin::new(n), seed), seed)`), keeping the optimum's value and what the wrapped problem provides: its gradient (by the chain rule through the rotation), constraint values and their Jacobian. Constrained, with fitness `(score, violation)` and `constraints(&x)` (`g <= 0`, then `h = 0`): `problems::cec2006::{G01, …, G24}` (equalities met within `EQUALITY_TOLERANCE` = 1e-4; `with_tolerance(δ)` for the problems with equalities, e.g. `G03::with_tolerance(δ)`), and `problems::engineering::{WeldedBeam, WeldedBeamRagsdell, PressureVessel, TensionCompressionSpring, SpeedReducer, ThreeBarTruss, CantileverBeam, CarSideImpact}`. `PressureVessel` and `SpeedReducer` round their discrete genes when evaluated; `design(&x)` gives the rounded design. `engineering::GearTrain` has an `Integer` genome and isn't in `all()`. `Optimum::is_proven()` is false for a best known value. On bit strings (`Binary` genomes, `Bits`), maximized, not in `all()`: `problems::binary::{OneMax, LeadingOnes}::new(n)`, `Trap::new(blocks, k)` (Deb and Goldberg's trap: a = k − 1, b = k, z = k − 1; `with_values(blocks, k, a, b, z)?`), `RoyalRoad::{r1, r2}()` (`new(blocks, size)`, `hierarchical(blocks, size)`), `NkLandscape::new(n, k, Neighborhood::{Adjacent, Random}, seed)?` and the 0/1 knapsack, `Knapsack::generator(KnapsackClass::..., items).seed(s).generate()?` (Pisinger's classes, fitness `(profit, violation)`; `Knapsack::new(weights, profits, capacity)?` for given items); the NK landscapes' and knapsacks' `optimum()` is computed exactly (dynamic programming, or every string of a small NK landscape), `None` when too large. Python: `gx.problems.binary`. - **Extras:** return `Evaluated::new(value, info)` (`value` any of the above, `info` any `Send + Sync + 'static` type, e.g. a struct with a penalty's terms) to keep what the fitness function computed. Read it by type: `outcome.best_info::()`, `snapshot.info::(genome)` and `snapshot.best_info::()` in `.on_generation`, `hall_of_fame.info::(genome)`, and in `MultiEngine` `snapshot.info` and `outcome.info(genome)` for the front; `None` for another type. Never used by the search. Kept by genome for the population, the discarded and the best (copies share it); not in checkpoints. - **Gradients:** `Differentiable(|x: &Reals, gradient: &mut [f64]| value)`, for gradient-based methods; see [Gradients](#gradients-supplying-them). - **Constraint values:** `Constrained::new(m, |x: &Reals, g: &mut [f64]| score)` writes the values of m constraints `gᵢ(x) <= 0`; `Constrained::differentiable(m, |x, gradient, g, jacobian| score)` also the gradient and the Jacobian (`jacobian[i * n + j]` = ∂gᵢ/∂xⱼ). Either is `(score, Σ max(0, gᵢ))` for any algorithm, and gives the values one by one to those that use them (`Mma`). The CEC 2006 problems with inequalities only and the engineering problems give their values (`problem.provides().inequalities`), shifted or rotated too, not their gradients. diff --git a/docs/features.md b/docs/features.md index 03fbbc5e..9cc6d768 100644 --- a/docs/features.md +++ b/docs/features.md @@ -74,6 +74,7 @@ What genoxide has on main; [docs.rs](https://docs.rs/genoxide) documents the lat - **Integer genomes:** the genes rounded inside the kernel (Garrido-Merchán and Hernández-Lobato 2020), the acquisition maximized on the lattice by a hill climb (every point of a small lattice at once), the run finished once the lattice is evaluated. - **Gaussian processes** (`model::gp`, unstable for one release): regression with a constant mean, an ARD Matérn 5/2 or squared exponential kernel, no noise by default (the model interpolates a deterministic function) or learned noise, inputs scaled to the unit cube and outputs standardized; hyperparameters by maximum marginal likelihood (Rasmussen and Williams, eq. 2.30 and 5.9) with genoxide's L-BFGS-B from fixed and seeded random starts; the posterior mean and variance with their gradients; Cholesky factorizations with a growing jitter. In Python as `gx.model.gp`. - **Test problems** (`problems`): Sphere, the axis-parallel ellipsoid, Schwefel 1.2 and 2.26, Rastrigin, Rosenbrock, Ackley, Griewank, Levy, Zakharov, Styblinski-Tang, Michalewicz, Himmelblau, Branin, Goldstein-Price, the six-hump camel, Hartmann's functions in 3 and 6 dimensions, Shekel's with 5, 7 and 10 wells, Easom, the eggholder, Schaffer's F6, Schwefel 2.21 and 2.22, Dixon-Price, Trid, Powell's singular function, Beale, Booth, Matyas, Bohachevsky's three functions, the three-hump camel, Langermann, Shekel's foxholes, Kowalik, the sum of different powers, the step function, the quartic (with or without noise), Yao, Liu and Lin's two penalized functions, the high-conditioned elliptic, the bent cigar, the discus, BBOB's different powers, Büche-Rastrigin, the non-continuous Rastrigin, Weierstrass, Katsuura, HappyCat, HGBat, Schaffer's F7 and the rotated hyper-ellipsoid, each with its bounds, known optimum (or best known, for those found numerically) and reference, in Rust and Python, and the analytic gradient of all but seven (the eggholder, Schwefel 2.21 and 2.22, the step function, the non-continuous Rastrigin, Katsuura and the noisy quartic, whose derivative is undefined or 0 on sets of positive measure); and shift and rotation wrappers, generated from a seed, for CEC- and BBOB-style instances of any of them, which pass on the wrapped problem's gradient (by the chain rule), constraint values and Jacobian. +- **Binary and combinatorial test problems** (`problems::binary`), maximized, on bit strings: OneMax and LeadingOnes (Droste, Jansen and Wegener), Deb and Goldberg's deceptive trap in blocks of any size, with any slopes (Ackley's included), Mitchell, Forrest and Holland's royal roads R1 and R2 and their forms of any size, Kauffman and Weinberger's NK landscapes with adjacent or random neighbors, and the 0/1 knapsack with Pisinger's eleven generated instance classes (uncorrelated, weakly, strongly, inverse strongly and almost strongly correlated, subset sum, similar weights, spanner, multiple strongly correlated, profit ceiling and circle) or given items. NK landscapes and knapsack instances are drawn from a seed with the portable random numbers, the same on every platform, and their optimum is computed exactly: by dynamic programming (the knapsack, and NK landscapes with adjacent neighbors) or by evaluating every string in Gray code order (small NK landscapes). In Rust and Python. - **Constrained test problems:** CEC 2006's g01-g24 (`problems::cec2006`), and the engineering design problems (`problems::engineering`): the welded beam in two forms, the pressure vessel, the tension/compression spring, the speed reducer, the gear train (integer), the three-bar truss, the cantilever beam and the car side impact, each with its optimum or best known solution and references, in Rust and Python. Those with inequalities only give their constraints' values one by one to the algorithms that use them. ## Multi-objective @@ -120,7 +121,12 @@ Each example in [examples/](../examples/) is a folder with the same program in R ```text cargo run --release --example one_max # binary genome, GA -cargo run --release --example knapsack # a constraint with Deb's feasibility rules +cargo run --release --example leading_ones # the (1+1) evolutionary algorithm, Θ(n²) steps +cargo run --release --example deceptive_trap # fully deceptive blocks, solved by two-point crossover +cargo run --release --example royal_road_r1 # hill climbing beats a GA on the royal road R1 +cargo run --release --example royal_road_r2 # the hierarchical royal road R2, a GA +cargo run --release --example nk_landscape # an NK landscape, iterated local search against exhaustive search +cargo run --release --example knapsack # Pisinger's knapsack instances, Deb's rules and dynamic programming cargo run --release --example n_queens # permutation, (μ+λ) cargo run --release --example tsp_berlin52 # TSPLIB berlin52, simulated annealing with 2-opt moves cargo run --release --example jobshop_ft06 # job shop ft06, permutation with repetition diff --git a/docs/problems-plan.md b/docs/problems-plan.md index 45bb4b02..0684feb1 100644 --- a/docs/problems-plan.md +++ b/docs/problems-plan.md @@ -127,12 +127,42 @@ exact suites need their data files, which is a separate decision. | Problem | Original | Optimum | Status | |---|---|---|---| -| OneMax | folklore; analyzed by Mühlenbein, H. (1992). How genetic algorithms really work: mutation and hillclimbing. PPSN II | n at 1ⁿ | U (no single origin) | -| LeadingOnes | Rudolph, G. (1997). *Convergence Properties of Evolutionary Algorithms.* Kovač, Hamburg | n at 1ⁿ | U | -| Deceptive trap (k bits per block) | Ackley (1987); Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. FOGA 2: 93-108 | n at 1ⁿ (attractor 0ⁿ) | U | -| Royal road R1, R2 | Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms. Proc. 1st ECAL: 245-254; Forrest and Mitchell (1993), FOGA 2 | 64 at 1⁶⁴ | U | -| NK landscapes | Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and its application to maturation of the immune response. *J. Theor. Biol.* 141(2): 211-245. doi:10.1016/S0022-5193(89)80019-0 | instance-specific; exact by dynamic programming for adjacent neighborhoods | U | -| 0/1 knapsack instance classes | Pisinger, D. (2005). Where are the hard knapsack problems? *Computers & Operations Research* 32(9): 2271-2284. doi:10.1016/j.cor.2004.03.002 | from the generated instance (dynamic programming in the test) | U | +| OneMax | folklore (no single origin); Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary algorithm. *TCS* 276(1-2): 51-81. doi:10.1016/S0304-3975(01)00182-7, Definition 9; Ackley (1987, section 3.3.1) tests ten times it; Mühlenbein (1992), PPSN II, analyzes it (unread) | n at 1ⁿ | **VO** (Droste et al.) | +| LeadingOnes | Droste et al. (2002), Definition 16 and Theorem 17, from Rudolph, G. (1997). *Convergence Properties of Evolutionary Algorithms.* Kovač, Hamburg (unread), whom they credit | n at 1ⁿ | **VO** (Droste et al.) | +| Deceptive trap (k bits per block) | Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. FOGA 2: 93-108. doi:10.1016/B978-0-08-094832-4.50012-X, equation 1; after Ackley, D. H. (1987). *A Connectionist Machine for Genetic Hillclimbing.* Kluwer. doi:10.1007/978-1-4613-1997-9, section 3.3.3 | blocks × b at 1ⁿ (attractor 0ⁿ) | **VO** (both); the sum over blocks isn't written out in either | +| Royal road R1, R2 | R2: Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms. Proc. 1st ECAL: 245-254, Figure 1; R1: Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a genetic algorithm outperform hill climbing? NIPS 6: 51-58, Figure 1; Forrest and Mitchell (1993), FOGA 2 (unread) | 64 (R1), 256 (R2) at 1⁶⁴ | **VO** | +| NK landscapes | Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and its application to maturation of the immune response. *J. Theor. Biol.* 141(2): 211-245. doi:10.1016/S0022-5193(89)80019-0 | instance-specific: exact by dynamic programming (adjacent neighborhoods) or exhaustive search (N up to about 25) | **VO** | +| 0/1 knapsack instance classes | Pisinger, D. (2005). Where are the hard knapsack problems? *Computers & Operations Research* 32(9): 2271-2284. doi:10.1016/j.cor.2004.03.002 | instance-specific: exact by dynamic programming | **VO** (the classes; the generator's random numbers are genoxide's) | + +**Checked in batch 12** (the binary and combinatorial problems, `problems::binary`). Read: Droste, +Jansen and Wegener (2002): ONEMAX as the linear function with all weights 1 (Definition 9, and the +Θ(n log n) of Lemma 10), LEADINGONES as `Σᵢ Πⱼ≤ᵢ xⱼ` (Definition 16, from Rudolph's 1997 example, +which they credit) and its Θ(n²), at most e n² (Theorem 17); Rudolph's book and Mühlenbein (1992) +weren't available. Ackley (1987): "One Max" is 10 times the number of ones (section 3.3.1), and his +"Trap" (section 3.3.3) is `(8n/z)(z − c)` for c ≤ z = ⌊3n/4⌋, else `(10n/(n − z))(c − z)`, which is +Deb and Goldberg's equation 1 with a = 8n, b = 10n. Deb and Goldberg (1993), rendered pages: +equation 1, Theorem 1 (full deception iff every order-(ℓ − 1) schema is misleading), inequality 16 +(`r ≥ (2 − 1/(ℓ − z)) / (2 − 1/z)`) and eq. 20 (`r_min = (ℓ − 1)/(2ℓ − 3)` at z = ℓ − 1). genoxide's +default a = k − 1, b = k, z = k − 1 has r = (k − 1)/k ≥ r_min, fully deceptive for k ≥ 3, checked by +averaging over every schema of a block. The paper says Ackley's traps are "fully deceptive only for +ℓ < 7"; by its own inequality 16, and by enumerating the schemas, only those of 3 and 4 bits are +(5 and 6 aren't); its conclusion that none of Ackley's sizes (8 to 20) is fully deceptive stands. +The concatenated form, blocks of k adjacent bits summed, isn't written out in the paper. Mitchell, +Forrest and Holland (1992): Figure 1's 15 schemas, c_s = order(s), the optimum 256, and Table 1 (GA +mean 590, median 542 generations over 50 runs, with population 128, single-point crossover 0.7, +mutation 0.005 and sigma scaling); Mitchell, Holland and Forrest (1994): Figure 1's R1, Table 1 +(RMHC mean 6,179 and median 5,775 evaluations, GA 61,334 and 54,208) and equation 1 (E(8, 8) ≈ +6,549). Kauffman and Weinberger (1989): two states per site, contributions uniform on (0, 1) for +the 2^(K+1) combinations, W the mean (equation 1), K/2 flanking neighbors on each side on a circle +(Table 1) or K random ones (Table 2); genoxide gives an odd K one more neighbor after the site. +Pisinger (2005), rendered pages: the classes of section 3 (uncorrelated, weakly, strongly, inverse +strongly and almost strongly correlated, subset sum, similar weights) and 3.3 (span(v, m) with +⌈2p/m⌉ and ⌈2w/m⌉, mstr(k₁, k₂, d), pceil(d) = d⌈w/d⌉, circle(d) = d √(4R² − (w − 2R)²)), and the +capacity of eq. 5; the paper doesn't say how "p ≥ 1" is kept in the weakly correlated class +(genoxide narrows the interval) or how the circle's profits are rounded (genoxide rounds down, +exactly). The optima: NK by dynamic programming over the circle (adjacent) or exhaustive search in +Gray code order, checked against brute force; the knapsack by Bellman's recursion, checked against +brute force for every class. DOIs checked with Crossref. **Pitfalls to settle per function when implementing:** Schwefel's numbering (1.2, 2.21, 2.22, 2.26, and the shifted 418.9829 n form); Ackley's bounds; Griewank's divisor and domain; the @@ -941,7 +971,7 @@ the papers report. No front file from another project is used. | Group | Core problems | Optional | Verified in the original (or its standard report) | Unverified or secondary only | |---|---|---|---|---| -| Single objective, unconstrained | 59 continuous (18 scalable unimodal, 21 scalable multimodal, 20 fixed-dimension) | 6 binary and combinatorial; whole CEC/BBOB suites | Goldstein-Price; the CEC 2005 and BBOB forms; citations of Rosenbrock, Griewank, Rastrigin (1991), Styblinski-Tang | most: the originals are books and reports that aren't online; the common forms are secondary (Yao, Liu and Lin 1999; CEC 2005) | +| Single objective, unconstrained | 59 continuous (18 scalable unimodal, 21 scalable multimodal, 20 fixed-dimension) | 6 binary and combinatorial (done in batch 12); whole CEC/BBOB suites | Goldstein-Price; the CEC 2005 and BBOB forms; citations of Rosenbrock, Griewank, Rastrigin (1991), Styblinski-Tang; the 6 binary and combinatorial problems | most: the originals are books and reports that aren't online; the common forms are secondary (Yao, Liu and Lin 1999; CEC 2005) | | Single objective, constrained (CEC 2006) | 24 | | all 24 (the September 2006 report) | g17's better value, g22's claimed better value, g04's variant, the "max" origins | | Multi objective, unconstrained | 25 (SCH1, SCH2, FON, KUR, POL, VNT1-3, ZDT5, DTLZ5-7, scaled DTLZ1/2, convex DTLZ2, inverted DTLZ1, WFG1-9) | MaF1-15, UF1-10, Deb's 1999 problems | ZDT5, DTLZ5-7, WFG (batch 4), scaled/convex/inverted DTLZ (batch 8) | Kursawe, Poloni, Viennet, SCH2's formula; FON secondary | | Multi objective, constrained | 45 (CONSTR, SRN, TNK, BNH, OSY, CTP1-8, C-DTLZ ×6, MW1-14, DAS-CMOP1-9, DTLZ8-9) | LIR-CMOP1-14, CF1-10, DC-DTLZ ×6, Viennet 4 / MOP-C | CONSTR, SRN, TNK (as NSGA-II restates them), CTP1-8 (the paper, its KanGAL report and Deb's book; g, n and bounds from the authors' code), C-DTLZ, MW, DAS-CMOP, DTLZ8-9, and the optional LIR-CMOP, CF, DC-DTLZ | BNH, OSY; DC2/DC3 parameters | @@ -1292,7 +1322,7 @@ page); a comparison belongs on the problems' own pages. | 10a | done | Remaining low-dimensional and classic scalable functions (section 1.1's "Checked in batch 10a") | Beale, Booth, Matyas, Bohachevsky 1-3, Three-hump camel, Dixon-Price, Trid, Powell, Langermann, Shekel's foxholes, Kowalik, Schwefel 2.21, Schwefel 2.22 (15) | an example per function: `beale`, `booth`, `matyas`, `bohachevsky1` to `bohachevsky3`, `three_hump_camel`, `langermann`, `shekel_foxholes` and `kowalik` (30 seeds each of CMA-ES with and without IPOP, DE, PSO or a GA), `dixon_price` (CMA-ES with IPOP, DE and PSO in 5 and 10 dimensions), and `schwefel_2_21`, `schwefel_2_22`, `trid` and `powell` (CMA-ES, sep-CMA-ES, DE, PSO and a GA to errors of 1 … 1e-8) | | 10b | done | CEC and BBOB-style functions, and the shift / rotation wrappers (section 1.1's "Checked in batch 10b") | `Shifted

`, `Rotated

`; Sum of different powers, Step, Quartic (deterministic: without noise, or with noise seeded from the genome, since fitness functions must be deterministic), Penalized 1 and 2, High-conditioned elliptic, Bent cigar, Discus, Büche-Rastrigin, Non-continuous Rastrigin, Weierstrass, Katsuura, HappyCat, HGBat, Schaffer F7, Rotated hyper-ellipsoid, BBOB different powers (17; the shifted and rotated Rastrigin of CEC 2005 and BBOB are the wrappers around `Rastrigin`) | an example per function: `sum_of_different_powers`, `step`, `quartic` (with and without noise), `rotated_hyper_ellipsoid`, `high_conditioned_elliptic`, `bent_cigar`, `discus` and `different_powers` (CMA-ES, sep-CMA-ES, DE, PSO and a GA to errors of 1 … 1e-8, as they are and shifted and rotated, or rotated), and `schaffer_f7`, `penalized1`, `penalized2`, `buche_rastrigin`, `non_continuous_rastrigin`, `weierstrass`, `katsuura`, `happy_cat` and `hg_bat` (10 seeds each of CMA-ES with and without IPOP, DE, PSO and a GA) | | 11 | done | Advanced constrained multi-objective suites (sections 1.3's and 1.4's "Checked in batch 11") | DAS-CMOP1-9 (with the 16 difficulty triplets as a parameter), DC-DTLZ (DC1-DC3 on DTLZ1/DTLZ3), DTLZ8, DTLZ9 (17) | an example per problem: `das_cmop1` to `das_cmop3` (MOEA/D-DE, and NSGA-II with the paper's settings), `das_cmop4` to `das_cmop6` (NSGA-II, the paper's settings and η = 5), `das_cmop7`, `das_cmop8` (NSGA-II and NSGA-III), `das_cmop9` (MOEA/D-DE, at its 300 weights' limit, and NSGA-II), `dc1_dtlz1_3obj`, `dc1_dtlz3_3obj` (NSGA-III and SMS-EMOA), `dc2_dtlz1_3obj`, `dc2_dtlz3_3obj`, `dc3_dtlz1_3obj`, `dc3_dtlz3_3obj` (NSGA-III with constrained dominance, and without the constraints, or the one on g), `dtlz8_3obj`, `dtlz9_3obj` (NSGA-II for the report's 500 generations, and SMS-EMOA) | -| 12 | | Binary and combinatorial problems | OneMax, LeadingOnes, deceptive trap, royal road, NK landscapes (seeded), 0/1 knapsack (generated instance classes) (6) | an example per problem; `one_max` and `knapsack` switch to the problems | +| 12 | done | Binary and combinatorial problems (`problems::binary`, section 1.1's "Checked in batch 12") | OneMax, LeadingOnes, deceptive trap, royal road R1 and R2, NK landscapes (seeded), 0/1 knapsack (Pisinger's generated instance classes) (6) | an example per problem: `one_max` (switched to `binary::OneMax`; GA), `leading_ones` (the (1+1) EA), `deceptive_trap` (GA with two-point crossover, against uniform crossover and hill climbing), `royal_road_r1` (random-mutation hill climbing against a GA, Table 1), `royal_road_r2` (GA with the 1992 settings), `nk_landscape` (iterated local search against exhaustive search, and a GA), `knapsack` (switched to an uncorrelated instance of 50 items; GA with Deb's rules against dynamic programming, and a strongly correlated instance as a contrast) | | 13 (optional) | | Competition suites whose definitions are long | LIR-CMOP1-14, CEC 2009 UF1-UF10 and CF1-CF10, MaF1-MaF15, Deb's 1999 two-objective problems, Van Veldhuizen's constrained problems; the deferred engineering problems (section 1.5) once their originals are read | none; used by the benchmark suite | Order rationale: batch 1 builds the machinery with the functions everyone starts with; batch 2 diff --git a/examples/README.md b/examples/README.md index a1746c8e..4d2b7019 100644 --- a/examples/README.md +++ b/examples/README.md @@ -31,6 +31,11 @@ python examples/tsp_berlin52/main.py | Example | Category | Languages | Interactive run | |---|---|---|---| | [OneMax](one_max/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/one-max) | +| [LeadingOnes](leading_ones/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/leading-ones) | +| [Deceptive trap](deceptive_trap/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/deceptive-trap) | +| [Royal road R1](royal_road_r1/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/royal-road-r1) | +| [Royal road R2](royal_road_r2/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/royal-road-r2) | +| [NK landscape](nk_landscape/) | binary | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/nk-landscape) | | [0/1 knapsack](knapsack/) | constrained | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/knapsack) | | [N-Queens 8×8](n_queens_8/) | permutation | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/n-queens-8) | | [N-Queens 16×16](n_queens_16/) | permutation | Rust, Python | [tachsin.gr](https://tachsin.gr/projects/genoxide/examples/n-queens-16) | diff --git a/examples/deceptive_trap/README.md b/examples/deceptive_trap/README.md new file mode 100644 index 00000000..abdd8f06 --- /dev/null +++ b/examples/deceptive_trap/README.md @@ -0,0 +1,90 @@ +--- +title: Deceptive trap +category: binary +summary: Maximize 10 blocks of 4 bits, each a trap function that leads away from its optimum, with a genetic algorithm whose two-point crossover keeps the blocks together, against uniform crossover and hill climbing. +reference: "Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. Foundations of Genetic Algorithms 2: 93-108." +reference_url: https://doi.org/10.1016/B978-0-08-094832-4.50012-X +optimum: "40 (all ones)" +languages: [rust, python] +order: 12 +--- + +# Deceptive trap + +## The problem + +A trap function scores a block of k bits by its number of ones u (Deb and Goldberg, 1993, equation +1): + +```text +f(u) = a (z − u) / z for u ≤ z + b (u − z) / (k − z) otherwise +``` + +It falls from a at u = 0 to 0 at the slope change z, and rises to b > a at u = k. Ackley (1987, *A +Connectionist Machine for Genetic Hillclimbing*, section 3.3.3) introduced it with a = 8k, b = 10k +and z = ⌊3k/4⌋; Deb and Goldberg analyze any a, b and z. Here a = k − 1, b = k and z = k − 1, so a +block of k = 4 bits scores 3, 2, 1 and 0 for 0 to 3 ones, and 4 for all four. The string has 10 +consecutive blocks, bits 0 to 3, 4 to 7 and so on, 40 bits, and scores the sum of its blocks: 40 +at most, for the string of all ones, and 30 for the string of all zeros. The function is genoxide's +`problems::binary::Trap::new(10, 4)`. + +## What makes it hard + +Every step up within a block leads to all zeros, except the last one: from 3 ones, the fourth gives +4, and anything else less. A block with u ones below 4 gains by losing a one. Deb and Goldberg prove +that a trap is *fully deceptive* (every schema of order below k, averaged over the strings it +contains, favors the all-zeros block) if and only if all schemata of order k − 1 are (Theorem 1), +and give the condition on r = a/b (inequality 16). With a = k − 1, b = k and z = k − 1, r = 3/4 is +above their limit (k − 1)/(2k − 3) = 3/5 for k = 4 (eq. 20): the blocks are fully deceptive. +genoxide's tests check this by averaging over every schema. A hill climber takes each block to its +attractor, all zeros, unless the block starts with all ones, or with three and the missing bit is +the first to flip. + +The 10 blocks are independent, so a search that finds the optimum of each block, and keeps it, +solves the whole. A genetic algorithm can, if crossover passes whole blocks from parent to child +and selection keeps the complete ones; the blocks are tight, 4 adjacent bits, so a point crossover +rarely cuts one. + +## Representation + +A `Binary` genome of 40 bits is the string itself. The fitness is the sum of the blocks' values, to +maximize. + +## Algorithm + +A genetic algorithm: + +- a population of 1000, enough to hold several copies of each block's optimum from the start; +- tournament selection of size 4; +- two-point crossover, which cuts the string at two points: of the 39 places it can cut, 9 are + between blocks, and a cut elsewhere breaks one block; +- bit-flip mutation at a rate of 1/40 per bit; +- the default generational scheme, which keeps the best individual. + +A run from seed 1 stops at 40, or after 300 generations; then runs from seeds 1 to 20. Two +contrasts run from the same seeds: the same GA with uniform crossover, which takes each bit from +either parent and so breaks the blocks apart, and hill climbing, `LocalSearch` flipping one bit at a +time and keeping the change if it's no worse, for 10,000 evaluations. + +## Output + +The table follows the run from seed 1: each generation's best score, its population's median score +and the blocks of the best string that are all ones. Then the generation and evaluations at which +it reaches 40. The last three lines give the runs from seeds 1 to 20: how many reach 40 with +two-point crossover and after how many generations (the median), how many with uniform crossover +and the median number of blocks of ones they end with, and hill climbing's median best and blocks +of ones. The function is evaluated in Rust in both languages, so both print the same. + +[The project page](https://tachsin.gr/projects/genoxide/examples/deceptive-trap) plays back the run +from seed 1: the first 16 strings of the population, blocks of ones spreading through them. + +## Good results + +The optimum is 40. The run from seed 1 reaches it at generation 15, after 14,687 evaluations, and so +do all 20 runs from seeds 1 to 20, after a median of 15 generations. Over seeds 1 to 300, every run +reached it, within 95 generations. + +The contrasts don't. Uniform crossover reaches it from none of the 20 seeds in 300 generations: its +runs end with a median of 3 blocks of ones of the 10. Hill climbing +ends at a median of 31: one block of ones and nine at the deceptive attractor, 3 each. diff --git a/examples/deceptive_trap/main.py b/examples/deceptive_trap/main.py new file mode 100644 index 00000000..409c5fd0 --- /dev/null +++ b/examples/deceptive_trap/main.py @@ -0,0 +1,115 @@ +"""Deceptive trap: maximize 10 blocks of Deb and Goldberg's trap function of 4 bits, which leads +each block away from its optimum, with a genetic algorithm whose two-point crossover keeps the +blocks together. + +A block of 4 bits scores 3 − u for u ones below 4, and 4 with all of them: fully deceptive, every +schema of order below 4 favoring all zeros. A run from seed 1 prints each generation; then runs +from seeds 1 to 20 count how often the GA reaches the optimum, 40. As contrasts: the same GA with +uniform crossover, which breaks the blocks apart, and hill climbing, one bit at a time, which +climbs to the deceptive attractor. The function is genoxide's problems.binary.Trap, which run +evaluates in Rust. + +With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's +page, with trace.py. + + python examples/deceptive_trap/main.py +""" + +import genoxide as gx + +from trace import Trace + +BLOCKS = 10 +K = 4 +BITS = BLOCKS * K +SEEDS = 20 +GENERATIONS = 300 +# the evaluations of a hill climb +CLIMB = 10_000 + + +def ga(problem, crossover, seed): + """The genetic algorithm with ``crossover``, from ``seed``.""" + return gx.Ga( + problem.genome, + population_size=1000, + select=gx.Tournament(4), + crossover=crossover, + mutation=gx.BitFlip(rate=1 / BITS), + seed=seed, + ) + + +def blocks_of_ones(genome): + """The blocks of ``genome`` that are all ones.""" + return sum(all(genome[block * K : (block + 1) * K]) for block in range(BLOCKS)) + + +def median(values): + """The median of ``values``.""" + values = sorted(values) + middle = len(values) // 2 + return values[middle] if len(values) % 2 else (values[middle - 1] + values[middle]) / 2 + + +problem = gx.problems.binary.Trap(BLOCKS, K) +optimum = problem.optimum.value +print(f"Deceptive trap: {BLOCKS} blocks of {K} bits, maximum {optimum:.0f}") +print("a GA with two-point crossover, from seed 1") +print("generation best median blocks of ones in the best") +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(BITS, optimum) + + +def progress(progress): + print( + f"{progress.generation:>10} {progress.best_fitness:>4.0f} " + f"{median(progress.scores):>6g} {blocks_of_ones(progress.best_genome):>2}" + ) + trace.record(progress) + + +result = ga(problem, gx.PointCrossover(2), 1).run( + problem, target=optimum, generations=GENERATIONS, on_generation=progress +) +print( + f"{result.best_fitness:.0f} after {result.generations} generations and " + f"{result.evaluations} evaluations" +) + +# two-point crossover from seeds 1 to 20, and uniform crossover as a contrast +reached = [] +for seed in range(1, SEEDS + 1): + result = ga(problem, gx.PointCrossover(2), seed).run( + problem, target=optimum, generations=GENERATIONS + ) + if result.stop_reason == "target": + reached.append(result.generations) +print( + f"\nseeds 1 to {SEEDS}, two-point crossover: {len(reached)} of {SEEDS} reach {optimum:.0f}, " + f"after a median of {median(reached):.1f} generations" +) +uniform, bests = 0, [] +for seed in range(1, SEEDS + 1): + result = ga(problem, gx.UniformCrossover(), seed).run( + problem, target=optimum, generations=GENERATIONS + ) + uniform += result.stop_reason == "target" + bests.append(blocks_of_ones(result.best_genome)) +print( + f"contrast, uniform crossover: {uniform} of {SEEDS} reach it in {GENERATIONS} generations; " + f"a median of {median(bests):.1f} blocks of ones" +) + +# hill climbing: one bit flipped at a time, kept if no worse +scores, blocks = [], [] +for seed in range(1, SEEDS + 1): + search = gx.LocalSearch(problem.genome, neighbor=gx.BitFlip(count=1), seed=seed) + result = search.run(problem, target=optimum, evaluations=CLIMB) + scores.append(result.best_fitness) + blocks.append(blocks_of_ones(result.best_genome)) +print( + f"contrast, hill climbing ({CLIMB} evaluations): a median best of {median(scores):.1f}, " + f"with {median(blocks):.1f} blocks of ones" +) +trace.write() diff --git a/examples/deceptive_trap/main.rs b/examples/deceptive_trap/main.rs new file mode 100644 index 00000000..03aadd7b --- /dev/null +++ b/examples/deceptive_trap/main.rs @@ -0,0 +1,152 @@ +//! Deceptive trap: maximize 10 blocks of Deb and Goldberg's trap function of 4 bits, which leads +//! each block away from its optimum, with a genetic algorithm whose two-point crossover keeps +//! the blocks together. +//! +//! A block of 4 bits scores 3 − u for u ones below 4, and 4 with all of them: fully deceptive, +//! every schema of order below 4 favoring all zeros. A run from seed 1 prints each generation; +//! then runs from seeds 1 to 20 count how often the GA reaches the optimum, 40. As contrasts: the +//! same GA with uniform crossover, which breaks the blocks apart, and hill climbing, one bit at a +//! time, which climbs to the deceptive attractor. The function is genoxide's +//! `problems::binary::Trap`. +//! +//! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's +//! page, with `trace.rs`. +//! +//! ```text +//! cargo run --release --example deceptive_trap +//! ``` + +mod trace; + +use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::Trap; + +const BLOCKS: usize = 10; +const K: usize = 4; +const BITS: usize = BLOCKS * K; +const SEEDS: u64 = 20; +const GENERATIONS: u64 = 300; +// the evaluations of a hill climb +const CLIMB: u64 = 10_000; + +// the genetic algorithm with `crossover`, from `seed` +fn ga>( + problem: &Trap, + crossover: C, + seed: u64, +) -> Result> { + Ga::builder(problem.representation()) + .population_size(1000) + .select(Tournament::new(4)?) + .crossover(crossover) + .mutate(BitFlip::per_gene(1.0 / BITS as f64)?) + .seed(seed) + .build() +} + +// the blocks of `genome` that are all ones +fn blocks_of_ones(genome: &Bits) -> usize { + (0..BLOCKS) + .filter(|block| (block * K..(block + 1) * K).all(|i| genome.get(i) == Some(true))) + .count() +} + +// the median of `values` +fn median(mut values: Vec) -> f64 { + values.sort_by(f64::total_cmp); + let middle = values.len() / 2; + if values.len() % 2 == 1 { + values[middle] + } else { + (values[middle - 1] + values[middle]) / 2.0 + } +} + +fn main() -> Result<()> { + let problem = Trap::new(BLOCKS, K); + let optimum = problem.optimum().expect("known").value(); + println!("Deceptive trap: {BLOCKS} blocks of {K} bits, maximum {optimum}"); + println!("a GA with two-point crossover, from seed 1"); + println!("generation best median blocks of ones in the best"); + // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page + let mut trace = trace::Trace::from_env(BITS, optimum); + let outcome = Engine::new(ga(&problem, PointCrossover::two_point(), 1)?, problem) + .stop_when(Stop::target(optimum).or(Stop::generations(GENERATIONS))) + .on_generation(|snapshot| { + let progress = snapshot.progress(); + let scores = snapshot.population().iter(); + let scores = scores.filter_map(|individual| individual.fitness()?.score()); + println!( + "{:>10} {:>4} {:>6} {:>2}", + progress.generation(), + snapshot + .best() + .fitness() + .and_then(Fitness::score) + .unwrap_or(0.0), + median(scores.collect()), + blocks_of_ones(snapshot.best().genome()) + ); + }) + .on_generation(|snapshot| trace.record(snapshot)) + .run()?; + println!( + "{} after {} generations and {} evaluations", + outcome.best_fitness(), + outcome.generations(), + outcome.evaluations() + ); + + // two-point crossover from seeds 1 to 20, and uniform crossover as a contrast + let mut reached = Vec::new(); + for seed in 1..=SEEDS { + let outcome = Engine::new(ga(&problem, PointCrossover::two_point(), seed)?, problem) + .stop_when(Stop::target(optimum).or(Stop::generations(GENERATIONS))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + reached.push(outcome.generations() as f64); + } + } + println!( + "\nseeds 1 to {SEEDS}, two-point crossover: {} of {SEEDS} reach {optimum}, after a median \ + of {:.1} generations", + reached.len(), + median(reached) + ); + let (mut uniform, mut bests) = (0, Vec::new()); + for seed in 1..=SEEDS { + let outcome = Engine::new(ga(&problem, UniformCrossover::new(), seed)?, problem) + .stop_when(Stop::target(optimum).or(Stop::generations(GENERATIONS))) + .run()?; + uniform += usize::from(outcome.stop_reason() == StopReason::Target); + bests.push(blocks_of_ones(outcome.best_genome()) as f64); + } + println!( + "contrast, uniform crossover: {uniform} of {SEEDS} reach it in {GENERATIONS} generations; \ + a median of {:.1} blocks of ones", + median(bests) + ); + + // hill climbing: one bit flipped at a time, kept if no worse + let (mut scores, mut blocks) = (Vec::new(), Vec::new()); + for seed in 1..=SEEDS { + let search = LocalSearch::builder(problem.representation()) + .neighbor(BitFlip::count(1)?) + .seed(seed) + .build()?; + let outcome = Engine::new(search, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(CLIMB))) + .run()?; + scores.push(outcome.best_fitness().score().expect("valid")); + blocks.push(blocks_of_ones(outcome.best_genome()) as f64); + } + println!( + "contrast, hill climbing ({CLIMB} evaluations): a median best of {:.1}, with {:.1} \ + blocks of ones", + median(scores), + median(blocks) + ); + trace.write(); + Ok(()) +} diff --git a/examples/deceptive_trap/output.txt b/examples/deceptive_trap/output.txt new file mode 100644 index 00000000..f15dae92 --- /dev/null +++ b/examples/deceptive_trap/output.txt @@ -0,0 +1,24 @@ +Deceptive trap: 10 blocks of 4 bits, maximum 40 +a GA with two-point crossover, from seed 1 +generation best median blocks of ones in the best + 0 23 13 2 + 1 30 16 4 + 2 30 19 4 + 3 34 20 7 + 4 34 22 7 + 5 34 24 7 + 6 34 25 7 + 7 36 25 7 + 8 36 26 7 + 9 38 27 8 + 10 38 28 8 + 11 38 28 8 + 12 38 29 8 + 13 38 30 8 + 14 39 30 9 + 15 40 31 10 +40 after 15 generations and 14687 evaluations + +seeds 1 to 20, two-point crossover: 20 of 20 reach 40, after a median of 15.0 generations +contrast, uniform crossover: 0 of 20 reach it in 300 generations; a median of 3.0 blocks of ones +contrast, hill climbing (10000 evaluations): a median best of 31.0, with 1.0 blocks of ones diff --git a/examples/deceptive_trap/trace.json b/examples/deceptive_trap/trace.json new file mode 100644 index 00000000..2441aba9 --- /dev/null +++ b/examples/deceptive_trap/trace.json @@ -0,0 +1,18 @@ +{"example":"deceptive_trap","format":1,"log_y":false,"objective":"maximize","optimum":40.0,"plot":"bits","problem":{"length":40},"x_label":"generations","y_label":"score","frames":[ +{"best":23.0,"evaluations":1000,"generation":0,"median":13.0,"state":{"population":["1000110110110000001001010011000101010111","1101011001110001011100000011111100011011","1010011001100000111000000110110011000000","0100001010110000000100001100011000111011","1010011111010011110110001010101111001010","0001101000000100010001111001011101101100","1110100001010001111111110111111000000111","0111000100010010101000111010110101000010","1100100000000110100011010101000011011000","0001000010010110000000010011010000001101","1000111011000010001100001110000001001101","1100011110010010000010001000111101111010","1100010001110001110110111010101110100111","1000010110000111100101111001111111010000","0100011111011111011001010100111011111101","1011100011100110100010011100111001011011"]}}, +{"best":30.0,"evaluations":1941,"generation":1,"median":16.0,"state":{"population":["1111000000000001110100001111100001101010","1101100010000101000001000011000110000100","1000000101110100000101110000000111001000","1111111111111011111100111010011111101110","0100111111111110101100101011111011111000","0000000101000001000101101010000100001111","0010100000011010001011111110000101101100","0000001011010000000000101100100100110111","0101100000110111010111101111101000110101","1001001101000011010010000000011011100100","0100000000100011010010001000100001010111","1001000111011011111001000010111111110011","1100001100001000001111000011111110001100","0000011101011011101011010111111100100100","0111110000001011011001101111001010100011","0010011110101101111000111000111110011101"]}}, +{"best":30.0,"evaluations":2873,"generation":2,"median":19.0,"state":{"population":["1111000000000001010100001111111111110100","0100110010111000010111011111100111111000","0001000010111111001011111100111101111001","0111000000001111110011000010011100001011","1000101001011100111010001111111110100011","1101010011110001000111000011011010011110","0001000001000000000010010001110001000010","0010000011111111101111111000111011110111","1110010111110000000010001000001110011000","0010000010101100000110000000110101001000","1111000111111011110010001100001000101110","1111010000001100100001001111000001001000","0100001000000101110010001111111100100010","1000011100110001101111110110011100011001","0111111101101110111111111000111110011101","0011000000101110001000010010010011101011"]}}, +{"best":34.0,"evaluations":3812,"generation":3,"median":20.0,"state":{"population":["1111000000000001010100001111111111110100","1111000001000100001010010000010001000010","1111000110100100011001100110101011011101","1111100001110101110010101001100011111011","1111100011110000100000011000100000010001","1100010000000110101101000001000001001001","1111110100010100110010110110100011111001","1111000110100100011111110000111111111111","0101110111111100011001100001110001101000","0010111111110000011110011111001000000010","1101010000011000010000011000100001001001","0001111111111001111011110011101111101001","1111100111110011000100011010000111110001","0010000001110000000010011000001111110001","1110000010011111101111111000111110011110","1100111111011111101000000001111110001001"]}}, +{"best":34.0,"evaluations":4760,"generation":4,"median":22.0,"state":{"population":["1111010011111111111101100000111111111111","1111011010000010110001000100111100000000","0000000011111111000000010100110011011111","1111000001001101101000101111110111110110","1110011111110000000011111000111111000000","0010000011111110101111011000100011101111","1111010001001101010001001111000001001000","1011110111111111100011111100100000100010","1111001001011101111000100111100100100010","1111000110100100001111110000111111111111","1111011010000010110101000100111100000000","1111000001000000110111001101111011110010","1101111110010111001110110110010001010011","0000000001110000000011110000011000010010","0100001000000001100100001000111111111000","1111000101110000010001011100000111010111"]}}, +{"best":34.0,"evaluations":5678,"generation":5,"median":24.0,"state":{"population":["1111010011111111111101100000111111111111","1111111101111000111101000110111010000001","0110111111110000001010001111001000011000","1111000110000101111111110000111111111111","1111000010000100010111111010111110100010","0000111111011111111101100000111111011111","1111010010011011100010001111001000000001","0010111100011111100010010000111111001111","0000000111001111110101010000011001111111","1001110010000101111100001111000100011111","1110001111110010010011110001100011110010","0000100011110000101000011111101000101111","1111010011110001000001011001101111110101","1111000000000001110001110010001101110100","1100001011110000001100101111111111110000","0111000110001110110101000000111011001111"]}}, +{"best":34.0,"evaluations":6611,"generation":6,"median":25.0,"state":{"population":["1111010011111111111101100000111111111111","0000010011110001000100010000100010101101","0010110111110001011011110010001000100010","0000000111110000110100011000000011111100","1000111111111111111111110111100000010001","0111000000000001010100001111100011110000","1111000000000011111111110111010111110110","1111000010011111100011111101000011100000","0010000011110111111111110000001010101111","0000000111110001001011111111011111110000","1110000110000101111101000000111100101111","0000000111011111000011110000101111111111","0101001010110000101101011000110001001010","1111011111111111100000000010011111110100","0010100010111111100000000010100110111111","1111010001000000111100111001111111110010"]}}, +{"best":36.0,"evaluations":7546,"generation":7,"median":25.0,"state":{"population":["1111010011111111111101100000111111111111","0001011111101111100000011111010011111111","1111001000000000100000000010111111110101","1110000011110111100010011101010011111111","1111000000000101111111010011111111111100","1111000000001111111101101111111100001111","1111001111110111111001000111000111111000","0000100111111111101011110000111110110000","0011000010111111101000011110111110111111","0010000000001111100011111001111111101111","0011101111111011111111110000111111111000","1111000011000010111000001111111111110101","0010000101111101010100001111111100101111","1111000101011001111011111111111111110001","1110000111111111001000100010111100001100","1110111111101111011100101111111011111101"]}}, +{"best":36.0,"evaluations":8458,"generation":8,"median":26.0,"state":{"population":["1111000011110000000111111111111111111111","0111000010000001111101100000110111110110","1111000000011111111111110000111011111101","1110000111111111000000101111000111111111","1111100100110011111100100001111011111000","1101011010011111100101001000011111100000","0100000000010100111010000111000011100000","1111000011110100000111111000111101101111","0000000001001011100011111001000000001111","0010000011110000000011111111000100001101","1000000111111101000011101111000001010000","1111110101010100000011111101111111010000","0001101110000010001000001111111111110001","1111001001110000000100000111111111111111","0111000000000001011011111111111100000100","1111000000000001000010010000111110110010"]}}, +{"best":38.0,"evaluations":9371,"generation":9,"median":27.0,"state":{"population":["1111000011110000000111111111111111111111","1111000110000000000011111111000111100000","0111000001111000000000011011011111110100","1101000000001111111101100001000111110000","1111100001000000110011111111000111110000","1011000011111010000001000000111110111111","1111001100000000100010001111111100011111","0000000011110000101011110000111111110111","1010111100001111111100001111111011111100","1111000011110000011111010000011111110000","1111000011110000001011111000111100000000","1111001011111001001011110000110011111111","1111010000000000000000001111011111110100","1111001011111111111100100000111111110111","1000101111111101101010111111111111111100","1111000000010011011110000000101111110110"]}}, +{"best":38.0,"evaluations":10282,"generation":10,"median":28.0,"state":{"population":["1111000000001111111111111111111111111111","0000000001001101101111110000010100001111","1111010111111111111101111111111111111101","0010111100011010000110001011000111111111","0010111110010111111111111000111111111111","1111001010011111011000001000111111011111","1101111111111111100000011111110111110000","1101010011111100111111111010001011011111","1101000000100100000000010111111111110100","1111010011111100001000111010000011111111","1111011111111111111111100011111111110000","1111101111111111010111110011101111111100","1111000001000000001011111011011101100100","1111100011110011111111011111111111111110","1111000111111111111001000000111111111111","1111001011110000100111111111111111101111"]}}, +{"best":38.0,"evaluations":11189,"generation":11,"median":28.0,"state":{"population":["1111000000001111111111111111111111111111","1101000011111111111111111111010100001111","1111000000011111111111110000111110001110","1110111100000000001000011111111001111111","1001000111110100000111110111000010111111","1111110011111100111101001111111111111111","1111000000010011111101010001111111111111","0000111100000000101000011111111111111110","1110000011111111000000010111111111111110","1111001111111111110011001000111111110101","0000111100111111111101011111101111111111","1111001111110000110101001111110111111111","1111001000000010111101100000111111111111","1111100100111111010011111111111111111111","1111011011000001111111111111111100011111","1001000011110001000101101111111111110000"]}}, +{"best":38.0,"evaluations":12085,"generation":12,"median":29.0,"state":{"population":["1111000000001111111111111111111111111111","0111000010111111010011111000101111111111","1111000011110111000111111111110011111111","0010111101111111000101100111111111101111","1011000011111111111000111111111111111111","1111001100001111010011110000111111111111","1111000011111101000000001111111111111111","0010110011110000000011111110111111111111","1111000111110011111100001111111111111111","1010111100001111000011111100111111111111","1111001010001111111111111000111011111111","1111000011111101000000011111111111111111","1110000010001111111100011111111100001111","1111111101111111111111100000111111111111","1111000100000001000010010000111111111111","0000000011110100010100110100111111111111"]}}, +{"best":38.0,"evaluations":12975,"generation":13,"median":30.0,"state":{"population":["1111000000001111111111111111111111111111","1111001000000001111111111111111111101111","1110000011110000000111111111111111111111","1111000011111110001011111100111111111111","1000000011111111000111111111111111111111","1111111100011111111100001111111110111111","1111000001100100111101011101111111111111","0011000111111111000000000110001001111111","1001000011111111111111111001111111111111","1111000111111111111111111000111110111111","1111111100001111001011111111111111110111","0110111101111111111011111111111111110111","1111110010000111100111110000111111110100","1001110000010001111111111111111111111111","0000111100011101111111110000111111110000","1111000000000011100011111101110111111111"]}}, +{"best":39.0,"evaluations":13843,"generation":14,"median":30.0,"state":{"population":["1111000000001111111111111111111111111111","1111101011111101111111111011111111111111","0001111011111111110011111111111111111111","1111001011111111111101101000111111111111","1111111111111111000000000101111111111111","1111000011110000000111111111111101011111","1111010111111111111111111111111111111000","0111010011111111110100001101111111111111","1111111101111100111111011111111111111111","0111000011011100111101100000111111111101","1101111110000000001000011111111111111111","1111000011111111101100110000111111111111","1111100000011111111100011111111111111111","1111100100000011111111110101111111111011","1111000111110001011100011111111111111111","0001110011110000010011111001111011111111"]}}, +{"best":40.0,"evaluations":14687,"generation":15,"median":31.0,"state":{"population":["1111000011111111111111111111111111111111","1111111100000100110100111101111111111111","1000111111110000111110110111111111101111","1111001111111001111111111101111111111111","1111000001101111010010111111111111111101","1111010011111111111110010000111111111111","1111111111111111000000011111111111111010","1111010000000000000011111111111111111111","1011000011110000000100000000111111111111","1111000011110101111111111000111111111111","1011111111111111111100011111111111111111","1111000011111111111000010010111011111111","1111000011111111000000000100111111111111","1111000000000100111111110100111111111111","1111000011110011000011001111111111111111","1111101010111111011111111111111111111111"]}} +]} diff --git a/examples/deceptive_trap/trace.py b/examples/deceptive_trap/trace.py new file mode 100644 index 00000000..359c27bc --- /dev/null +++ b/examples/deceptive_trap/trace.py @@ -0,0 +1,183 @@ +"""The trace of the run for the plot on the example's page, written to the file that +``GENOXIDE_TRACE`` names: the first 16 strings of the population, in at most 32 generations. The Rust example writes the same +file.""" + +import json +import math +import os + + +class Trace: + """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" + + def __init__(self, length, optimum): + self.path = os.environ.get("GENOXIDE_TRACE") + self.frames = Frames(32) + self.length, self.optimum = length, optimum + + @property + def on_generation(self): + """The callback for ``run``: None without a trace to record.""" + return self.record if self.path else None + + def record(self, progress): + """Records a generation: the first 16 strings of the population.""" + if self.path: + rows = ["".join("1" if one else "0" for one in row) for row in progress.population[:16]] + self.frames.push(frame(progress, {"population": rows})) + + def write(self): + """Writes the trace, if there's one.""" + if self.path: + settings = { + "format": 1, + "example": "deceptive_trap", + "objective": "maximize", + "x_label": "generations", + "y_label": "score", + "log_y": False, + "optimum": float(self.optimum), + "plot": "bits", + "problem": {"length": self.length}, + } + write(self.path, settings, self.frames.to_list()) + + +# ---- the same in every example's trace ---------------------------------------------------------- + + +class Frames: + """The frames of at most ``most`` generations, from the part of the run where what the page + plots changes: the frames after the last change are left out (a run that reached its target, + or a front that no longer moves), and the rest are spread evenly over the generations up to + it. While the run goes, up to 8 × ``most`` frames are kept: every ``every``-th generation, + with ``every`` doubling whenever there are that many, and the last one.""" + + def __init__(self, most): + self.most, self.every, self.kept, self.last = most, 1, [], None + + def push(self, frame): + if frame["generation"] % self.every: + self.last = frame + return + self.kept.append(frame) + self.last = None + if len(self.kept) == 8 * self.most: + self.every *= 2 + self.kept = [kept for kept in self.kept if kept["generation"] % self.every == 0] + + def to_list(self): + frames = self.kept + ([self.last] if self.last else []) + active = frames[: last_change(frames) + 1] + count, most = len(active), max(self.most, 2) + if count <= most: + return active + return [active[(i * (count - 1) + (most - 1) // 2) // (most - 1)] for i in range(most)] + + +def last_change(frames): + """The index of the frame after which nothing the page plots changes. To 3 significant + digits, as a plot shows them: the best, the median and, for a single objective (a numeric + best), the state; to within a hundredth of their range over the run: a front's + hypervolumes, in the state or in a grid's series.""" + if not frames: + return 0 + last = len(frames) - 1 + number = lambda value: isinstance(value, (int, float)) and not isinstance(value, bool) + single = any(number(frame.get("best")) for frame in frames) + + def measures(frame): + values = [] + state = frame.get("state") + hypervolume = state.get("hypervolume") if isinstance(state, dict) else None + for value in (hypervolume, frame.get("series")): + if number(value): + values.append(float(value)) + elif isinstance(value, dict): + values.extend(float(v) for _, v in sorted(value.items()) if number(v)) + return values + + measured = [measures(frame) for frame in frames] + end = measured[last] + tolerance = [] + for k in range(len(end)): + values = [values[k] for values in measured if k < len(values)] + tolerance.append((max(values) - min(values)) / 100.0) + + def same(frame, final, key, flush=False): + return coarse(frame.get(key), flush) == coarse(final.get(key), flush) + + def settled(i): + frame, final = frames[i], frames[last] + return ( + same(frame, final, "best") + and same(frame, final, "median") + and (not single or same(frame, final, "state", flush=True)) + and len(measured[i]) == len(end) + and all(abs(v - e) <= t for v, e, t in zip(measured[i], end, tolerance)) + ) + + first = last + while first > 0 and settled(first - 1): + first -= 1 + return first + + +def frame(progress, state): + """The frame of a generation: its progress, the median score of its population and + ``state``.""" + return { + "generation": progress.generation, + "evaluations": progress.evaluations, + "best": progress.best_fitness, + "median": median(progress.scores), + "state": state, + } + + +def median(scores): + """The median of the valid scores, None without any.""" + scores = sorted(float(score) for score in scores if not math.isnan(score)) + middle = len(scores) // 2 + if not scores: + return None + return scores[middle] if len(scores) % 2 else (scores[middle - 1] + scores[middle]) / 2 + + +def coarse(value, flush=False): + """``value`` with its numbers to 3 significant digits, as precisely as a plot shows them: two + frames whose plotted values agree to that precision look the same. With ``flush``, for the + solutions a plot draws on their ranges, numbers below 1e-6 in size count as 0.""" + if isinstance(value, float): + if flush and abs(value) < 1e-6: + value = 0.0 + return f"{value:.2e}" + if isinstance(value, (list, tuple)): + return "[" + ",".join(coarse(item, flush) for item in value) + "]" + if isinstance(value, dict): + items = sorted(value.items()) + return "{" + ",".join(f"{key}:{coarse(item, flush)}" for key, item in items) + "}" + return json.dumps(value) + + +def write(path, settings, frames): + """Writes the settings and the frames to ``path``, a frame per line.""" + lines = ",\n".join(map(to_json, frames)) + with open(path, "w", encoding="utf-8", newline="\n") as file: + file.write(f'{to_json(settings)[:-1]},"frames":[\n{lines}\n]}}\n') + + +def to_json(value): + """Compact JSON with sorted keys, and numbers rounded to 6 significant digits, as the Rust + example writes it.""" + return json.dumps(rounded(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def rounded(value): + if isinstance(value, dict): + return {key: rounded(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [rounded(item) for item in value] + if isinstance(value, float): + return float(f"{value:.5e}") if math.isfinite(value) else None + return value diff --git a/examples/deceptive_trap/trace.rs b/examples/deceptive_trap/trace.rs new file mode 100644 index 00000000..0c33a410 --- /dev/null +++ b/examples/deceptive_trap/trace.rs @@ -0,0 +1,264 @@ +//! The trace of the run for the plot on the example's page, written to the file that +//! `GENOXIDE_TRACE` names: the first 16 strings of the population, in at most 32 generations. The Python example writes the same +//! file. + +use genoxide::observer::Snapshot; +use genoxide::prelude::*; +use serde_json::{Value, json}; + +pub struct Trace { + path: Option, + frames: Frames, + length: usize, + optimum: f64, +} + +impl Trace { + // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set + pub fn from_env(length: usize, optimum: f64) -> Self { + let path = std::env::var("GENOXIDE_TRACE").ok(); + let frames = Frames::new(32); + Self { + path, + frames, + length, + optimum, + } + } + + // records a generation: the first 16 strings of the population + pub fn record(&mut self, snapshot: &Snapshot<'_, Bits>) { + if self.path.is_none() { + return; + } + let bits = |genome: &Bits| { + genome + .iter() + .map(|one| if one { '1' } else { '0' }) + .collect() + }; + let rows = snapshot.population().iter().take(16); + let rows: Vec = rows.map(|row| bits(row.genome())).collect(); + self.frames + .push(frame(snapshot, json!({ "population": rows }))); + } + + // writes the trace, if there's one + pub fn write(self) { + let Some(path) = self.path else { return }; + let settings = json!({ + "format": 1, + "example": "deceptive_trap", + "objective": "maximize", + "x_label": "generations", + "y_label": "score", + "log_y": false, + "optimum": self.optimum, + "plot": "bits", + "problem": { "length": self.length }, + }); + write(&path, settings, self.frames.into_vec()); + } +} + +// ---- the same in every example's trace --------------------------------------------------------- + +// the frames of at most `most` generations, from the part of the run where what the page plots +// changes: the frames after the last change are left out (a run that reached its target, or a +// front that no longer moves), and the rest are spread evenly over the generations up to it. While +// the run goes, up to 8 × `most` frames are kept: every `every`-th generation, with `every` +// doubling whenever there are that many, and the last one. +struct Frames { + most: usize, + every: u64, + kept: Vec<(u64, Value)>, + last: Option<(u64, Value)>, +} + +impl Frames { + fn new(most: usize) -> Self { + let (every, kept, last) = (1, Vec::new(), None); + Self { + most, + every, + kept, + last, + } + } + + fn push(&mut self, frame: Value) { + let generation = frame["generation"].as_u64().expect("a generation"); + if !generation.is_multiple_of(self.every) { + self.last = Some((generation, frame)); + return; + } + self.kept.push((generation, frame)); + self.last = None; + if self.kept.len() == 8 * self.most { + self.every *= 2; + let every = self.every; + self.kept.retain(|(generation, _)| generation % every == 0); + } + } + + fn into_vec(self) -> Vec { + let frames = self.kept.into_iter().chain(self.last); + let frames: Vec = frames.map(|(_, frame)| frame).collect(); + let active = &frames[..=last_change(&frames)]; + let (count, most) = (active.len(), self.most.max(2)); + if count <= most { + return active.to_vec(); + } + let at = |i: usize| active[(i * (count - 1) + (most - 1) / 2) / (most - 1)].clone(); + (0..most).map(at).collect() + } +} + +// the index of the frame after which nothing the page plots changes. To 3 significant digits, as +// a plot shows them: the best, the median and, for a single objective (a numeric best), the state; +// to within a hundredth of their range over the run: a front's hypervolumes, in the state or in a +// grid's series +fn last_change(frames: &[Value]) -> usize { + let Some(last) = frames.len().checked_sub(1) else { + return 0; + }; + let single = frames.iter().any(|frame| frame["best"].is_number()); + let measures = |frame: &Value| -> Vec { + let mut values = Vec::new(); + for value in [&frame["state"]["hypervolume"], &frame["series"]] { + match value { + Value::Number(number) => values.extend(number.as_f64()), + Value::Object(map) => values.extend(map.values().filter_map(Value::as_f64)), + _ => {} + } + } + values + }; + let measured: Vec> = frames.iter().map(measures).collect(); + let end = &measured[last]; + let tolerance: Vec = (0..end.len()) + .map(|k| { + let values = measured.iter().filter_map(|values| values.get(k).copied()); + let (low, high) = values.fold((f64::INFINITY, f64::NEG_INFINITY), |(low, high), v| { + (low.min(v), high.max(v)) + }); + (high - low) / 100.0 + }) + .collect(); + let settled = |i: usize| { + let (frame, final_frame) = (&frames[i], &frames[last]); + let same = + |key: &str, flush: bool| coarse(&frame[key], flush) == coarse(&final_frame[key], flush); + same("best", false) + && same("median", false) + && (!single || same("state", true)) + && measured[i].len() == end.len() + && measured[i] + .iter() + .zip(end) + .zip(&tolerance) + .all(|((value, end), tolerance)| (value - end).abs() <= *tolerance) + }; + let mut first = last; + while first > 0 && settled(first - 1) { + first -= 1; + } + first +} + +// the frame of a generation: its progress, the median score of its population and `state` +fn frame(snapshot: &Snapshot<'_, G>, state: Value) -> Value { + let progress = snapshot.progress(); + let population = snapshot.population().iter(); + let scores = population.filter_map(|individual| individual.fitness()?.score()); + json!({ + "generation": progress.generation(), + "evaluations": progress.evaluations(), + "best": progress.best().and_then(Fitness::score), + "median": median(scores.collect()), + "state": state, + }) +} + +// the median of the scores, None without any +fn median(mut scores: Vec) -> Option { + scores.sort_by(f64::total_cmp); + let middle = scores.len() / 2; + match scores.len() { + 0 => None, + n if n % 2 == 1 => Some(scores[middle]), + _ => Some((scores[middle - 1] + scores[middle]) / 2.0), + } +} + +// `value` with its numbers to 3 significant digits, as precisely as a plot shows them: two frames +// whose plotted values agree to that precision look the same. With `flush`, for the solutions a +// plot draws on their ranges, numbers below 1e-6 in size count as 0 +fn coarse(value: &Value, flush: bool) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => { + let number = number.as_f64().expect("f64"); + let number = if flush && number.abs() < 1e-6 { + 0.0 + } else { + number + }; + format!("{number:.2e}") + } + Value::Array(items) => { + format!( + "[{}]", + join(items.iter().map(|item| coarse(item, flush)).collect()) + ) + } + Value::Object(map) => { + let entry = |(key, item): (&String, &Value)| format!("{key}:{}", coarse(item, flush)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +// writes the settings and the frames to `path`, a frame per line +fn write(path: &str, settings: Value, frames: Vec) { + let frames: Vec = frames.iter().map(to_json).collect(); + let settings = to_json(&settings); + let head = &settings[..settings.len() - 1]; + let text = format!("{head},\"frames\":[\n{}\n]}}\n", frames.join(",\n")); + std::fs::write(path, text).expect("the trace is written"); +} + +// compact JSON with sorted keys, and numbers rounded to 6 significant digits and written as +// Python writes them (7542.0, 1e-08): the Python example writes the same file +fn to_json(value: &Value) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => python_float(number.as_f64().expect("f64")), + Value::Array(items) => format!("[{}]", join(items.iter().map(to_json).collect())), + Value::Object(map) => { + let entry = + |(key, item): (&String, &Value)| format!("{}:{}", json!(key), to_json(item)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +fn python_float(value: f64) -> String { + let rounded: f64 = format!("{value:.5e}").parse().expect("a number"); + let shortest = format!("{rounded:e}"); + let (mantissa, exponent) = shortest.split_once('e').expect("an exponent"); + let exponent: i32 = exponent.parse().expect("an exponent"); + if (-4..16).contains(&exponent) { + let text = rounded.to_string(); + if text.contains('.') { + text + } else { + text + ".0" + } + } else { + let sign = if exponent < 0 { '-' } else { '+' }; + format!("{mantissa}e{sign}{:02}", exponent.abs()) + } +} diff --git a/examples/knapsack/README.md b/examples/knapsack/README.md index 3f782ab6..282739b5 100644 --- a/examples/knapsack/README.md +++ b/examples/knapsack/README.md @@ -1,10 +1,10 @@ --- title: 0/1 knapsack category: constrained -summary: Choose the items with the highest total value whose total weight fits a capacity. -reference: "Martello, S. and Toth, P. (1990). Knapsack Problems: Algorithms and Computer Implementations. Wiley." -reference_url: "https://silvano333.github.io/kp.html" -optimum: "625 (total value)" +summary: Choose the items with the highest total value whose total weight fits a capacity, on an instance of 50 items drawn from one of Pisinger's classes, checked against dynamic programming. +reference: "Pisinger, D. (2005). Where are the hard knapsack problems? Computers & Operations Research 32(9): 2271-2284." +reference_url: https://doi.org/10.1016/j.cor.2004.03.002 +optimum: "20849 (total value, by dynamic programming)" languages: [rust, python] order: 20 --- @@ -13,38 +13,42 @@ order: 20 ## The problem -The 0/1 knapsack problem has items, each with a weight and a value, and a knapsack with a capacity. -It chooses the items with the highest total value whose total weight is at most the capacity. Each -item is taken whole or not at all. Martello and Toth (1990) treat it and its variants in a book. +The 0/1 knapsack problem has items, each with a weight and a value (a profit), and a knapsack with +a capacity. It chooses the items with the highest total value whose total weight is at most the +capacity; each item is taken whole or not at all. Martello and Toth (1990, *Knapsack Problems: +Algorithms and Computer Implementations*, Wiley) treat it and its variants in a book. -The example has 20 items and a capacity of 400. The first ten are the test data of Martello and -Toth's program for the multiple knapsack problem (Algorithm 632, 1985, ACM Transactions on -Mathematical Software 11(2): 135-140), where one knapsack of capacity 165 holds a value of at most -309. The other ten items and the capacity are this example's own. The first item weighs 23 and is -worth 92; the eighth weighs 85 and is worth 84. All 20 together weigh 896, more than twice the -capacity. +The instance is drawn from the first of Pisinger's (2005) classes of generated instances, the +uncorrelated one: 50 items whose weights and values are drawn independently and uniformly from 1 +to R = 1000. The capacity is half the total weight, rounded down, as instance 50 of Pisinger's +series of 100 has it: c = ⌊50/101 Σ w⌋ (his eq. 5), 12,572 of 25,397. The instance is genoxide's +`problems::binary::Knapsack::generator(KnapsackClass::Uncorrelated, 50).seed(1)`: its items are +drawn from seed 1 with genoxide's portable random numbers, the same on every platform and in Python. +The first item weighs 613 and is worth 705; the second weighs 858 and is worth 45. ## What makes it hard -The problem is NP-hard (Martello and Toth, 1990), though dynamic programming solves it in time -proportional to the number of items times the capacity. With 20 items, there are 2²⁰ = 1,048,576 -selections, and 356,115 of them (34%) fit. +The problem is NP-hard, though only weakly: dynamic programming over the capacities solves it in +time proportional to the number of items times the capacity (Bellman's recursion), and genoxide's +`optimum` does, exactly. With 50 items there are 2⁵⁰ ≈ 10¹⁵ selections, far too many to try. -The capacity constraint binds. The best selection is unique and weighs 395 of 400, and adding any -item to it breaks the limit. Taking items in order of value per unit of weight, while they fit, -gives 604, not the optimum. +The capacity binds: the best selection weighs 12,551 of 12,572, and taking any further item breaks +the limit: the lightest item left out weighs 209. Pisinger's classes are hard in different ways. +Uncorrelated instances are "generally easy to solve" for exact algorithms; in the strongly +correlated class, whose values are the weights plus R/10, sorting by value per unit of weight is +sorting by weight, and the continuous optimum is far from the integer one. ## Representation -A `Binary` genome of 20 bits, a bit per item: 1 takes the item. Every bit string is a selection, but +A `Binary` genome of 50 bits, a bit per item: 1 takes the item. Every bit string is a selection, but not every selection fits. -The fitness function returns two numbers: the total value, and how far the weight exceeds the -capacity (0 when it fits). genoxide compares them with Deb's feasibility rules (Deb, 2000, Computer -Methods in Applied Mechanics and Engineering 186: 311-338). A selection that fits beats one that -doesn't. Two that fit compare by value. Two that don't compare by their excess weight. So overweight -selections still guide the search: the less overweight, the better. There is no penalty weight to -tune. +The fitness has two numbers: the total value, and how far the weight exceeds the capacity (0 when +it fits), `constraint::at_most(weight, capacity)`. genoxide compares them with Deb's feasibility +rules (Deb, 2000, Computer Methods in Applied Mechanics and Engineering 186: 311-338). A selection +that fits beats one that doesn't. Two that fit compare by value. Two that don't compare by their +excess weight. So overweight selections still guide the search: the less overweight, the better. +There is no penalty weight to tune. ## Algorithm @@ -53,26 +57,33 @@ A genetic algorithm: - a population of 200; - tournament selection of size 3; genoxide's guide recommends tournaments with constraints, since roulette selection gives infeasible selections no weight; -- two-point crossover, which swaps a segment of items between the parents; -- bit-flip mutation at a rate of 1/20 per bit, one flip per child on average. +- uniform crossover, which takes each item's bit from either parent; +- bit-flip mutation at a rate of 1/50 per bit, one flip per child on average. -It stops as soon as it reaches the optimum, 625, which dynamic programming finds beforehand; or, -if it never does, after 200 generations without improvement, or after 2,000 generations. +It stops as soon as it reaches the optimum, which dynamic programming finds beforehand; or, if it +never does, after 200 generations without improvement, or after 2,000 generations. A run from seed +1, then runs from seeds 1 to 20. As a contrast, the same runs on an instance of Pisinger's strongly +correlated class, 50 items from seed 1. ## Output -The first line lists the chosen items, numbered from 0. The second gives their total value and -weight, and the capacity. The third gives the evaluations, and the optimum that dynamic programming -finds for comparison. - -The run stops at the generation that finds the optimum, the 36th, after 6,588 evaluations. Copies -of a parent inherit its score, so a generation of 200 needs fewer than 200 evaluations. +The first line gives the instance: its total weight and capacity. The next three give the run from +seed 1: the chosen items, numbered from 0, their total value and weight, and the evaluations, with +the optimum that dynamic programming finds. The last two lines give the runs from seeds 1 to 20: +how many reach the optimum, and after how many evaluations (the median); and how many reach the +optimum of the strongly correlated instance. Copies of a parent inherit its score, so a generation +of 200 needs fewer than 200 evaluations. The problem is evaluated in Rust in both languages, so both +print the same. [The project page](https://tachsin.gr/projects/genoxide/examples/knapsack) plays this run back. ## Good results -The optimum is 625: 13 items that weigh 395. The run finds it, and the dynamic programming line -confirms it. Over seeds 1 to 300, every run found it, after 4,988 evaluations at the median -and at most 33,544. With a population of 60, 18 of seeds 1 to 100 stopped short of it: -17 at 624, a selection that fills the knapsack exactly, and one at 619. +The optimum is 20,849: 32 items that weigh 12,551. The run from seed 1 finds it after 6,864 +evaluations, and so do all 20 runs from seeds 1 to 20, after a median of 6,936. Over seeds 1 to 200, +every run found it, within 16,536 evaluations; on the instances of seeds 1 to 5, 997 of 1,000 runs +(200 each) did. + +The strongly correlated instance, optimum 16,020, is another matter: the same GA reaches it from 2 +of the 20 seeds, and the other runs stall from 1 to 115 below it, within 0.7%. Pisinger's hard +classes are hard for genetic algorithms too. diff --git a/examples/knapsack/main.py b/examples/knapsack/main.py index f4831f50..f75d574d 100644 --- a/examples/knapsack/main.py +++ b/examples/knapsack/main.py @@ -1,8 +1,12 @@ -"""0/1 knapsack: choose items with the highest total value that fit in the knapsack. +"""0/1 knapsack: choose items with the highest total value that fit in the knapsack, on an +instance of 50 items drawn from Pisinger's uncorrelated class. -Shows a constraint with Deb's feasibility rules: the fitness function returns the value and how -far the weight exceeds the capacity, so overweight selections still guide the search towards the -feasible ones. The result is checked against the optimum found by dynamic programming. +Shows a constraint with Deb's feasibility rules: the fitness is the value and how far the weight +exceeds the capacity, so overweight selections still guide the search towards the feasible ones. +The instance is genoxide's problems.binary.Knapsack, generated from seed 1, which run evaluates in +Rust, and whose optimum dynamic programming finds. A run from seed 1 stops at it; then runs from +seeds 1 to 20 count how often the GA reaches it, and, as a contrast, how often it reaches the +optimum of an instance of Pisinger's strongly correlated class, which is harder. With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's page, with trace.py. @@ -16,71 +20,78 @@ from trace import Trace -# (weight, value): the first ten are the test data of Martello and Toth's Algorithm 632 (ACM TOMS -# 11(2), 1985), the other ten this example's own -ITEMS = [ - (23, 92), - (31, 57), - (29, 49), - (44, 68), - (53, 60), - (38, 43), - (63, 67), - (85, 84), - (89, 87), - (82, 72), - (12, 31), - (17, 29), - (41, 52), - (35, 38), - (27, 44), - (58, 61), - (19, 26), - (46, 55), - (71, 70), - (33, 41), -] -CAPACITY = 400 -WEIGHTS = np.array([weight for weight, _ in ITEMS]) -VALUES = np.array([value for _, value in ITEMS]) - - -def value(selection): - """The total value, and how much the weight exceeds the capacity (0 if the items fit).""" - weight = WEIGHTS[selection].sum() - return float(VALUES[selection].sum()), float(max(0, weight - CAPACITY)) - - -def optimum(): - """The best value that fits, by dynamic programming over the capacities.""" - best = [0] * (CAPACITY + 1) - for weight, value in ITEMS: - for capacity in range(CAPACITY, weight - 1, -1): - best[capacity] = max(best[capacity], best[capacity - weight] + value) - return best[CAPACITY] - - -ga = gx.Ga( - gx.Binary(len(ITEMS)), - population_size=200, - select=gx.Tournament(3), - crossover=gx.PointCrossover(2), - mutation=gx.BitFlip(rate=1 / len(ITEMS)), - seed=7, -) -# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page -trace = Trace(ITEMS, CAPACITY, optimum()) -# stops at the optimum that dynamic programming finds, or once the search stalls -result = ga.run( - value, - target=optimum(), - stagnation=200, - generations=2_000, - on_generation=trace.on_generation, +ITEMS = 50 +SEEDS = 20 + + +def ga(knapsack, seed): + """The genetic algorithm from ``seed``.""" + return gx.Ga( + knapsack.genome, + population_size=200, + select=gx.Tournament(3), + crossover=gx.UniformCrossover(), + mutation=gx.BitFlip(rate=1 / ITEMS), + seed=seed, + ) + + +def run(knapsack, seed, optimum, on_generation=None): + """A run from ``seed``: it stops at the optimum that dynamic programming finds, or once the + search stalls.""" + return ga(knapsack, seed).run( + knapsack, + target=optimum, + stagnation=200, + generations=2_000, + on_generation=on_generation, + ) + + +def seeds(knapsack, optimum): + """The runs from seeds 1 to 20 that reach the optimum, and their median evaluations.""" + evaluations = sorted( + result.evaluations + for result in (run(knapsack, seed, optimum) for seed in range(1, SEEDS + 1)) + if result.stop_reason == "target" + ) + middle = len(evaluations) // 2 + if not evaluations: + return 0, float("nan") + if len(evaluations) % 2: + return len(evaluations), float(evaluations[middle]) + return len(evaluations), (evaluations[middle - 1] + evaluations[middle]) / 2 + + +knapsack = gx.problems.binary.Knapsack(ITEMS, "uncorrelated", seed=1) +optimum = knapsack.optimum.value +weights, values = knapsack.weights, knapsack.profits +print( + f"{ITEMS} uncorrelated items (R = 1000, seed 1): total weight {weights.sum()}, " + f"capacity {knapsack.capacity}" ) +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(knapsack, optimum) +result = run(knapsack, 1, optimum, trace.on_generation) best = result.best_genome print(f"items {np.flatnonzero(best).tolist()}") -print(f"value {VALUES[best].sum()}, weight {WEIGHTS[best].sum()} of {CAPACITY}") -print(f"after {result.evaluations} evaluations; the optimum is {optimum()}") +print(f"value {values[best].sum()}, weight {weights[best].sum()} of {knapsack.capacity}") +print( + f"after {result.evaluations} evaluations; the optimum, by dynamic programming, is " + f"{optimum:.0f}" +) + +reached, median = seeds(knapsack, optimum) +print( + f"\nseeds 1 to {SEEDS}: {reached} reach {optimum:.0f}, after a median of {median:.1f} " + "evaluations" +) +strong = gx.problems.binary.Knapsack(ITEMS, "strongly_correlated", seed=1) +strong_optimum = strong.optimum.value +reached, _ = seeds(strong, strong_optimum) +print( + f"contrast, {ITEMS} strongly correlated items (seed 1): {reached} of {SEEDS} reach its " + f"optimum, {strong_optimum:.0f}" +) trace.write() diff --git a/examples/knapsack/main.rs b/examples/knapsack/main.rs index c43f3c79..89ca6e85 100644 --- a/examples/knapsack/main.rs +++ b/examples/knapsack/main.rs @@ -1,8 +1,12 @@ -//! 0/1 knapsack: choose items with the highest total value that fit in the knapsack. +//! 0/1 knapsack: choose items with the highest total value that fit in the knapsack, on an +//! instance of 50 items drawn from Pisinger's uncorrelated class. //! -//! Shows a constraint with Deb's feasibility rules: the fitness function returns the value and how -//! far the weight exceeds the capacity, so overweight selections still guide the search towards -//! the feasible ones. The result is checked against the optimum found by dynamic programming. +//! Shows a constraint with Deb's feasibility rules: the fitness is the value and how far the +//! weight exceeds the capacity, so overweight selections still guide the search towards the +//! feasible ones. The instance is genoxide's `problems::binary::Knapsack`, generated from seed 1, +//! whose optimum dynamic programming finds. A run from seed 1 stops at it; then runs from seeds 1 +//! to 20 count how often the GA reaches it, and, as a contrast, how often it reaches the optimum +//! of an instance of Pisinger's strongly correlated class, which is harder. //! //! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's //! page, with `trace.rs`. @@ -14,98 +18,92 @@ mod trace; use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::{Knapsack, KnapsackClass}; -// (weight, value): the first ten are the test data of Martello and Toth's Algorithm 632 (ACM TOMS -// 11(2), 1985), the other ten this example's own -const ITEMS: [(u32, u32); 20] = [ - (23, 92), - (31, 57), - (29, 49), - (44, 68), - (53, 60), - (38, 43), - (63, 67), - (85, 84), - (89, 87), - (82, 72), - (12, 31), - (17, 29), - (41, 52), - (35, 38), - (27, 44), - (58, 61), - (19, 26), - (46, 55), - (71, 70), - (33, 41), -]; -const CAPACITY: u32 = 400; +const ITEMS: usize = 50; +const SEEDS: u64 = 20; -// the total weight and value of the selected items -fn totals(selection: &Bits) -> (u32, u32) { - let (mut weight, mut value) = (0, 0); - for (item, selected) in selection.iter().enumerate() { - if selected { - weight += ITEMS[item].0; - value += ITEMS[item].1; - } - } - (weight, value) +// the genetic algorithm from `seed` +fn ga(knapsack: &Knapsack, seed: u64) -> Result> { + Ga::builder(knapsack.representation()) + .population_size(200) + .select(Tournament::new(3)?) + .crossover(UniformCrossover::new()) + .mutate(BitFlip::per_gene(1.0 / ITEMS as f64)?) + .seed(seed) + .build() } -// the total value, and how much the weight exceeds the capacity (0 if the items fit) -fn value(selection: &Bits) -> (f64, f64) { - let (weight, value) = totals(selection); - ( - f64::from(value), - constraint::at_most(f64::from(weight), f64::from(CAPACITY)), - ) +// stops at the optimum that dynamic programming finds, or once the search stalls +fn stop(optimum: f64) -> Stop { + Stop::target(optimum) + .or(Stop::stagnation(200)) + .or(Stop::generations(2_000)) } -// the best value that fits, by dynamic programming over the capacities -fn optimum() -> u32 { - let mut best = [0; CAPACITY as usize + 1]; - for (weight, value) in ITEMS { - for capacity in (weight as usize..=CAPACITY as usize).rev() { - best[capacity] = best[capacity].max(best[capacity - weight as usize] + value); +// the runs from seeds 1 to 20 that reach the optimum, and their median evaluations +fn seeds(knapsack: &Knapsack, optimum: f64) -> Result<(usize, f64)> { + let mut evaluations = Vec::new(); + for seed in 1..=SEEDS { + let outcome = Engine::new(ga(knapsack, seed)?, knapsack.clone()) + .stop_when(stop(optimum)) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations() as f64); } } - best[CAPACITY as usize] + evaluations.sort_by(f64::total_cmp); + let (count, middle) = (evaluations.len(), evaluations.len() / 2); + let median = match count { + 0 => f64::NAN, + n if n % 2 == 1 => evaluations[middle], + _ => (evaluations[middle - 1] + evaluations[middle]) / 2.0, + }; + Ok((count, median)) } fn main() -> Result<()> { - let ga = Ga::builder(Binary::new(ITEMS.len())?) - .population_size(200) - .select(Tournament::new(3)?) - .crossover(PointCrossover::two_point()) - .mutate(BitFlip::per_gene(1.0 / ITEMS.len() as f64)?) - .seed(7) - .build()?; + let knapsack = Knapsack::generator(KnapsackClass::Uncorrelated, ITEMS) + .seed(1) + .generate()?; + let optimum = knapsack.optimum().expect("small enough").value(); + let total: u64 = knapsack.weights().iter().sum(); + println!( + "{ITEMS} uncorrelated items (R = 1000, seed 1): total weight {total}, capacity {}", + knapsack.capacity() + ); // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page - let mut trace = trace::Trace::from_env(); - // stops at the optimum that dynamic programming finds, or once the search stalls - let target = Stop::target(f64::from(optimum())); - let outcome = Engine::new(ga, value) - .stop_when( - target - .or(Stop::stagnation(200)) - .or(Stop::generations(2_000)), - ) + let mut trace = trace::Trace::from_env(&knapsack, optimum); + let outcome = Engine::new(ga(&knapsack, 1)?, knapsack.clone()) + .stop_when(stop(optimum)) .on_generation(|snapshot| trace.record(snapshot)) .run()?; - let best = outcome.best_genome(); - let items: Vec = (0..ITEMS.len()) + let items: Vec = (0..ITEMS) .filter(|&item| best.get(item) == Some(true)) .collect(); - let (weight, value) = totals(best); + let (weight, value) = knapsack.totals(best); println!("items {items:?}"); - println!("value {value}, weight {weight} of {CAPACITY}"); + println!("value {value}, weight {weight} of {}", knapsack.capacity()); + println!( + "after {} evaluations; the optimum, by dynamic programming, is {optimum}", + outcome.evaluations() + ); + + let (reached, median) = seeds(&knapsack, optimum)?; + println!( + "\nseeds 1 to {SEEDS}: {reached} reach {optimum}, after a median of {median:.1} evaluations" + ); + let strong = Knapsack::generator(KnapsackClass::StronglyCorrelated, ITEMS) + .seed(1) + .generate()?; + let strong_optimum = strong.optimum().expect("small enough").value(); + let (reached, _) = seeds(&strong, strong_optimum)?; println!( - "after {} evaluations; the optimum is {}", - outcome.evaluations(), - optimum() + "contrast, {ITEMS} strongly correlated items (seed 1): {reached} of {SEEDS} reach its \ + optimum, {strong_optimum}" ); trace.write(); Ok(()) diff --git a/examples/knapsack/output.txt b/examples/knapsack/output.txt index 0ae56ed8..5b9406df 100644 --- a/examples/knapsack/output.txt +++ b/examples/knapsack/output.txt @@ -1,3 +1,7 @@ -items [0, 1, 2, 3, 5, 10, 11, 12, 13, 14, 16, 17, 19] -value 625, weight 395 of 400 -after 6588 evaluations; the optimum is 625 +50 uncorrelated items (R = 1000, seed 1): total weight 25397, capacity 12572 +items [0, 2, 5, 6, 7, 10, 14, 15, 16, 17, 18, 19, 21, 22, 26, 28, 29, 30, 31, 32, 33, 34, 36, 37, 39, 40, 43, 45, 46, 47, 48, 49] +value 20849, weight 12551 of 12572 +after 6864 evaluations; the optimum, by dynamic programming, is 20849 + +seeds 1 to 20: 20 reach 20849, after a median of 6936.0 evaluations +contrast, 50 strongly correlated items (seed 1): 2 of 20 reach its optimum, 16020 diff --git a/examples/knapsack/trace.json b/examples/knapsack/trace.json index f6dbe356..b39d77b4 100644 --- a/examples/knapsack/trace.json +++ b/examples/knapsack/trace.json @@ -1,39 +1,38 @@ -{"example":"knapsack","format":1,"log_y":true,"objective":"maximize","optimum":0.0,"plot":"knapsack","problem":{"capacity":400,"items":[{"value":92,"weight":23},{"value":57,"weight":31},{"value":49,"weight":29},{"value":68,"weight":44},{"value":60,"weight":53},{"value":43,"weight":38},{"value":67,"weight":63},{"value":84,"weight":85},{"value":87,"weight":89},{"value":72,"weight":82},{"value":31,"weight":12},{"value":29,"weight":17},{"value":52,"weight":41},{"value":38,"weight":35},{"value":44,"weight":27},{"value":61,"weight":58},{"value":26,"weight":19},{"value":55,"weight":46},{"value":70,"weight":71},{"value":41,"weight":33}]},"x_label":"generations","y_label":"value missing from the optimum","frames":[ -{"best":59.0,"evaluations":200,"generation":0,"median":59.5,"state":{"best":[1,1,0,1,0,0,1,0,1,0,0,1,0,0,1,0,1,1,0,1]}}, -{"best":24.0,"evaluations":375,"generation":1,"median":121.0,"state":{"best":[1,1,1,1,1,1,0,0,1,0,1,1,0,0,1,0,0,0,0,1]}}, -{"best":24.0,"evaluations":555,"generation":2,"median":134.5,"state":{"best":[1,1,1,1,1,1,0,0,1,0,1,1,0,0,1,0,0,0,0,1]}}, -{"best":24.0,"evaluations":741,"generation":3,"median":127.0,"state":{"best":[1,1,1,1,1,1,0,0,1,0,1,1,0,0,1,0,0,0,0,1]}}, -{"best":24.0,"evaluations":922,"generation":4,"median":119.5,"state":{"best":[1,1,1,1,1,1,0,0,1,0,1,1,0,0,1,0,0,0,0,1]}}, -{"best":24.0,"evaluations":1100,"generation":5,"median":108.5,"state":{"best":[1,1,1,1,1,1,0,0,1,0,1,1,0,0,1,0,0,0,0,1]}}, -{"best":23.0,"evaluations":1276,"generation":6,"median":104.0,"state":{"best":[1,1,1,1,0,1,0,1,0,0,1,1,0,1,1,0,1,0,0,1]}}, -{"best":23.0,"evaluations":1453,"generation":7,"median":88.0,"state":{"best":[1,1,1,1,0,1,0,1,0,0,1,1,0,1,1,0,1,0,0,1]}}, -{"best":23.0,"evaluations":1639,"generation":8,"median":88.5,"state":{"best":[1,1,1,1,0,1,0,1,0,0,1,1,0,1,1,0,1,0,0,1]}}, -{"best":15.0,"evaluations":1814,"generation":9,"median":101.0,"state":{"best":[1,1,1,1,1,0,0,0,1,0,1,1,1,0,1,0,0,0,0,1]}}, -{"best":3.0,"evaluations":1989,"generation":10,"median":89.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":2168,"generation":11,"median":84.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":2351,"generation":12,"median":74.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":2534,"generation":13,"median":87.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":2712,"generation":14,"median":80.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":2885,"generation":15,"median":79.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3064,"generation":16,"median":74.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3253,"generation":17,"median":85.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3433,"generation":18,"median":79.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3607,"generation":19,"median":86.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3783,"generation":20,"median":85.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":3961,"generation":21,"median":70.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":4124,"generation":22,"median":81.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":4308,"generation":23,"median":73.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":4480,"generation":24,"median":78.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":4654,"generation":25,"median":76.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":4829,"generation":26,"median":75.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5005,"generation":27,"median":76.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5188,"generation":28,"median":81.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5361,"generation":29,"median":78.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5538,"generation":30,"median":77.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5715,"generation":31,"median":73.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":5886,"generation":32,"median":63.5,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":6061,"generation":33,"median":47.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":3.0,"evaluations":6234,"generation":34,"median":66.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,0,1,1,0,1,0,1]}}, -{"best":1.0,"evaluations":6408,"generation":35,"median":64.0,"state":{"best":[1,1,1,1,1,0,0,0,0,0,1,1,1,0,1,1,1,1,0,0]}}, -{"best":0.0,"evaluations":6588,"generation":36,"median":66.0,"state":{"best":[1,1,1,1,0,1,0,0,0,0,1,1,1,1,1,0,1,1,0,1]}} +{"example":"knapsack","format":1,"log_y":true,"objective":"maximize","optimum":0.0,"plot":"knapsack","problem":{"capacity":12572,"items":[{"value":705,"weight":613},{"value":45,"weight":858},{"value":998,"weight":688},{"value":22,"weight":724},{"value":438,"weight":989},{"value":648,"weight":146},{"value":622,"weight":93},{"value":821,"weight":553},{"value":48,"weight":486},{"value":547,"weight":873},{"value":495,"weight":110},{"value":193,"weight":698},{"value":3,"weight":861},{"value":323,"weight":907},{"value":479,"weight":393},{"value":997,"weight":241},{"value":348,"weight":386},{"value":364,"weight":84},{"value":344,"weight":327},{"value":968,"weight":907},{"value":47,"weight":209},{"value":647,"weight":917},{"value":909,"weight":539},{"value":105,"weight":614},{"value":24,"weight":936},{"value":237,"weight":700},{"value":401,"weight":51},{"value":164,"weight":745},{"value":951,"weight":275},{"value":131,"weight":173},{"value":646,"weight":181},{"value":265,"weight":211},{"value":829,"weight":99},{"value":774,"weight":919},{"value":785,"weight":822},{"value":520,"weight":811},{"value":894,"weight":301},{"value":856,"weight":118},{"value":184,"weight":381},{"value":763,"weight":423},{"value":220,"weight":259},{"value":103,"weight":502},{"value":382,"weight":652},{"value":828,"weight":621},{"value":309,"weight":900},{"value":708,"weight":334},{"value":822,"weight":452},{"value":499,"weight":349},{"value":824,"weight":749},{"value":308,"weight":217}]},"x_label":"generations","y_label":"value missing from the optimum","frames":[ +{"best":5834.0,"evaluations":200,"generation":0,"median":8840.5,"state":{"best":[0,0,1,0,0,1,1,1,1,1,0,1,0,0,1,0,0,1,1,1,0,1,1,0,0,0,0,1,0,0,1,0,1,0,1,0,0,1,0,1,1,0,1,1,0,0,1,0,1,1]}}, +{"best":5480.0,"evaluations":391,"generation":1,"median":9231.5,"state":{"best":[1,1,0,0,0,0,1,1,1,0,1,1,0,0,1,0,1,1,0,0,0,0,0,0,1,0,1,0,1,1,1,1,1,1,1,1,1,1,1,0,0,0,0,1,0,1,1,1,1,1]}}, +{"best":5083.0,"evaluations":579,"generation":2,"median":8592.0,"state":{"best":[0,1,1,0,1,0,1,1,0,0,1,1,0,0,0,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,0,0,1,0,1,1,1,0,0,1,1,0,1,0,1,0,1,1,0]}}, +{"best":4510.0,"evaluations":769,"generation":3,"median":7906.5,"state":{"best":[1,0,0,0,0,1,1,1,0,1,0,1,0,0,1,1,1,1,0,1,1,0,1,0,1,1,0,0,1,1,1,0,1,1,0,0,1,1,1,0,0,0,0,1,0,1,1,1,0,1]}}, +{"best":3918.0,"evaluations":959,"generation":4,"median":7222.0,"state":{"best":[0,0,1,0,0,0,1,0,1,0,1,0,0,1,0,1,1,1,1,1,0,0,1,1,0,0,1,1,1,1,1,0,1,0,0,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":3166.0,"evaluations":1150,"generation":5,"median":6489.5,"state":{"best":[1,0,0,0,0,0,1,1,0,0,1,1,0,0,0,1,1,1,1,0,1,0,1,0,0,0,1,1,1,0,1,0,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":2361.0,"evaluations":1341,"generation":6,"median":6066.5,"state":{"best":[0,1,1,0,1,1,1,1,0,0,1,0,0,0,1,1,1,1,1,0,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,1,1,1,0,1,1,1,0,1,0,0,1,1,1,0]}}, +{"best":2361.0,"evaluations":1533,"generation":7,"median":5364.0,"state":{"best":[0,1,1,0,1,1,1,1,0,0,1,0,0,0,1,1,1,1,1,0,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,1,1,1,0,1,1,1,0,1,0,0,1,1,1,0]}}, +{"best":2361.0,"evaluations":1725,"generation":8,"median":4956.0,"state":{"best":[0,1,1,0,1,1,1,1,0,0,1,0,0,0,1,1,1,1,1,0,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,1,1,1,0,1,1,1,0,1,0,0,1,1,1,0]}}, +{"best":2306.0,"evaluations":1917,"generation":9,"median":4511.5,"state":{"best":[0,0,1,0,0,1,1,1,0,0,1,0,0,1,1,1,0,1,1,0,0,1,1,0,0,0,1,0,1,1,1,0,1,1,1,0,1,1,0,1,1,1,1,0,0,1,1,1,1,1]}}, +{"best":1842.0,"evaluations":2111,"generation":10,"median":4162.5,"state":{"best":[0,0,1,0,0,1,1,1,0,0,1,0,0,1,1,1,0,0,1,0,0,1,1,0,0,0,1,0,1,1,1,0,1,1,1,0,1,1,0,1,1,1,1,1,0,1,1,1,1,1]}}, +{"best":1842.0,"evaluations":2303,"generation":11,"median":3735.5,"state":{"best":[0,0,1,0,0,1,1,1,0,0,1,0,0,1,1,1,0,0,1,0,0,1,1,0,0,0,1,0,1,1,1,0,1,1,1,0,1,1,0,1,1,1,1,1,0,1,1,1,1,1]}}, +{"best":1135.0,"evaluations":2488,"generation":12,"median":3520.5,"state":{"best":[0,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,0,0,1,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,1,1,1,0,1,1,1,1,1]}}, +{"best":1135.0,"evaluations":2682,"generation":13,"median":3229.5,"state":{"best":[0,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,0,0,1,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,1,1,1,0,1,1,1,1,1]}}, +{"best":1031.0,"evaluations":2873,"generation":14,"median":3097.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,0,1,0,0,1,1,0,0,0,1,0,1,1,1,0,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":1015.0,"evaluations":3067,"generation":15,"median":2914.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,0,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,0]}}, +{"best":556.0,"evaluations":3257,"generation":16,"median":2521.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,0,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":3447,"generation":17,"median":2543.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":3639,"generation":18,"median":2474.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":3834,"generation":19,"median":2350.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":4026,"generation":20,"median":2152.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":4211,"generation":21,"median":2244.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":4404,"generation":22,"median":2002.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":4600,"generation":23,"median":2032.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":396.0,"evaluations":4789,"generation":24,"median":1817.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":4973,"generation":25,"median":1865.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":5165,"generation":26,"median":1743.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":5357,"generation":27,"median":1708.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":5543,"generation":28,"median":1660.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":5732,"generation":29,"median":1731.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":5923,"generation":30,"median":1508.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":6107,"generation":31,"median":1500.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":6296,"generation":32,"median":1471.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":212.0,"evaluations":6484,"generation":33,"median":1384.5,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,0,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,1,1,1,0,1,1,0,1,1,1,1,1]}}, +{"best":131.0,"evaluations":6679,"generation":34,"median":1303.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,1,1,0,0,0,1,0,1,0,1,1,1,1,1,0,1,1,0,1,1,0,0,1,0,1,1,1,1,1]}}, +{"best":0.0,"evaluations":6864,"generation":35,"median":1275.0,"state":{"best":[1,0,1,0,0,1,1,1,0,0,1,0,0,0,1,1,1,1,1,1,0,1,1,0,0,0,1,0,1,1,1,1,1,1,1,0,1,1,0,1,1,0,0,1,0,1,1,1,1,1]}} ]} diff --git a/examples/knapsack/trace.py b/examples/knapsack/trace.py index 106d8efd..59e70826 100644 --- a/examples/knapsack/trace.py +++ b/examples/knapsack/trace.py @@ -11,10 +11,11 @@ class Trace: """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" - def __init__(self, items, capacity, optimum): + def __init__(self, knapsack, optimum): self.path = os.environ.get("GENOXIDE_TRACE") self.frames = Frames(200) - self.items, self.capacity, self.optimum = items, capacity, optimum + self.items = list(zip(knapsack.weights.tolist(), knapsack.profits.tolist())) + self.capacity, self.optimum = knapsack.capacity, optimum @property def on_generation(self): diff --git a/examples/knapsack/trace.rs b/examples/knapsack/trace.rs index 45deed62..c01b56e0 100644 --- a/examples/knapsack/trace.rs +++ b/examples/knapsack/trace.rs @@ -3,22 +3,36 @@ //! and median values as what they miss of the optimum, for a log axis. The Python example writes //! the same file. -use crate::{CAPACITY, ITEMS, optimum}; use genoxide::observer::Snapshot; use genoxide::prelude::*; +use genoxide::problems::binary::Knapsack; use serde_json::{Value, json}; pub struct Trace { path: Option, frames: Frames, + items: Vec, + capacity: u64, + optimum: f64, } impl Trace { // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set - pub fn from_env() -> Self { + pub fn from_env(knapsack: &Knapsack, optimum: f64) -> Self { let path = std::env::var("GENOXIDE_TRACE").ok(); let frames = Frames::new(200); - Self { path, frames } + let items = knapsack.weights().iter().zip(knapsack.profits()); + let items = items + .map(|(weight, value)| json!({ "weight": weight, "value": value })) + .collect(); + let capacity = knapsack.capacity(); + Self { + path, + frames, + items, + capacity, + optimum, + } } // records a generation: the best selection so far @@ -33,10 +47,6 @@ impl Trace { // writes the trace, if there's one pub fn write(self) { let Some(path) = self.path else { return }; - let items: Vec = ITEMS - .iter() - .map(|&(weight, value)| json!({ "weight": weight, "value": value })) - .collect(); let settings = json!({ "format": 1, "example": "knapsack", @@ -46,12 +56,12 @@ impl Trace { "log_y": true, "optimum": 0.0, "plot": "knapsack", - "problem": { "capacity": CAPACITY, "items": items }, + "problem": { "capacity": self.capacity, "items": self.items }, }); write( &path, settings, - errors(self.frames.into_vec(), f64::from(optimum())), + errors(self.frames.into_vec(), self.optimum), ); } } diff --git a/examples/leading_ones/README.md b/examples/leading_ones/README.md new file mode 100644 index 00000000..04841180 --- /dev/null +++ b/examples/leading_ones/README.md @@ -0,0 +1,76 @@ +--- +title: LeadingOnes +category: binary +summary: Maximize the number of ones before the first zero of a string of 100 bits, with the (1+1) evolutionary algorithm, whose Θ(n²) time on it Droste, Jansen and Wegener proved. +reference: "Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary algorithm. Theoretical Computer Science 276(1-2): 51-81." +reference_url: https://doi.org/10.1016/S0304-3975(01)00182-7 +optimum: "100 (all ones)" +languages: [rust, python] +order: 11 +--- + +# LeadingOnes + +## The problem + +LeadingOnes counts the ones of a bit string before its first zero: + +```text +LeadingOnes(x) = Σᵢ₌₁ⁿ Πⱼ₌₁ⁱ xⱼ +``` + +The string 1101 0111 scores 2. Here the strings have 100 bits, so the best score is 100, for the +string of all ones. The definition is Droste, Jansen and Wegener's (2002, Definition 16), who took +the function from Rudolph's book (1997, *Convergence Properties of Evolutionary Algorithms*, Kovač, +Hamburg, not read). The function is genoxide's `problems::binary::LeadingOnes`. + +## What makes it hard + +The landscape is unimodal: appending a one to the leading ones always improves a string, so there +are no local optima. But only one bit can improve a string, the first zero, and nothing before it +may change. The bits after it don't count, and drift at random until the leading ones reach them. + +Droste et al. used the function to disprove a remark, which they attribute to Mühlenbein, that +every unimodal function takes the (1+1) evolutionary algorithm O(n log n) steps, as +[OneMax](../one_max/) does. Their Theorem 17 proves that LeadingOnes takes Θ(n²) steps on average: +at most e n² (27,183 for n = 100), and at least n²/6 (1,667) except with a probability exponentially +small in n. + +## Representation + +A `Binary` genome of 100 bits is the string itself. The fitness is its number of leading ones, to +maximize. + +## Algorithm + +The (1+1) evolutionary algorithm of Droste et al.: one string, each of whose bits is flipped with +probability 1/n, the child replacing it if it's no worse. In genoxide, that's `LocalSearch` with +bit-flip mutation at a rate of 1/100 as its neighbor, one neighbor per step, and the default +acceptance of neighbors that are no worse. + +`LocalSearch` draws a neighbor again when the mutation flips nothing, so each evaluation is a step +of the (1+1) EA that changes the string. A step of the (1+1) EA flips nothing with probability +(1 − 1/n)ⁿ = 0.366 for n = 100, and such steps don't change the string, so the search is the +(1+1) EA's, without its idle steps: the (1+1) EA takes on average 1 / 0.634 times as many steps +as there are evaluations here. + +A run from seed 1 stops at the optimum, or after 1,000,000 evaluations. Then runs from seeds 1 to +100 count the evaluations to the optimum. + +## Output + +The first lines give the run from seed 1: the evaluations at which the leading ones first reach 10, +20, … 100, and the evaluations to the optimum. The last two lines give the runs from seeds 1 to 100: +how many reach the optimum, the mean of their evaluations, the fewest, the median and the most, and +n² for comparison. The function is evaluated in Rust in both languages, so both print the same. + +[The project page](https://tachsin.gr/projects/genoxide/examples/leading-ones) plays back the run +from seed 1: the string, its leading ones growing from the left while the bits after them change at +random. + +## Good results + +The optimum is 100. The run from seed 1 reaches it after 4,452 evaluations. Every run from seeds 1 +to 100 reaches it, after 5,332 evaluations on average, from 3,503 to 7,802, with a median of +5,312.5. Counted with its idle steps, the (1+1) EA takes about 5,332 / 0.634 ≈ 8,410 steps on +average, between Droste et al.'s n²/6 and e n². diff --git a/examples/leading_ones/main.py b/examples/leading_ones/main.py new file mode 100644 index 00000000..8ce05019 --- /dev/null +++ b/examples/leading_ones/main.py @@ -0,0 +1,73 @@ +"""LeadingOnes: maximize the number of ones before the first zero of a string of 100 bits, with +the (1+1) evolutionary algorithm. + +The (1+1) EA keeps one string, flips each of its bits with probability 1/n and keeps the child if +it's no worse: genoxide's LocalSearch with BitFlip(rate=1/n) as its neighbor. A run from seed 1 +prints the evaluations at which the leading ones reach 10, 20, … 100; then runs from seeds 1 to +100 give the evaluations to the optimum, which Droste, Jansen and Wegener (2002) proved to be +Θ(n²). The function is genoxide's problems.binary.LeadingOnes, which run evaluates in Rust. + +With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's +page, with trace.py. + + python examples/leading_ones/main.py +""" + +import genoxide as gx + +from trace import Trace + +BITS = 100 +SEEDS = 100 +# the most evaluations of a run +BUDGET = 1_000_000 + + +def one_plus_one(problem, seed): + """The (1+1) evolutionary algorithm from ``seed``.""" + return gx.LocalSearch(problem.genome, neighbor=gx.BitFlip(rate=1 / BITS), seed=seed) + + +problem = gx.problems.binary.LeadingOnes(BITS) +optimum = problem.optimum.value +print(f"LeadingOnes of {BITS} bits, the (1+1) EA from seed 1") +print("leading ones evaluations") +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(BITS, optimum) +next_tenth = 10 + + +def progress(progress): + global next_tenth + best = progress.best_fitness or 0.0 + # the evaluations at which the leading ones first reach each tenth of the string + while best >= next_tenth: + print(f"{next_tenth:>12} {progress.evaluations:>11}") + next_tenth += 10 + trace.record(progress) + + +result = one_plus_one(problem, 1).run( + problem, target=optimum, evaluations=BUDGET, on_generation=progress +) +print( + f"{result.best_fitness:.0f} leading ones after {result.evaluations} evaluations " + f"(the optimum: {BITS})" +) + +# the evaluations to the optimum from seeds 1 to 100 +evaluations = [] +for seed in range(1, SEEDS + 1): + result = one_plus_one(problem, seed).run(problem, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + evaluations.append(result.evaluations) +evaluations.sort() +count = len(evaluations) +mean = sum(evaluations) / count +if count % 2: + median = float(evaluations[count // 2]) +else: + median = (evaluations[count // 2 - 1] + evaluations[count // 2]) / 2 +print(f"\nseeds 1 to {SEEDS}: {count} reach the optimum, after {mean:.0f} evaluations on average") +print(f"(fewest {evaluations[0]}, median {median:.1f}, most {evaluations[-1]}); n² = {BITS * BITS}") +trace.write() diff --git a/examples/leading_ones/main.rs b/examples/leading_ones/main.rs new file mode 100644 index 00000000..9d87aa95 --- /dev/null +++ b/examples/leading_ones/main.rs @@ -0,0 +1,92 @@ +//! LeadingOnes: maximize the number of ones before the first zero of a string of 100 bits, with +//! the (1+1) evolutionary algorithm. +//! +//! The (1+1) EA keeps one string, flips each of its bits with probability 1/n and keeps the child +//! if it's no worse: genoxide's `LocalSearch` with `BitFlip::per_gene(1/n)` as its neighbor. A run +//! from seed 1 prints the evaluations at which the leading ones reach 10, 20, … 100; then runs +//! from seeds 1 to 100 give the evaluations to the optimum, which Droste, Jansen and Wegener +//! (2002) proved to be Θ(n²). The function is genoxide's `problems::binary::LeadingOnes`. +//! +//! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's +//! page, with `trace.rs`. +//! +//! ```text +//! cargo run --release --example leading_ones +//! ``` + +mod trace; + +use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::LeadingOnes; + +const BITS: usize = 100; +const SEEDS: u64 = 100; +// the most evaluations of a run +const BUDGET: u64 = 1_000_000; + +// the (1+1) evolutionary algorithm from `seed` +fn one_plus_one(problem: &LeadingOnes, seed: u64) -> Result> { + LocalSearch::builder(problem.representation()) + .neighbor(BitFlip::per_gene(1.0 / BITS as f64)?) + .seed(seed) + .build() +} + +fn main() -> Result<()> { + let problem = LeadingOnes::new(BITS); + let optimum = problem.optimum().expect("known").value(); + println!("LeadingOnes of {BITS} bits, the (1+1) EA from seed 1"); + println!("leading ones evaluations"); + // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page + let mut trace = trace::Trace::from_env(BITS, optimum); + let mut next = 10.0; + let outcome = Engine::new(one_plus_one(&problem, 1)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .on_generation(|snapshot| { + let progress = snapshot.progress(); + let best = progress.best().and_then(Fitness::score).unwrap_or(0.0); + // the evaluations at which the leading ones first reach each tenth of the string + while best >= next { + println!("{next:>12.0} {:>11}", progress.evaluations()); + next += 10.0; + } + }) + .on_generation(|snapshot| trace.record(snapshot)) + .run()?; + println!( + "{} leading ones after {} evaluations (the optimum: {BITS})", + outcome.best_fitness(), + outcome.evaluations() + ); + + // the evaluations to the optimum from seeds 1 to 100 + let mut evaluations = Vec::new(); + for seed in 1..=SEEDS { + let outcome = Engine::new(one_plus_one(&problem, seed)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations()); + } + } + evaluations.sort_unstable(); + let count = evaluations.len(); + let mean = evaluations.iter().sum::() as f64 / count as f64; + let median = if count % 2 == 1 { + evaluations[count / 2] as f64 + } else { + (evaluations[count / 2 - 1] + evaluations[count / 2]) as f64 / 2.0 + }; + println!( + "\nseeds 1 to {SEEDS}: {count} reach the optimum, after {mean:.0} evaluations on average" + ); + println!( + "(fewest {}, median {median:.1}, most {}); n² = {}", + evaluations[0], + evaluations[count - 1], + BITS * BITS + ); + trace.write(); + Ok(()) +} diff --git a/examples/leading_ones/output.txt b/examples/leading_ones/output.txt new file mode 100644 index 00000000..30450f9c --- /dev/null +++ b/examples/leading_ones/output.txt @@ -0,0 +1,16 @@ +LeadingOnes of 100 bits, the (1+1) EA from seed 1 +leading ones evaluations + 10 291 + 20 414 + 30 947 + 40 1332 + 50 1927 + 60 2102 + 70 3313 + 80 3525 + 90 3952 + 100 4452 +100 leading ones after 4452 evaluations (the optimum: 100) + +seeds 1 to 100: 100 reach the optimum, after 5332 evaluations on average +(fewest 3503, median 5312.5, most 7802); n² = 10000 diff --git a/examples/leading_ones/trace.json b/examples/leading_ones/trace.json new file mode 100644 index 00000000..da04ee73 --- /dev/null +++ b/examples/leading_ones/trace.json @@ -0,0 +1,34 @@ +{"example":"leading_ones","format":1,"log_y":false,"objective":"maximize","optimum":100.0,"plot":"bits","problem":{"length":100},"x_label":"evaluations","y_label":"leading ones","frames":[ +{"best":1.0,"evaluations":1,"generation":0,"median":1.0,"state":{"population":["1000110110110000001001010011000101010111001100101001000011100110110101100111000101110000001111110001"]}}, +{"best":2.0,"evaluations":161,"generation":160,"median":2.0,"state":{"population":["1100011011011101001010001100110001110000100100010010001010100110001010010001101111001001000111000011"]}}, +{"best":7.0,"evaluations":289,"generation":288,"median":7.0,"state":{"population":["1111111010011110110001001011111011011110000001001010011000111000100111101010000001010001110001001011"]}}, +{"best":21.0,"evaluations":449,"generation":448,"median":21.0,"state":{"population":["1111111111111111111110100001011010000100001100000100100000011000001000010100111001010101110001001011"]}}, +{"best":21.0,"evaluations":577,"generation":576,"median":21.0,"state":{"population":["1111111111111111111110101000001010100011000011000011010010011000001010110101010101010111101010111011"]}}, +{"best":22.0,"evaluations":737,"generation":736,"median":22.0,"state":{"population":["1111111111111111111111000000101010111111010100111011110011110000111100011000001001010110010100001010"]}}, +{"best":23.0,"evaluations":865,"generation":864,"median":23.0,"state":{"population":["1111111111111111111111101111000101101111001100111011011100110110111100000111001111011110111010101111"]}}, +{"best":32.0,"evaluations":1025,"generation":1024,"median":32.0,"state":{"population":["1111111111111111111111111111111100010011100110111011110000101000110010010001101101011011010110000100"]}}, +{"best":33.0,"evaluations":1153,"generation":1152,"median":33.0,"state":{"population":["1111111111111111111111111111111110010100011010101001010111100111110000010001101111110111010000110000"]}}, +{"best":38.0,"evaluations":1313,"generation":1312,"median":38.0,"state":{"population":["1111111111111111111111111111111111111101011100001101111001110010010011100110111110010111000001100100"]}}, +{"best":40.0,"evaluations":1441,"generation":1440,"median":40.0,"state":{"population":["1111111111111111111111111111111111111111001010011101001100111000111111110100000000010110100111010111"]}}, +{"best":45.0,"evaluations":1601,"generation":1600,"median":45.0,"state":{"population":["1111111111111111111111111111111111111111111110010010111010001000101100001101111101010101001111111101"]}}, +{"best":45.0,"evaluations":1729,"generation":1728,"median":45.0,"state":{"population":["1111111111111111111111111111111111111111111110100000111110011010011110110101100111110010100100001011"]}}, +{"best":46.0,"evaluations":1889,"generation":1888,"median":46.0,"state":{"population":["1111111111111111111111111111111111111111111111001100101010000110100110101101111010111001100100111010"]}}, +{"best":54.0,"evaluations":2017,"generation":2016,"median":54.0,"state":{"population":["1111111111111111111111111111111111111111111111111111110011100000011100111101101010101001001111010010"]}}, +{"best":63.0,"evaluations":2177,"generation":2176,"median":63.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111110111110111011000000011001101110000000"]}}, +{"best":63.0,"evaluations":2305,"generation":2304,"median":63.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111110010000001110010101001100101001101101"]}}, +{"best":64.0,"evaluations":2465,"generation":2464,"median":64.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111000001011111110001001000101111110010"]}}, +{"best":64.0,"evaluations":2593,"generation":2592,"median":64.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111000100001001011101111101011110010001"]}}, +{"best":64.0,"evaluations":2753,"generation":2752,"median":64.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111010000101000111111011101101101000000"]}}, +{"best":64.0,"evaluations":2881,"generation":2880,"median":64.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111001000100100111110111010100111100100"]}}, +{"best":67.0,"evaluations":3041,"generation":3040,"median":67.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111000100011001010101011100100110101"]}}, +{"best":67.0,"evaluations":3169,"generation":3168,"median":67.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111001101110110111101001100001110101"]}}, +{"best":70.0,"evaluations":3329,"generation":3328,"median":70.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111010100101001110010100011000101"]}}, +{"best":76.0,"evaluations":3457,"generation":3456,"median":76.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111001100011110101000000001"]}}, +{"best":82.0,"evaluations":3617,"generation":3616,"median":82.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111000011000111101011"]}}, +{"best":82.0,"evaluations":3745,"generation":3744,"median":82.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111000111010101001010"]}}, +{"best":84.0,"evaluations":3905,"generation":3904,"median":84.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111110001011101001011"]}}, +{"best":92.0,"evaluations":4033,"generation":4032,"median":92.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111101110010"]}}, +{"best":96.0,"evaluations":4193,"generation":4192,"median":96.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111110000"]}}, +{"best":99.0,"evaluations":4321,"generation":4320,"median":99.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111110"]}}, +{"best":100.0,"evaluations":4452,"generation":4451,"median":100.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111111"]}} +]} diff --git a/examples/leading_ones/trace.py b/examples/leading_ones/trace.py new file mode 100644 index 00000000..b1870b59 --- /dev/null +++ b/examples/leading_ones/trace.py @@ -0,0 +1,183 @@ +"""The trace of the run for the plot on the example's page, written to the file that +``GENOXIDE_TRACE`` names: the string of the search, in at most 32 steps. The Rust example writes the same +file.""" + +import json +import math +import os + + +class Trace: + """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" + + def __init__(self, length, optimum): + self.path = os.environ.get("GENOXIDE_TRACE") + self.frames = Frames(32) + self.length, self.optimum = length, optimum + + @property + def on_generation(self): + """The callback for ``run``: None without a trace to record.""" + return self.record if self.path else None + + def record(self, progress): + """Records a step: the current string.""" + if self.path: + rows = ["".join("1" if one else "0" for one in row) for row in progress.population[:16]] + self.frames.push(frame(progress, {"population": rows})) + + def write(self): + """Writes the trace, if there's one.""" + if self.path: + settings = { + "format": 1, + "example": "leading_ones", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "leading ones", + "log_y": False, + "optimum": float(self.optimum), + "plot": "bits", + "problem": {"length": self.length}, + } + write(self.path, settings, self.frames.to_list()) + + +# ---- the same in every example's trace ---------------------------------------------------------- + + +class Frames: + """The frames of at most ``most`` generations, from the part of the run where what the page + plots changes: the frames after the last change are left out (a run that reached its target, + or a front that no longer moves), and the rest are spread evenly over the generations up to + it. While the run goes, up to 8 × ``most`` frames are kept: every ``every``-th generation, + with ``every`` doubling whenever there are that many, and the last one.""" + + def __init__(self, most): + self.most, self.every, self.kept, self.last = most, 1, [], None + + def push(self, frame): + if frame["generation"] % self.every: + self.last = frame + return + self.kept.append(frame) + self.last = None + if len(self.kept) == 8 * self.most: + self.every *= 2 + self.kept = [kept for kept in self.kept if kept["generation"] % self.every == 0] + + def to_list(self): + frames = self.kept + ([self.last] if self.last else []) + active = frames[: last_change(frames) + 1] + count, most = len(active), max(self.most, 2) + if count <= most: + return active + return [active[(i * (count - 1) + (most - 1) // 2) // (most - 1)] for i in range(most)] + + +def last_change(frames): + """The index of the frame after which nothing the page plots changes. To 3 significant + digits, as a plot shows them: the best, the median and, for a single objective (a numeric + best), the state; to within a hundredth of their range over the run: a front's + hypervolumes, in the state or in a grid's series.""" + if not frames: + return 0 + last = len(frames) - 1 + number = lambda value: isinstance(value, (int, float)) and not isinstance(value, bool) + single = any(number(frame.get("best")) for frame in frames) + + def measures(frame): + values = [] + state = frame.get("state") + hypervolume = state.get("hypervolume") if isinstance(state, dict) else None + for value in (hypervolume, frame.get("series")): + if number(value): + values.append(float(value)) + elif isinstance(value, dict): + values.extend(float(v) for _, v in sorted(value.items()) if number(v)) + return values + + measured = [measures(frame) for frame in frames] + end = measured[last] + tolerance = [] + for k in range(len(end)): + values = [values[k] for values in measured if k < len(values)] + tolerance.append((max(values) - min(values)) / 100.0) + + def same(frame, final, key, flush=False): + return coarse(frame.get(key), flush) == coarse(final.get(key), flush) + + def settled(i): + frame, final = frames[i], frames[last] + return ( + same(frame, final, "best") + and same(frame, final, "median") + and (not single or same(frame, final, "state", flush=True)) + and len(measured[i]) == len(end) + and all(abs(v - e) <= t for v, e, t in zip(measured[i], end, tolerance)) + ) + + first = last + while first > 0 and settled(first - 1): + first -= 1 + return first + + +def frame(progress, state): + """The frame of a generation: its progress, the median score of its population and + ``state``.""" + return { + "generation": progress.generation, + "evaluations": progress.evaluations, + "best": progress.best_fitness, + "median": median(progress.scores), + "state": state, + } + + +def median(scores): + """The median of the valid scores, None without any.""" + scores = sorted(float(score) for score in scores if not math.isnan(score)) + middle = len(scores) // 2 + if not scores: + return None + return scores[middle] if len(scores) % 2 else (scores[middle - 1] + scores[middle]) / 2 + + +def coarse(value, flush=False): + """``value`` with its numbers to 3 significant digits, as precisely as a plot shows them: two + frames whose plotted values agree to that precision look the same. With ``flush``, for the + solutions a plot draws on their ranges, numbers below 1e-6 in size count as 0.""" + if isinstance(value, float): + if flush and abs(value) < 1e-6: + value = 0.0 + return f"{value:.2e}" + if isinstance(value, (list, tuple)): + return "[" + ",".join(coarse(item, flush) for item in value) + "]" + if isinstance(value, dict): + items = sorted(value.items()) + return "{" + ",".join(f"{key}:{coarse(item, flush)}" for key, item in items) + "}" + return json.dumps(value) + + +def write(path, settings, frames): + """Writes the settings and the frames to ``path``, a frame per line.""" + lines = ",\n".join(map(to_json, frames)) + with open(path, "w", encoding="utf-8", newline="\n") as file: + file.write(f'{to_json(settings)[:-1]},"frames":[\n{lines}\n]}}\n') + + +def to_json(value): + """Compact JSON with sorted keys, and numbers rounded to 6 significant digits, as the Rust + example writes it.""" + return json.dumps(rounded(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def rounded(value): + if isinstance(value, dict): + return {key: rounded(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [rounded(item) for item in value] + if isinstance(value, float): + return float(f"{value:.5e}") if math.isfinite(value) else None + return value diff --git a/examples/leading_ones/trace.rs b/examples/leading_ones/trace.rs new file mode 100644 index 00000000..4685f35e --- /dev/null +++ b/examples/leading_ones/trace.rs @@ -0,0 +1,264 @@ +//! The trace of the run for the plot on the example's page, written to the file that +//! `GENOXIDE_TRACE` names: the string of the search, in at most 32 steps. The Python example writes the same +//! file. + +use genoxide::observer::Snapshot; +use genoxide::prelude::*; +use serde_json::{Value, json}; + +pub struct Trace { + path: Option, + frames: Frames, + length: usize, + optimum: f64, +} + +impl Trace { + // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set + pub fn from_env(length: usize, optimum: f64) -> Self { + let path = std::env::var("GENOXIDE_TRACE").ok(); + let frames = Frames::new(32); + Self { + path, + frames, + length, + optimum, + } + } + + // records a step: the current string + pub fn record(&mut self, snapshot: &Snapshot<'_, Bits>) { + if self.path.is_none() { + return; + } + let bits = |genome: &Bits| { + genome + .iter() + .map(|one| if one { '1' } else { '0' }) + .collect() + }; + let rows = snapshot.population().iter().take(16); + let rows: Vec = rows.map(|row| bits(row.genome())).collect(); + self.frames + .push(frame(snapshot, json!({ "population": rows }))); + } + + // writes the trace, if there's one + pub fn write(self) { + let Some(path) = self.path else { return }; + let settings = json!({ + "format": 1, + "example": "leading_ones", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "leading ones", + "log_y": false, + "optimum": self.optimum, + "plot": "bits", + "problem": { "length": self.length }, + }); + write(&path, settings, self.frames.into_vec()); + } +} + +// ---- the same in every example's trace --------------------------------------------------------- + +// the frames of at most `most` generations, from the part of the run where what the page plots +// changes: the frames after the last change are left out (a run that reached its target, or a +// front that no longer moves), and the rest are spread evenly over the generations up to it. While +// the run goes, up to 8 × `most` frames are kept: every `every`-th generation, with `every` +// doubling whenever there are that many, and the last one. +struct Frames { + most: usize, + every: u64, + kept: Vec<(u64, Value)>, + last: Option<(u64, Value)>, +} + +impl Frames { + fn new(most: usize) -> Self { + let (every, kept, last) = (1, Vec::new(), None); + Self { + most, + every, + kept, + last, + } + } + + fn push(&mut self, frame: Value) { + let generation = frame["generation"].as_u64().expect("a generation"); + if !generation.is_multiple_of(self.every) { + self.last = Some((generation, frame)); + return; + } + self.kept.push((generation, frame)); + self.last = None; + if self.kept.len() == 8 * self.most { + self.every *= 2; + let every = self.every; + self.kept.retain(|(generation, _)| generation % every == 0); + } + } + + fn into_vec(self) -> Vec { + let frames = self.kept.into_iter().chain(self.last); + let frames: Vec = frames.map(|(_, frame)| frame).collect(); + let active = &frames[..=last_change(&frames)]; + let (count, most) = (active.len(), self.most.max(2)); + if count <= most { + return active.to_vec(); + } + let at = |i: usize| active[(i * (count - 1) + (most - 1) / 2) / (most - 1)].clone(); + (0..most).map(at).collect() + } +} + +// the index of the frame after which nothing the page plots changes. To 3 significant digits, as +// a plot shows them: the best, the median and, for a single objective (a numeric best), the state; +// to within a hundredth of their range over the run: a front's hypervolumes, in the state or in a +// grid's series +fn last_change(frames: &[Value]) -> usize { + let Some(last) = frames.len().checked_sub(1) else { + return 0; + }; + let single = frames.iter().any(|frame| frame["best"].is_number()); + let measures = |frame: &Value| -> Vec { + let mut values = Vec::new(); + for value in [&frame["state"]["hypervolume"], &frame["series"]] { + match value { + Value::Number(number) => values.extend(number.as_f64()), + Value::Object(map) => values.extend(map.values().filter_map(Value::as_f64)), + _ => {} + } + } + values + }; + let measured: Vec> = frames.iter().map(measures).collect(); + let end = &measured[last]; + let tolerance: Vec = (0..end.len()) + .map(|k| { + let values = measured.iter().filter_map(|values| values.get(k).copied()); + let (low, high) = values.fold((f64::INFINITY, f64::NEG_INFINITY), |(low, high), v| { + (low.min(v), high.max(v)) + }); + (high - low) / 100.0 + }) + .collect(); + let settled = |i: usize| { + let (frame, final_frame) = (&frames[i], &frames[last]); + let same = + |key: &str, flush: bool| coarse(&frame[key], flush) == coarse(&final_frame[key], flush); + same("best", false) + && same("median", false) + && (!single || same("state", true)) + && measured[i].len() == end.len() + && measured[i] + .iter() + .zip(end) + .zip(&tolerance) + .all(|((value, end), tolerance)| (value - end).abs() <= *tolerance) + }; + let mut first = last; + while first > 0 && settled(first - 1) { + first -= 1; + } + first +} + +// the frame of a generation: its progress, the median score of its population and `state` +fn frame(snapshot: &Snapshot<'_, G>, state: Value) -> Value { + let progress = snapshot.progress(); + let population = snapshot.population().iter(); + let scores = population.filter_map(|individual| individual.fitness()?.score()); + json!({ + "generation": progress.generation(), + "evaluations": progress.evaluations(), + "best": progress.best().and_then(Fitness::score), + "median": median(scores.collect()), + "state": state, + }) +} + +// the median of the scores, None without any +fn median(mut scores: Vec) -> Option { + scores.sort_by(f64::total_cmp); + let middle = scores.len() / 2; + match scores.len() { + 0 => None, + n if n % 2 == 1 => Some(scores[middle]), + _ => Some((scores[middle - 1] + scores[middle]) / 2.0), + } +} + +// `value` with its numbers to 3 significant digits, as precisely as a plot shows them: two frames +// whose plotted values agree to that precision look the same. With `flush`, for the solutions a +// plot draws on their ranges, numbers below 1e-6 in size count as 0 +fn coarse(value: &Value, flush: bool) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => { + let number = number.as_f64().expect("f64"); + let number = if flush && number.abs() < 1e-6 { + 0.0 + } else { + number + }; + format!("{number:.2e}") + } + Value::Array(items) => { + format!( + "[{}]", + join(items.iter().map(|item| coarse(item, flush)).collect()) + ) + } + Value::Object(map) => { + let entry = |(key, item): (&String, &Value)| format!("{key}:{}", coarse(item, flush)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +// writes the settings and the frames to `path`, a frame per line +fn write(path: &str, settings: Value, frames: Vec) { + let frames: Vec = frames.iter().map(to_json).collect(); + let settings = to_json(&settings); + let head = &settings[..settings.len() - 1]; + let text = format!("{head},\"frames\":[\n{}\n]}}\n", frames.join(",\n")); + std::fs::write(path, text).expect("the trace is written"); +} + +// compact JSON with sorted keys, and numbers rounded to 6 significant digits and written as +// Python writes them (7542.0, 1e-08): the Python example writes the same file +fn to_json(value: &Value) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => python_float(number.as_f64().expect("f64")), + Value::Array(items) => format!("[{}]", join(items.iter().map(to_json).collect())), + Value::Object(map) => { + let entry = + |(key, item): (&String, &Value)| format!("{}:{}", json!(key), to_json(item)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +fn python_float(value: f64) -> String { + let rounded: f64 = format!("{value:.5e}").parse().expect("a number"); + let shortest = format!("{rounded:e}"); + let (mantissa, exponent) = shortest.split_once('e').expect("an exponent"); + let exponent: i32 = exponent.parse().expect("an exponent"); + if (-4..16).contains(&exponent) { + let text = rounded.to_string(); + if text.contains('.') { + text + } else { + text + ".0" + } + } else { + let sign = if exponent < 0 { '-' } else { '+' }; + format!("{mantissa}e{sign}{:02}", exponent.abs()) + } +} diff --git a/examples/nk_landscape/README.md b/examples/nk_landscape/README.md new file mode 100644 index 00000000..d968f902 --- /dev/null +++ b/examples/nk_landscape/README.md @@ -0,0 +1,83 @@ +--- +title: NK landscape +category: binary +summary: Maximize a rugged NK landscape of 20 bits, each interacting with 4 others, with iterated local search, checked against the optimum found by evaluating every string. +reference: "Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and its application to maturation of the immune response. Journal of Theoretical Biology 141(2): 211-245." +reference_url: https://doi.org/10.1016/S0022-5193(89)80019-0 +optimum: "0.742541 (landscape seed 1, by exhaustive search)" +languages: [rust, python] +order: 15 +--- + +# NK landscape + +## The problem + +Kauffman and Weinberger's NK model scores a string of N bits as the mean of N contributions. The +contribution wᵢ of bit i depends on its own value and on those of K other bits, its neighbors: +each of the 2^(K+1) combinations gets a value drawn independently and uniformly from (0, 1), and + +```text +W(x) = (1/N) Σᵢ wᵢ(xᵢ, x_{i₁}, …, x_{i_K}) +``` + +(their equation 1). K tunes the landscape: at K = 0 the bits are independent and there's a single +optimum; as K grows towards N − 1, the landscape gets more rugged, with more local optima and lower +ones, until at K = N − 1 every one-bit change gives a new random fitness. + +Here N = 20 and K = 4, and each bit's 4 neighbors are drawn at random from the other 19 (their +Table 2; "adjacent" neighbors, a bit's flanking ones on a circle, are their Table 1). The landscape +is genoxide's `problems::binary::NkLandscape::new(20, 4, Neighborhood::Random, 1)`: its neighbors +and its 20 tables of 32 values are drawn from seed 1 with genoxide's portable random numbers, so +it's the same landscape on every platform and in Python. Its contributions are odd multiples of 2⁻⁵³, +added up exactly, so the fitness is the same to the bit everywhere. + +## What makes it hard + +The optimum isn't known in closed form: each landscape has its own. genoxide's `optimum` computes it +exactly, here by evaluating all 2²⁰ = 1,048,576 strings, changing one bit at a time in Gray code +order so that only the contributions that depend on it change (a hundredth of a second). For +adjacent neighbors it uses dynamic programming instead, which solves landscapes of hundreds of bits. + +With K = 4, a bit's best value depends on its neighbors' values, which depend on theirs: changing +one bit changes its own contribution and those of the bits it bears on, five on average. The +landscape has many local optima, strings that no single flip improves. A search that only climbs +stops at one of them. + +## Representation + +A `Binary` genome of 20 bits is the string itself. The fitness is W, to maximize. + +## Algorithm + +Iterated local search: `LocalSearch` flips one bit at a time and keeps the change if the string is +no worse, and after 100 steps without a better best, restarts from the best string, changed by 5 +random flips, enough to leave its basin and few enough to keep most of it. A run from seed 1 stops +at the optimum, or after 500,000 evaluations. + +Then, on the landscapes of seeds 1 to 5, iterated local search and, as a contrast, a genetic +algorithm run from seeds 1 to 20: a population of 500, tournaments of 2, two-point crossover and +bit-flip mutation at 1/20 per bit, for 200 generations. + +## Output + +The first lines give the landscape and its optimum, with the string that reaches it, bit 0 first. +The table follows the run from seed 1: the evaluations at which its best improves, to 6 decimals, +and then the string it ends with. The last table gives, for each landscape, its optimum, how many +runs of iterated local search from seeds 1 to 20 reach it and their median evaluations, and how many +runs of the GA do. The landscapes are evaluated in Rust in both languages, so both print the same. + +[The project page](https://tachsin.gr/projects/genoxide/examples/nk-landscape) plays back the run +from seed 1: the string, as single flips climb and restarts kick it. + +## Good results + +The optimum of the landscape of seed 1 is 0.742541. The run from seed 1 reaches it after 927 +evaluations, at the same string as exhaustive search. On the 5 landscapes, iterated local search +reaches the optimum from all 20 seeds, after a median of 1,109 to 4,910 evaluations. Over landscapes +1 to 10 and seeds 1 to 100 each, it reached the optimum in 997 of 1,000 runs within 200,000 +evaluations. + +The genetic algorithm reaches it on the easier landscapes but not the others: from 5 of 20 seeds on +landscape 2 and 6 of 20 on landscape 5, where its other runs end below the optimum within 200 +generations. Each landscape is a new instance, so a method has to be judged over several. diff --git a/examples/nk_landscape/main.py b/examples/nk_landscape/main.py new file mode 100644 index 00000000..f4c92769 --- /dev/null +++ b/examples/nk_landscape/main.py @@ -0,0 +1,105 @@ +"""NK landscape: maximize a landscape of Kauffman and Weinberger's NK model, N = 20 bits each +interacting with K = 4 others chosen at random, with iterated local search, and check it against +the optimum found by evaluating all 2^20 strings. + +The landscape is drawn from seed 1 with genoxide's portable random numbers. Iterated local search +flips one bit at a time, keeping changes that are no worse, and after 100 steps without a better +best restarts from the best, changed by 5 random flips. A run from seed 1 prints its improvements; +then, on the landscapes of seeds 1 to 5, runs from seeds 1 to 20 count how often it reaches the +optimum, against a genetic algorithm as a contrast. The landscapes are genoxide's +problems.binary.NkLandscape, which run evaluates in Rust, and whose optimum searches every string. + +With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's +page, with trace.py. + + python examples/nk_landscape/main.py +""" + +import genoxide as gx + +from trace import Trace + +N = 20 +K = 4 +LANDSCAPES = 5 +SEEDS = 20 +# the most evaluations of a run of iterated local search, and generations of the GA +BUDGET = 500_000 +GENERATIONS = 200 + + +def ils(landscape, seed): + """Iterated local search from ``seed``.""" + return gx.LocalSearch( + landscape.genome, neighbor=gx.BitFlip(count=1), restart=(100, 5), seed=seed + ) + + +def median(values): + """The median of ``values``.""" + values = sorted(values) + middle = len(values) // 2 + return values[middle] if len(values) % 2 else (values[middle - 1] + values[middle]) / 2 + + +def text(bits): + """A bit string as 0s and 1s.""" + return "".join("1" if bit else "0" for bit in bits) + + +landscape = gx.problems.binary.NkLandscape(N, K, "random", seed=1) +optimum = landscape.optimum +print(f"NK landscape: N = {N}, K = {K}, random neighbors, drawn from seed 1") +print( + f"the optimum, over all 2^{N} strings: {optimum.value:.6f} at {text(optimum.solutions[0])}" +) +print("iterated local search from seed 1") +print("evaluations best") +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(N, optimum.value) +last = 0.0 + + +def progress(progress): + global last + best = progress.best_fitness or 0.0 + if best > last: + print(f"{progress.evaluations:>11} {best:.6f}") + last = best + trace.record(progress) + + +result = ils(landscape, 1).run( + landscape, target=optimum.value, evaluations=BUDGET, on_generation=progress +) +print( + f"{result.best_fitness:.6f} after {result.evaluations} evaluations, at " + f"{text(result.best_genome)}" +) + +# iterated local search and, as a contrast, a GA on landscapes 1 to 5, from seeds 1 to 20 +print("\nlandscape optimum ILS reaches median evaluations GA reaches") +for landscape_seed in range(1, LANDSCAPES + 1): + landscape = gx.problems.binary.NkLandscape(N, K, "random", seed=landscape_seed) + optimum = landscape.optimum.value + evaluations = [] + ga_reached = 0 + for seed in range(1, SEEDS + 1): + result = ils(landscape, seed).run(landscape, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + evaluations.append(result.evaluations) + ga = gx.Ga( + landscape.genome, + population_size=500, + select=gx.Tournament(2), + crossover=gx.PointCrossover(2), + mutation=gx.BitFlip(rate=1 / N), + seed=seed, + ) + result = ga.run(landscape, target=optimum, generations=GENERATIONS) + ga_reached += result.stop_reason == "target" + print( + f"{landscape_seed:>9} {optimum:.6f} {len(evaluations):>8}/{SEEDS} " + f"{median(evaluations):>18.1f} {ga_reached:>7}/{SEEDS}" + ) +trace.write() diff --git a/examples/nk_landscape/main.rs b/examples/nk_landscape/main.rs new file mode 100644 index 00000000..626b5dc0 --- /dev/null +++ b/examples/nk_landscape/main.rs @@ -0,0 +1,120 @@ +//! NK landscape: maximize a landscape of Kauffman and Weinberger's NK model, N = 20 bits each +//! interacting with K = 4 others chosen at random, with iterated local search, and check it +//! against the optimum found by evaluating all 2^20 strings. +//! +//! The landscape is drawn from seed 1 with genoxide's portable random numbers. Iterated local +//! search flips one bit at a time, keeping changes that are no worse, and after 100 steps without +//! a better best restarts from the best, changed by 5 random flips. A run from seed 1 prints its +//! improvements; then, on the landscapes of seeds 1 to 5, runs from seeds 1 to 20 count how often +//! it reaches the optimum, against a genetic algorithm as a contrast. The landscapes are genoxide's +//! `problems::binary::NkLandscape`, whose `optimum` searches every string. +//! +//! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's +//! page, with `trace.rs`. +//! +//! ```text +//! cargo run --release --example nk_landscape +//! ``` + +mod trace; + +use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::{Neighborhood, NkLandscape}; + +const N: usize = 20; +const K: usize = 4; +const LANDSCAPES: u64 = 5; +const SEEDS: u64 = 20; +// the most evaluations of a run of iterated local search, and generations of the GA +const BUDGET: u64 = 500_000; +const GENERATIONS: u64 = 200; + +// iterated local search from `seed` +fn ils(landscape: &NkLandscape, seed: u64) -> Result> { + LocalSearch::builder(landscape.representation()) + .neighbor(BitFlip::count(1)?) + .restart(100, 5) + .seed(seed) + .build() +} + +// the median of `values` +fn median(mut values: Vec) -> f64 { + values.sort_by(f64::total_cmp); + let middle = values.len() / 2; + if values.len() % 2 == 1 { + values[middle] + } else { + (values[middle - 1] + values[middle]) / 2.0 + } +} + +fn main() -> Result<()> { + let landscape = NkLandscape::new(N, K, Neighborhood::Random, 1)?; + let optimum = landscape.optimum().expect("small enough"); + println!("NK landscape: N = {N}, K = {K}, random neighbors, drawn from seed 1"); + println!( + "the optimum, over all 2^{N} strings: {:.6} at {}", + optimum.value(), + optimum.solutions()[0] + ); + println!("iterated local search from seed 1"); + println!("evaluations best"); + // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page + let mut trace = trace::Trace::from_env(N, optimum.value()); + let mut last = 0.0; + let outcome = Engine::new(ils(&landscape, 1)?, landscape.clone()) + .stop_when(Stop::target(optimum.value()).or(Stop::evaluations(BUDGET))) + .on_generation(|snapshot| { + let progress = snapshot.progress(); + let best = progress.best().and_then(Fitness::score).unwrap_or(0.0); + if best > last { + println!("{:>11} {best:.6}", progress.evaluations()); + last = best; + } + }) + .on_generation(|snapshot| trace.record(snapshot)) + .run()?; + println!( + "{:.6} after {} evaluations, at {}", + outcome.best_fitness().score().expect("valid"), + outcome.evaluations(), + outcome.best_genome() + ); + + // iterated local search and, as a contrast, a GA on landscapes 1 to 5, from seeds 1 to 20 + println!("\nlandscape optimum ILS reaches median evaluations GA reaches"); + for landscape_seed in 1..=LANDSCAPES { + let landscape = NkLandscape::new(N, K, Neighborhood::Random, landscape_seed)?; + let optimum = landscape.optimum().expect("small enough").value(); + let mut evaluations = Vec::new(); + let mut ga_reached = 0; + for seed in 1..=SEEDS { + let outcome = Engine::new(ils(&landscape, seed)?, landscape.clone()) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations() as f64); + } + let ga = Ga::builder(landscape.representation()) + .population_size(500) + .select(Tournament::new(2)?) + .crossover(PointCrossover::two_point()) + .mutate(BitFlip::per_gene(1.0 / N as f64)?) + .seed(seed) + .build()?; + let outcome = Engine::new(ga, landscape.clone()) + .stop_when(Stop::target(optimum).or(Stop::generations(GENERATIONS))) + .run()?; + ga_reached += usize::from(outcome.stop_reason() == StopReason::Target); + } + println!( + "{landscape_seed:>9} {optimum:.6} {:>8}/{SEEDS} {:>18.1} {ga_reached:>7}/{SEEDS}", + evaluations.len(), + median(evaluations) + ); + } + trace.write(); + Ok(()) +} diff --git a/examples/nk_landscape/output.txt b/examples/nk_landscape/output.txt new file mode 100644 index 00000000..90fc3b15 --- /dev/null +++ b/examples/nk_landscape/output.txt @@ -0,0 +1,25 @@ +NK landscape: N = 20, K = 4, random neighbors, drawn from seed 1 +the optimum, over all 2^20 strings: 0.742541 at 10101011100001100010 +iterated local search from seed 1 +evaluations best + 1 0.495078 + 3 0.543267 + 7 0.556021 + 8 0.617134 + 18 0.619018 + 20 0.625517 + 21 0.650122 + 31 0.691331 + 32 0.701480 + 46 0.723895 + 171 0.735152 + 907 0.736484 + 927 0.742541 +0.742541 after 927 evaluations, at 10101011100001100010 + +landscape optimum ILS reaches median evaluations GA reaches + 1 0.742541 20/20 1952.0 20/20 + 2 0.768337 20/20 2909.5 5/20 + 3 0.811498 20/20 3209.0 19/20 + 4 0.787638 20/20 1109.0 20/20 + 5 0.777704 20/20 4910.0 6/20 diff --git a/examples/nk_landscape/trace.json b/examples/nk_landscape/trace.json new file mode 100644 index 00000000..7d603016 --- /dev/null +++ b/examples/nk_landscape/trace.json @@ -0,0 +1,34 @@ +{"example":"nk_landscape","format":1,"log_y":false,"objective":"maximize","optimum":0.742541,"plot":"bits","problem":{"length":20},"x_label":"evaluations","y_label":"fitness W","frames":[ +{"best":0.495078,"evaluations":1,"generation":0,"median":0.495078,"state":{"population":["10001101101100000010"]}}, +{"best":0.650122,"evaluations":29,"generation":28,"median":0.650122,"state":{"population":["10111111111000000000"]}}, +{"best":0.723895,"evaluations":61,"generation":60,"median":0.723895,"state":{"population":["10111111111001000011"]}}, +{"best":0.723895,"evaluations":89,"generation":88,"median":0.723895,"state":{"population":["10111111111001000011"]}}, +{"best":0.723895,"evaluations":121,"generation":120,"median":0.723895,"state":{"population":["10111111111001000011"]}}, +{"best":0.723895,"evaluations":149,"generation":148,"median":0.641382,"state":{"population":["10111011111011101011"]}}, +{"best":0.735152,"evaluations":181,"generation":180,"median":0.735152,"state":{"population":["10101011110111100011"]}}, +{"best":0.735152,"evaluations":209,"generation":208,"median":0.735152,"state":{"population":["10101011110111100011"]}}, +{"best":0.735152,"evaluations":241,"generation":240,"median":0.735152,"state":{"population":["10101011110111100011"]}}, +{"best":0.735152,"evaluations":269,"generation":268,"median":0.735152,"state":{"population":["10101011110111100011"]}}, +{"best":0.735152,"evaluations":301,"generation":300,"median":0.67858,"state":{"population":["10101110110111100111"]}}, +{"best":0.735152,"evaluations":329,"generation":328,"median":0.713149,"state":{"population":["10101010110111100011"]}}, +{"best":0.735152,"evaluations":361,"generation":360,"median":0.735152,"state":{"population":["10101011110111100011"]}}, +{"best":0.735152,"evaluations":389,"generation":388,"median":0.63619,"state":{"population":["01101111110111100011"]}}, +{"best":0.735152,"evaluations":421,"generation":420,"median":0.677206,"state":{"population":["01111011100111100001"]}}, +{"best":0.735152,"evaluations":449,"generation":448,"median":0.677206,"state":{"population":["01111011100111100001"]}}, +{"best":0.735152,"evaluations":481,"generation":480,"median":0.658542,"state":{"population":["10001111110110100101"]}}, +{"best":0.735152,"evaluations":509,"generation":508,"median":0.706045,"state":{"population":["10101011100010100111"]}}, +{"best":0.735152,"evaluations":541,"generation":540,"median":0.724668,"state":{"population":["10101011100011100011"]}}, +{"best":0.735152,"evaluations":569,"generation":568,"median":0.724668,"state":{"population":["10101011100011100011"]}}, +{"best":0.735152,"evaluations":601,"generation":600,"median":0.662971,"state":{"population":["10111001010111100101"]}}, +{"best":0.735152,"evaluations":629,"generation":628,"median":0.662971,"state":{"population":["10111001010111100101"]}}, +{"best":0.735152,"evaluations":661,"generation":660,"median":0.662971,"state":{"population":["10111001010111100101"]}}, +{"best":0.735152,"evaluations":689,"generation":688,"median":0.692451,"state":{"population":["00111011100111110011"]}}, +{"best":0.735152,"evaluations":721,"generation":720,"median":0.692451,"state":{"population":["00111011100111110011"]}}, +{"best":0.735152,"evaluations":749,"generation":748,"median":0.692451,"state":{"population":["00111011100111110011"]}}, +{"best":0.735152,"evaluations":781,"generation":780,"median":0.681559,"state":{"population":["10101111100111100111"]}}, +{"best":0.735152,"evaluations":809,"generation":808,"median":0.688651,"state":{"population":["10101101100111100111"]}}, +{"best":0.735152,"evaluations":841,"generation":840,"median":0.688651,"state":{"population":["10101101100111100111"]}}, +{"best":0.735152,"evaluations":869,"generation":868,"median":0.688651,"state":{"population":["10101101100111100111"]}}, +{"best":0.735152,"evaluations":901,"generation":900,"median":0.729431,"state":{"population":["10101011110010100010"]}}, +{"best":0.742541,"evaluations":927,"generation":926,"median":0.742541,"state":{"population":["10101011100001100010"]}} +]} diff --git a/examples/nk_landscape/trace.py b/examples/nk_landscape/trace.py new file mode 100644 index 00000000..13bd1acc --- /dev/null +++ b/examples/nk_landscape/trace.py @@ -0,0 +1,183 @@ +"""The trace of the run for the plot on the example's page, written to the file that +``GENOXIDE_TRACE`` names: the string of the search, in at most 32 steps. The Rust example writes the same +file.""" + +import json +import math +import os + + +class Trace: + """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" + + def __init__(self, length, optimum): + self.path = os.environ.get("GENOXIDE_TRACE") + self.frames = Frames(32) + self.length, self.optimum = length, optimum + + @property + def on_generation(self): + """The callback for ``run``: None without a trace to record.""" + return self.record if self.path else None + + def record(self, progress): + """Records a step: the current string.""" + if self.path: + rows = ["".join("1" if one else "0" for one in row) for row in progress.population[:16]] + self.frames.push(frame(progress, {"population": rows})) + + def write(self): + """Writes the trace, if there's one.""" + if self.path: + settings = { + "format": 1, + "example": "nk_landscape", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "fitness W", + "log_y": False, + "optimum": float(self.optimum), + "plot": "bits", + "problem": {"length": self.length}, + } + write(self.path, settings, self.frames.to_list()) + + +# ---- the same in every example's trace ---------------------------------------------------------- + + +class Frames: + """The frames of at most ``most`` generations, from the part of the run where what the page + plots changes: the frames after the last change are left out (a run that reached its target, + or a front that no longer moves), and the rest are spread evenly over the generations up to + it. While the run goes, up to 8 × ``most`` frames are kept: every ``every``-th generation, + with ``every`` doubling whenever there are that many, and the last one.""" + + def __init__(self, most): + self.most, self.every, self.kept, self.last = most, 1, [], None + + def push(self, frame): + if frame["generation"] % self.every: + self.last = frame + return + self.kept.append(frame) + self.last = None + if len(self.kept) == 8 * self.most: + self.every *= 2 + self.kept = [kept for kept in self.kept if kept["generation"] % self.every == 0] + + def to_list(self): + frames = self.kept + ([self.last] if self.last else []) + active = frames[: last_change(frames) + 1] + count, most = len(active), max(self.most, 2) + if count <= most: + return active + return [active[(i * (count - 1) + (most - 1) // 2) // (most - 1)] for i in range(most)] + + +def last_change(frames): + """The index of the frame after which nothing the page plots changes. To 3 significant + digits, as a plot shows them: the best, the median and, for a single objective (a numeric + best), the state; to within a hundredth of their range over the run: a front's + hypervolumes, in the state or in a grid's series.""" + if not frames: + return 0 + last = len(frames) - 1 + number = lambda value: isinstance(value, (int, float)) and not isinstance(value, bool) + single = any(number(frame.get("best")) for frame in frames) + + def measures(frame): + values = [] + state = frame.get("state") + hypervolume = state.get("hypervolume") if isinstance(state, dict) else None + for value in (hypervolume, frame.get("series")): + if number(value): + values.append(float(value)) + elif isinstance(value, dict): + values.extend(float(v) for _, v in sorted(value.items()) if number(v)) + return values + + measured = [measures(frame) for frame in frames] + end = measured[last] + tolerance = [] + for k in range(len(end)): + values = [values[k] for values in measured if k < len(values)] + tolerance.append((max(values) - min(values)) / 100.0) + + def same(frame, final, key, flush=False): + return coarse(frame.get(key), flush) == coarse(final.get(key), flush) + + def settled(i): + frame, final = frames[i], frames[last] + return ( + same(frame, final, "best") + and same(frame, final, "median") + and (not single or same(frame, final, "state", flush=True)) + and len(measured[i]) == len(end) + and all(abs(v - e) <= t for v, e, t in zip(measured[i], end, tolerance)) + ) + + first = last + while first > 0 and settled(first - 1): + first -= 1 + return first + + +def frame(progress, state): + """The frame of a generation: its progress, the median score of its population and + ``state``.""" + return { + "generation": progress.generation, + "evaluations": progress.evaluations, + "best": progress.best_fitness, + "median": median(progress.scores), + "state": state, + } + + +def median(scores): + """The median of the valid scores, None without any.""" + scores = sorted(float(score) for score in scores if not math.isnan(score)) + middle = len(scores) // 2 + if not scores: + return None + return scores[middle] if len(scores) % 2 else (scores[middle - 1] + scores[middle]) / 2 + + +def coarse(value, flush=False): + """``value`` with its numbers to 3 significant digits, as precisely as a plot shows them: two + frames whose plotted values agree to that precision look the same. With ``flush``, for the + solutions a plot draws on their ranges, numbers below 1e-6 in size count as 0.""" + if isinstance(value, float): + if flush and abs(value) < 1e-6: + value = 0.0 + return f"{value:.2e}" + if isinstance(value, (list, tuple)): + return "[" + ",".join(coarse(item, flush) for item in value) + "]" + if isinstance(value, dict): + items = sorted(value.items()) + return "{" + ",".join(f"{key}:{coarse(item, flush)}" for key, item in items) + "}" + return json.dumps(value) + + +def write(path, settings, frames): + """Writes the settings and the frames to ``path``, a frame per line.""" + lines = ",\n".join(map(to_json, frames)) + with open(path, "w", encoding="utf-8", newline="\n") as file: + file.write(f'{to_json(settings)[:-1]},"frames":[\n{lines}\n]}}\n') + + +def to_json(value): + """Compact JSON with sorted keys, and numbers rounded to 6 significant digits, as the Rust + example writes it.""" + return json.dumps(rounded(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def rounded(value): + if isinstance(value, dict): + return {key: rounded(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [rounded(item) for item in value] + if isinstance(value, float): + return float(f"{value:.5e}") if math.isfinite(value) else None + return value diff --git a/examples/nk_landscape/trace.rs b/examples/nk_landscape/trace.rs new file mode 100644 index 00000000..18252334 --- /dev/null +++ b/examples/nk_landscape/trace.rs @@ -0,0 +1,264 @@ +//! The trace of the run for the plot on the example's page, written to the file that +//! `GENOXIDE_TRACE` names: the string of the search, in at most 32 steps. The Python example writes the same +//! file. + +use genoxide::observer::Snapshot; +use genoxide::prelude::*; +use serde_json::{Value, json}; + +pub struct Trace { + path: Option, + frames: Frames, + length: usize, + optimum: f64, +} + +impl Trace { + // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set + pub fn from_env(length: usize, optimum: f64) -> Self { + let path = std::env::var("GENOXIDE_TRACE").ok(); + let frames = Frames::new(32); + Self { + path, + frames, + length, + optimum, + } + } + + // records a step: the current string + pub fn record(&mut self, snapshot: &Snapshot<'_, Bits>) { + if self.path.is_none() { + return; + } + let bits = |genome: &Bits| { + genome + .iter() + .map(|one| if one { '1' } else { '0' }) + .collect() + }; + let rows = snapshot.population().iter().take(16); + let rows: Vec = rows.map(|row| bits(row.genome())).collect(); + self.frames + .push(frame(snapshot, json!({ "population": rows }))); + } + + // writes the trace, if there's one + pub fn write(self) { + let Some(path) = self.path else { return }; + let settings = json!({ + "format": 1, + "example": "nk_landscape", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "fitness W", + "log_y": false, + "optimum": self.optimum, + "plot": "bits", + "problem": { "length": self.length }, + }); + write(&path, settings, self.frames.into_vec()); + } +} + +// ---- the same in every example's trace --------------------------------------------------------- + +// the frames of at most `most` generations, from the part of the run where what the page plots +// changes: the frames after the last change are left out (a run that reached its target, or a +// front that no longer moves), and the rest are spread evenly over the generations up to it. While +// the run goes, up to 8 × `most` frames are kept: every `every`-th generation, with `every` +// doubling whenever there are that many, and the last one. +struct Frames { + most: usize, + every: u64, + kept: Vec<(u64, Value)>, + last: Option<(u64, Value)>, +} + +impl Frames { + fn new(most: usize) -> Self { + let (every, kept, last) = (1, Vec::new(), None); + Self { + most, + every, + kept, + last, + } + } + + fn push(&mut self, frame: Value) { + let generation = frame["generation"].as_u64().expect("a generation"); + if !generation.is_multiple_of(self.every) { + self.last = Some((generation, frame)); + return; + } + self.kept.push((generation, frame)); + self.last = None; + if self.kept.len() == 8 * self.most { + self.every *= 2; + let every = self.every; + self.kept.retain(|(generation, _)| generation % every == 0); + } + } + + fn into_vec(self) -> Vec { + let frames = self.kept.into_iter().chain(self.last); + let frames: Vec = frames.map(|(_, frame)| frame).collect(); + let active = &frames[..=last_change(&frames)]; + let (count, most) = (active.len(), self.most.max(2)); + if count <= most { + return active.to_vec(); + } + let at = |i: usize| active[(i * (count - 1) + (most - 1) / 2) / (most - 1)].clone(); + (0..most).map(at).collect() + } +} + +// the index of the frame after which nothing the page plots changes. To 3 significant digits, as +// a plot shows them: the best, the median and, for a single objective (a numeric best), the state; +// to within a hundredth of their range over the run: a front's hypervolumes, in the state or in a +// grid's series +fn last_change(frames: &[Value]) -> usize { + let Some(last) = frames.len().checked_sub(1) else { + return 0; + }; + let single = frames.iter().any(|frame| frame["best"].is_number()); + let measures = |frame: &Value| -> Vec { + let mut values = Vec::new(); + for value in [&frame["state"]["hypervolume"], &frame["series"]] { + match value { + Value::Number(number) => values.extend(number.as_f64()), + Value::Object(map) => values.extend(map.values().filter_map(Value::as_f64)), + _ => {} + } + } + values + }; + let measured: Vec> = frames.iter().map(measures).collect(); + let end = &measured[last]; + let tolerance: Vec = (0..end.len()) + .map(|k| { + let values = measured.iter().filter_map(|values| values.get(k).copied()); + let (low, high) = values.fold((f64::INFINITY, f64::NEG_INFINITY), |(low, high), v| { + (low.min(v), high.max(v)) + }); + (high - low) / 100.0 + }) + .collect(); + let settled = |i: usize| { + let (frame, final_frame) = (&frames[i], &frames[last]); + let same = + |key: &str, flush: bool| coarse(&frame[key], flush) == coarse(&final_frame[key], flush); + same("best", false) + && same("median", false) + && (!single || same("state", true)) + && measured[i].len() == end.len() + && measured[i] + .iter() + .zip(end) + .zip(&tolerance) + .all(|((value, end), tolerance)| (value - end).abs() <= *tolerance) + }; + let mut first = last; + while first > 0 && settled(first - 1) { + first -= 1; + } + first +} + +// the frame of a generation: its progress, the median score of its population and `state` +fn frame(snapshot: &Snapshot<'_, G>, state: Value) -> Value { + let progress = snapshot.progress(); + let population = snapshot.population().iter(); + let scores = population.filter_map(|individual| individual.fitness()?.score()); + json!({ + "generation": progress.generation(), + "evaluations": progress.evaluations(), + "best": progress.best().and_then(Fitness::score), + "median": median(scores.collect()), + "state": state, + }) +} + +// the median of the scores, None without any +fn median(mut scores: Vec) -> Option { + scores.sort_by(f64::total_cmp); + let middle = scores.len() / 2; + match scores.len() { + 0 => None, + n if n % 2 == 1 => Some(scores[middle]), + _ => Some((scores[middle - 1] + scores[middle]) / 2.0), + } +} + +// `value` with its numbers to 3 significant digits, as precisely as a plot shows them: two frames +// whose plotted values agree to that precision look the same. With `flush`, for the solutions a +// plot draws on their ranges, numbers below 1e-6 in size count as 0 +fn coarse(value: &Value, flush: bool) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => { + let number = number.as_f64().expect("f64"); + let number = if flush && number.abs() < 1e-6 { + 0.0 + } else { + number + }; + format!("{number:.2e}") + } + Value::Array(items) => { + format!( + "[{}]", + join(items.iter().map(|item| coarse(item, flush)).collect()) + ) + } + Value::Object(map) => { + let entry = |(key, item): (&String, &Value)| format!("{key}:{}", coarse(item, flush)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +// writes the settings and the frames to `path`, a frame per line +fn write(path: &str, settings: Value, frames: Vec) { + let frames: Vec = frames.iter().map(to_json).collect(); + let settings = to_json(&settings); + let head = &settings[..settings.len() - 1]; + let text = format!("{head},\"frames\":[\n{}\n]}}\n", frames.join(",\n")); + std::fs::write(path, text).expect("the trace is written"); +} + +// compact JSON with sorted keys, and numbers rounded to 6 significant digits and written as +// Python writes them (7542.0, 1e-08): the Python example writes the same file +fn to_json(value: &Value) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => python_float(number.as_f64().expect("f64")), + Value::Array(items) => format!("[{}]", join(items.iter().map(to_json).collect())), + Value::Object(map) => { + let entry = + |(key, item): (&String, &Value)| format!("{}:{}", json!(key), to_json(item)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +fn python_float(value: f64) -> String { + let rounded: f64 = format!("{value:.5e}").parse().expect("a number"); + let shortest = format!("{rounded:e}"); + let (mantissa, exponent) = shortest.split_once('e').expect("an exponent"); + let exponent: i32 = exponent.parse().expect("an exponent"); + if (-4..16).contains(&exponent) { + let text = rounded.to_string(); + if text.contains('.') { + text + } else { + text + ".0" + } + } else { + let sign = if exponent < 0 { '-' } else { '+' }; + format!("{mantissa}e{sign}{:02}", exponent.abs()) + } +} diff --git a/examples/one_max/README.md b/examples/one_max/README.md index a9a4eecd..e0a10dbc 100644 --- a/examples/one_max/README.md +++ b/examples/one_max/README.md @@ -2,8 +2,8 @@ title: OneMax category: binary summary: Find the bit string with the most ones, the "hello world" of genetic algorithms. -reference: "Ackley, D. H. (1987). A Connectionist Machine for Genetic Hillclimbing. Kluwer Academic Publishers." -reference_url: https://doi.org/10.1007/978-1-4613-1997-9 +reference: "Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary algorithm. Theoretical Computer Science 276(1-2): 51-81." +reference_url: https://doi.org/10.1016/S0304-3975(01)00182-7 optimum: "500 (all ones)" languages: [rust, python] order: 10 @@ -13,9 +13,12 @@ order: 10 ## The problem -OneMax scores a bit string by the number of its ones. Ackley (1987) used it as one of the test -functions of his genetic hill climber. The string 10110010 scores 4. Here the strings have 500 bits, -so the best score is 500, for the string of all ones. +OneMax scores a bit string by the number of its ones, `Σ xᵢ`: Droste, Jansen and Wegener (2002, +Definition 9) define it as the linear function whose weights are all 1. It has no single origin; +Ackley (1987, *A Connectionist Machine for Genetic Hillclimbing*, section 3.3.1) tested his genetic +hill climber on a "One Max" that scores ten times the number of ones. The string 10110010 scores 4. +Here the strings have 500 bits, so the best score is 500, for the string of all ones. The function +is genoxide's `problems::binary::OneMax`. ## What makes it hard @@ -26,8 +29,7 @@ flips, so there are no local optima. What the problem measures is how fast an al The slow part is the end. At a flip rate of 1/500 per bit, a mutation flips a given zero with probability 0.2%, and it can flip a one at the same time. For the (1+1) evolutionary algorithm, which keeps one string and flips each bit with probability 1/n, the expected number of evaluations -is of the order of n log n (Droste, Jansen and Wegener, 2002, Theoretical Computer Science 276(1-2): -51-81). +is of the order of n log n (Droste et al., 2002, Lemma 10). ## Representation diff --git a/examples/one_max/main.py b/examples/one_max/main.py index 031c0d4b..fa5e176d 100644 --- a/examples/one_max/main.py +++ b/examples/one_max/main.py @@ -1,7 +1,8 @@ """OneMax: find the bit string with the most ones. The "hello world" of genetic algorithms: a binary genome, tournament selection, uniform crossover -and bit-flip mutation, with the best count printed every 50 generations. +and bit-flip mutation, with the best count printed every 50 generations. The function is +genoxide's problems.binary.OneMax, which run evaluates in Rust. With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's page, with trace.py. @@ -15,8 +16,9 @@ LEN = 500 +problem = gx.problems.binary.OneMax(LEN) ga = gx.Ga( - gx.Binary(LEN), + problem.genome, population_size=100, select=gx.Tournament(3), crossover=gx.UniformCrossover(), @@ -36,7 +38,9 @@ def progress(progress): print("generation best") -result = ga.run(lambda bits: bits.sum(), target=LEN, generations=10_000, on_generation=progress) +result = ga.run( + problem, target=problem.optimum.value, generations=10_000, on_generation=progress +) print( f"\n{result.best_fitness:.0f} ones after {result.generations} generations and " f"{result.evaluations} evaluations (the optimum: {LEN})" diff --git a/examples/one_max/main.rs b/examples/one_max/main.rs index ada54464..52598fdf 100644 --- a/examples/one_max/main.rs +++ b/examples/one_max/main.rs @@ -1,7 +1,8 @@ //! OneMax: find the bit string with the most ones. //! //! The "hello world" of genetic algorithms: a binary genome, tournament selection, uniform -//! crossover and bit-flip mutation, with the best count printed every 50 generations. +//! crossover and bit-flip mutation, with the best count printed every 50 generations. The function +//! is genoxide's `problems::binary::OneMax`. //! //! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's //! page, with `trace.rs`. @@ -13,11 +14,15 @@ mod trace; use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::OneMax; const LEN: usize = 500; fn main() -> Result<()> { - let ga = Ga::builder(Binary::new(LEN)?) + let problem = OneMax::new(LEN); + let optimum = problem.optimum().expect("known").value(); + let ga = Ga::builder(problem.representation()) .population_size(100) .select(Tournament::new(3)?) .crossover(UniformCrossover::new()) @@ -28,8 +33,8 @@ fn main() -> Result<()> { // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page let mut trace = trace::Trace::from_env(); println!("generation best"); - let outcome = Engine::new(ga, |genome: &Bits| genome.count_ones() as f64) - .stop_when(Stop::target(LEN as f64).or(Stop::generations(10_000))) + let outcome = Engine::new(ga, problem) + .stop_when(Stop::target(optimum).or(Stop::generations(10_000))) .on_generation(|snapshot| { let progress = snapshot.progress(); if progress.generation() % 50 == 0 { diff --git a/examples/royal_road_r1/README.md b/examples/royal_road_r1/README.md new file mode 100644 index 00000000..2e929807 --- /dev/null +++ b/examples/royal_road_r1/README.md @@ -0,0 +1,80 @@ +--- +title: Royal road R1 +category: binary +summary: Maximize 8 blocks of 8 ones that score only when complete, with random-mutation hill climbing, which Mitchell, Holland and Forrest found nearly ten times faster on it than their genetic algorithm. +reference: "Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a genetic algorithm outperform hill climbing? Advances in Neural Information Processing Systems 6: 51-58." +reference_url: https://proceedings.neurips.cc/paper_files/paper/1993/hash/ab88b15733f543179858600245108dd8-Abstract.html +optimum: "64 (all ones)" +languages: [rust, python] +order: 13 +family: Royal road +tab: R1 +--- + +# Royal road R1 + +## The problem + +R1 is a sum over 8 schemas, each a block of 8 consecutive bits that must all be ones, s₁ = +11111111∗∗…∗ to s₈ = ∗∗…∗11111111, each with the coefficient cᵢ = 8, its order (Mitchell, Holland +and Forrest, 1994, Figure 1): + +```text +R1(x) = Σᵢ cᵢ δᵢ(x), δᵢ(x) = 1 if x is an instance of sᵢ, 0 otherwise +``` + +A string scores 8 for each complete block: a string with two complete blocks scores 16, and the +string of 64 ones, the optimum, 64. The function is genoxide's `problems::binary::RoyalRoad::r1()`. +Its hierarchical sibling, [R2](../royal_road_r2/), scores the blocks' combinations too. + +## What makes it hard + +The royal road functions were designed to be easy for a genetic algorithm: crossover would combine +complete blocks into strings with more of them, a "royal road" to the optimum, while a hill climber +would have to set 8 bits at once to complete a block. Both expectations failed (Forrest and +Mitchell, 1993, as summarized by Mitchell et al.). A block short of complete scores nothing, so a string's +fitness says nothing about its 7 of 8 ones: the search drifts on plateaus. In a genetic algorithm, +a string with a new block takes over the population, and the zeros next to the block hitchhike with +it, slowing the discovery of the blocks beside it. + +A hill climber that flips one bit at a time and keeps any change that's no worse drifts freely +inside the incomplete blocks without losing the complete ones. Mitchell et al. analyze it: the +expected time to find one block of K ones is slightly above 2ᴷ, 301.2 evaluations for K = 8, and to +find N blocks about E(K, 1) N (1 + 1/2 + … + 1/N), 6,549 evaluations for R1 (their equation 1). + +## Representation + +A `Binary` genome of 64 bits is the string itself. The fitness is R1, to maximize. + +## Algorithm + +Random-mutation hill climbing (RMHC), Mitchell et al.'s: start from a random string, flip one bit +chosen at random, and keep the change if the string is no worse. In genoxide, that's `LocalSearch` +with `BitFlip::count(1)` as its neighbor, one neighbor per step, and the default acceptance of +neighbors that are no worse. A run from seed 1 stops at 64, or after 256,000 evaluations, the +paper's limit; then 200 runs, seeds 1 to 200, as in the paper's Table 1. + +For comparison, a genetic algorithm with the paper's population of 128, single-point crossover at a +rate of 0.7 and mutation of 0.005 per bit (Mitchell, Forrest and Holland, 1992), with tournament +selection of size 2 in place of their fitness-proportionate selection with sigma scaling, from seeds +1 to 50 with the same limit. + +## Output + +The first lines follow the run from seed 1: the evaluations at which it completes a block, its +score rising by 8, and the evaluations to 64. Then two lines for the 200 runs of RMHC, how many +reach 64 and the mean and median of their evaluations, with the paper's for comparison, and two for +the 50 runs of the genetic algorithm. The function is evaluated in Rust in both languages, so both +print the same. + +[The project page](https://tachsin.gr/projects/genoxide/examples/royal-road-r1) plays back the run +from seed 1: the string, its blocks filling with ones one after another. + +## Good results + +The optimum is 64. The run from seed 1 reaches it after 10,101 evaluations, and every one of the 200 +runs does, after 6,793 evaluations on average, with a median of 6,144. Mitchell et al.'s 200 runs +took 6,179 on average (with a standard error of 186) and 5,775 at the median, against their expected +6,549. The genetic algorithm reaches 64 from all 50 seeds too, after 34,303 evaluations on average, +five times as many as hill climbing; the paper's GA took 61,334 (and genoxide doesn't evaluate a +child that is a copy of its parent again). diff --git a/examples/royal_road_r1/main.py b/examples/royal_road_r1/main.py new file mode 100644 index 00000000..111e8e35 --- /dev/null +++ b/examples/royal_road_r1/main.py @@ -0,0 +1,98 @@ +"""Royal road R1: maximize 8 blocks of 8 ones, each scoring only when complete, with +random-mutation hill climbing, which Mitchell, Holland and Forrest (1994) found faster on it than +their genetic algorithm. + +Random-mutation hill climbing (RMHC) flips one bit, chosen at random, and keeps the change if it's +no worse: genoxide's LocalSearch with BitFlip(count=1). A run from seed 1 prints the evaluations +at which each block is completed; then 200 runs, as in the paper's Table 1, give the mean and +median evaluations to the optimum, 64. As a comparison, a genetic algorithm with the paper's +population, crossover and mutation (and tournament selection) from 50 seeds. The function is +genoxide's problems.binary.RoyalRoad.r1(), which run evaluates in Rust. + +With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's +page, with trace.py. + + python examples/royal_road_r1/main.py +""" + +import genoxide as gx + +from trace import Trace + +BITS = 64 +# the runs of RMHC, as in the paper's Table 1, and of the genetic algorithm +RUNS = 200 +GA_RUNS = 50 +# the most evaluations of a run, the paper's +BUDGET = 256_000 + + +def rmhc(problem, seed): + """Random-mutation hill climbing from ``seed``.""" + return gx.LocalSearch(problem.genome, neighbor=gx.BitFlip(count=1), seed=seed) + + +def mean_and_median(values): + """The mean and the median of ``values``.""" + values = sorted(values) + middle = len(values) // 2 + median = values[middle] if len(values) % 2 else (values[middle - 1] + values[middle]) / 2 + return sum(values) / len(values), median + + +problem = gx.problems.binary.RoyalRoad.r1() +optimum = problem.optimum.value +print(f"Royal road R1: 8 blocks of 8 bits, maximum {optimum:.0f}") +print("random-mutation hill climbing from seed 1") +print("R1 evaluations") +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(BITS, optimum) +last = 0.0 + + +def progress(progress): + global last + best = progress.best_fitness or 0.0 + # the evaluations at which a block is completed + if best > last: + print(f"{best:>2.0f} {progress.evaluations:>11}") + last = best + trace.record(progress) + + +result = rmhc(problem, 1).run(problem, target=optimum, evaluations=BUDGET, on_generation=progress) +print(f"{result.best_fitness:.0f} after {result.evaluations} evaluations") + +# RMHC from seeds 1 to 200, and the genetic algorithm from seeds 1 to 50 +evaluations = [] +for seed in range(1, RUNS + 1): + result = rmhc(problem, seed).run(problem, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + evaluations.append(result.evaluations) +mean, median = mean_and_median(evaluations) +print( + f"\nRMHC, seeds 1 to {RUNS}: {len(evaluations)} reach {optimum:.0f}; mean {mean:.0f}, " + f"median {median:.0f} evaluations" +) +print(" (the paper's 200 runs: mean 6179, median 5775)") +evaluations = [] +for seed in range(1, GA_RUNS + 1): + ga = gx.Ga( + problem.genome, + population_size=128, + select=gx.Tournament(2), + crossover=gx.PointCrossover(1), + crossover_rate=0.7, + mutation=gx.BitFlip(rate=0.005), + seed=seed, + ) + result = ga.run(problem, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + evaluations.append(result.evaluations) +mean, median = mean_and_median(evaluations) +print( + f"GA, seeds 1 to {GA_RUNS}: {len(evaluations)} reach {optimum:.0f}; mean {mean:.0f}, " + f"median {median:.0f} evaluations" +) +print(" (the paper's GA, 200 runs: mean 61334, median 54208)") +trace.write() diff --git a/examples/royal_road_r1/main.rs b/examples/royal_road_r1/main.rs new file mode 100644 index 00000000..af619e6b --- /dev/null +++ b/examples/royal_road_r1/main.rs @@ -0,0 +1,123 @@ +//! Royal road R1: maximize 8 blocks of 8 ones, each scoring only when complete, with +//! random-mutation hill climbing, which Mitchell, Holland and Forrest (1994) found faster on it +//! than their genetic algorithm. +//! +//! Random-mutation hill climbing (RMHC) flips one bit, chosen at random, and keeps the change if +//! it's no worse: genoxide's `LocalSearch` with `BitFlip::count(1)`. A run from seed 1 prints the +//! evaluations at which each block is completed; then 200 runs, as in the paper's Table 1, give +//! the mean and median evaluations to the optimum, 64. As a comparison, a genetic algorithm with +//! the paper's population, crossover and mutation (and tournament selection) from 50 seeds. The +//! function is genoxide's `problems::binary::RoyalRoad::r1`. +//! +//! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's +//! page, with `trace.rs`. +//! +//! ```text +//! cargo run --release --example royal_road_r1 +//! ``` + +mod trace; + +use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::RoyalRoad; + +const BITS: usize = 64; +// the runs of RMHC, as in the paper's Table 1, and of the genetic algorithm +const RUNS: u64 = 200; +const GA_RUNS: u64 = 50; +// the most evaluations of a run, the paper's +const BUDGET: u64 = 256_000; + +// random-mutation hill climbing from `seed` +fn rmhc(problem: &RoyalRoad, seed: u64) -> Result> { + LocalSearch::builder(problem.representation()) + .neighbor(BitFlip::count(1)?) + .seed(seed) + .build() +} + +// the mean and the median of `values` +fn mean_and_median(mut values: Vec) -> (f64, f64) { + values.sort_by(f64::total_cmp); + let middle = values.len() / 2; + let median = if values.len() % 2 == 1 { + values[middle] + } else { + (values[middle - 1] + values[middle]) / 2.0 + }; + (values.iter().sum::() / values.len() as f64, median) +} + +fn main() -> Result<()> { + let problem = RoyalRoad::r1(); + let optimum = problem.optimum().expect("known").value(); + println!("Royal road R1: 8 blocks of 8 bits, maximum {optimum}"); + println!("random-mutation hill climbing from seed 1"); + println!("R1 evaluations"); + // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page + let mut trace = trace::Trace::from_env(BITS, optimum); + let mut last = 0.0; + let outcome = Engine::new(rmhc(&problem, 1)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .on_generation(|snapshot| { + let progress = snapshot.progress(); + let best = progress.best().and_then(Fitness::score).unwrap_or(0.0); + // the evaluations at which a block is completed + if best > last { + println!("{best:>2} {:>11}", progress.evaluations()); + last = best; + } + }) + .on_generation(|snapshot| trace.record(snapshot)) + .run()?; + println!( + "{} after {} evaluations", + outcome.best_fitness(), + outcome.evaluations() + ); + + // RMHC from seeds 1 to 200, and the genetic algorithm from seeds 1 to 50 + let mut evaluations = Vec::new(); + for seed in 1..=RUNS { + let outcome = Engine::new(rmhc(&problem, seed)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations() as f64); + } + } + let reached = evaluations.len(); + let (mean, median) = mean_and_median(evaluations); + println!( + "\nRMHC, seeds 1 to {RUNS}: {reached} reach {optimum}; mean {mean:.0}, median {median:.0} \ + evaluations" + ); + println!(" (the paper's 200 runs: mean 6179, median 5775)"); + let mut evaluations = Vec::new(); + for seed in 1..=GA_RUNS { + let ga = Ga::builder(problem.representation()) + .population_size(128) + .select(Tournament::new(2)?) + .crossover(PointCrossover::one_point()) + .crossover_rate(0.7) + .mutate(BitFlip::per_gene(0.005)?) + .seed(seed) + .build()?; + let outcome = Engine::new(ga, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations() as f64); + } + } + let reached = evaluations.len(); + let (mean, median) = mean_and_median(evaluations); + println!( + "GA, seeds 1 to {GA_RUNS}: {reached} reach {optimum}; mean {mean:.0}, median {median:.0} \ + evaluations" + ); + println!(" (the paper's GA, 200 runs: mean 61334, median 54208)"); + trace.write(); + Ok(()) +} diff --git a/examples/royal_road_r1/output.txt b/examples/royal_road_r1/output.txt new file mode 100644 index 00000000..19e605d7 --- /dev/null +++ b/examples/royal_road_r1/output.txt @@ -0,0 +1,17 @@ +Royal road R1: 8 blocks of 8 bits, maximum 64 +random-mutation hill climbing from seed 1 +R1 evaluations + 8 451 +16 625 +24 638 +32 1615 +40 1656 +48 1837 +56 3717 +64 10101 +64 after 10101 evaluations + +RMHC, seeds 1 to 200: 200 reach 64; mean 6793, median 6144 evaluations + (the paper's 200 runs: mean 6179, median 5775) +GA, seeds 1 to 50: 50 reach 64; mean 34303, median 30996 evaluations + (the paper's GA, 200 runs: mean 61334, median 54208) diff --git a/examples/royal_road_r1/trace.json b/examples/royal_road_r1/trace.json new file mode 100644 index 00000000..fdae951b --- /dev/null +++ b/examples/royal_road_r1/trace.json @@ -0,0 +1,34 @@ +{"example":"royal_road_r1","format":1,"log_y":false,"objective":"maximize","optimum":64.0,"plot":"bits","problem":{"length":64},"x_label":"evaluations","y_label":"R1","frames":[ +{"best":0.0,"evaluations":1,"generation":0,"median":0.0,"state":{"population":["1000110110110000001001010011000101010111001100101001000011100110"]}}, +{"best":0.0,"evaluations":321,"generation":320,"median":0.0,"state":{"population":["1101110101010001111001110101001110010011111100010100111011000110"]}}, +{"best":24.0,"evaluations":641,"generation":640,"median":24.0,"state":{"population":["1010111010010111111010111111111111111111011100001111111101111000"]}}, +{"best":24.0,"evaluations":961,"generation":960,"median":24.0,"state":{"population":["0001110111010001010011011111111111111111111001001111111111001101"]}}, +{"best":24.0,"evaluations":1281,"generation":1280,"median":24.0,"state":{"population":["1011001011001000001010111111111111111111001110011111111111101001"]}}, +{"best":24.0,"evaluations":1601,"generation":1600,"median":24.0,"state":{"population":["1001010011111110110010111111111111111111010001001111111110101011"]}}, +{"best":48.0,"evaluations":1985,"generation":1984,"median":48.0,"state":{"population":["1010100111111111111111111111111111111111001101101111111111111111"]}}, +{"best":48.0,"evaluations":2305,"generation":2304,"median":48.0,"state":{"population":["0100010011111111111111111111111111111111000100101111111111111111"]}}, +{"best":48.0,"evaluations":2625,"generation":2624,"median":48.0,"state":{"population":["0011100011111111111111111111111111111111110000111111111111111111"]}}, +{"best":48.0,"evaluations":2945,"generation":2944,"median":48.0,"state":{"population":["1110110011111111111111111111111111111111001111101111111111111111"]}}, +{"best":48.0,"evaluations":3265,"generation":3264,"median":48.0,"state":{"population":["1000110011111111111111111111111111111111110111001111111111111111"]}}, +{"best":48.0,"evaluations":3585,"generation":3584,"median":48.0,"state":{"population":["0011001111111111111111111111111111111111001000111111111111111111"]}}, +{"best":56.0,"evaluations":3905,"generation":3904,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111101001111111111111111111"]}}, +{"best":56.0,"evaluations":4225,"generation":4224,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000111001111111111111111"]}}, +{"best":56.0,"evaluations":4545,"generation":4544,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111101010001111111111111111"]}}, +{"best":56.0,"evaluations":4865,"generation":4864,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000011111111111111111111"]}}, +{"best":56.0,"evaluations":5249,"generation":5248,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000110101111111111111111"]}}, +{"best":56.0,"evaluations":5569,"generation":5568,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111111111011111111111111111"]}}, +{"best":56.0,"evaluations":5889,"generation":5888,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111110101101111111111111111"]}}, +{"best":56.0,"evaluations":6209,"generation":6208,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111101001011111111111111111"]}}, +{"best":56.0,"evaluations":6529,"generation":6528,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000100111111111111111111"]}}, +{"best":56.0,"evaluations":6849,"generation":6848,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111101010111111111111111111"]}}, +{"best":56.0,"evaluations":7169,"generation":7168,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111001001001111111111111111"]}}, +{"best":56.0,"evaluations":7489,"generation":7488,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111110001001111111111111111"]}}, +{"best":56.0,"evaluations":7809,"generation":7808,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000011001111111111111111"]}}, +{"best":56.0,"evaluations":8129,"generation":8128,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111011101111111111111111111"]}}, +{"best":56.0,"evaluations":8513,"generation":8512,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000011011111111111111111"]}}, +{"best":56.0,"evaluations":8833,"generation":8832,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111001101101111111111111111"]}}, +{"best":56.0,"evaluations":9153,"generation":9152,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000010011111111111111111"]}}, +{"best":56.0,"evaluations":9473,"generation":9472,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111011011011111111111111111"]}}, +{"best":56.0,"evaluations":9793,"generation":9792,"median":56.0,"state":{"population":["1111111111111111111111111111111111111111000101101111111111111111"]}}, +{"best":64.0,"evaluations":10101,"generation":10100,"median":64.0,"state":{"population":["1111111111111111111111111111111111111111111111111111111111111111"]}} +]} diff --git a/examples/royal_road_r1/trace.py b/examples/royal_road_r1/trace.py new file mode 100644 index 00000000..6b924b9b --- /dev/null +++ b/examples/royal_road_r1/trace.py @@ -0,0 +1,183 @@ +"""The trace of the run for the plot on the example's page, written to the file that +``GENOXIDE_TRACE`` names: the string of the search, in at most 32 steps. The Rust example writes the same +file.""" + +import json +import math +import os + + +class Trace: + """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" + + def __init__(self, length, optimum): + self.path = os.environ.get("GENOXIDE_TRACE") + self.frames = Frames(32) + self.length, self.optimum = length, optimum + + @property + def on_generation(self): + """The callback for ``run``: None without a trace to record.""" + return self.record if self.path else None + + def record(self, progress): + """Records a step: the current string.""" + if self.path: + rows = ["".join("1" if one else "0" for one in row) for row in progress.population[:16]] + self.frames.push(frame(progress, {"population": rows})) + + def write(self): + """Writes the trace, if there's one.""" + if self.path: + settings = { + "format": 1, + "example": "royal_road_r1", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "R1", + "log_y": False, + "optimum": float(self.optimum), + "plot": "bits", + "problem": {"length": self.length}, + } + write(self.path, settings, self.frames.to_list()) + + +# ---- the same in every example's trace ---------------------------------------------------------- + + +class Frames: + """The frames of at most ``most`` generations, from the part of the run where what the page + plots changes: the frames after the last change are left out (a run that reached its target, + or a front that no longer moves), and the rest are spread evenly over the generations up to + it. While the run goes, up to 8 × ``most`` frames are kept: every ``every``-th generation, + with ``every`` doubling whenever there are that many, and the last one.""" + + def __init__(self, most): + self.most, self.every, self.kept, self.last = most, 1, [], None + + def push(self, frame): + if frame["generation"] % self.every: + self.last = frame + return + self.kept.append(frame) + self.last = None + if len(self.kept) == 8 * self.most: + self.every *= 2 + self.kept = [kept for kept in self.kept if kept["generation"] % self.every == 0] + + def to_list(self): + frames = self.kept + ([self.last] if self.last else []) + active = frames[: last_change(frames) + 1] + count, most = len(active), max(self.most, 2) + if count <= most: + return active + return [active[(i * (count - 1) + (most - 1) // 2) // (most - 1)] for i in range(most)] + + +def last_change(frames): + """The index of the frame after which nothing the page plots changes. To 3 significant + digits, as a plot shows them: the best, the median and, for a single objective (a numeric + best), the state; to within a hundredth of their range over the run: a front's + hypervolumes, in the state or in a grid's series.""" + if not frames: + return 0 + last = len(frames) - 1 + number = lambda value: isinstance(value, (int, float)) and not isinstance(value, bool) + single = any(number(frame.get("best")) for frame in frames) + + def measures(frame): + values = [] + state = frame.get("state") + hypervolume = state.get("hypervolume") if isinstance(state, dict) else None + for value in (hypervolume, frame.get("series")): + if number(value): + values.append(float(value)) + elif isinstance(value, dict): + values.extend(float(v) for _, v in sorted(value.items()) if number(v)) + return values + + measured = [measures(frame) for frame in frames] + end = measured[last] + tolerance = [] + for k in range(len(end)): + values = [values[k] for values in measured if k < len(values)] + tolerance.append((max(values) - min(values)) / 100.0) + + def same(frame, final, key, flush=False): + return coarse(frame.get(key), flush) == coarse(final.get(key), flush) + + def settled(i): + frame, final = frames[i], frames[last] + return ( + same(frame, final, "best") + and same(frame, final, "median") + and (not single or same(frame, final, "state", flush=True)) + and len(measured[i]) == len(end) + and all(abs(v - e) <= t for v, e, t in zip(measured[i], end, tolerance)) + ) + + first = last + while first > 0 and settled(first - 1): + first -= 1 + return first + + +def frame(progress, state): + """The frame of a generation: its progress, the median score of its population and + ``state``.""" + return { + "generation": progress.generation, + "evaluations": progress.evaluations, + "best": progress.best_fitness, + "median": median(progress.scores), + "state": state, + } + + +def median(scores): + """The median of the valid scores, None without any.""" + scores = sorted(float(score) for score in scores if not math.isnan(score)) + middle = len(scores) // 2 + if not scores: + return None + return scores[middle] if len(scores) % 2 else (scores[middle - 1] + scores[middle]) / 2 + + +def coarse(value, flush=False): + """``value`` with its numbers to 3 significant digits, as precisely as a plot shows them: two + frames whose plotted values agree to that precision look the same. With ``flush``, for the + solutions a plot draws on their ranges, numbers below 1e-6 in size count as 0.""" + if isinstance(value, float): + if flush and abs(value) < 1e-6: + value = 0.0 + return f"{value:.2e}" + if isinstance(value, (list, tuple)): + return "[" + ",".join(coarse(item, flush) for item in value) + "]" + if isinstance(value, dict): + items = sorted(value.items()) + return "{" + ",".join(f"{key}:{coarse(item, flush)}" for key, item in items) + "}" + return json.dumps(value) + + +def write(path, settings, frames): + """Writes the settings and the frames to ``path``, a frame per line.""" + lines = ",\n".join(map(to_json, frames)) + with open(path, "w", encoding="utf-8", newline="\n") as file: + file.write(f'{to_json(settings)[:-1]},"frames":[\n{lines}\n]}}\n') + + +def to_json(value): + """Compact JSON with sorted keys, and numbers rounded to 6 significant digits, as the Rust + example writes it.""" + return json.dumps(rounded(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def rounded(value): + if isinstance(value, dict): + return {key: rounded(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [rounded(item) for item in value] + if isinstance(value, float): + return float(f"{value:.5e}") if math.isfinite(value) else None + return value diff --git a/examples/royal_road_r1/trace.rs b/examples/royal_road_r1/trace.rs new file mode 100644 index 00000000..f80a14ab --- /dev/null +++ b/examples/royal_road_r1/trace.rs @@ -0,0 +1,264 @@ +//! The trace of the run for the plot on the example's page, written to the file that +//! `GENOXIDE_TRACE` names: the string of the search, in at most 32 steps. The Python example writes the same +//! file. + +use genoxide::observer::Snapshot; +use genoxide::prelude::*; +use serde_json::{Value, json}; + +pub struct Trace { + path: Option, + frames: Frames, + length: usize, + optimum: f64, +} + +impl Trace { + // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set + pub fn from_env(length: usize, optimum: f64) -> Self { + let path = std::env::var("GENOXIDE_TRACE").ok(); + let frames = Frames::new(32); + Self { + path, + frames, + length, + optimum, + } + } + + // records a step: the current string + pub fn record(&mut self, snapshot: &Snapshot<'_, Bits>) { + if self.path.is_none() { + return; + } + let bits = |genome: &Bits| { + genome + .iter() + .map(|one| if one { '1' } else { '0' }) + .collect() + }; + let rows = snapshot.population().iter().take(16); + let rows: Vec = rows.map(|row| bits(row.genome())).collect(); + self.frames + .push(frame(snapshot, json!({ "population": rows }))); + } + + // writes the trace, if there's one + pub fn write(self) { + let Some(path) = self.path else { return }; + let settings = json!({ + "format": 1, + "example": "royal_road_r1", + "objective": "maximize", + "x_label": "evaluations", + "y_label": "R1", + "log_y": false, + "optimum": self.optimum, + "plot": "bits", + "problem": { "length": self.length }, + }); + write(&path, settings, self.frames.into_vec()); + } +} + +// ---- the same in every example's trace --------------------------------------------------------- + +// the frames of at most `most` generations, from the part of the run where what the page plots +// changes: the frames after the last change are left out (a run that reached its target, or a +// front that no longer moves), and the rest are spread evenly over the generations up to it. While +// the run goes, up to 8 × `most` frames are kept: every `every`-th generation, with `every` +// doubling whenever there are that many, and the last one. +struct Frames { + most: usize, + every: u64, + kept: Vec<(u64, Value)>, + last: Option<(u64, Value)>, +} + +impl Frames { + fn new(most: usize) -> Self { + let (every, kept, last) = (1, Vec::new(), None); + Self { + most, + every, + kept, + last, + } + } + + fn push(&mut self, frame: Value) { + let generation = frame["generation"].as_u64().expect("a generation"); + if !generation.is_multiple_of(self.every) { + self.last = Some((generation, frame)); + return; + } + self.kept.push((generation, frame)); + self.last = None; + if self.kept.len() == 8 * self.most { + self.every *= 2; + let every = self.every; + self.kept.retain(|(generation, _)| generation % every == 0); + } + } + + fn into_vec(self) -> Vec { + let frames = self.kept.into_iter().chain(self.last); + let frames: Vec = frames.map(|(_, frame)| frame).collect(); + let active = &frames[..=last_change(&frames)]; + let (count, most) = (active.len(), self.most.max(2)); + if count <= most { + return active.to_vec(); + } + let at = |i: usize| active[(i * (count - 1) + (most - 1) / 2) / (most - 1)].clone(); + (0..most).map(at).collect() + } +} + +// the index of the frame after which nothing the page plots changes. To 3 significant digits, as +// a plot shows them: the best, the median and, for a single objective (a numeric best), the state; +// to within a hundredth of their range over the run: a front's hypervolumes, in the state or in a +// grid's series +fn last_change(frames: &[Value]) -> usize { + let Some(last) = frames.len().checked_sub(1) else { + return 0; + }; + let single = frames.iter().any(|frame| frame["best"].is_number()); + let measures = |frame: &Value| -> Vec { + let mut values = Vec::new(); + for value in [&frame["state"]["hypervolume"], &frame["series"]] { + match value { + Value::Number(number) => values.extend(number.as_f64()), + Value::Object(map) => values.extend(map.values().filter_map(Value::as_f64)), + _ => {} + } + } + values + }; + let measured: Vec> = frames.iter().map(measures).collect(); + let end = &measured[last]; + let tolerance: Vec = (0..end.len()) + .map(|k| { + let values = measured.iter().filter_map(|values| values.get(k).copied()); + let (low, high) = values.fold((f64::INFINITY, f64::NEG_INFINITY), |(low, high), v| { + (low.min(v), high.max(v)) + }); + (high - low) / 100.0 + }) + .collect(); + let settled = |i: usize| { + let (frame, final_frame) = (&frames[i], &frames[last]); + let same = + |key: &str, flush: bool| coarse(&frame[key], flush) == coarse(&final_frame[key], flush); + same("best", false) + && same("median", false) + && (!single || same("state", true)) + && measured[i].len() == end.len() + && measured[i] + .iter() + .zip(end) + .zip(&tolerance) + .all(|((value, end), tolerance)| (value - end).abs() <= *tolerance) + }; + let mut first = last; + while first > 0 && settled(first - 1) { + first -= 1; + } + first +} + +// the frame of a generation: its progress, the median score of its population and `state` +fn frame(snapshot: &Snapshot<'_, G>, state: Value) -> Value { + let progress = snapshot.progress(); + let population = snapshot.population().iter(); + let scores = population.filter_map(|individual| individual.fitness()?.score()); + json!({ + "generation": progress.generation(), + "evaluations": progress.evaluations(), + "best": progress.best().and_then(Fitness::score), + "median": median(scores.collect()), + "state": state, + }) +} + +// the median of the scores, None without any +fn median(mut scores: Vec) -> Option { + scores.sort_by(f64::total_cmp); + let middle = scores.len() / 2; + match scores.len() { + 0 => None, + n if n % 2 == 1 => Some(scores[middle]), + _ => Some((scores[middle - 1] + scores[middle]) / 2.0), + } +} + +// `value` with its numbers to 3 significant digits, as precisely as a plot shows them: two frames +// whose plotted values agree to that precision look the same. With `flush`, for the solutions a +// plot draws on their ranges, numbers below 1e-6 in size count as 0 +fn coarse(value: &Value, flush: bool) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => { + let number = number.as_f64().expect("f64"); + let number = if flush && number.abs() < 1e-6 { + 0.0 + } else { + number + }; + format!("{number:.2e}") + } + Value::Array(items) => { + format!( + "[{}]", + join(items.iter().map(|item| coarse(item, flush)).collect()) + ) + } + Value::Object(map) => { + let entry = |(key, item): (&String, &Value)| format!("{key}:{}", coarse(item, flush)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +// writes the settings and the frames to `path`, a frame per line +fn write(path: &str, settings: Value, frames: Vec) { + let frames: Vec = frames.iter().map(to_json).collect(); + let settings = to_json(&settings); + let head = &settings[..settings.len() - 1]; + let text = format!("{head},\"frames\":[\n{}\n]}}\n", frames.join(",\n")); + std::fs::write(path, text).expect("the trace is written"); +} + +// compact JSON with sorted keys, and numbers rounded to 6 significant digits and written as +// Python writes them (7542.0, 1e-08): the Python example writes the same file +fn to_json(value: &Value) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => python_float(number.as_f64().expect("f64")), + Value::Array(items) => format!("[{}]", join(items.iter().map(to_json).collect())), + Value::Object(map) => { + let entry = + |(key, item): (&String, &Value)| format!("{}:{}", json!(key), to_json(item)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +fn python_float(value: f64) -> String { + let rounded: f64 = format!("{value:.5e}").parse().expect("a number"); + let shortest = format!("{rounded:e}"); + let (mantissa, exponent) = shortest.split_once('e').expect("an exponent"); + let exponent: i32 = exponent.parse().expect("an exponent"); + if (-4..16).contains(&exponent) { + let text = rounded.to_string(); + if text.contains('.') { + text + } else { + text + ".0" + } + } else { + let sign = if exponent < 0 { '-' } else { '+' }; + format!("{mantissa}e{sign}{:02}", exponent.abs()) + } +} diff --git a/examples/royal_road_r2/README.md b/examples/royal_road_r2/README.md new file mode 100644 index 00000000..822d5a17 --- /dev/null +++ b/examples/royal_road_r2/README.md @@ -0,0 +1,84 @@ +--- +title: Royal road R2 +category: binary +summary: Maximize 8 blocks of 8 ones and their pairs, quadruples and whole, each scoring when complete, with the genetic algorithm of Mitchell, Forrest and Holland's settings. +reference: "Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms: fitness landscapes and GA performance. Proceedings of the First European Conference on Artificial Life: 245-254." +reference_url: https://www.osti.gov/biblio/6107786 +optimum: "256 (all ones)" +languages: [rust, python] +order: 14 +family: Royal road +tab: R2 +--- + +# Royal road R2 + +## The problem + +R2 is the royal road function of Mitchell, Forrest and Holland's (1992) Figure 1, which Forrest and +Mitchell later called R2. It's a sum over 15 schemas s, each scoring its order, its number of +defined bits, c_s = order(s), when a string is an instance of it: + +```text +F(x) = Σₛ c_s σ_s(x), σ_s(x) = 1 if x is an instance of s, 0 otherwise +``` + +The schemas form a tree over 64 bits: 8 blocks of 8 consecutive ones (c = 8), the 4 pairs of +adjacent blocks, 16 ones each (c = 16), the 2 halves of 32 ones (c = 32) and the whole string +(c = 64). The string of all ones is an instance of all 15 and scores 8 · 8 + 4 · 16 + 2 · 32 + 64 += 256. A string whose first 16 bits are ones scores 8 + 8 + 16 = 32. The function is genoxide's +`problems::binary::RoyalRoad::r2()`; [R1](../royal_road_r1/) is its lowest level alone. + +## What makes it hard + +The intermediate schemas were meant as stepping stones: a genetic algorithm would combine two +complete blocks into a pair, two pairs into a half, and climb the tree by crossover. A hill climber +gains nothing from them: a step that doesn't complete or break a block changes neither R1 nor R2, +so a climber that keeps changes that are no worse takes the same steps on both. The difficulty is +R1's: a block short of complete scores nothing, and a string that completes one spreads through the +population, its zeros elsewhere hitchhiking with it. + +Mitchell et al. ran their genetic algorithm 50 times: it found the optimum after 590 generations on +average (with a standard error of 50), 542 at the median, and 1,022 without crossover (their Table +1). Their iterated hill climber never found it in 256,000 evaluations; later, Mitchell, Holland and +Forrest (1994) found random-mutation hill climbing faster than the GA on R1. + +## Representation + +A `Binary` genome of 64 bits is the string itself. The fitness is R2, to maximize. + +## Algorithm + +A genetic algorithm with Mitchell et al.'s settings: + +- a population of 128; +- tournament selection of size 2, in place of their fitness-proportionate selection with sigma + scaling (at most 1.5 expected offspring), which genoxide doesn't have; +- single-point crossover at a rate of 0.7 per pair of parents; +- bit-flip mutation with probability 0.005 per bit; +- the default generational scheme, which keeps the best individual. + +A run from seed 1 stops at 256, or after 256,000 evaluations, the paper's 2,000 generations of 128; +then runs from seeds 1 to 50, as in the paper. For comparison, random-mutation hill climbing +(`LocalSearch` flipping one bit at a time and keeping the change if it's no worse) from seeds 1 to +200, as on [R1](../royal_road_r1/). + +## Output + +The table follows the run from seed 1: the generations at which its best score improves. Then the +generations and evaluations at which it reaches 256. The last three lines give the 50 runs of the +GA, how many reach 256 and the mean and median of their generations, with the paper's for +comparison, and the 200 runs of hill climbing, with their evaluations. The function is evaluated in +Rust in both languages, so both print the same. + +[The project page](https://tachsin.gr/projects/genoxide/examples/royal-road-r2) plays back the run +from seed 1: the first 16 strings of the population, as blocks of ones complete and spread. + +## Good results + +The optimum is 256. The run from seed 1 reaches it at generation 69, and all 50 runs from seeds 1 to +50 do, after 661 generations on average and 478 at the median, close to the paper's 590 and 542. +Over seeds 1 to 300, every run reached it within 256,000 evaluations, after 27,296 at the median +(genoxide doesn't evaluate a child that is a copy of its parent again). Random-mutation hill +climbing reaches it in all 200 runs after 6,793 evaluations on average and 6,144 at the median, the +same as on R1, step for step: on R2 too, it needs about a quarter of the GA's evaluations. diff --git a/examples/royal_road_r2/main.py b/examples/royal_road_r2/main.py new file mode 100644 index 00000000..8a4f4fd3 --- /dev/null +++ b/examples/royal_road_r2/main.py @@ -0,0 +1,100 @@ +"""Royal road R2: maximize 8 blocks of 8 ones, and the pairs, quadruples and whole string above +them, each scoring its number of bits when complete, with a genetic algorithm with the settings +of Mitchell, Forrest and Holland (1992). + +The GA has their population of 128, single-point crossover at a rate of 0.7 and a mutation +probability of 0.005 per bit, with tournament selection of size 2 in place of their +fitness-proportionate selection with sigma scaling. A run from seed 1 prints the generations at +which its best improves; then runs from seeds 1 to 50, as in their Table 1, give the generations +to the optimum, 256. As a comparison, random-mutation hill climbing, one bit at a time. The +function is genoxide's problems.binary.RoyalRoad.r2(), which run evaluates in Rust. + +With ``GENOXIDE_TRACE=``, it also writes a trace of its run for the plot on the example's +page, with trace.py. + + python examples/royal_road_r2/main.py +""" + +import genoxide as gx + +from trace import Trace + +BITS = 64 +# the runs of the GA, as in the paper's Table 1, and of hill climbing +RUNS = 50 +CLIMBS = 200 +# the most evaluations of a run, the paper's 2000 generations of 128 +BUDGET = 256_000 + + +def ga(problem, seed): + """The genetic algorithm from ``seed``.""" + return gx.Ga( + problem.genome, + population_size=128, + select=gx.Tournament(2), + crossover=gx.PointCrossover(1), + crossover_rate=0.7, + mutation=gx.BitFlip(rate=0.005), + seed=seed, + ) + + +def mean_and_median(values): + """The mean and the median of ``values``.""" + values = sorted(values) + middle = len(values) // 2 + median = values[middle] if len(values) % 2 else (values[middle - 1] + values[middle]) / 2 + return sum(values) / len(values), median + + +problem = gx.problems.binary.RoyalRoad.r2() +optimum = problem.optimum.value +print(f"Royal road R2: 8 blocks of 8 bits and the levels above, maximum {optimum:.0f}") +print("a GA with single-point crossover, from seed 1") +print("generation best") +# with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page +trace = Trace(BITS, optimum) +last = -1.0 + + +def progress(progress): + global last + best = progress.best_fitness or 0.0 + # the generations at which the best improves + if best > last: + print(f"{progress.generation:>10} {best:>4.0f}") + last = best + trace.record(progress) + + +result = ga(problem, 1).run(problem, target=optimum, evaluations=BUDGET, on_generation=progress) +print( + f"{result.best_fitness:.0f} after {result.generations} generations and " + f"{result.evaluations} evaluations" +) + +# the GA from seeds 1 to 50, and hill climbing from seeds 1 to 200 +generations = [] +for seed in range(1, RUNS + 1): + result = ga(problem, seed).run(problem, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + generations.append(result.generations) +mean, median = mean_and_median(generations) +print( + f"\nGA, seeds 1 to {RUNS}: {len(generations)} reach {optimum:.0f}; mean {mean:.0f}, " + f"median {median:.0f} generations" +) +print(" (the paper's GA, 50 runs: mean 590, median 542)") +evaluations = [] +for seed in range(1, CLIMBS + 1): + search = gx.LocalSearch(problem.genome, neighbor=gx.BitFlip(count=1), seed=seed) + result = search.run(problem, target=optimum, evaluations=BUDGET) + if result.stop_reason == "target": + evaluations.append(result.evaluations) +mean, median = mean_and_median(evaluations) +print( + f"hill climbing, seeds 1 to {CLIMBS}: {len(evaluations)} reach {optimum:.0f}; " + f"mean {mean:.0f}, median {median:.0f} evaluations" +) +trace.write() diff --git a/examples/royal_road_r2/main.rs b/examples/royal_road_r2/main.rs new file mode 100644 index 00000000..68bebe68 --- /dev/null +++ b/examples/royal_road_r2/main.rs @@ -0,0 +1,123 @@ +//! Royal road R2: maximize 8 blocks of 8 ones, and the pairs, quadruples and whole string above +//! them, each scoring its number of bits when complete, with a genetic algorithm with the +//! settings of Mitchell, Forrest and Holland (1992). +//! +//! The GA has their population of 128, single-point crossover at a rate of 0.7 and a mutation +//! probability of 0.005 per bit, with tournament selection of size 2 in place of their +//! fitness-proportionate selection with sigma scaling. A run from seed 1 prints the generations at +//! which its best improves; then runs from seeds 1 to 50, as in their Table 1, give the +//! generations to the optimum, 256. As a comparison, random-mutation hill climbing, one bit at a time. The function +//! is genoxide's `problems::binary::RoyalRoad::r2`. +//! +//! With `GENOXIDE_TRACE=`, it also writes a trace of its run for the plot on the example's +//! page, with `trace.rs`. +//! +//! ```text +//! cargo run --release --example royal_road_r2 +//! ``` + +mod trace; + +use genoxide::prelude::*; +use genoxide::problems::Problem; +use genoxide::problems::binary::RoyalRoad; + +const BITS: usize = 64; +// the runs of the GA, as in the paper's Table 1, and of hill climbing +const RUNS: u64 = 50; +const CLIMBS: u64 = 200; +// the most evaluations of a run, the paper's 2000 generations of 128 +const BUDGET: u64 = 256_000; + +// the genetic algorithm from `seed` +fn ga(problem: &RoyalRoad, seed: u64) -> Result> { + Ga::builder(problem.representation()) + .population_size(128) + .select(Tournament::new(2)?) + .crossover(PointCrossover::one_point()) + .crossover_rate(0.7) + .mutate(BitFlip::per_gene(0.005)?) + .seed(seed) + .build() +} + +// the mean and the median of `values` +fn mean_and_median(mut values: Vec) -> (f64, f64) { + values.sort_by(f64::total_cmp); + let middle = values.len() / 2; + let median = if values.len() % 2 == 1 { + values[middle] + } else { + (values[middle - 1] + values[middle]) / 2.0 + }; + (values.iter().sum::() / values.len() as f64, median) +} + +fn main() -> Result<()> { + let problem = RoyalRoad::r2(); + let optimum = problem.optimum().expect("known").value(); + println!("Royal road R2: 8 blocks of 8 bits and the levels above, maximum {optimum}"); + println!("a GA with single-point crossover, from seed 1"); + println!("generation best"); + // with GENOXIDE_TRACE=, a trace of the run for the plot on the example's page + let mut trace = trace::Trace::from_env(BITS, optimum); + let mut last = -1.0; + let outcome = Engine::new(ga(&problem, 1)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .on_generation(|snapshot| { + let progress = snapshot.progress(); + let best = progress.best().and_then(Fitness::score).unwrap_or(0.0); + // the generations at which the best improves + if best > last { + println!("{:>10} {best:>4}", progress.generation()); + last = best; + } + }) + .on_generation(|snapshot| trace.record(snapshot)) + .run()?; + println!( + "{} after {} generations and {} evaluations", + outcome.best_fitness(), + outcome.generations(), + outcome.evaluations() + ); + + // the GA from seeds 1 to 50, and hill climbing from seeds 1 to 200 + let mut generations = Vec::new(); + for seed in 1..=RUNS { + let outcome = Engine::new(ga(&problem, seed)?, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + generations.push(outcome.generations() as f64); + } + } + let reached = generations.len(); + let (mean, median) = mean_and_median(generations); + println!( + "\nGA, seeds 1 to {RUNS}: {reached} reach {optimum}; mean {mean:.0}, median {median:.0} \ + generations" + ); + println!(" (the paper's GA, 50 runs: mean 590, median 542)"); + let mut evaluations = Vec::new(); + for seed in 1..=CLIMBS { + let search = LocalSearch::builder(problem.representation()) + .neighbor(BitFlip::count(1)?) + .seed(seed) + .build()?; + let outcome = Engine::new(search, problem) + .stop_when(Stop::target(optimum).or(Stop::evaluations(BUDGET))) + .run()?; + if outcome.stop_reason() == StopReason::Target { + evaluations.push(outcome.evaluations() as f64); + } + } + let reached = evaluations.len(); + let (mean, median) = mean_and_median(evaluations); + println!( + "hill climbing, seeds 1 to {CLIMBS}: {reached} reach {optimum}; mean {mean:.0}, median \ + {median:.0} evaluations" + ); + trace.write(); + Ok(()) +} diff --git a/examples/royal_road_r2/output.txt b/examples/royal_road_r2/output.txt new file mode 100644 index 00000000..32ff5f33 --- /dev/null +++ b/examples/royal_road_r2/output.txt @@ -0,0 +1,16 @@ +Royal road R2: 8 blocks of 8 bits and the levels above, maximum 256 +a GA with single-point crossover, from seed 1 +generation best + 0 16 + 6 40 + 10 48 + 12 64 + 20 72 + 26 80 + 52 136 + 69 256 +256 after 69 generations and 4767 evaluations + +GA, seeds 1 to 50: 50 reach 256; mean 661, median 478 generations + (the paper's GA, 50 runs: mean 590, median 542) +hill climbing, seeds 1 to 200: 200 reach 256; mean 6793, median 6144 evaluations diff --git a/examples/royal_road_r2/trace.json b/examples/royal_road_r2/trace.json new file mode 100644 index 00000000..23004939 --- /dev/null +++ b/examples/royal_road_r2/trace.json @@ -0,0 +1,34 @@ +{"example":"royal_road_r2","format":1,"log_y":false,"objective":"maximize","optimum":256.0,"plot":"bits","problem":{"length":64},"x_label":"generations","y_label":"R2","frames":[ +{"best":16.0,"evaluations":128,"generation":0,"median":0.0,"state":{"population":["1000110110110000001001010011000101010111001100101001000011100110","1101011001110001011100000011111100011011011000000010100100101000","1010011001100000111000000110110011000000110101000001110100011001","0100001010110000000100001100011000111011111001011010010000011100","1010011111010011110110001010101111001010010011111111100100010010","0001101000000100010001111001011101101100101100010101011110101101","1110100001010001111111110111111000000111111110111001000001101110","0111000100010010101000111010110101000010111111000010100011100100","1100100000000110100011010101000011011000111001011000111111111011","0001000010010110000000010011010000001101000010101010011100101101","1000111011000010001100001110000001001101101111010110110111011110","1100011110010010000010001000111101111010010111111100001111001001","1100010001110001110110111010101110100111101101000110111010011000","1000010110000111100101111001111111010000010110000000100111000010","0100011111011111011001010100111011111101111101111010100000101110","1011100011100110100010011100111001011011100010001011110111100010"]}}, +{"best":16.0,"evaluations":327,"generation":2,"median":0.0,"state":{"population":["1110001001101100110100011111111110000111000000110110000011111111","0100101111101101100011010101011010111100110101100011100101000100","0111011110101111101111111000111110111101001011100111011110010000","0010001110010001000111101111011110111010010000000111101100000010","0101001010001001001110001110011101001001011001100010110110111111","1100000100001110111110001111111111000110111011101001000101101111","0011101000100010101111010111010101101100010001111101111001101101","1011110001110010100011010101111010111101110101100011101011000001","0111001111001000111110011000001010000000011001010010011110000111","1000110011010100111111010101111000000111111001010111010100001000","0101101001111000000001100011110101101001011110111001000001101110","1110011111010011110110001010101111001010010011111111100100001101","0101001000101010010010001110000110011100010111111100011000110010","0101111100111000001100111011000110111111101100100110001000010111","0110100110111111111011100001101010010011001011111010011011110100","0001110111001011011011111110101111111001011000011001100110101001"]}}, +{"best":16.0,"evaluations":493,"generation":4,"median":8.0,"state":{"population":["1110001001101100110100011111111110000111000000110110000011111111","1110001001101100110100011111111110000111000000110110000011111111","0010010111000101110000111000001001000111001001011101101110010101","0100010101010010010100101000101111000110111011101001000101101111","0010011110111100010011101111111111000110001010100010010100100010","1001101010101111100110100110010100111010000110011111011110010010","0010010101010010010100101000101111000110001010100010010100100000","1101101010100011111111110011111101100100110111011100000011111111","1110001001101100110111110011111101101100110110110110101100000100","1001110001001100101011001100100100111101000100100100100011011101","0010011101101100001010011111011111000110011000100101100010100010","1010011001100010111111110010111101101100110110110100000111000000","1101101010100000111110001111111111000111001011100111011110010000","1100000100001110111110001111111111000110111011101001000101101111","0100011101010010010000101111111111000110011011101001000101101111","0010011110111100010011101111111111000110110000110110000011111111"]}}, +{"best":40.0,"evaluations":773,"generation":7,"median":16.0,"state":{"population":["0100101111101011111111111111111111000110001000110110000011111111","1110100001010000110100001111111110000111000000110110000011111110","1110001001101101111111110111111001000110000001101001000101101111","1110100001010001111111110111111000001100110110110110000011111111","0100101111101011111111111111111111000110001000110110000011111111","1110100000001110111100001111111111000110111011101001001111010000","1110101001101100110011101111111111000110001000110110000011011101","1110001001101100110100001111111110000111000000110101000011111111","0100000100001100111111110111111111000110111011101000000011111111","1010011110100011111000101111111111000110011000100100101011011101","0101101010111100010011101111111111000110111011100110000011111111","0100011101011100110100011111111110000111000000110110000011111111","1110000001100010010000101111111110000111000000110110100011111111","1110001111001110110100011111111110000111000000110100000011111111","1110001001101100110111110101010110011011111111111100101111111011","1100000100001110111110001111011110011011000000110110000011110111"]}}, +{"best":40.0,"evaluations":946,"generation":9,"median":16.0,"state":{"population":["0100101111101011111111111111111111000110001000110110000011111111","0010011101101110111110001111111101000111001011110110100011111111","0100011101010001110010001111111111001100000000110110000011111111","0100011101010100010011101111111111000110111011100110000011111111","1010011110111100110100011111111110010111000000110110000011111111","0100011101010010010000101111110111000110111111111100101111111111","0100011101010010010001111111111111000110000010110110000011111111","0101101010100011111111110011111101101100111111111100101111111111","0100011101010010010000101111111111000110110110100110011110001010","1100001001101100110100111111111110011011000000110110000011111111","1110001001101100110100011111111110000111000000111110000011111111","1110001001101100110100011111111110000111000001110110000011111111","0100011101010000110111101111111111001110110110110110000011111111","0101011101100011111111110111111110000111000000110110101011111111","0010011101010010010001111111111111000110000011110110000011111111","1000101000001100010011101111111111000110111011100110000011111111"]}}, +{"best":48.0,"evaluations":1117,"generation":11,"median":16.0,"state":{"population":["0100101111101011111111111111111111000110111111111100101111111111","1110100001010011111111111111111111000110110111111100101110111011","0100011101010001111111111111111111000110001000110110000011111111","0100011101010010010000101111111111000110111000101000000011111111","0100010101010010010000001111111110000111011011110110000011111111","0100101111101011111111111111111111000100110110010011111110010000","0101101111101011111111111111111111000110001010110110000011111111","0100101111101011111111111111111111000110100000110110000011111111","0100101111101011111111110111111110000111000000111010000011111111","0100101111101011111111111111111111000111000010110110000011111111","0100101101101100110100011111111110000111000000110110000011111111","0100101111101011111111111111111111000110100001110010000011111111","1100100001100010001010011111111111000110011000110110000011111111","0101101111101011111111111111111111000110001000110110000011111111","0100011101010010010000001111111110000111011011101000000011111111","0100101111101011110100011111111110000111111111111000101111111111"]}}, +{"best":64.0,"evaluations":1288,"generation":13,"median":40.0,"state":{"population":["0101101111101011111111111111111111000110001011101111111111111111","0101101111101011111111111111111111000110100000110110000011111111","0100101011101011111111111111111110000110001000110110000011111111","0100101111101011111111111111111111000110000010100110001011111111","0101101111101011111111111110111111000110111111111100101111111111","0100101111101011111111111111111111000100110110010011111110010000","0100111111101011111111111111111111000110110110100101000011111111","0100101111101011111111111111111111000110001000110110000011111111","0100101111111011111111111111111111000110101111111100101111111111","1101110100010011110100011111111111000111001111101111111111111111","1110100000111101110100011111111110001111111011111100101111111110","0100101111101011111111111111111111000111000000110110000011111110","0100101111101011111111111111111111000110000000110111000011111111","1110001111101011111111111111111110000111000010111100001011111111","0100011101010001111111111111011110000110100000110110000011111111","0101101111101001111111111111111111000110001000110110000011111111"]}}, +{"best":64.0,"evaluations":1518,"generation":16,"median":40.0,"state":{"population":["0101101111101011111111111111111111000110001011101111111111111111","0100101111101011111111111111111100000010111100111110000011111111","0000101111101011111111111111111110001111101010110110000011111111","1101101111101011111111111111111110000110001000110110000011111111","0101101111101011111011111111111111000110001000110110000011111111","0100101111101011111111111111111111000110111110110110000011111111","0101001111101011111111111111111111000110001011101111111111111110","0101101011101011111111111111111110000110001000110110000011111111","1110101111101011111111111111111111000010000011101111111111111111","0101101011101011111111111111111111000110111100110110000011111111","0100101111101011111111111111111111000111111111110110000011111111","0100101110100011111111111111111100000110111000010111000011111111","0101101111101011111111111111111111000100011000110110000011111111","0101101111101011111111111111111110001111101000110110000011111111","0100101111101011111111111111111111000110110110100110001011111111","0101101111101011111111111111111111000110111011100110000011111111"]}}, +{"best":64.0,"evaluations":1678,"generation":18,"median":40.0,"state":{"population":["0101101111101011111111111111111111000110001011101111111111111111","0100101001010001111111110111111111000110001011110110000011111111","1110101111101011111111111111111111000110101000110110010011111111","0100101111101011111111111111111111000110111100110110000011111101","0101101111101011111111111111111111000110111011100110100011111111","0100101111101011111111111111111111000110111100110110000011111111","0101101111101011111111111111111111000110001000010100000011111111","0101101111101011111111111111111111000110001000111110000011111111","0100011111010001111111111111111111000111111111110110000011111111","0101101111101011111111111111111011000110111100110110000011111111","0100101001010001111111111011111111001110111100110110000011111111","0101111111101011111111111111111111000011001111101111111111111111","0100111111111111111111111111111111100110011000110110000011111111","0101101111101011111111111111111111000110001011101111111111111111","0101101111101011111111111111111111000110111011110110000011111111","0100111111111011111111111111111011100110011000110110000011111111"]}}, +{"best":72.0,"evaluations":1841,"generation":20,"median":48.0,"state":{"population":["0101101111101011111111111111111111000110001011101111111111111111","0100101111101011111111111111111110000111011110101111111111111111","0000101111101011111111111111111111000110111011110110001011111111","0101001111101011111111111111111101000110101000110110000011111111","1101101111101011111111111111111110000110001000110110001011111111","0100101111101011111111111111111111000110111011110110001011111111","0110001101101011111111111111111111000010000011101111111111111111","0101101111101011111111111111111111000010000001101111111111111111","0101101111101011111111111111111111000110001011101111111111111111","0101101111101011111111111110111111000111000010110110000011111111","0101101001010001111111111111111111000111011100110110000011111111","0101101101101011111111111111111111000110001011101111111111111111","0101101111101011111111111111111111000010000011101111111111111111","0101101111100011111111111111111111000110110110100110000011111111","0101101111101011111111111111111111000110001011101111111111101111","0101101101101011111111110111111111000110001000110110000011111111"]}}, +{"best":72.0,"evaluations":1992,"generation":22,"median":64.0,"state":{"population":["0100111111111111111111111111111111000011001111101111111111111111","0101111111111011111111111111111111100111111111111111111111111111","0101101101010001111110111111111111000110001011101111111111111111","0101101101101011111111111111111111100110001011101111111111111111","0101101111111011111111111111111111100110001101101111111111111111","0101101111101011111111111111111111000110001011101111111111111111","0101101111101011111111111111111111100110001110101111111111111111","0101101111101011111111111111111111000111111111110111111111111111","0101101111101001111111111111111111000010000011101111111111111111","0101101111101011111111111111111111100110001101101111111111111111","0100000101010001111111111111111111000010000011101111111111111111","0101101111101011111111111111111111000010001011101111111111111111","0101101111101011111111111111111101100110000011101111111111111111","0100011101010001111111111111111111000010000011101111111111111111","0101101111101001111111111111111111000010101101101111111111111111","0100111111111111111111111111111111000011001111101111111111111111"]}}, +{"best":72.0,"evaluations":2143,"generation":24,"median":64.0,"state":{"population":["0100111111111111111111111111111111000011001111101111111111111111","1100101111101011111111111111111111000011001111101111111111111111","0100111111111111111111111111111111000110001111101111111111111111","0101101101011011111111111111111111000110001011101111111111111111","0101101011100001111111111111111111000111111111111111111111111111","0101101111101011111111111111111111100110001110101111111111111111","0101100111101001111111111111111111000010101101101111111111111111","0100000101011001111111111111111111000110000011101111111111111111","0101101011101011111111111111111111000010100011101111111111111111","0101101011101011111111111111111111000110000011101111111111111111","0100101011101011111111111011111111000010000011101111111111111111","0100111111111111111111111111111111000011001111101111111111111111","0100111111111111111111111111111111000011001111101111111111111111","0101101101010001111111111111111111000111111111111111111111111111","0101101101010001111111111111111111000111111111111111111111111111","0100111111111111111111111111111111000011001111101111111111111111"]}}, +{"best":80.0,"evaluations":2350,"generation":27,"median":64.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100101111101111111111111111111111000110001011101111111111111111","0100111111111011111111111111111111000011001111101111111111111111","0101101111101011111111111111111111000010000011101111111111111111","0100111111110001111111111111111111000111111111101111111111111111","0100011101000001111111111111111111000011001111101111111111111111","0101101101010001111111111111111101100110001111101111111111111111","0101101111111111111111111111111111000011001111101111111111111111","0101101110101011111111111111111111000110001011101111111111111111","0100111111111111111111111111111111000011001111101111111111111111","0101011101010001111111111111111111000011001111101111111111111111","0101101101010001111111111111111101100110001111101111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0101101111101011111111111111111110000101000011101111111111111111","1110101111101011111101111111111111000110001011111111111111111111","0110101111101011111111111111111111000010001011101111111111111111"]}}, +{"best":80.0,"evaluations":2467,"generation":29,"median":72.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011001111101111111111111111","0100111111111111111111111111111111000011001111101111111111111111","0100110111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000011001110101111111111111111","1100111101010011111111111111111111001111111111111111101111111111","0100111111111111111111111111111111000011001111111111111111111111","0101101111111111111111111111111111000011001111101111111111111111","0100111111111111111111101111111111000110001011101111111111111111","0101101111111111111111111111111101000111111111111111111111111111","0100110111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000010000011101111111111111111","0100111111111111111111111111111111000110001111101111111111111111","0100111111101011111111111111111111001111111111111111111111111111","0100010101010001111101111111111111000011101111101111111111111111","0100111111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":2588,"generation":31,"median":72.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000010001011101111111111111111","0101101111111111111111111111111101000111111111111111111111111111","0100111111111111111111111111111111001110001111101111111111111111","0100111111100111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111000111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111011111","0100111111111111111111111111111111001111111111101111111111111111","0100111111111111111111111111111111000110001111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0100111011111111111111111111111111000111111111111111111111111111","0101111111111111111111111111111111000111001111111111111111111111","0101111111111111111111111111111111001101111111111111111111111111","0101101111111111111111111111111111001111111111111111111111101111","0101111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":2699,"generation":33,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100101101111111111111111111111101100111111111111110111111111111","0100111111111111111111111111111101000111111111111111111111111111","0101101111111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111000111111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0101101111111111111101111111111111001111111111111111111111111111","0101101111111111111111111111111101000111111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0100111111111111110111111111111111001111111111111111111111111111","0101111111111111111111111111111111001101111111111111111111111111","0101101111111111111111111111111111001111111111111110111111111111","0100111111111111111111111111111101000111111111101111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0100111111111110111111111111111111000011111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":2877,"generation":36,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111111000111111111111111111111111111","0100111111111111111111111111111101110111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0100111101101111111111111111111111000010111111111111111111111111","1100110110111111111111111111111111000010011111111111111111111111","0100111111111111111111111111111111000111111111111111111111111111","0101111111111111111111111111111111011111111111111111111111111111","0101100111111111111111111111111111000011111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111101101001111111111111111111111111","0001101111111111111111111111111111001101111111111111111111111111","0101100111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":2996,"generation":38,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0100101111111111111111111111111111000111111111111111111111111111","0001101111111111111111111111111111000111111111111111111111111111","0100111111111111111111111111111110000011111111111111111111111111","0101101111111111111111111111111111000001111111111111111111111111","0101111111111111111111111111111111001011111111111111111111111111","0101101111111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111001110111111111111111111111111","0101100111111111111111111111111111001111111111111111111111111111","0100111111111111111111011111111111001111111111111111111111111111","0001101111111111111111111111111111001111111111111111111111111111","0100101111111111111111111111111111000010111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000111111111111111111111111111","0001101111111111111111111110111111000111111111111111111111111111"]}}, +{"best":80.0,"evaluations":3128,"generation":40,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100111101111111111111111111111111000011111101111111111111111111","0100111111111111111111111111111110000111111111111111111111111111","0001101111111111111111111111111111000100111111111111111111111111","0100111111111111111111111111111101101011111111011111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000011111111111111111111101111","0101111111111111111111111111111111000011111111111111111111111111","0101101111101111111111111111111111000111101111111111111111111111","0011101111111111111111111111111111000111111111111111111111111111","0101101111111111111111111111111111001011111111111111111111111111","0101100111111111111111111110111101000111111111101111111101111111","1111101111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":3237,"generation":42,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0111001111111111111111111111111110000111111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111001001111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0101110111111111111111111111111111001110111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","1111101111111111111111111111011111001111111111111111111111111111","1111101111111111111111111111111111001111111111111111111111111111","0110111111111111111111111111111111000100111111111111011111111111","1111101111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001110111111111111111111111111","0101101111111111111111101111111111001001111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000111111111111111111111111111"]}}, +{"best":80.0,"evaluations":3411,"generation":45,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","1111101111111111111111111111111111001111111111101111111111111111","0100111111111111111111111111111111001110111111111111111111111111","1111101111111111111111111111111111000011111111111111111111111111","0100111011111111111111111111111111000111111111111111111111111111","0101111111111111111111111111111111000111111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0110101111111111111111111111111111001111111111111111111111111111","0111111111111111111111111111111111000110111111111111111111111111","1100111011111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111111000011111111111111111111111111","0110111111111111111111111111111111000110111111111111111111111110","0101101111111111111111111111111111001111111111111111111111111111","0111111111111111111111111111111111001101111111111111111111111111","0100101111111111111111111111111111001111111111101111111111111111"]}}, +{"best":80.0,"evaluations":3530,"generation":47,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0101111111111111111111111111111101100111111111111111111111111111","0111101111111111111110111111111111010111111111111111110111111111","0100111111111111111111111111111111000011111111111111111111111111","1111101111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111000011111111111011111111111111","0111111111111111111111111111111111000011111111111111111111111111","0100101111101111111111111111111111000100111111111111111111111111","0101101111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0100111011111111111111111111111111001111111111111111111111110111","0101101111111111111111111111111111000011111111111111111111111111","1100111011111111111111111111111111000011111111111111111111111111","1100111011111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111"]}}, +{"best":80.0,"evaluations":3656,"generation":49,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0100111011111111111111111111111111000011110111111111111011110111","0100111111111110111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000110011111111111111111111111","0100101111111111111111111111111111000001111111111111111111111111","0001111111111111111111111111111101100111111111111111111111111111","0100111011111111111111111111111111000001111111111111111011111111","1100111011111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001101111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111001111111111111111111111111111","0100111111111110111111111111111111000110111111111111111111111111","0101101111111111111111111111111111000011111011111111111111111111"]}}, +{"best":80.0,"evaluations":3766,"generation":51,"median":80.0,"state":{"population":["0100111111111111111111111111111111001111111111111111111111111111","0001111111111111111111111111111111000111111111111111111111111111","0101101111111111111111111111111111000010111111101111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0100101111111111111111111111111111000001111111111111111111111111","1100111011111111111111111111111111000011111111111111111111111111","0101101111111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001011111111111111111111111111","0100111111111111111111111111111111000110111111111111111111111111","0100110011111111111111111111111111000111111111111111111111111111","0001111111111111111111111111111101000111111111111111111111111111","1111101111111111111111111111111111000001111111111111111111111111","0100101111111111111111111111111111000011111111111111111101111111","0100101011111111111111111111111111000011111111111111111111111111","0100111111111111111111111111111111001111111111111111111111111111","0101111111111111111111111111111101100111111111111111111111111111"]}}, +{"best":136.0,"evaluations":3907,"generation":53,"median":80.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","0101101011111111111111111111111111001110111111111111111111111111","0100111111111111111111111111111111001011111111111111111111111111","0100110111111111111111101111111111000011111111111111111111111111","0001101111111111111111111111111111000011111111111111111111111111","0100110011111111111111111111111111000011111111111111111111111111","0101101011111111111111111111111111000110111111111111111111111111","0101101011111111111111111111111111000110111111111111111111111111","0001101111111111111111111111111111001111111111111111111111111111","0101111111111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111110001010111111111111111111111111","0101111011111111111111111111111111000011111111111111111111111111","0100111111110111111111111111111111001111111111111111111111111111","0101101111111111111111111111111111000111111111111111111111111111","1100111011111111111111111111111111000010111111111111111111111111","0001110011111111111111111111111111001011111111111111111110111111"]}}, +{"best":136.0,"evaluations":4112,"generation":56,"median":80.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111101100111111111111111111111111111","0001101111111111111111111111111101100011111111111111111111111111","1111001111111111111111111111111101000111111111111111111111111111","0101101011111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111111100111111111110111111111111111","1100111111111111111111111111111111001011111111111111111111111111","0101101111111111111111111111111111000111111111111111111111111110","0001101111111111111111111111111111000110111111111111111111111111","0100101111111111111111111111111111001101111111111111111111111111","0101101011111111111111111111111111001111111111111111111111111111","0100101111111111111111111111111111000111111111111111111111111111","0100111111111111111111111111111111001111111011111111111111111111","0110110011111111111111011111111111000111111111111111111111111111","0100111111111111111111111111111111000011111111110111111111111111","0001101111111111111111111111111111000011111111111111111111111111"]}}, +{"best":136.0,"evaluations":4244,"generation":58,"median":80.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","0100101111111111111111011111111111001111111111111111111111111111","1100111011111111111111111111111111001011111111111111111111111111","0100101111111111111111111111111111001111111111111111111111111111","0001110011111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111011000011111111111111111111111111","0100101111111111111111111111111111000111111111111111111111111111","0100101111111111111111111111111111100011111111111111111111111111","1100111011111111111111110111111111001101111111111111111111111111","0101111111111111111111111101111111001111111111111111111111110111","1111111111111111111111111111111111001111111111110111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111101111111111111111111111111111001111111111111111111111111111","1101111111111111111111111111111111001111111111111111111011111111","1100111011111111111111111111111111100001111111111111111111111111","0001111111111111101111111111111111000001111111111111111111111111"]}}, +{"best":136.0,"evaluations":4370,"generation":60,"median":104.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","1100111011111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111101111111111111111111111110111","0100111111111111111111111111111111000011111111111111111111111111","0100111011111111111111111111111111001111111111111111111111111111","0100111111111111111111111111111111001011111111111111111111111111","1111101111111111111011111111111101100100111111111111111111111111","1111111110111111111111111111111111001111111111111101111111111111","1100101111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","0001101111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111011111","0101111011111111111111111111111101001111111111111111111111111111","1111111111111111111111111111111111001011111111111111111111111111"]}}, +{"best":136.0,"evaluations":4480,"generation":62,"median":136.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111000001111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","0100110111111111111111111111111111001111111111111111011111111111","1111111111111111111111111111111111101111111111111011111111111111","1111111111111111111111111111111111001111111111111111101111111111","1111111111111111111111111111111111001111111111111011111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001011111111111111111111111111","1111111111111111111111111111111111101111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111011111","1111111111111111111111111111111110001011111111101110111111111111","1111111111111111111111111111111111001101111111111111111111111111","1111111111111111111111111111111111001111111111111111111011111111"]}}, +{"best":136.0,"evaluations":4607,"generation":65,"median":136.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111011011111111111111111111111111","1111111111111111111101111111111101000001111111111111111111111111","1111111111111111111111111111111110001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1011111111111111011111111111111111101111111111111111111111111111","1111111111111111111111111111111111000001111111111111111111111111","1111111111111111111111111111111111001010111111111111111111111111","1111111111111111111111111111111111101101111111111111101111111111","1111111111111111111111111111111111001011111111111111111011111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001011111111111111111011111111","1111111111111111111101111111111111110111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111000001111111111111111111111111"]}}, +{"best":136.0,"evaluations":4686,"generation":67,"median":136.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001101111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111110001111111111011111111111111111","1111111111111111111111111111111111001101111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001111110111111101111111111111","1111111111111111111111111111111101001111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111111001011111111111111111111111111","1111111111111111111111111111111101001011111111111111111111111111","1111111111111111111111111111111111001011011111111111111111111111","1111111011111111111111111111111111101111111111111111111111111111","1111111011111111111111111111111111001010111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111"]}}, +{"best":256.0,"evaluations":4767,"generation":69,"median":136.0,"state":{"population":["1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111110100111111111111111111111111111","1111111111111111111111011111111111001111111111111111111111111111","1111111111111111111111111111111111101101111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111111111111111110001011111111111111111111111111","1111110111111111111111111111111111001011111111111111111111111111","1111111111111111111111111111111111001011111111111111111111111111","1111111111111111111101111111111110100111111111111111111111111111","1111111111111111111111111111111111001111111111111111011111111111","1111111111111111111111111111111111001101111111111111111111111111","1111111111111111111111111111111101001111111111111111111111111111","1111111111111111111111111111111111101111111111111111111111111111","1111111111111111111111111111111111001111111111111111111111111111","1111111111111111111101111111111111001101111111111111111111111111","1111111111111111111111111111111111100111111111111111111111111111"]}} +]} diff --git a/examples/royal_road_r2/trace.py b/examples/royal_road_r2/trace.py new file mode 100644 index 00000000..19381ef0 --- /dev/null +++ b/examples/royal_road_r2/trace.py @@ -0,0 +1,183 @@ +"""The trace of the run for the plot on the example's page, written to the file that +``GENOXIDE_TRACE`` names: the first 16 strings of the population, in at most 32 generations. The Rust example writes the same +file.""" + +import json +import math +import os + + +class Trace: + """Records the run through ``on_generation`` when ``GENOXIDE_TRACE`` is set.""" + + def __init__(self, length, optimum): + self.path = os.environ.get("GENOXIDE_TRACE") + self.frames = Frames(32) + self.length, self.optimum = length, optimum + + @property + def on_generation(self): + """The callback for ``run``: None without a trace to record.""" + return self.record if self.path else None + + def record(self, progress): + """Records a generation: the first 16 strings of the population.""" + if self.path: + rows = ["".join("1" if one else "0" for one in row) for row in progress.population[:16]] + self.frames.push(frame(progress, {"population": rows})) + + def write(self): + """Writes the trace, if there's one.""" + if self.path: + settings = { + "format": 1, + "example": "royal_road_r2", + "objective": "maximize", + "x_label": "generations", + "y_label": "R2", + "log_y": False, + "optimum": float(self.optimum), + "plot": "bits", + "problem": {"length": self.length}, + } + write(self.path, settings, self.frames.to_list()) + + +# ---- the same in every example's trace ---------------------------------------------------------- + + +class Frames: + """The frames of at most ``most`` generations, from the part of the run where what the page + plots changes: the frames after the last change are left out (a run that reached its target, + or a front that no longer moves), and the rest are spread evenly over the generations up to + it. While the run goes, up to 8 × ``most`` frames are kept: every ``every``-th generation, + with ``every`` doubling whenever there are that many, and the last one.""" + + def __init__(self, most): + self.most, self.every, self.kept, self.last = most, 1, [], None + + def push(self, frame): + if frame["generation"] % self.every: + self.last = frame + return + self.kept.append(frame) + self.last = None + if len(self.kept) == 8 * self.most: + self.every *= 2 + self.kept = [kept for kept in self.kept if kept["generation"] % self.every == 0] + + def to_list(self): + frames = self.kept + ([self.last] if self.last else []) + active = frames[: last_change(frames) + 1] + count, most = len(active), max(self.most, 2) + if count <= most: + return active + return [active[(i * (count - 1) + (most - 1) // 2) // (most - 1)] for i in range(most)] + + +def last_change(frames): + """The index of the frame after which nothing the page plots changes. To 3 significant + digits, as a plot shows them: the best, the median and, for a single objective (a numeric + best), the state; to within a hundredth of their range over the run: a front's + hypervolumes, in the state or in a grid's series.""" + if not frames: + return 0 + last = len(frames) - 1 + number = lambda value: isinstance(value, (int, float)) and not isinstance(value, bool) + single = any(number(frame.get("best")) for frame in frames) + + def measures(frame): + values = [] + state = frame.get("state") + hypervolume = state.get("hypervolume") if isinstance(state, dict) else None + for value in (hypervolume, frame.get("series")): + if number(value): + values.append(float(value)) + elif isinstance(value, dict): + values.extend(float(v) for _, v in sorted(value.items()) if number(v)) + return values + + measured = [measures(frame) for frame in frames] + end = measured[last] + tolerance = [] + for k in range(len(end)): + values = [values[k] for values in measured if k < len(values)] + tolerance.append((max(values) - min(values)) / 100.0) + + def same(frame, final, key, flush=False): + return coarse(frame.get(key), flush) == coarse(final.get(key), flush) + + def settled(i): + frame, final = frames[i], frames[last] + return ( + same(frame, final, "best") + and same(frame, final, "median") + and (not single or same(frame, final, "state", flush=True)) + and len(measured[i]) == len(end) + and all(abs(v - e) <= t for v, e, t in zip(measured[i], end, tolerance)) + ) + + first = last + while first > 0 and settled(first - 1): + first -= 1 + return first + + +def frame(progress, state): + """The frame of a generation: its progress, the median score of its population and + ``state``.""" + return { + "generation": progress.generation, + "evaluations": progress.evaluations, + "best": progress.best_fitness, + "median": median(progress.scores), + "state": state, + } + + +def median(scores): + """The median of the valid scores, None without any.""" + scores = sorted(float(score) for score in scores if not math.isnan(score)) + middle = len(scores) // 2 + if not scores: + return None + return scores[middle] if len(scores) % 2 else (scores[middle - 1] + scores[middle]) / 2 + + +def coarse(value, flush=False): + """``value`` with its numbers to 3 significant digits, as precisely as a plot shows them: two + frames whose plotted values agree to that precision look the same. With ``flush``, for the + solutions a plot draws on their ranges, numbers below 1e-6 in size count as 0.""" + if isinstance(value, float): + if flush and abs(value) < 1e-6: + value = 0.0 + return f"{value:.2e}" + if isinstance(value, (list, tuple)): + return "[" + ",".join(coarse(item, flush) for item in value) + "]" + if isinstance(value, dict): + items = sorted(value.items()) + return "{" + ",".join(f"{key}:{coarse(item, flush)}" for key, item in items) + "}" + return json.dumps(value) + + +def write(path, settings, frames): + """Writes the settings and the frames to ``path``, a frame per line.""" + lines = ",\n".join(map(to_json, frames)) + with open(path, "w", encoding="utf-8", newline="\n") as file: + file.write(f'{to_json(settings)[:-1]},"frames":[\n{lines}\n]}}\n') + + +def to_json(value): + """Compact JSON with sorted keys, and numbers rounded to 6 significant digits, as the Rust + example writes it.""" + return json.dumps(rounded(value), sort_keys=True, separators=(",", ":"), ensure_ascii=False) + + +def rounded(value): + if isinstance(value, dict): + return {key: rounded(item) for key, item in value.items()} + if isinstance(value, (list, tuple)): + return [rounded(item) for item in value] + if isinstance(value, float): + return float(f"{value:.5e}") if math.isfinite(value) else None + return value diff --git a/examples/royal_road_r2/trace.rs b/examples/royal_road_r2/trace.rs new file mode 100644 index 00000000..40ca5a79 --- /dev/null +++ b/examples/royal_road_r2/trace.rs @@ -0,0 +1,264 @@ +//! The trace of the run for the plot on the example's page, written to the file that +//! `GENOXIDE_TRACE` names: the first 16 strings of the population, in at most 32 generations. The Python example writes the same +//! file. + +use genoxide::observer::Snapshot; +use genoxide::prelude::*; +use serde_json::{Value, json}; + +pub struct Trace { + path: Option, + frames: Frames, + length: usize, + optimum: f64, +} + +impl Trace { + // a trace for the file that GENOXIDE_TRACE names, or nothing to record if it isn't set + pub fn from_env(length: usize, optimum: f64) -> Self { + let path = std::env::var("GENOXIDE_TRACE").ok(); + let frames = Frames::new(32); + Self { + path, + frames, + length, + optimum, + } + } + + // records a generation: the first 16 strings of the population + pub fn record(&mut self, snapshot: &Snapshot<'_, Bits>) { + if self.path.is_none() { + return; + } + let bits = |genome: &Bits| { + genome + .iter() + .map(|one| if one { '1' } else { '0' }) + .collect() + }; + let rows = snapshot.population().iter().take(16); + let rows: Vec = rows.map(|row| bits(row.genome())).collect(); + self.frames + .push(frame(snapshot, json!({ "population": rows }))); + } + + // writes the trace, if there's one + pub fn write(self) { + let Some(path) = self.path else { return }; + let settings = json!({ + "format": 1, + "example": "royal_road_r2", + "objective": "maximize", + "x_label": "generations", + "y_label": "R2", + "log_y": false, + "optimum": self.optimum, + "plot": "bits", + "problem": { "length": self.length }, + }); + write(&path, settings, self.frames.into_vec()); + } +} + +// ---- the same in every example's trace --------------------------------------------------------- + +// the frames of at most `most` generations, from the part of the run where what the page plots +// changes: the frames after the last change are left out (a run that reached its target, or a +// front that no longer moves), and the rest are spread evenly over the generations up to it. While +// the run goes, up to 8 × `most` frames are kept: every `every`-th generation, with `every` +// doubling whenever there are that many, and the last one. +struct Frames { + most: usize, + every: u64, + kept: Vec<(u64, Value)>, + last: Option<(u64, Value)>, +} + +impl Frames { + fn new(most: usize) -> Self { + let (every, kept, last) = (1, Vec::new(), None); + Self { + most, + every, + kept, + last, + } + } + + fn push(&mut self, frame: Value) { + let generation = frame["generation"].as_u64().expect("a generation"); + if !generation.is_multiple_of(self.every) { + self.last = Some((generation, frame)); + return; + } + self.kept.push((generation, frame)); + self.last = None; + if self.kept.len() == 8 * self.most { + self.every *= 2; + let every = self.every; + self.kept.retain(|(generation, _)| generation % every == 0); + } + } + + fn into_vec(self) -> Vec { + let frames = self.kept.into_iter().chain(self.last); + let frames: Vec = frames.map(|(_, frame)| frame).collect(); + let active = &frames[..=last_change(&frames)]; + let (count, most) = (active.len(), self.most.max(2)); + if count <= most { + return active.to_vec(); + } + let at = |i: usize| active[(i * (count - 1) + (most - 1) / 2) / (most - 1)].clone(); + (0..most).map(at).collect() + } +} + +// the index of the frame after which nothing the page plots changes. To 3 significant digits, as +// a plot shows them: the best, the median and, for a single objective (a numeric best), the state; +// to within a hundredth of their range over the run: a front's hypervolumes, in the state or in a +// grid's series +fn last_change(frames: &[Value]) -> usize { + let Some(last) = frames.len().checked_sub(1) else { + return 0; + }; + let single = frames.iter().any(|frame| frame["best"].is_number()); + let measures = |frame: &Value| -> Vec { + let mut values = Vec::new(); + for value in [&frame["state"]["hypervolume"], &frame["series"]] { + match value { + Value::Number(number) => values.extend(number.as_f64()), + Value::Object(map) => values.extend(map.values().filter_map(Value::as_f64)), + _ => {} + } + } + values + }; + let measured: Vec> = frames.iter().map(measures).collect(); + let end = &measured[last]; + let tolerance: Vec = (0..end.len()) + .map(|k| { + let values = measured.iter().filter_map(|values| values.get(k).copied()); + let (low, high) = values.fold((f64::INFINITY, f64::NEG_INFINITY), |(low, high), v| { + (low.min(v), high.max(v)) + }); + (high - low) / 100.0 + }) + .collect(); + let settled = |i: usize| { + let (frame, final_frame) = (&frames[i], &frames[last]); + let same = + |key: &str, flush: bool| coarse(&frame[key], flush) == coarse(&final_frame[key], flush); + same("best", false) + && same("median", false) + && (!single || same("state", true)) + && measured[i].len() == end.len() + && measured[i] + .iter() + .zip(end) + .zip(&tolerance) + .all(|((value, end), tolerance)| (value - end).abs() <= *tolerance) + }; + let mut first = last; + while first > 0 && settled(first - 1) { + first -= 1; + } + first +} + +// the frame of a generation: its progress, the median score of its population and `state` +fn frame(snapshot: &Snapshot<'_, G>, state: Value) -> Value { + let progress = snapshot.progress(); + let population = snapshot.population().iter(); + let scores = population.filter_map(|individual| individual.fitness()?.score()); + json!({ + "generation": progress.generation(), + "evaluations": progress.evaluations(), + "best": progress.best().and_then(Fitness::score), + "median": median(scores.collect()), + "state": state, + }) +} + +// the median of the scores, None without any +fn median(mut scores: Vec) -> Option { + scores.sort_by(f64::total_cmp); + let middle = scores.len() / 2; + match scores.len() { + 0 => None, + n if n % 2 == 1 => Some(scores[middle]), + _ => Some((scores[middle - 1] + scores[middle]) / 2.0), + } +} + +// `value` with its numbers to 3 significant digits, as precisely as a plot shows them: two frames +// whose plotted values agree to that precision look the same. With `flush`, for the solutions a +// plot draws on their ranges, numbers below 1e-6 in size count as 0 +fn coarse(value: &Value, flush: bool) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => { + let number = number.as_f64().expect("f64"); + let number = if flush && number.abs() < 1e-6 { + 0.0 + } else { + number + }; + format!("{number:.2e}") + } + Value::Array(items) => { + format!( + "[{}]", + join(items.iter().map(|item| coarse(item, flush)).collect()) + ) + } + Value::Object(map) => { + let entry = |(key, item): (&String, &Value)| format!("{key}:{}", coarse(item, flush)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +// writes the settings and the frames to `path`, a frame per line +fn write(path: &str, settings: Value, frames: Vec) { + let frames: Vec = frames.iter().map(to_json).collect(); + let settings = to_json(&settings); + let head = &settings[..settings.len() - 1]; + let text = format!("{head},\"frames\":[\n{}\n]}}\n", frames.join(",\n")); + std::fs::write(path, text).expect("the trace is written"); +} + +// compact JSON with sorted keys, and numbers rounded to 6 significant digits and written as +// Python writes them (7542.0, 1e-08): the Python example writes the same file +fn to_json(value: &Value) -> String { + let join = |items: Vec| items.join(","); + match value { + Value::Number(number) if number.is_f64() => python_float(number.as_f64().expect("f64")), + Value::Array(items) => format!("[{}]", join(items.iter().map(to_json).collect())), + Value::Object(map) => { + let entry = + |(key, item): (&String, &Value)| format!("{}:{}", json!(key), to_json(item)); + format!("{{{}}}", join(map.iter().map(entry).collect())) + } + other => other.to_string(), + } +} + +fn python_float(value: f64) -> String { + let rounded: f64 = format!("{value:.5e}").parse().expect("a number"); + let shortest = format!("{rounded:e}"); + let (mantissa, exponent) = shortest.split_once('e').expect("an exponent"); + let exponent: i32 = exponent.parse().expect("an exponent"); + if (-4..16).contains(&exponent) { + let text = rounded.to_string(); + if text.contains('.') { + text + } else { + text + ".0" + } + } else { + let sign = if exponent < 0 { '-' } else { '+' }; + format!("{mantissa}e{sign}{:02}", exponent.abs()) + } +} diff --git a/python/README.md b/python/README.md index ee3a6954..45cf367e 100644 --- a/python/README.md +++ b/python/README.md @@ -48,7 +48,8 @@ print(result.best_genome, result.best_fitness) More in [examples/](https://github.com/tachsin/genoxide/tree/main/examples), each the same program in Python and Rust (`python examples//main.py` in the repository), on the [docs site](https://tachsin.github.io/genoxide/examples/), and played back with charts made for each problem on [tachsin.gr](https://tachsin.gr/projects/genoxide/examples): -- OneMax, a knapsack with a constraint, and N-Queens +- OneMax, LeadingOnes, the deceptive trap, the royal roads, an NK landscape, a knapsack with a + constraint, and N-Queens - the travelling salesman (TSPLIB berlin52) and job shop scheduling (ft06) - Rastrigin with CMA-ES and L-SHADE, and the pressure vessel and welded beam designs with constraints - the gear train design, an integer problem @@ -279,6 +280,27 @@ objectives), `CarSideImpact()`, `RocketInjector()`, `VehicleCrashworthiness()` a `ideal_point`, and for two objectives their `nadir_point`. `DiscBrake` and `SpeedReducer` round their integer gene, and `design(x)` gives the rounded design. +`gx.problems.binary` has problems of bit strings, all maximized: `OneMax(bits)`, +`LeadingOnes(bits)`, the deceptive `Trap(blocks, k)` (with any `a`, `b` and `z`), the royal roads +`RoyalRoad.r1()` and `RoyalRoad.r2()`, `NkLandscape(n, k, neighborhood, seed)` and the 0/1 +`Knapsack(items, instance_class, seed=...)` of Pisinger's generated classes (`KnapsackItems(weights, +profits, capacity)` for given items). The NK landscapes and the knapsacks are drawn from their seed, +the same as in Rust, and compute their `optimum` exactly, by dynamic programming or exhaustive +search. + +```python +problem = gx.problems.binary.Knapsack(50, "uncorrelated", seed=1) +ga = gx.Ga( + problem.genome, + population_size=200, + select=gx.Tournament(3), + crossover=gx.UniformCrossover(), + mutation=gx.BitFlip(rate=1 / 50), + seed=1, +) +result = ga.run(problem, target=problem.optimum.value, generations=2_000) +print(result.best_fitness, result.violation, problem.capacity) +``` ```python problem = gx.problems.engineering.WeldedBeam() diff --git a/python/genoxide/_genoxide.pyi b/python/genoxide/_genoxide.pyi index c55ab5c8..f5261e3e 100644 --- a/python/genoxide/_genoxide.pyi +++ b/python/genoxide/_genoxide.pyi @@ -65,6 +65,8 @@ class Snapshot: def das_dennis(objectives: int, divisions: int) -> np.ndarray: ... def problem_info(problem: str) -> dict[str, Any]: ... +def problem_optimum(problem: str) -> dict[str, Any] | None: ... +def nk_tables(problem: str) -> tuple[np.ndarray, np.ndarray]: ... def evaluate(problem: str, genomes: np.ndarray) -> Any: ... def constraints(problem: str, genome: np.ndarray) -> np.ndarray: ... def optimal_front(problem: str, points: int) -> np.ndarray | None: ... diff --git a/python/genoxide/problems/__init__.py b/python/genoxide/problems/__init__.py index b91b3a90..ffabe39c 100644 --- a/python/genoxide/problems/__init__.py +++ b/python/genoxide/problems/__init__.py @@ -64,13 +64,22 @@ pass the problem's gradient on (turned by the rotation), and a constrained problem's constraint values, which :class:`genoxide.Bo` models. -All problems here are minimized, on :class:`genoxide.Real` genomes except the gear train's -:class:`genoxide.Integer` and :class:`Zdt5`'s :class:`genoxide.Binary`. Each class's docstring -gives -the function, its bounds, its optimum or front and its source. Many originals are books, reports -or proceedings that aren't online, and some functions have no known origin: their definitions -are taken from later papers that restate them, named in the docstrings, and are still to be -checked against the originals (https://github.com/tachsin/genoxide/issues/168), among them: +:mod:`genoxide.problems.binary` holds problems of bit strings, maximized: OneMax, LeadingOnes, +the deceptive trap, the royal roads, NK landscapes and the 0/1 knapsack:: + + problem = gx.problems.binary.NkLandscape(20, 4, seed=1) + search = gx.LocalSearch( + problem.genome, neighbor=gx.BitFlip(count=1), restart=(100, 5), seed=1 + ) + result = search.run(problem, target=problem.optimum.value, evaluations=500_000) + +All the other problems here are minimized, on :class:`genoxide.Real` genomes except the gear +train's :class:`genoxide.Integer` and :class:`Zdt5`'s :class:`genoxide.Binary`. Each class's +docstring gives the function, its bounds, its optimum or front and its source. Many originals +are books, reports or proceedings that aren't online, and some functions have no known origin: +their definitions are taken from later papers that restate them, named in the docstrings, and +are still to be checked against the originals (https://github.com/tachsin/genoxide/issues/168), +among them: - Yao, X., Liu, Y. and Lin, G. (1999). Evolutionary programming made faster. IEEE Transactions on Evolutionary Computation 3(2): 82-102. doi:10.1109/4235.771163 @@ -264,7 +273,8 @@ class Optimum: known.""" -# the genome of a problem: Real for most, Integer for the gear train, Binary for ZDT5 +# the genome of a problem: Real for most, Integer for the gear train, Binary for ZDT5 and the +# problems of `binary` _Genome = TypeVar("_Genome", bound="Real | Integer | Binary", covariant=True) @@ -364,17 +374,24 @@ class Problem(_Described[_Genome]): @property def objective(self) -> ObjectiveName: - """Whether the score is minimized or maximized: "minimize" for every problem here.""" + """Whether the score is minimized or maximized: "minimize" for every problem but those of + :mod:`genoxide.problems.binary`, which are maximized.""" return cast(ObjectiveName, self._info["objectives"][0]) @property def optimum(self) -> Optimum | None: - """The global optimum, or None if it isn't known for this size.""" - optimum = self._info["optimum"] + """The global optimum, or None if it isn't known for this size. The NK landscapes and + the knapsack of :mod:`genoxide.problems.binary` compute theirs, once, when it's first + asked for.""" + optimum = self._info["optimum"] if "optimum" in self._info else self._computed_optimum if optimum is None: return None return Optimum(optimum["value"], optimum["solutions"], optimum["proven"]) + @cached_property + def _computed_optimum(self) -> dict[str, Any] | None: + return _genoxide.problem_optimum(self._json()) + def evaluate(self, genomes: Any) -> Any: """The scores of ``genomes``, a 2-D array with a genome per row, as a 1-D array; for a constrained problem, a tuple of it and an array of constraint violations. @@ -3402,4 +3419,4 @@ class DasCmop9(_DasCmop): _least: ClassVar[int] = 3 -from . import cec2006, control, engineering, multi_engineering # noqa: E402 +from . import binary, cec2006, control, engineering, multi_engineering # noqa: E402 diff --git a/python/genoxide/problems/binary.py b/python/genoxide/problems/binary.py new file mode 100644 index 00000000..3f485d8e --- /dev/null +++ b/python/genoxide/problems/binary.py @@ -0,0 +1,486 @@ +"""Binary and combinatorial test problems, on :class:`genoxide.Binary` genomes, all maximized and +evaluated in Rust:: + + import genoxide as gx + + problem = gx.problems.binary.Trap(10, 4) + ga = gx.Ga( + problem.genome, + population_size=1000, + select=gx.Tournament(4), + crossover=gx.PointCrossover(2), + mutation=gx.BitFlip(rate=1 / problem.dimensions), + seed=1, + ) + result = ga.run(problem, target=problem.optimum.value, generations=300) + +The problems are :class:`OneMax`, :class:`LeadingOnes`, the deceptive :class:`Trap`, the royal +roads R1 and R2 (:class:`RoyalRoad`), :class:`NkLandscape` and the 0/1 :class:`Knapsack` with +Pisinger's generated instance classes (:class:`KnapsackItems` for given items). NK landscapes and +knapsack instances are drawn from a seed with genoxide's portable random numbers: the same seed +gives the same problem on every platform, and the same as in Rust. Their ``optimum`` is computed, +exactly: by dynamic programming (NK landscapes with adjacent neighborhoods, and the knapsack) or by +evaluating every bit string (small NK landscapes); None when that would take too long. + +Each class's docstring gives the definition and its source, read for it: + +- Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary + algorithm. Theoretical Computer Science 276(1-2): 51-81. doi:10.1016/S0304-3975(01)00182-7 +- Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. Foundations of + Genetic Algorithms 2: 93-108. doi:10.1016/B978-0-08-094832-4.50012-X +- Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms: + fitness landscapes and GA performance. Proceedings of the First European Conference on + Artificial Life: 245-254. +- Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a genetic algorithm outperform + hill climbing? Advances in Neural Information Processing Systems 6: 51-58. +- Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and its + application to maturation of the immune response. Journal of Theoretical Biology 141(2): + 211-245. doi:10.1016/S0022-5193(89)80019-0 +- Pisinger, D. (2005). Where are the hard knapsack problems? Computers & Operations Research + 32(9): 2271-2284. doi:10.1016/j.cor.2004.03.002 +""" + +from __future__ import annotations + +from dataclasses import dataclass +from functools import cached_property +from typing import Any, ClassVar, Literal, Union, cast + +import numpy as np + +from .. import Binary, _genoxide, _number, _whole +from . import Problem + +__all__ = [ + "OneMax", + "LeadingOnes", + "Trap", + "RoyalRoad", + "NkLandscape", + "Knapsack", + "KnapsackItems", + "Spanner", + "MultipleStronglyCorrelated", + "ProfitCeiling", + "Circle", +] + +_MAX_BITS = 2**24 + + +@dataclass(frozen=True) +class OneMax(Problem[Binary]): + """OneMax: the number of ones, ``Σ xᵢ``, the simplest function of bit strings. + + Every string but the optimum has a better neighbor one flip away. The (1+1) evolutionary + algorithm needs Θ(n log n) evaluations on average (Droste et al., Lemma 10). + + Maximum ``bits`` at all ones; ``bits`` is 1 to 2^24. + + Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary + algorithm. Theoretical Computer Science 276(1-2): 51-81, Definition 9. The function has no + single origin; Ackley (1987, A Connectionist Machine for Genetic Hillclimbing, section 3.3.1) + tests a "One Max" that is ten times the number of ones. + """ + + bits: int = 100 + _type: ClassVar[str] = "one_max" + + def _describe(self) -> dict[str, Any]: + bits = _whole("OneMax.bits", self.bits, minimum=1, maximum=_MAX_BITS) + return {"type": self._type, "bits": bits} + + +@dataclass(frozen=True) +class LeadingOnes(Problem[Binary]): + """LeadingOnes: the number of ones before the first zero, ``Σᵢ Πⱼ≤ᵢ xⱼ``. + + Unimodal, but only the first zero can improve a string. The (1+1) evolutionary algorithm + needs Θ(n²) evaluations on average, and at most e n² (Droste et al., Theorem 17). + + Maximum ``bits`` at all ones; ``bits`` is 1 to 2^24. + + Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary + algorithm. Theoretical Computer Science 276(1-2): 51-81, Definition 16, from Rudolph, G. + (1997). Convergence Properties of Evolutionary Algorithms. Kovač, Hamburg, whom they credit. + """ + + bits: int = 100 + _type: ClassVar[str] = "leading_ones" + + def _describe(self) -> dict[str, Any]: + bits = _whole("LeadingOnes.bits", self.bits, minimum=1, maximum=_MAX_BITS) + return {"type": self._type, "bits": bits} + + +@dataclass(frozen=True) +class Trap(Problem[Binary]): + """The deceptive trap: ``blocks`` consecutive blocks of ``k`` bits, each scored by Deb and + Goldberg's trap function of its number of ones u, summed. + + ``f(u) = a (z − u) / z`` for u ≤ z, and ``b (u − z) / (k − z)`` otherwise: it falls from a at + no ones, the deceptive attractor, to 0 at z, and rises to b > a with all ones. a, b and z are + k − 1, k and k − 1 unless given (each on its own): a block scores k − 1 − u below k ones, + and k with all of them, fully deceptive for k ≥ 3. Ackley's trap of n bits is + ``Trap(1, n, a=8n, b=10n, z=3n // 4)``. + + Maximum ``blocks * b`` at all ones; ``blocks`` at least 1, ``k`` at least 2, at most 2^24 + bits; z from 1 to k − 1, and 0 ≤ a < b. + + Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. Foundations of + Genetic Algorithms 2: 93-108, equation 1, after Ackley, D. H. (1987). A Connectionist Machine + for Genetic Hillclimbing. Kluwer, section 3.3.3. + """ + + blocks: int = 10 + k: int = 4 + a: float | None = None + b: float | None = None + z: int | None = None + _type: ClassVar[str] = "trap" + + def _describe(self) -> dict[str, Any]: + return { + "type": self._type, + "blocks": _whole("Trap.blocks", self.blocks, minimum=1, maximum=_MAX_BITS), + "k": _whole("Trap.k", self.k, minimum=2, maximum=_MAX_BITS), + "a": None if self.a is None else _number("Trap.a", self.a), + "b": None if self.b is None else _number("Trap.b", self.b), + "z": None if self.z is None else _whole("Trap.z", self.z, maximum=_MAX_BITS), + } + + +@dataclass(frozen=True) +class RoyalRoad(Problem[Binary]): + """The royal road functions: ``blocks`` blocks of ``block_size`` consecutive ones, each + scoring ``block_size`` when complete (R1), and with ``hierarchical``, the complete pairs, + quadruples and so on up to all the blocks scoring again, each its number of bits (R2). + + :meth:`r1` is R1, 8 blocks of 8 bits, maximum 64 (Mitchell, Holland and Forrest, 1994, + Figure 1); :meth:`r2` is R2, maximum 8 · 8 + 4 · 16 + 2 · 32 + 64 = 256 (Mitchell, Forrest and + Holland, 1992, Figure 1). ``blocks`` is a power of 2 with ``hierarchical``. + + Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms: + fitness landscapes and GA performance. Proceedings of the First European Conference on + Artificial Life: 245-254; Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a + genetic algorithm outperform hill climbing? Advances in Neural Information Processing Systems + 6: 51-58. + """ + + blocks: int = 8 + block_size: int = 8 + hierarchical: bool = False + _type: ClassVar[str] = "royal_road" + + @classmethod + def r1(cls) -> RoyalRoad: + """R1: 8 blocks of 8 bits, maximum 64.""" + return cls(8, 8) + + @classmethod + def r2(cls) -> RoyalRoad: + """R2: 8 blocks of 8 bits and the levels above them, maximum 256.""" + return cls(8, 8, hierarchical=True) + + def _describe(self) -> dict[str, Any]: + if not isinstance(self.hierarchical, (bool, np.bool_)): + raise ValueError(f"RoyalRoad.hierarchical is True or False, not {self.hierarchical!r}") + return { + "type": self._type, + "blocks": _whole("RoyalRoad.blocks", self.blocks, minimum=1, maximum=_MAX_BITS), + "block_size": _whole( + "RoyalRoad.block_size", self.block_size, minimum=1, maximum=_MAX_BITS + ), + "hierarchical": bool(self.hierarchical), + } + + +@dataclass(frozen=True) +class NkLandscape(Problem[Binary]): + """An NK landscape: ``n`` bits, each contributing a value drawn uniformly from (0, 1) for each + combination of its bit and those of ``k`` others, the fitness their mean. + + ``neighborhood`` "adjacent" takes a site's flanking sites on a circle (k/2 on each side; for + an odd k, one more after it), "random" k distinct other sites drawn for each. The neighbors + and contributions are drawn from ``seed``: the same seed gives the same landscape on every + platform and in Rust. :attr:`neighbors` and :attr:`contributions` give them. + + ``optimum`` is computed exactly: by dynamic programming for adjacent neighborhoods, or by + evaluating every string, whichever is less work; None when both take more than 2^30 steps. + + ``n`` is 1 to 2^24, ``k`` below ``n``, and the tables at most 2^24 values (n 2^(k+1)). + + Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and + its application to maturation of the immune response. Journal of Theoretical Biology 141(2): + 211-245. + """ + + n: int + k: int + neighborhood: Literal["adjacent", "random"] = "random" + seed: int = 0 + _type: ClassVar[str] = "nk_landscape" + + def _describe(self) -> dict[str, Any]: + if self.neighborhood not in ("adjacent", "random"): + raise ValueError( + f'NkLandscape.neighborhood is "adjacent" or "random", not {self.neighborhood!r}' + ) + return { + "type": self._type, + "n": _whole("NkLandscape.n", self.n, minimum=1, maximum=_MAX_BITS), + "k": _whole("NkLandscape.k", self.k, maximum=_MAX_BITS), + "neighborhood": self.neighborhood, + "seed": _whole("NkLandscape.seed", self.seed), + } + + @cached_property + def _tables(self) -> tuple[np.ndarray, np.ndarray]: + return _genoxide.nk_tables(self._json()) + + @property + def neighbors(self) -> np.ndarray: + """The k sites that bear on each site, a row per site, in the order of the bits of its + table's index.""" + return self._tables[0] + + @property + def contributions(self) -> np.ndarray: + """The contributions of each site, a row per site with 2^(k+1) values: the site's own bit + is bit k of the index (the highest), and its j-th neighbor's bit k − 1 − j.""" + return self._tables[1] + + +@dataclass(frozen=True) +class Spanner: + """span(v, m) knapsack instances: every item a multiple (1 to ``m``) of one of ``v`` spanner + items, drawn with weights in [1, R] and profits by ``distribution``, then scaled down to + ⌈2p/m⌉ and ⌈2w/m⌉. The paper uses span(2, 10).""" + + v: int = 2 + m: int = 10 + distribution: Literal["uncorrelated", "weakly_correlated", "strongly_correlated"] = ( + "uncorrelated" + ) + + def _describe(self) -> dict[str, Any]: + if self.distribution not in ("uncorrelated", "weakly_correlated", "strongly_correlated"): + raise ValueError( + 'Spanner.distribution is "uncorrelated", "weakly_correlated" or ' + f'"strongly_correlated", not {self.distribution!r}' + ) + return { + "type": "spanner", + "v": _whole("Spanner.v", self.v, minimum=1, maximum=_MAX_BITS), + "m": _whole("Spanner.m", self.m, minimum=1, maximum=2**32), + "distribution": self.distribution, + } + + +@dataclass(frozen=True) +class MultipleStronglyCorrelated: + """mstr(k1, k2, d) knapsack instances: weights in [1, R], profits w + k1 for the weights + divisible by d and w + k2 for the others. The paper uses mstr(3R/10, 2R/10, 6), which None + for ``k1`` and ``k2`` gives.""" + + k1: int | None = None + k2: int | None = None + d: int = 6 + + def _describe(self, data_range: int) -> dict[str, Any]: + k1 = 3 * data_range // 10 if self.k1 is None else self.k1 + k2 = 2 * data_range // 10 if self.k2 is None else self.k2 + return { + "type": "multiple_strongly_correlated", + "k1": _whole("MultipleStronglyCorrelated.k1", k1, maximum=2**32), + "k2": _whole("MultipleStronglyCorrelated.k2", k2, maximum=2**32), + "d": _whole("MultipleStronglyCorrelated.d", self.d, minimum=1, maximum=2**32), + } + + +@dataclass(frozen=True) +class ProfitCeiling: + """pceil(d) knapsack instances: weights in [1, R], profits d⌈w/d⌉. The paper uses + pceil(3).""" + + d: int = 3 + + def _describe(self) -> dict[str, Any]: + return { + "type": "profit_ceiling", + "d": _whole("ProfitCeiling.d", self.d, minimum=1, maximum=2**32), + } + + +@dataclass(frozen=True) +class Circle: + """circle(d) knapsack instances: weights in [1, R], profits ``d √(4R² − (w − 2R)²)`` rounded + down, with d = ``numerator / denominator``. The paper uses circle(2/3).""" + + numerator: int = 2 + denominator: int = 3 + + def _describe(self) -> dict[str, Any]: + return { + "type": "circle", + "numerator": _whole("Circle.numerator", self.numerator, minimum=1, maximum=2**16), + "denominator": _whole( + "Circle.denominator", self.denominator, minimum=1, maximum=2**16 + ), + } + + +KnapsackClassName = Literal[ + "uncorrelated", + "weakly_correlated", + "strongly_correlated", + "inverse_strongly_correlated", + "almost_strongly_correlated", + "subset_sum", + "uncorrelated_similar_weights", + "spanner", + "multiple_strongly_correlated", + "profit_ceiling", + "circle", +] +"""The names of Pisinger's instance classes; the last four with the paper's parameters.""" + +_SIMPLE = ( + "uncorrelated", + "weakly_correlated", + "strongly_correlated", + "inverse_strongly_correlated", + "almost_strongly_correlated", + "subset_sum", + "uncorrelated_similar_weights", +) + +KnapsackClass = Union[ + KnapsackClassName, Spanner, MultipleStronglyCorrelated, ProfitCeiling, Circle +] +"""An instance class: a name, or one of the classes with parameters.""" + + +class _KnapsackProblem(Problem[Binary]): + """What a knapsack has: its items and capacity.""" + + @property + def weights(self) -> np.ndarray: + """The weight of each item (int64).""" + return np.array(self._info["weights"], dtype=np.int64) + + @property + def profits(self) -> np.ndarray: + """The profit of each item (int64).""" + return np.array(self._info["profits"], dtype=np.int64) + + @property + def capacity(self) -> int: + """The capacity.""" + return cast(int, self._info["capacity"]) + + +@dataclass(frozen=True) +class Knapsack(_KnapsackProblem): + """The 0/1 knapsack problem on a generated instance of ``items`` items: the most profitable + selection, a bit per item, whose weight fits the capacity. + + The fitness is ``(profit, violation)``: the total profit, and how far the weight exceeds the + capacity, 0 when it fits (Deb's rules). ``instance_class`` is one of Pisinger's classes, "in + [x, y]" drawn uniformly and R/10, R/500 rounded down: + + - "uncorrelated": weights and profits in [1, R]; + - "weakly_correlated": weights in [1, R], profits in [max(1, w − R/10), w + R/10]; + - "strongly_correlated": weights in [1, R], profits w + R/10; + - "inverse_strongly_correlated": profits in [1, R], weights p + R/10; + - "almost_strongly_correlated": weights in [1, R], profits in [w + R/10 − R/500, + w + R/10 + R/500]; + - "subset_sum": weights in [1, R], profits equal to them; + - "uncorrelated_similar_weights": weights in [100 000, 100 100], profits in [1, 1000]; + - :class:`Spanner`, :class:`MultipleStronglyCorrelated`, :class:`ProfitCeiling` and + :class:`Circle`, or their names for the paper's parameters: span(2, 10) of uncorrelated + items, mstr(3R/10, 2R/10, 6), pceil(3) and circle(2/3). + + ``data_range`` is R (1 to 2^32), and the capacity is ⌊h/(H + 1) Σ w⌋ for ``instance`` h of + ``instances`` H (eq. 5): about half the total weight by default. The items are drawn from + ``seed``, the same on every platform and in Rust. ``optimum`` is computed by dynamic + programming, None when the items times the capacity are above 2^28. + + Pisinger, D. (2005). Where are the hard knapsack problems? Computers & Operations Research + 32(9): 2271-2284, sections 3 and 3.3. + """ + + items: int + instance_class: KnapsackClass = "uncorrelated" + data_range: int = 1000 + instance: int = 50 + instances: int = 100 + seed: int = 0 + _type: ClassVar[str] = "knapsack" + + def _describe(self) -> dict[str, Any]: + data_range = _whole("Knapsack.data_range", self.data_range, minimum=1, maximum=2**32) + kind = self.instance_class + if isinstance(kind, str): + if kind in _SIMPLE: + described: dict[str, Any] = {"type": kind} + elif kind == "spanner": + described = Spanner()._describe() + elif kind == "multiple_strongly_correlated": + described = MultipleStronglyCorrelated()._describe(data_range) + elif kind == "profit_ceiling": + described = ProfitCeiling()._describe() + elif kind == "circle": + described = Circle()._describe() + else: + raise ValueError(f"Knapsack.instance_class: no class {kind!r}") + elif isinstance(kind, MultipleStronglyCorrelated): + described = kind._describe(data_range) + elif isinstance(kind, (Spanner, ProfitCeiling, Circle)): + described = kind._describe() + else: + raise ValueError( + "Knapsack.instance_class is a class name, Spanner, MultipleStronglyCorrelated, " + f"ProfitCeiling or Circle, not {kind!r}" + ) + return { + "type": self._type, + "class": described, + "items": _whole("Knapsack.items", self.items, minimum=1, maximum=_MAX_BITS), + "range": data_range, + "instance": _whole("Knapsack.instance", self.instance, minimum=1, maximum=2**32), + "instances": _whole("Knapsack.instances", self.instances, minimum=1, maximum=2**32), + "seed": _whole("Knapsack.seed", self.seed), + } + + +class KnapsackItems(_KnapsackProblem): + """The 0/1 knapsack problem on given items: ``weights`` and ``profits``, whole numbers of at + least 0, one each per item, and ``capacity``; as :class:`Knapsack`, whose fitness and optimum + it has. The totals and the capacity are at most 2^53.""" + + _type: ClassVar[str] = "knapsack_items" + + def __init__(self, weights: Any, profits: Any, capacity: int) -> None: + self._weights = tuple(_wholes("KnapsackItems.weights", weights)) + self._profits = tuple(_wholes("KnapsackItems.profits", profits)) + self._capacity = _whole("KnapsackItems.capacity", capacity, maximum=2**53) + + def __repr__(self) -> str: + return ( + f"KnapsackItems({list(self._weights)}, {list(self._profits)}, {self._capacity})" + ) + + def _describe(self) -> dict[str, Any]: + return { + "type": self._type, + "weights": list(self._weights), + "profits": list(self._profits), + "capacity": self._capacity, + } + + +def _wholes(name: str, values: Any) -> list[int]: + """``values`` as whole numbers of at least 0.""" + values = np.asarray(values).tolist() + return [_whole(name, value, maximum=2**53, plural=True) for value in values] diff --git a/python/src/fitness.rs b/python/src/fitness.rs index 437520ab..9bdae238 100644 --- a/python/src/fitness.rs +++ b/python/src/fitness.rs @@ -22,7 +22,7 @@ //! which cost as much as the new array. use crate::genes::{GenomeContext, PyGenome}; -use crate::problems::{IntegerProblem, MultiNative}; +use crate::problems::{BinaryProblem, IntegerProblem, MultiNative}; use crate::tasks::Balance; use crate::tree_problems::TreeFitness; use genoxide::Fitness; @@ -716,12 +716,13 @@ pub struct Single<'a> { pub problem: Option>, } -/// A single-objective test problem evaluated in Rust: on real or on integer genomes; or a +/// A single-objective test problem evaluated in Rust: on real, integer or binary genomes; or a /// network's weights balancing poles; or a fitness of trees. #[derive(Clone, Copy)] pub enum Native<'a> { Real(&'a dyn DynProblem), Integer(&'a dyn IntegerProblem), + Binary(&'a dyn BinaryProblem), Balance(&'a Balance), Tree(&'a TreeFitness), } @@ -746,6 +747,11 @@ impl FitnessFunction for Single<'_> { Value::Native(problem.evaluate(genome)) }); } + Some(Native::Binary(problem)) => { + return genome.bits().map_or(Value::Invalid, |genome| { + Value::Native(problem.evaluate(genome)) + }); + } Some(Native::Balance(balance)) => { return genome.reals().map_or(Value::Invalid, |weights| { Value::Native(balance.evaluate(weights)) diff --git a/python/src/lib.rs b/python/src/lib.rs index dd8f94a2..78f3f839 100644 --- a/python/src/lib.rs +++ b/python/src/lib.rs @@ -33,6 +33,8 @@ fn _genoxide(module: &Bound<'_, PyModule>) -> PyResult<()> { module.add_class::()?; module.add_function(wrap_pyfunction!(run::das_dennis, module)?)?; module.add_function(wrap_pyfunction!(problems::problem_info, module)?)?; + module.add_function(wrap_pyfunction!(problems::problem_optimum, module)?)?; + module.add_function(wrap_pyfunction!(problems::nk_tables, module)?)?; module.add_function(wrap_pyfunction!(problems::evaluate, module)?)?; module.add_function(wrap_pyfunction!(problems::constraints, module)?)?; module.add_function(wrap_pyfunction!(problems::optimal_front, module)?)?; diff --git a/python/src/problems.rs b/python/src/problems.rs index ee0ebb55..eb507813 100644 --- a/python/src/problems.rs +++ b/python/src/problems.rs @@ -7,6 +7,7 @@ use genoxide::engine::{Extras, FitnessFunction, IntoFitness, Provided}; use genoxide::genome::{Binary, Bits, Integer, Integers, Real, Reals, Representation}; use genoxide::multi::problems::{self as multi, DynMultiProblem, MultiProblem, try_boxed}; use genoxide::multi::{IntoScores, MultiFitnessFunction, Scores}; +use genoxide::problems::binary::{self, Knapsack, KnapsackClass, NkLandscape}; use genoxide::problems::{ self, Constraints, DynProblem, Optimum, Problem as _, cec2006, engineering, }; @@ -217,10 +218,141 @@ pub enum Config { ThreeBarTruss {}, CantileverBeam {}, CarSideImpact {}, + OneMax { + bits: usize, + }, + LeadingOnes { + bits: usize, + }, + Trap { + blocks: usize, + k: usize, + a: Option, + b: Option, + z: Option, + }, + RoyalRoad { + blocks: usize, + block_size: usize, + hierarchical: bool, + }, + NkLandscape { + n: usize, + k: usize, + neighborhood: NeighborhoodConfig, + seed: u64, + }, + Knapsack { + class: KnapsackClassConfig, + items: usize, + range: u64, + instance: u64, + instances: u64, + seed: u64, + }, + KnapsackItems { + weights: Vec, + profits: Vec, + capacity: u64, + }, #[serde(untagged)] Multi(MultiConfig), } +/// The neighborhoods of an NK landscape. +#[derive(Clone, Copy, Debug, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum NeighborhoodConfig { + Adjacent, + Random, +} + +impl NeighborhoodConfig { + fn neighborhood(self) -> binary::Neighborhood { + match self { + Self::Adjacent => binary::Neighborhood::Adjacent, + Self::Random => binary::Neighborhood::Random, + } + } +} + +/// How the spanner set of a spanner knapsack instance is drawn. +#[derive(Clone, Copy, Debug, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum SpannerConfig { + Uncorrelated, + WeaklyCorrelated, + StronglyCorrelated, +} + +/// A class of generated knapsack instances. +#[derive(Clone, Copy, Debug, Deserialize)] +#[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] +pub enum KnapsackClassConfig { + Uncorrelated {}, + WeaklyCorrelated {}, + StronglyCorrelated {}, + InverseStronglyCorrelated {}, + AlmostStronglyCorrelated {}, + SubsetSum {}, + UncorrelatedSimilarWeights {}, + Spanner { + v: usize, + m: u64, + distribution: SpannerConfig, + }, + MultipleStronglyCorrelated { + k1: u64, + k2: u64, + d: u64, + }, + ProfitCeiling { + d: u64, + }, + Circle { + numerator: u64, + denominator: u64, + }, +} + +impl KnapsackClassConfig { + fn class(self) -> KnapsackClass { + match self { + Self::Uncorrelated {} => KnapsackClass::Uncorrelated, + Self::WeaklyCorrelated {} => KnapsackClass::WeaklyCorrelated, + Self::StronglyCorrelated {} => KnapsackClass::StronglyCorrelated, + Self::InverseStronglyCorrelated {} => KnapsackClass::InverseStronglyCorrelated, + Self::AlmostStronglyCorrelated {} => KnapsackClass::AlmostStronglyCorrelated, + Self::SubsetSum {} => KnapsackClass::SubsetSum, + Self::UncorrelatedSimilarWeights {} => KnapsackClass::UncorrelatedSimilarWeights, + Self::Spanner { v, m, distribution } => KnapsackClass::Spanner { + v, + m, + distribution: match distribution { + SpannerConfig::Uncorrelated => binary::SpannerDistribution::Uncorrelated, + SpannerConfig::WeaklyCorrelated => { + binary::SpannerDistribution::WeaklyCorrelated + } + SpannerConfig::StronglyCorrelated => { + binary::SpannerDistribution::StronglyCorrelated + } + }, + }, + Self::MultipleStronglyCorrelated { k1, k2, d } => { + KnapsackClass::MultipleStronglyCorrelated { k1, k2, d } + } + Self::ProfitCeiling { d } => KnapsackClass::ProfitCeiling { d }, + Self::Circle { + numerator, + denominator, + } => KnapsackClass::Circle { + numerator, + denominator, + }, + } + } +} + /// A multi-objective problem, as `_describe()` of a `gx.problems` class gives it. #[derive(Clone, Copy, Debug, Deserialize)] #[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] @@ -1517,9 +1649,158 @@ pub enum Problem { Single(Box), /// A single-objective problem on integer genomes. Integer(Box), + /// A single-objective problem on binary genomes. + Binary(Box), Multi(MultiConfig), } +/// A single-objective problem on [`Binary`] genomes, as a trait object: what [`DynProblem`] is +/// for real genomes. +pub trait BinaryProblem: Send + Sync { + fn name(&self) -> &'static str; + fn binary(&self) -> Binary; + fn objective(&self) -> Objective; + fn evaluate(&self, genome: &Bits) -> Fitness; + fn optimum(&self) -> Option>; + fn reference(&self) -> &'static str; + fn reference_url(&self) -> Option<&'static str>; + fn constraints(&self, genome: &Bits) -> Constraints; +} + +impl

BinaryProblem for P +where + P: problems::Problem + Send + Sync, +{ + fn name(&self) -> &'static str { + problems::Problem::name(self) + } + + fn binary(&self) -> Binary { + self.representation() + } + + fn objective(&self) -> Objective { + problems::Problem::objective(self) + } + + fn evaluate(&self, genome: &Bits) -> Fitness { + FitnessFunction::evaluate(self, genome) + .into_fitness() + .unwrap_or_else(|_| Fitness::invalid()) + } + + fn optimum(&self) -> Option> { + problems::Problem::optimum(self) + } + + fn reference(&self) -> &'static str { + problems::Problem::reference(self) + } + + fn reference_url(&self) -> Option<&'static str> { + problems::Problem::reference_url(self) + } + + fn constraints(&self, genome: &Bits) -> Constraints { + problems::Problem::constraints(self, genome) + } +} + +// the knapsack that `config` describes, if it's one: an error for settings out of bounds +fn knapsack(config: &Config) -> Option> { + let knapsack = match config { + &Config::Knapsack { + class, + items, + range, + instance, + instances, + seed, + } => Knapsack::generator(class.class(), items) + .range(range) + .instance(instance) + .instances(instances) + .seed(seed) + .generate(), + Config::KnapsackItems { + weights, + profits, + capacity, + } => Knapsack::new(weights.clone(), profits.clone(), *capacity), + _ => return None, + }; + Some(knapsack.map_err(|error| error.to_string())) +} + +// the binary problem that `config` describes, if it's one; a size out of bounds is an error, not +// the panic of the constructor +fn build_binary(config: &Config) -> Option> { + fn boxed

(problem: P) -> Problem + where + P: problems::Problem + Send + Sync + 'static, + { + Problem::Binary(Box::new(problem)) + } + let message = |error: genoxide::Error| error.to_string(); + let problem = match config { + &Config::OneMax { bits } => { + at_least(bits, 1, "OneMax", "bits").map(|bits| boxed(binary::OneMax::new(bits))) + } + &Config::LeadingOnes { bits } => at_least(bits, 1, "LeadingOnes", "bits") + .map(|bits| boxed(binary::LeadingOnes::new(bits))), + &Config::Trap { blocks, k, a, b, z } => { + // a = k − 1, b = k and z = k − 1 unless given + let a = a.unwrap_or(k.saturating_sub(1) as f64); + let b = b.unwrap_or(k as f64); + let z = z.unwrap_or(k.saturating_sub(1)); + binary::Trap::with_values(blocks, k, a, b, z) + .map(boxed) + .map_err(message) + } + &Config::RoyalRoad { + blocks, + block_size, + hierarchical, + } => royal_road(blocks, block_size, hierarchical), + &Config::NkLandscape { + n, + k, + neighborhood, + seed, + } => NkLandscape::new(n, k, neighborhood.neighborhood(), seed) + .map(boxed) + .map_err(message), + Config::Knapsack { .. } | Config::KnapsackItems { .. } => { + knapsack(config).expect("a knapsack").map(boxed) + } + _ => return None, + }; + Some(problem) +} + +// a royal road, its sizes checked +fn royal_road(blocks: usize, block_size: usize, hierarchical: bool) -> Result { + at_least(blocks, 1, "RoyalRoad", "blocks")?; + at_least(block_size, 1, "RoyalRoad", "bits per block")?; + let bits = blocks.saturating_mul(block_size); + if bits > MAX_GENES { + return Err(format!( + "RoyalRoad takes at most {MAX_GENES} (2^24) bits, not {bits}" + )); + } + if hierarchical && !blocks.is_power_of_two() { + return Err(format!( + "a hierarchical royal road needs a power of 2 blocks, not {blocks}" + )); + } + let road = if hierarchical { + binary::RoyalRoad::hierarchical(blocks, block_size) + } else { + binary::RoyalRoad::new(blocks, block_size) + }; + Ok(Problem::Binary(Box::new(road))) +} + /// A single-objective problem on [`Integer`] genomes, as a trait object: what [`DynProblem`] is /// for real genomes. pub trait IntegerProblem: Send + Sync { @@ -1570,6 +1851,9 @@ where // the problem that `config` describes; a size below the minimum is an error, not the panic of // the constructor fn build(config: Config) -> Result { + if let Some(problem) = build_binary(&config) { + return problem; + } let at_least = |dimensions: usize, minimum: usize, name: &str| { at_least(dimensions, minimum, name, "dimensions") }; @@ -1800,6 +2084,13 @@ fn build(config: Config) -> Result { config.check()?; return Ok(Problem::Multi(config)); } + Config::OneMax { .. } + | Config::LeadingOnes { .. } + | Config::Trap { .. } + | Config::RoyalRoad { .. } + | Config::NkLandscape { .. } + | Config::Knapsack { .. } + | Config::KnapsackItems { .. } => unreachable!("built above"), })) } @@ -2025,6 +2316,27 @@ pub fn problem_info<'py>(py: Python<'py>, problem: &str) -> PyResult { + info.set_item("name", problem.name())?; + info.set_item("genome", "binary")?; + let bits = problem.binary().genome_len(); + info.set_item("bounds", vec![(0, 1); bits])?; + let objective = match problem.objective() { + Objective::Maximize => "maximize", + Objective::Minimize => "minimize", + }; + info.set_item("objectives", vec![objective])?; + info.set_item("constraints", problem.constraints(&Bits::zeros(bits)).len())?; + info.set_item("reference", problem.reference())?; + info.set_item("reference_url", problem.reference_url())?; + // the optimum can take seconds to compute: `problem_optimum` gives it + if let Some(knapsack) = knapsack(&config(description)?) { + let knapsack = knapsack.map_err(PyValueError::new_err)?; + info.set_item("weights", knapsack.weights().to_vec())?; + info.set_item("profits", knapsack.profits().to_vec())?; + info.set_item("capacity", knapsack.capacity())?; + } + } Problem::Multi(config) => { let task = MultiInfo { py, @@ -2129,6 +2441,29 @@ pub fn evaluate<'py>( }); Ok(PyArray1::from_vec(py, scores).into_any()) } + Problem::Binary(problem) => { + let genomes = bit_rows(problem.name(), &problem.binary(), &genomes)?; + let bits = problem.binary().genome_len(); + let constrained = !problem.constraints(&Bits::zeros(bits)).is_empty(); + let (scores, violations): (Vec, Vec) = py.detach(|| { + genomes + .iter() + .map(|genome| { + let fitness = problem.evaluate(genome); + match fitness.score() { + Some(score) => (score, fitness.violation()), + None => (f64::NAN, f64::NAN), + } + }) + .unzip() + }); + let scores = PyArray1::from_vec(py, scores).into_any(); + if constrained { + (scores, PyArray1::from_vec(py, violations)).into_bound_py_any(py) + } else { + Ok(scores) + } + } Problem::Multi(config) => run_with( config, MultiEvaluate { @@ -2275,6 +2610,20 @@ pub fn constraints<'py>( } Constraints::none() } + Problem::Binary(problem) => { + let (name, bits) = (problem.name(), problem.binary().genome_len()); + let genome = genome + .iter() + .map(|&gene| bit(name, gene)) + .collect::>()?; + if genome.len() != bits { + return Err(PyValueError::new_err(format!( + "{name} takes genomes of {bits} bits, not {}", + genome.len() + ))); + } + problem.constraints(&genome) + } Problem::Multi(config) => run_with( config, MultiConstraints { @@ -2354,6 +2703,10 @@ pub fn optimal_front<'py>( "{} has one objective, and an optimum instead of a front", problem.name() ))), + Problem::Binary(problem) => Err(PyValueError::new_err(format!( + "{} has one objective, and an optimum instead of a front", + problem.name() + ))), Problem::Multi(config) => run_with(config, MultiFront { py, config, points })?, } } @@ -2422,6 +2775,67 @@ pub fn design<'py>( Ok(PyArray1::from_vec(py, design)) } +/// The optimum of the binary problem that `problem` (JSON) describes: None, or its value, +/// solutions a row each (booleans), and whether it's proven. Computed at each call, without the +/// GIL: by dynamic programming or exhaustive search for the NK landscapes and the knapsack. +#[pyfunction] +pub fn problem_optimum<'py>(py: Python<'py>, problem: &str) -> PyResult> { + let Problem::Binary(problem) = parse(problem)? else { + return Err(PyValueError::new_err( + "problem_optimum is for the problems of genoxide.problems.binary", + )); + }; + let Some(optimum) = py.detach(|| problem.optimum()) else { + return Ok(py.None().into_bound(py)); + }; + let bits = problem.binary().genome_len(); + let genes: Vec = optimum.solutions().iter().flat_map(|x| x.iter()).collect(); + let solutions = Array2::from_shape_vec((optimum.solutions().len(), bits), genes) + .map_err(|error| PyValueError::new_err(error.to_string()))?; + let description = PyDict::new(py); + description.set_item("value", optimum.value())?; + description.set_item("solutions", solutions.into_pyarray(py))?; + description.set_item("proven", optimum.is_proven())?; + Ok(description.into_any()) +} + +/// The neighbors of each site of the NK landscape that `problem` (JSON) describes, a row per +/// site (2-D, int64), and its contributions, a row per site with a value per index of its table +/// (2-D, float64). +#[pyfunction] +pub fn nk_tables<'py>( + py: Python<'py>, + problem: &str, +) -> PyResult<(Bound<'py, PyAny>, Bound<'py, PyAny>)> { + let Config::NkLandscape { + n, + k, + neighborhood, + seed, + } = config(problem)? + else { + return Err(PyValueError::new_err("nk_tables is for an NK landscape")); + }; + let landscape = NkLandscape::new(n, k, neighborhood.neighborhood(), seed) + .map_err(|error| PyValueError::new_err(error.to_string()))?; + let neighbors: Vec = (0..n) + .flat_map(|site| landscape.neighbors(site).iter().map(|&j| j as i64)) + .collect(); + let entries = 1usize << (k + 1); + let tables: Vec = (0..n) + .flat_map(|site| (0..entries).map(move |index| (site, index))) + .map(|(site, index)| landscape.contribution(site, index)) + .collect(); + let neighbors = Array2::from_shape_vec((n, k), neighbors) + .map_err(|error| PyValueError::new_err(error.to_string()))?; + let tables = Array2::from_shape_vec((n, entries), tables) + .map_err(|error| PyValueError::new_err(error.to_string()))?; + Ok(( + neighbors.into_pyarray(py).into_any(), + tables.into_pyarray(py).into_any(), + )) +} + /// The names of the problems of `genoxide::problems::all()`, in its order. #[pyfunction] pub fn problem_names() -> Vec<&'static str> { diff --git a/python/src/run.rs b/python/src/run.rs index 1898c142..50de1ddb 100644 --- a/python/src/run.rs +++ b/python/src/run.rs @@ -289,8 +289,36 @@ fn minimized(name: &str, run: &config::Run) -> Result<()> { Ok(()) } -// a test problem runs with its objectives, minimized, and a genome of its type and dimensions +// a binary test problem is maximized, the algorithms' default: minimizing it would optimize the +// wrong way without a warning +fn maximized(name: &str, run: &config::Run) -> Result<()> { + if run.objectives.contains(&config::Objective::Minimize) { + return Err(format!( + "{name} maximizes its objective: pass objective=\"maximize\" (the problem's objective)" + )); + } + Ok(()) +} + +// a test problem runs with its objectives, and a genome of its type and dimensions fn check_problem(problem: &problems::Problem, run: &config::Run) -> Result<()> { + if let problems::Problem::Binary(problem) = problem { + let name = problem.name(); + if run.objectives.len() != 1 { + return Err(format!( + "{name} has one objective: use a single-objective algorithm" + )); + } + maximized(name, run)?; + let bits = problem.binary().genome_len(); + return match &run.genome { + config::Genome::Binary { length } if *length == bits => Ok(()), + config::Genome::Binary { length } => Err(format!( + "{name} has {bits} bits, but the genome has {length}" + )), + _ => Err(format!("{name} needs a Binary genome")), + }; + } if let problems::Problem::Integer(problem) = problem { let name = problem.name(); if run.objectives.len() != 1 { @@ -314,7 +342,7 @@ fn check_problem(problem: &problems::Problem, run: &config::Run) -> Result<()> { (problem.name(), 1, problem.real().genome_len(), false) } // checked above - problems::Problem::Integer(_) => return Ok(()), + problems::Problem::Integer(_) | problems::Problem::Binary(_) => return Ok(()), problems::Problem::Multi(config) => { let (name, dimensions, binary) = problems::name_and_dimensions(*config); (name, config.objectives(), dimensions, binary) @@ -1827,6 +1855,7 @@ where let problem = match &context.problem { Some(problems::Problem::Single(problem)) => Some(Native::Real(problem.as_ref())), Some(problems::Problem::Integer(problem)) => Some(Native::Integer(problem.as_ref())), + Some(problems::Problem::Binary(problem)) => Some(Native::Binary(problem.as_ref())), _ => match &context.tree { Some(tree) => Some(Native::Tree(tree)), None => context.balance.as_ref().map(Native::Balance), diff --git a/python/tests/test_binary_problems.py b/python/tests/test_binary_problems.py new file mode 100644 index 00000000..45e40792 --- /dev/null +++ b/python/tests/test_binary_problems.py @@ -0,0 +1,241 @@ +"""gx.problems.binary: the problems of bit strings, evaluated in Rust.""" + +import typing + +import numpy as np +import pytest + +import genoxide as gx + +binary = gx.problems.binary + + +def bits(text): + return np.array([bit == "1" for bit in text]) + + +def test_the_module_lists_its_problems(): + assert binary.__all__ == [ + "OneMax", + "LeadingOnes", + "Trap", + "RoyalRoad", + "NkLandscape", + "Knapsack", + "KnapsackItems", + "Spanner", + "MultipleStronglyCorrelated", + "ProfitCeiling", + "Circle", + ] + + +@pytest.mark.parametrize( + "problem", + [ + binary.OneMax(), + binary.LeadingOnes(), + binary.Trap(), + binary.RoyalRoad(), + binary.NkLandscape(12, 2), + binary.Knapsack(20), + binary.KnapsackItems([1, 2], [3, 4], 2), + ], +) +def test_every_problem_describes_itself(problem): + assert problem.objective == "maximize" + assert isinstance(problem.genome, gx.Binary) + assert problem.reference + assert problem.reference_url is None or problem.reference_url.startswith("https://") + optimum = problem.optimum + assert optimum.proven + assert optimum.solutions.dtype == np.bool_ + assert optimum.solutions.shape == (1, problem.dimensions) + value = problem(optimum.solutions[0]) + if isinstance(value, tuple): + assert value == (optimum.value, 0.0) + else: + assert value == optimum.value + # Problem[Binary], for type checkers + declared = next( + typing.get_args(base)[0] + for ancestor in type(problem).__mro__ + for base in getattr(ancestor, "__orig_bases__", ()) + if typing.get_origin(base) is gx.problems.Problem + ) + assert declared is gx.Binary + + +def test_the_definitions(): + assert binary.OneMax(5)(bits("10110")) == 3.0 + assert binary.LeadingOnes(6)(bits("110111")) == 2.0 + assert binary.LeadingOnes(6)(bits("011111")) == 0.0 + # k = 5: 4 − u below 5 ones, 5 with all of them + assert binary.Trap(3, 5)(bits("111110000000100")) == 5.0 + 4.0 + 3.0 + # a = 6, b = 10, z = 2: 6, 3, 0, 5, 10 + trap = binary.Trap(5, 4, a=6, b=10, z=2) + assert trap(bits("0000" "1000" "1100" "1110" "1111")) == 24.0 + assert trap.optimum.value == 50.0 + first_two = bits("1" * 16 + "0" * 48) + assert binary.RoyalRoad.r1()(first_two) == 16.0 + assert binary.RoyalRoad.r2()(first_two) == 32.0 + assert binary.RoyalRoad.r1().optimum.value == 64.0 + assert binary.RoyalRoad.r2().optimum.value == 256.0 + assert binary.RoyalRoad.r2().name == "RoyalRoadR2" + assert binary.RoyalRoad(16, 3, hierarchical=True).optimum.value == 48.0 * 5 + + +def test_an_nk_landscape_is_the_same_as_in_rust(): + # the values of the Rust tests, to the bit + landscape = binary.NkLandscape(10, 3, "random", seed=42) + assert landscape.neighbors[0].tolist() == [2, 6, 7] + assert landscape.neighbors[9].tolist() == [4, 5, 8] + assert landscape.contributions.shape == (10, 16) + assert landscape.contributions[0, 0].hex() == "0x1.208b86307d686p-2" + assert landscape.contributions[9, 15].hex() == "0x1.bfb4ddb16b0c9p-1" + assert landscape(bits("1100101101")).hex() == "0x1.5562443ce9d63p-2" + # the mean of the contributions + x = bits("1100101101") + total = 0.0 + for site in range(10): + index = int(x[site]) + for neighbor in landscape.neighbors[site]: + index = index << 1 | int(x[neighbor]) + total += landscape.contributions[site, index] + assert landscape(x) == pytest.approx(total / 10, abs=1e-15) + adjacent = binary.NkLandscape(8, 4, "adjacent") + assert adjacent.neighbors[0].tolist() == [6, 7, 1, 2] + + +def test_the_optimum_of_an_nk_landscape_is_the_best_string(): + landscape = binary.NkLandscape(10, 3, "adjacent", seed=3) + strings = np.array([[(s >> i) & 1 for i in range(10)] for s in range(1 << 10)]) + assert landscape.optimum.value == landscape.evaluate(strings).max() + assert binary.NkLandscape(200, 4, "random").optimum is None + assert binary.NkLandscape(200, 4, "adjacent").optimum.proven + + +def test_a_knapsack_is_the_same_as_in_rust(): + knapsack = binary.Knapsack(5, seed=1) + assert knapsack.weights.tolist() == [613, 858, 688, 724, 989] + assert knapsack.profits.tolist() == [705, 45, 998, 22, 438] + assert knapsack.capacity == 1916 + circle = binary.Knapsack(3, "circle", seed=2) + assert circle.weights.tolist() == [898, 956, 782] + assert circle.profits.tolist() == [1112, 1137, 1057] + assert circle.constraint_count == 1 + + +def test_the_knapsack_classes(): + for kind, holds in [ + ("strongly_correlated", lambda w, p: p == w + 100), + ("inverse_strongly_correlated", lambda w, p: w == p + 100), + ("subset_sum", lambda w, p: p == w), + ("profit_ceiling", lambda w, p: p % 3 == 0 and w <= p < w + 3), + (binary.ProfitCeiling(5), lambda w, p: p % 5 == 0 and w <= p < w + 5), + ("multiple_strongly_correlated", lambda w, p: p == w + (300 if w % 6 == 0 else 200)), + ( + binary.MultipleStronglyCorrelated(10, 20, 4), + lambda w, p: p == w + (10 if w % 4 == 0 else 20), + ), + ("uncorrelated_similar_weights", lambda w, p: 100_000 <= w <= 100_100 and 1 <= p <= 1000), + ]: + knapsack = binary.Knapsack(200, kind, seed=4) + for w, p in zip(knapsack.weights.tolist(), knapsack.profits.tolist()): + assert holds(w, p), (kind, w, p) + spanner = binary.Knapsack(100, binary.Spanner(2, 10, "strongly_correlated"), seed=5) + reduced = { + (w // np.gcd(w, p), p // np.gcd(w, p)) + for w, p in zip(spanner.weights.tolist(), spanner.profits.tolist()) + } + assert len(reduced) <= 2 + # eq. 5: instance 30 of 100 + knapsack = binary.Knapsack(50, "weakly_correlated", data_range=10_000, instance=30, seed=6) + assert knapsack.capacity == knapsack.weights.sum() * 30 // 101 + assert knapsack.weights.max() <= 10_000 + + +def test_the_knapsack_fitness_is_the_profit_and_the_excess_weight(): + knapsack = binary.KnapsackItems([5, 4, 3], [10, 40, 30], 7) + assert knapsack(bits("011")) == (70.0, 0.0) + assert knapsack(bits("111")) == (80.0, 5.0) + scores, violations = knapsack.evaluate(np.array([[0, 1, 1], [1, 1, 1]])) + assert scores.tolist() == [70.0, 80.0] and violations.tolist() == [0.0, 5.0] + assert knapsack.constraints(bits("111")).tolist() == [5.0] + assert knapsack.optimum.value == 70.0 + assert knapsack.weights.tolist() == [5, 4, 3] + assert knapsack.capacity == 7 + + +def test_a_native_run_equals_a_run_with_python_calls(): + for problem in (binary.Trap(5, 4), binary.Knapsack(20, seed=1), binary.NkLandscape(12, 2)): + ga = gx.Ga( + problem.genome, + population_size=40, + select=gx.Tournament(3), + crossover=gx.UniformCrossover(), + mutation=gx.BitFlip(rate=1 / problem.dimensions), + seed=3, + ) + native = ga.run(problem, generations=20) + python = ga.run(lambda x, problem=problem: problem(x), generations=20) + assert native.best_fitness == python.best_fitness + assert np.array_equal(native.best_genome, python.best_genome) + + +def test_the_examples_methods_reach_the_optimum(): + # the (1+1) evolutionary algorithm on LeadingOnes, and random-mutation hill climbing on R1 + leading_ones = binary.LeadingOnes(30) + search = gx.LocalSearch(leading_ones.genome, neighbor=gx.BitFlip(rate=1 / 30), seed=1) + result = search.run(leading_ones, target=30, evaluations=100_000) + assert result.best_fitness == 30 + r1 = binary.RoyalRoad.r1() + search = gx.LocalSearch(r1.genome, neighbor=gx.BitFlip(count=1), seed=1) + assert search.run(r1, target=64, evaluations=256_000).best_fitness == 64 + + +@pytest.mark.parametrize( + "make, message", + [ + (lambda: binary.OneMax(0), "OneMax.bits is at least 1"), + (lambda: binary.Trap(1, 1), "Trap.k is at least 2"), + (lambda: binary.Trap(1, 4, z=4), "must be between 1 and k − 1"), + (lambda: binary.Trap(1, 4, a=5), "0 ≤ a < b"), + (lambda: binary.RoyalRoad(6, 8, hierarchical=True), "a power of 2 blocks, not 6"), + (lambda: binary.RoyalRoad(8, 8, hierarchical=1), "True or False"), + (lambda: binary.NkLandscape(5, 5), "must be below n = 5"), + (lambda: binary.NkLandscape(100, 30), r"values, more than 2\^24"), + (lambda: binary.NkLandscape(5, 2, "ring"), '"adjacent" or "random"'), + (lambda: binary.Knapsack(0), "Knapsack.items is at least 1"), + (lambda: binary.Knapsack(5, "nope"), "no class 'nope'"), + (lambda: binary.Knapsack(5, instance=101), "instance"), + (lambda: binary.Knapsack(5, binary.Spanner(0)), "Spanner.v is at least 1"), + (lambda: binary.Knapsack(5, binary.Spanner(distribution="x")), "Spanner.distribution"), + (lambda: binary.Knapsack(5, binary.Circle(2, 0)), "Circle.denominator is at least 1"), + (lambda: binary.KnapsackItems([1, 2], [1], 3), "must have a profit per item"), + (lambda: binary.KnapsackItems([1, -2], [1, 1], 3), "at least 0"), + ], +) +def test_wrong_settings_are_errors(make, message): + with pytest.raises(ValueError, match=message): + make().name + + +def test_a_binary_problem_needs_its_genome_and_objective(): + problem = binary.OneMax(10) + with pytest.raises(ValueError, match="OneMax maximizes its objective"): + gx.LocalSearch( + problem.genome, neighbor=gx.BitFlip(count=1), objective="minimize", seed=1 + ).run(problem, generations=1) + with pytest.raises(ValueError, match="OneMax has 10 bits, but the genome has 12"): + gx.LocalSearch(gx.Binary(12), neighbor=gx.BitFlip(count=1), seed=1).run( + problem, generations=1 + ) + with pytest.raises(ValueError, match="OneMax needs a Binary genome"): + gx.De(gx.Real((0, 1), length=10), seed=1).run(problem, generations=1) + with pytest.raises(ValueError, match="OneMax takes genomes of 10 bits, not 3"): + problem([1, 0, 1]) + with pytest.raises(ValueError, match="OneMax takes bits, 0 or 1, as genes, not 0.5"): + problem([0.5] * 10) + with pytest.raises(ValueError, match="one objective, and an optimum instead of a front"): + gx._genoxide.optimal_front(problem._json(), 10) diff --git a/python/tests/test_docs.py b/python/tests/test_docs.py index 1bb364a1..ce414580 100644 --- a/python/tests/test_docs.py +++ b/python/tests/test_docs.py @@ -41,7 +41,8 @@ def test_the_module_docstring_example_runs(): def test_the_submodule_docstring_examples_run(): - modules = (gx.problems, gx.problems.cec2006, gx.problems.engineering, gx.neat) + modules = (gx.problems, gx.problems.binary, gx.problems.cec2006, gx.problems.engineering) + modules += (gx.neat,) modules += (gx.gp, gx.gp.regression, gx.gp.regression.problems, gx.gp.boolean) for module in modules + (gx.problems.multi_engineering, gx.indicators, gx.model.gp): # later examples use what earlier ones define diff --git a/site/lib/projects/genoxide/examples.js b/site/lib/projects/genoxide/examples.js index 6bf3668a..98783937 100644 --- a/site/lib/projects/genoxide/examples.js +++ b/site/lib/projects/genoxide/examples.js @@ -263,6 +263,8 @@ const FAMILY_SUMMARIES = { Penalized: "Yao, Liu and Lin's two penalized functions: a grid of shallow wells over the box, and a steep wall near its bounds.", Schaffer: "One variable and two objectives, from the paper of the first multi-objective genetic algorithm, VEGA.", Viennet: "Three objectives of two variables, with curved and split Pareto fronts.", + "Royal road": + "Blocks of ones that score only when complete, alone (R1) or with their pairs, quadruples and whole (R2): plateaus that a genetic algorithm was meant to cross by crossover.", "N-Queens": "N queens on an N×N chessboard with no two in a row, a column or a diagonal: the usual 8×8 board and larger ones, solved by the same search.", }; diff --git a/src/problems.rs b/src/problems.rs index 0d4f1b7b..8de6300d 100644 --- a/src/problems.rs +++ b/src/problems.rs @@ -22,7 +22,7 @@ //! //! # The problems //! -//! All are minimized. The classic functions are unconstrained, on [`Real`] genomes. `n` is the +//! The classic functions are minimized and unconstrained, on [`Real`] genomes. `n` is the //! number of dimensions. //! //! | Problem | n (default) | Bounds | Minimum | @@ -128,6 +128,12 @@ //! [`evaluate_with`](FitnessFunction::evaluate_with) writes them. Their gradients aren't given: //! [`Constrained::differentiable`](crate::constraint::Constrained::differentiable) adds them. //! +//! [`binary`] holds problems of bit strings, on [`Binary`](crate::genome::Binary) genomes and +//! maximized: OneMax, LeadingOnes, the deceptive trap, the royal roads R1 and R2, NK landscapes +//! and the 0/1 knapsack with Pisinger's generated instance classes. The NK landscapes and the +//! knapsack instances are drawn from a seed, and their optimum is computed exactly, by dynamic +//! programming or exhaustive search. +//! //! [`control`] holds control tasks instead of functions: the cart-pole and the double pole, with //! and without velocities, driven by a [`Policy`](control::Policy) such as a neural network of //! [`nn`](crate::nn), for neuroevolution. @@ -174,6 +180,7 @@ //! //! [`Engine`]: crate::Engine +pub mod binary; pub mod cec2006; mod classic; pub mod control; diff --git a/src/problems/binary.rs b/src/problems/binary.rs new file mode 100644 index 00000000..595e911a --- /dev/null +++ b/src/problems/binary.rs @@ -0,0 +1,989 @@ +//! Binary and combinatorial test problems: functions of bit strings, on [`Binary`] genomes, all +//! maximized. +//! +//! | Problem | Bits (default) | Maximum | +//! |---|---|---| +//! | [`OneMax`] | any (100) | n, at all ones | +//! | [`LeadingOnes`] | any (100) | n, at all ones | +//! | [`Trap`] | blocks × k (10 × 4) | blocks × b, at all ones; the deceptive attractor is all zeros | +//! | [`RoyalRoad`] | blocks × block size (8 × 8) | 64 for R1, 256 for R2, at all ones | +//! | [`NkLandscape`] | n | instance-specific: exhaustive search or dynamic programming | +//! | [`Knapsack`] | items | instance-specific: dynamic programming | +//! +//! Each is a [`FitnessFunction`] of [`Bits`] for an [`Engine`](crate::Engine) as it is, and a +//! [`Problem`] whose [`objective`](Problem::objective) is +//! [`Maximize`](crate::Objective::Maximize), the default of the algorithms: +//! +//! ``` +//! use genoxide::prelude::*; +//! use genoxide::problems::Problem; +//! use genoxide::problems::binary::LeadingOnes; +//! +//! // the (1+1) evolutionary algorithm of Droste, Jansen and Wegener: one string, each bit +//! // flipped with probability 1/n, the child kept if it's no worse +//! let problem = LeadingOnes::new(50); +//! let one_plus_one = LocalSearch::builder(problem.representation()) +//! .neighbor(BitFlip::per_gene(1.0 / 50.0)?) +//! .seed(1) +//! .build()?; +//! let optimum = problem.optimum().expect("known").value(); +//! let outcome = Engine::new(one_plus_one, problem) +//! .stop_when(Stop::target(optimum).or(Stop::evaluations(100_000))) +//! .run()?; +//! assert_eq!(outcome.best_fitness(), Fitness::new(50.0)); +//! # Ok::<(), genoxide::Error>(()) +//! ``` +//! +//! [`NkLandscape`] and [`Knapsack`] are instances generated from a seed with genoxide's portable +//! [`StreamRng`](crate::StreamRng): the same seed gives the same landscape or the same items on +//! every platform. Their optimum isn't known in closed form, and +//! [`optimum`](Problem::optimum) computes it, exactly: by dynamic programming (NK landscapes with +//! adjacent neighborhoods, and the knapsack), or by evaluating every bit string (small NK +//! landscapes). +//! +//! These problems aren't in [`all`](super::all), which lists the problems on real genomes. +//! +//! The definitions are taken from these papers, read for them: +//! +//! - Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary +//! algorithm. *Theoretical Computer Science* 276(1-2): 51-81. +//! doi:10.1016/S0304-3975(01)00182-7 (OneMax, Definition 9; LeadingOnes, Definition 16) +//! - Ackley, D. H. (1987). *A Connectionist Machine for Genetic Hillclimbing.* Kluwer Academic +//! Publishers. doi:10.1007/978-1-4613-1997-9 ("One Max" and "Trap", section 3.3) +//! - Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. In *Foundations +//! of Genetic Algorithms 2*, Morgan Kaufmann: 93-108. doi:10.1016/B978-0-08-094832-4.50012-X +//! - Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms: +//! fitness landscapes and GA performance. *Proceedings of the First European Conference on +//! Artificial Life*, MIT Press: 245-254 (R2, Figure 1) +//! - Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a genetic algorithm +//! outperform hill climbing? *Advances in Neural Information Processing Systems 6*, Morgan +//! Kaufmann: 51-58 (R1, Figure 1) +//! - Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and +//! its application to maturation of the immune response. *Journal of Theoretical Biology* +//! 141(2): 211-245. doi:10.1016/S0022-5193(89)80019-0 +//! - Pisinger, D. (2005). Where are the hard knapsack problems? *Computers & Operations Research* +//! 32(9): 2271-2284. doi:10.1016/j.cor.2004.03.002 + +mod knapsack; +mod nk; + +pub use knapsack::{Knapsack, KnapsackClass, KnapsackGenerator, SpannerDistribution}; +pub use nk::{Neighborhood, NkLandscape}; + +use super::{Optimum, Problem}; +use crate::Objective; +use crate::engine::FitnessFunction; +use crate::genome::{Binary, Bits}; + +// the most bits of a genome, as `Binary::new` accepts +const MAX_BITS: usize = 1 << 24; + +const DROSTE: &str = "Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the \ + (1+1) evolutionary algorithm. Theoretical Computer Science 276(1-2): 51-81."; +const DROSTE_URL: &str = "https://doi.org/10.1016/S0304-3975(01)00182-7"; + +// a genome length checked by the constructor +fn binary(bits: usize) -> Binary { + Binary::new(bits).expect("checked by the constructor") +} + +// the genome length of a problem, or a panic that names it +fn check_length(name: &str, x: &Bits, bits: usize) { + assert_eq!(x.len(), bits, "{name} takes genomes of {bits} bits"); +} + +// a number of bits from 1 to 2^24, or a panic that names the problem +fn check_bits(name: &str, bits: usize) -> usize { + assert!( + (1..=MAX_BITS).contains(&bits), + "{name} takes 1 to 2^24 bits, not {bits}" + ); + bits +} + +// the number of ones among the `len` bits of `x` from `start` +fn ones_in(x: &Bits, start: usize, len: usize) -> usize { + let words = x.as_words(); + let (mut count, mut at, end) = (0, start, start + len); + while at < end { + let (word, offset) = (at / 64, at % 64); + let take = (64 - offset).min(end - at); + let mask = if take == 64 { + u64::MAX + } else { + ((1u64 << take) - 1) << offset + }; + count += (words[word] & mask).count_ones() as usize; + at += take; + } + count +} + +// ---- OneMax ---------------------------------------------------------------------------------- + +/// OneMax: the number of ones, `Σ xᵢ`, the simplest function of bit strings. +/// +/// Each bit counts on its own, so every string but the optimum has a neighbor one flip away that +/// is better: a hill climber climbs straight to the top. It's the baseline of the binary problems. +/// The (1+1) evolutionary algorithm (one string, each bit flipped with probability 1/n, the child +/// kept if it's no worse) needs Θ(n log n) evaluations on average (Droste et al., 2002, Lemma 10, +/// for every linear function with nonzero weights). +/// +/// Maximum n at all ones; 100 bits by default. +/// +/// Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary +/// algorithm. *Theoretical Computer Science* 276(1-2): 51-81, Definition 9, where it's the linear +/// function with all weights 1. The function has no single origin; Ackley (1987, *A Connectionist +/// Machine for Genetic Hillclimbing*, section 3.3.1) tests a "One Max" that is ten times the +/// number of ones. Both read; Mühlenbein's (1992) analysis, which Droste et al. cite, wasn't. +/// +/// ``` +/// use genoxide::genome::Bits; +/// use genoxide::problems::binary::OneMax; +/// use genoxide::prelude::*; +/// +/// let x: Bits = [true, false, true, true].into_iter().collect(); +/// assert_eq!(OneMax::new(4).evaluate(&x), 3.0); +/// ``` +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct OneMax { + bits: usize, +} + +impl OneMax { + /// OneMax of `bits` bits. + /// + /// # Panics + /// + /// If `bits` is 0 or above 2^24. + pub fn new(bits: usize) -> Self { + Self { + bits: check_bits("OneMax", bits), + } + } + + /// The number of bits. + pub fn bits(&self) -> usize { + self.bits + } +} + +impl Default for OneMax { + /// OneMax of 100 bits. + fn default() -> Self { + Self::new(100) + } +} + +impl FitnessFunction for OneMax { + type Output = f64; + + /// The number of ones of `x`. + /// + /// # Panics + /// + /// If `x` doesn't have [`bits`](OneMax::bits) bits. + fn evaluate(&self, x: &Bits) -> f64 { + check_length("OneMax", x, self.bits); + x.count_ones() as f64 + } +} + +impl Problem for OneMax { + type Representation = Binary; + + fn name(&self) -> &'static str { + "OneMax" + } + + fn representation(&self) -> Binary { + binary(self.bits) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + fn optimum(&self) -> Option> { + Some(Optimum::proven( + self.bits as f64, + vec![Bits::ones(self.bits)], + )) + } + + fn reference(&self) -> &'static str { + DROSTE + } + + fn reference_url(&self) -> Option<&'static str> { + Some(DROSTE_URL) + } +} + +// ---- LeadingOnes ----------------------------------------------------------------------------- + +/// LeadingOnes: the number of ones before the first zero, `Σᵢ Πⱼ≤ᵢ xⱼ`. +/// +/// Unimodal, since appending a one to the leading ones always improves a string, but only one +/// bit at a time can: the first zero. The bits after it don't count until the leading ones reach +/// them. The (1+1) evolutionary algorithm (one string, each bit flipped with probability 1/n, the +/// child kept if it's no worse) needs Θ(n²) evaluations on average, and at most e n² (Droste et +/// al., 2002, Theorem 17), where OneMax needs Θ(n log n): Droste et al. give it to disprove the +/// remark, which they attribute to Mühlenbein, that every unimodal function takes O(n log n). +/// +/// Maximum n at all ones; 100 bits by default. +/// +/// Droste, S., Jansen, T. and Wegener, I. (2002). On the analysis of the (1+1) evolutionary +/// algorithm. *Theoretical Computer Science* 276(1-2): 51-81, Definition 16 and Theorem 17. They +/// take the function from Rudolph, G. (1997). *Convergence Properties of Evolutionary +/// Algorithms*, Kovač, Hamburg, who proved the O(n²) upper bound; Rudolph's book wasn't +/// available to check. +/// +/// ``` +/// use genoxide::genome::Bits; +/// use genoxide::problems::binary::LeadingOnes; +/// use genoxide::prelude::*; +/// +/// let x: Bits = [true, true, false, true].into_iter().collect(); +/// assert_eq!(LeadingOnes::new(4).evaluate(&x), 2.0); +/// ``` +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct LeadingOnes { + bits: usize, +} + +impl LeadingOnes { + /// LeadingOnes of `bits` bits. + /// + /// # Panics + /// + /// If `bits` is 0 or above 2^24. + pub fn new(bits: usize) -> Self { + Self { + bits: check_bits("LeadingOnes", bits), + } + } + + /// The number of bits. + pub fn bits(&self) -> usize { + self.bits + } +} + +impl Default for LeadingOnes { + /// LeadingOnes of 100 bits. + fn default() -> Self { + Self::new(100) + } +} + +impl FitnessFunction for LeadingOnes { + type Output = f64; + + /// The number of leading ones of `x`. + /// + /// # Panics + /// + /// If `x` doesn't have [`bits`](LeadingOnes::bits) bits. + fn evaluate(&self, x: &Bits) -> f64 { + check_length("LeadingOnes", x, self.bits); + let mut count = 0; + for &word in x.as_words() { + // bit i is bit i % 64 of word i / 64: the leading ones are a word's trailing ones + count += word.trailing_ones() as usize; + if word != u64::MAX { + break; + } + } + // the unused bits of the last word are zero + count.min(self.bits) as f64 + } +} + +impl Problem for LeadingOnes { + type Representation = Binary; + + fn name(&self) -> &'static str { + "LeadingOnes" + } + + fn representation(&self) -> Binary { + binary(self.bits) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + fn optimum(&self) -> Option> { + Some(Optimum::proven( + self.bits as f64, + vec![Bits::ones(self.bits)], + )) + } + + fn reference(&self) -> &'static str { + DROSTE + } + + fn reference_url(&self) -> Option<&'static str> { + Some(DROSTE_URL) + } +} + +// ---- deceptive trap -------------------------------------------------------------------------- + +/// The deceptive trap: blocks of k bits, each scored by a trap function of its number of ones, +/// that leads away from the optimum. +/// +/// A block with u ones scores +/// +/// ```text +/// f(u) = a (z − u) / z for u ≤ z, +/// b (u − z) / (k − z) otherwise, +/// ``` +/// +/// and the string scores the sum of its blocks, which are consecutive: bits 0 to k − 1, k to +/// 2k − 1 and so on. f falls from a at u = 0, the deceptive attractor, to 0 at the slope change +/// z, and rises to b > a at u = k, the optimum. Only a block with more than z ones leads up to +/// it; from fewer than z, every step up leads to all zeros. +/// +/// [`Trap::new`] takes the values most used since, a = k − 1, b = k and z = k − 1: a block scores +/// k − 1 − u below k ones, and k with all of them. For k ≥ 3, every schema of order below k +/// within a block then favors all zeros: the block is *fully deceptive* (Deb and Goldberg's +/// Theorem 1 and their inequality 16, r = a/b ≥ (2 − 1/(k − z)) / (2 − 1/z), which here is +/// (k − 1)/k ≥ (k − 1)/(2k − 3), their limiting ratio, eq. 20). [`Trap::with_values`] takes any +/// a, b and z: Ackley's 1987 trap of n bits, for one, is +/// `Trap::with_values(1, n, 8n, 10n, ⌊3n/4⌋)`. Deb and Goldberg find that none of Ackley's (of 8, +/// 12, 16 and 20 bits) is fully deceptive, and write that his traps are "fully deceptive only +/// for ℓ < 7"; by their inequality 16, and by enumerating the schemas, only those of 3 and 4 bits +/// are. +/// +/// Maximum blocks × b at all ones; the deceptive attractor, all zeros, scores blocks × a; 10 +/// blocks of 4 bits by default. +/// +/// Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. In *Foundations of +/// Genetic Algorithms 2*, Morgan Kaufmann: 93-108, equation 1, the function of a block, after +/// Ackley, D. H. (1987). *A Connectionist Machine for Genetic Hillclimbing.* Kluwer Academic +/// Publishers, section 3.3.3 ("Trap"). The paper analyzes one block of ℓ bits; the sum over +/// consecutive blocks, the form of test suites built from deceptive subfunctions, isn't written +/// out in it. +/// +/// ``` +/// use genoxide::genome::Bits; +/// use genoxide::problems::binary::Trap; +/// use genoxide::prelude::*; +/// +/// let trap = Trap::new(2, 4); +/// // a block of 4 ones scores 4; one of a single one, 2 +/// let x: Bits = "11110100".chars().map(|bit| bit == '1').collect(); +/// assert_eq!(trap.evaluate(&x), 6.0); +/// ``` +#[derive(Clone, Copy, Debug, PartialEq)] +pub struct Trap { + blocks: usize, + k: usize, + a: f64, + b: f64, + z: usize, +} + +impl Trap { + /// `blocks` blocks of `k` bits, with a = k − 1, b = k and z = k − 1: fully deceptive for + /// k ≥ 3. + /// + /// # Panics + /// + /// If `blocks` is 0, `k` is below 2, or the genome would have more than 2^24 bits. + pub fn new(blocks: usize, k: usize) -> Self { + assert!(blocks >= 1, "Trap needs at least 1 block"); + assert!(k >= 2, "Trap needs blocks of at least 2 bits, not {k}"); + let bits = blocks.saturating_mul(k); + check_bits("Trap", bits); + Self { + blocks, + k, + a: (k - 1) as f64, + b: k as f64, + z: k - 1, + } + } + + /// `blocks` blocks of `k` bits, with the values `a` at no ones and `b` at k ones and the + /// slope change `z`. + /// + /// # Errors + /// + /// [`Error::InvalidSetting`](crate::Error::InvalidSetting) if `blocks` is 0, `k` is below 2, + /// the genome would have more than 2^24 bits, `z` isn't between 1 and k − 1, or `a` and `b` + /// aren't finite with 0 ≤ a < b. + /// + /// ``` + /// use genoxide::problems::binary::Trap; + /// + /// // Ackley's trap of 20 bits: a local maximum 8n = 160 at all zeros, the global 10n = 200 + /// let ackley = Trap::with_values(1, 20, 160.0, 200.0, 15)?; + /// assert!(Trap::with_values(1, 20, 200.0, 160.0, 15).is_err()); + /// # Ok::<(), genoxide::Error>(()) + /// ``` + pub fn with_values(blocks: usize, k: usize, a: f64, b: f64, z: usize) -> crate::Result { + let invalid = |setting: &'static str, reason: String| { + Err(crate::Error::InvalidSetting { setting, reason }) + }; + if blocks == 0 { + return invalid("blocks", "a trap needs at least 1 block".to_string()); + } + if k < 2 { + return invalid("k", format!("a block has at least 2 bits, not {k}")); + } + if blocks.checked_mul(k).is_none_or(|bits| bits > MAX_BITS) { + return invalid( + "k", + format!("{blocks} blocks of {k} bits are more than 2^24 bits"), + ); + } + if !(1..k).contains(&z) { + return invalid( + "z", + format!("must be between 1 and k − 1 = {}, got {z}", k - 1), + ); + } + if !(a.is_finite() && b.is_finite() && 0.0 <= a && a < b) { + return invalid( + "a", + format!("a and b must be finite with 0 ≤ a < b, got a = {a} and b = {b}"), + ); + } + Ok(Self { blocks, k, a, b, z }) + } + + /// The number of blocks. + pub fn blocks(&self) -> usize { + self.blocks + } + + /// The number of bits of a block. + pub fn k(&self) -> usize { + self.k + } + + /// The value of a block with no ones, the deceptive attractor. + pub fn a(&self) -> f64 { + self.a + } + + /// The value of a block of ones, the optimum. + pub fn b(&self) -> f64 { + self.b + } + + /// The slope change: the number of ones at which a block scores 0. + pub fn z(&self) -> usize { + self.z + } + + /// The number of bits: blocks × k. + pub fn bits(&self) -> usize { + self.blocks * self.k + } + + /// The value of a block with `ones` ones, f(u). + /// + /// ``` + /// use genoxide::problems::binary::Trap; + /// + /// let trap = Trap::new(1, 4); + /// let values: Vec = (0..=4).map(|u| trap.block(u)).collect(); + /// assert_eq!(values, [3.0, 2.0, 1.0, 0.0, 4.0]); + /// ``` + /// + /// # Panics + /// + /// If `ones` is above k. + pub fn block(&self, ones: usize) -> f64 { + assert!(ones <= self.k, "a block has at most {} ones", self.k); + if ones <= self.z { + self.a * (self.z - ones) as f64 / self.z as f64 + } else { + self.b * (ones - self.z) as f64 / (self.k - self.z) as f64 + } + } +} + +impl Default for Trap { + /// 10 blocks of 4 bits, with a = 3, b = 4 and z = 3: 40 bits, maximum 40. + fn default() -> Self { + Self::new(10, 4) + } +} + +impl FitnessFunction for Trap { + type Output = f64; + + /// The sum of the values of the blocks of `x`, in their order. + /// + /// # Panics + /// + /// If `x` doesn't have [`bits`](Trap::bits) bits. + fn evaluate(&self, x: &Bits) -> f64 { + check_length("Trap", x, self.bits()); + (0..self.blocks) + .map(|block| self.block(ones_in(x, block * self.k, self.k))) + .sum() + } +} + +impl Problem for Trap { + type Representation = Binary; + + fn name(&self) -> &'static str { + "Trap" + } + + fn representation(&self) -> Binary { + binary(self.bits()) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + fn optimum(&self) -> Option> { + let ones = Bits::ones(self.bits()); + Some(Optimum::proven(self.evaluate(&ones), vec![ones])) + } + + fn reference(&self) -> &'static str { + "Deb, K. and Goldberg, D. E. (1993). Analyzing deception in trap functions. In Foundations \ + of Genetic Algorithms 2, Morgan Kaufmann: 93-108." + } + + fn reference_url(&self) -> Option<&'static str> { + Some("https://doi.org/10.1016/B978-0-08-094832-4.50012-X") + } +} + +// ---- royal road ------------------------------------------------------------------------------ + +/// The royal road functions R1 and R2: blocks of ones that score only when complete, and, in R2, +/// pairs, quadruples and so on of complete blocks that score again. +/// +/// The function is a sum over schemas s, `Σ c_s σ_s(x)`, with `σ_s(x)` 1 if x is an instance of +/// s (has ones wherever s defines a bit) and `c_s = order(s)`, its number of defined bits. +/// +/// - **R1** ([`RoyalRoad::r1`]): 8 schemas, each a block of 8 consecutive ones (bits 0 to 7, 8 +/// to 15, …): a string scores 8 per complete block, 64 at most (Mitchell, Holland and Forrest, +/// 1994, Figure 1). +/// - **R2** ([`RoyalRoad::r2`]): R1's 8 schemas and the 7 above them: the 4 blocks of 16 ones +/// (c = 16), the 2 of 32 (c = 32) and all 64 ones (c = 64), so a string of ones scores +/// 8 · 8 + 4 · 16 + 2 · 32 + 64 = 256 (Mitchell, Forrest and Holland, 1992, Figure 1). +/// +/// [`RoyalRoad::new`] and [`RoyalRoad::hierarchical`] make the same functions with any number of +/// blocks of any size. The functions were meant as a "royal road" for a genetic algorithm, +/// whose crossover would combine complete blocks into larger ones; but every string short of a +/// complete block scores alike, so the search drifts on plateaus. Mitchell, Holland and Forrest +/// found that random-mutation hill climbing (one bit flipped at a time, the flip kept if it's no +/// worse) solved R1 in 6,179 evaluations on average where their genetic algorithm took 61,334 +/// (Table 1). +/// +/// Maximum at all ones: blocks × block size for R1; blocks × block size × (log₂ blocks + 1) for +/// R2. R1 by default. +/// +/// Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic algorithms: +/// fitness landscapes and GA performance. *Proceedings of the First European Conference on +/// Artificial Life*, MIT Press: 245-254, Figure 1, the function with 15 schemas, which Forrest +/// and Mitchell (1993) later named R2; and Mitchell, M., Holland, J. H. and Forrest, S. (1994). +/// When will a genetic algorithm outperform hill climbing? *Advances in Neural Information +/// Processing Systems 6*, Morgan Kaufmann: 51-58, Figure 1, R1. Forrest and Mitchell (1993, +/// FOGA 2) wasn't available. +/// +/// ``` +/// use genoxide::genome::Bits; +/// use genoxide::problems::binary::RoyalRoad; +/// use genoxide::prelude::*; +/// +/// // the first two blocks complete: 16 in R1, and 8 + 8 + 16 = 32 in R2 +/// let x: Bits = (0..64).map(|i| i < 16).collect(); +/// assert_eq!(RoyalRoad::r1().evaluate(&x), 16.0); +/// assert_eq!(RoyalRoad::r2().evaluate(&x), 32.0); +/// ``` +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct RoyalRoad { + blocks: usize, + block_size: usize, + hierarchical: bool, +} + +impl RoyalRoad { + /// R1's form: `blocks` blocks of `block_size` ones, each scoring `block_size` when + /// complete. + /// + /// # Panics + /// + /// If `blocks` or `block_size` is 0, or the genome would have more than 2^24 bits. + pub fn new(blocks: usize, block_size: usize) -> Self { + Self::checked(blocks, block_size, false) + } + + /// R2's form: `blocks` blocks of `block_size` ones, and every level of complete pairs, + /// quadruples and so on up to all the blocks, each schema scoring its number of bits. + /// + /// # Panics + /// + /// If `blocks` isn't a power of 2, `block_size` is 0, or the genome would have more than + /// 2^24 bits. + pub fn hierarchical(blocks: usize, block_size: usize) -> Self { + assert!( + blocks.is_power_of_two(), + "a hierarchical royal road needs a power of 2 blocks, not {blocks}" + ); + Self::checked(blocks, block_size, true) + } + + /// R1: 8 blocks of 8 bits, maximum 64. + pub fn r1() -> Self { + Self::new(8, 8) + } + + /// R2: 8 blocks of 8 bits and the levels of 16, 32 and 64 bits above them, maximum 256. + pub fn r2() -> Self { + Self::hierarchical(8, 8) + } + + fn checked(blocks: usize, block_size: usize, hierarchical: bool) -> Self { + assert!(blocks >= 1, "a royal road needs at least 1 block"); + assert!(block_size >= 1, "a royal road's blocks have at least 1 bit"); + let bits = blocks.saturating_mul(block_size); + check_bits("RoyalRoad", bits); + Self { + blocks, + block_size, + hierarchical, + } + } + + /// The number of blocks at the lowest level. + pub fn blocks(&self) -> usize { + self.blocks + } + + /// The number of bits of a block at the lowest level. + pub fn block_size(&self) -> usize { + self.block_size + } + + /// Whether the function has R2's levels above the blocks. + pub fn is_hierarchical(&self) -> bool { + self.hierarchical + } + + /// The number of bits: blocks × block size. + pub fn bits(&self) -> usize { + self.blocks * self.block_size + } +} + +impl Default for RoyalRoad { + /// R1. + fn default() -> Self { + Self::r1() + } +} + +impl FitnessFunction for RoyalRoad { + type Output = f64; + + /// The sum of the orders of the schemas `x` is an instance of. + /// + /// # Panics + /// + /// If `x` doesn't have [`bits`](RoyalRoad::bits) bits. + fn evaluate(&self, x: &Bits) -> f64 { + check_length("RoyalRoad", x, self.bits()); + let mut total = 0; + // the blocks, then (in R2) the pairs, quadruples and so on: a schema of `len` bits scores + // `len` when they're all ones + let mut len = self.block_size; + loop { + total += (0..self.bits() / len) + .filter(|&schema| ones_in(x, schema * len, len) == len) + .count() + * len; + if !self.hierarchical || len == self.bits() { + break; + } + len *= 2; + } + total as f64 + } +} + +impl Problem for RoyalRoad { + type Representation = Binary; + + fn name(&self) -> &'static str { + if self.hierarchical { + "RoyalRoadR2" + } else { + "RoyalRoadR1" + } + } + + fn representation(&self) -> Binary { + binary(self.bits()) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + fn optimum(&self) -> Option> { + let ones = Bits::ones(self.bits()); + Some(Optimum::proven(self.evaluate(&ones), vec![ones])) + } + + fn reference(&self) -> &'static str { + if self.hierarchical { + "Mitchell, M., Forrest, S. and Holland, J. H. (1992). The royal road for genetic \ + algorithms: fitness landscapes and GA performance. Proceedings of the First European \ + Conference on Artificial Life, MIT Press: 245-254." + } else { + "Mitchell, M., Holland, J. H. and Forrest, S. (1994). When will a genetic algorithm \ + outperform hill climbing? Advances in Neural Information Processing Systems 6, \ + Morgan Kaufmann: 51-58." + } + } + + fn reference_url(&self) -> Option<&'static str> { + if self.hierarchical { + None + } else { + Some( + "https://proceedings.neurips.cc/paper_files/paper/1993/hash/\ + ab88b15733f543179858600245108dd8-Abstract.html", + ) + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::StreamRng; + use crate::genome::Representation; + + fn bits(text: &str) -> Bits { + text.chars().map(|bit| bit == '1').collect() + } + + #[test] + fn ones_are_counted_in_any_range() { + let mut rng = StreamRng::seed_from_u64(1); + let x = Binary::new(200).unwrap().random_genome(&mut rng); + for start in [0, 1, 63, 64, 65, 127, 130] { + for len in [0, 1, 5, 63, 64, 65, 70] { + let expected = (start..start + len) + .filter(|&i| x.get(i) == Some(true)) + .count(); + assert_eq!(ones_in(&x, start, len), expected, "{start} {len}"); + } + } + } + + #[test] + fn one_max_counts_ones() { + let problem = OneMax::new(5); + assert_eq!(problem.evaluate(&bits("10110")), 3.0); + assert_eq!(problem.evaluate(&bits("00000")), 0.0); + let optimum = problem.optimum().unwrap(); + assert_eq!(optimum.value(), 5.0); + assert_eq!(problem.evaluate(&optimum.solutions()[0]), 5.0); + assert_eq!(problem.objective(), Objective::Maximize); + assert_eq!(OneMax::default().bits(), 100); + } + + #[test] + fn leading_ones_stops_at_the_first_zero() { + let problem = LeadingOnes::new(6); + assert_eq!(problem.evaluate(&bits("110111")), 2.0); + assert_eq!(problem.evaluate(&bits("011111")), 0.0); + assert_eq!(problem.evaluate(&bits("111111")), 6.0); + // across words: Definition 16's sum of products, against the fast count + let mut rng = StreamRng::seed_from_u64(2); + for len in [1, 63, 64, 65, 128, 130] { + let problem = LeadingOnes::new(len); + for ones in [0, 1, len / 2, len - 1, len] { + let mut x = Binary::new(len).unwrap().random_genome(&mut rng); + for i in 0..ones { + x.set(i, true); + } + let definition: usize = (0..len) + .map(|i| { + (0..=i) + .map(|j| usize::from(x.get(j) == Some(true))) + .product::() + }) + .sum(); + assert_eq!(problem.evaluate(&x), definition as f64, "{len} {ones}"); + } + } + } + + #[test] + fn traps_follow_deb_and_goldberg() { + // k = 5: 4 − u below 5 ones, 5 with all of them + let trap = Trap::new(3, 5); + let values: Vec = (0..=5).map(|u| trap.block(u)).collect(); + assert_eq!(values, [4.0, 3.0, 2.0, 1.0, 0.0, 5.0]); + assert_eq!( + trap.evaluate(&bits("11111 00000 00100".replace(' ', "").as_str())), + 12.0 + ); + assert_eq!(trap.optimum().unwrap().value(), 15.0); + // equation 1 with other values: a = 6, b = 10, z = 2 for k = 4 + let trap = Trap::with_values(1, 4, 6.0, 10.0, 2).unwrap(); + let values: Vec = (0..=4).map(|u| trap.block(u)).collect(); + assert_eq!(values, [6.0, 3.0, 0.0, 5.0, 10.0]); + // Ackley's trap of 8 bits: z = 6, 8n (z − c)/z and 10n (c − z)/(n − z) + let ackley = Trap::with_values(1, 8, 64.0, 80.0, 6).unwrap(); + assert_eq!(ackley.block(0), 64.0); + assert_eq!(ackley.block(6), 0.0); + assert_eq!(ackley.block(7), 40.0); + assert_eq!(ackley.block(8), 80.0); + } + + // the mean value of a block over the strings that match a schema (its fixed bits), as Deb and + // Goldberg define a schema's fitness + fn schema_mean(trap: &Trap, fixed: u32, values: u32) -> f64 { + let k = trap.k(); + let mut sum = 0.0; + let mut count = 0.0; + for string in 0u32..1 << k { + if string & fixed == values { + sum += trap.block(string.count_ones() as usize); + count += 1.0; + } + } + sum / count + } + + // fully deceptive (Deb and Goldberg's definition): in every partition of order below k, the + // schema with zeros in its fixed bits is no worse than the others + fn fully_deceptive(trap: &Trap) -> bool { + let k = trap.k(); + (1u32..(1 << k) - 1).all(|fixed| { + let zeros = schema_mean(trap, fixed, 0); + (0u32..1 << k) + .filter(|values| values & !fixed == 0) + .all(|values| schema_mean(trap, fixed, values) <= zeros) + }) + } + + #[test] + fn the_default_traps_are_fully_deceptive_from_three_bits() { + for k in 3..=8 { + assert!(fully_deceptive(&Trap::new(1, k)), "k = {k}"); + // inequality 16 and its limiting ratio, eq. 20, at z = k − 1 + let (r, z) = ((k - 1) as f64 / k as f64, (k - 1) as f64); + let bound = (2.0 - 1.0 / (k as f64 - z)) / (2.0 - 1.0 / z); + assert!(r >= bound); + let limit = (k - 1) as f64 / (2 * k - 3) as f64; + assert!((bound - limit).abs() < 1e-15); + } + // k = 2 isn't: 1 − u below 2 ones and 2 with both, so a single one averages better + assert!(!fully_deceptive(&Trap::new(1, 2))); + // Ackley's traps (r = 0.8, z = ⌊3n/4⌋) are fully deceptive exactly where inequality 16 + // holds: for 3 and 4 bits, not 5 or 6 (the paper says "only for ℓ < 7"), and none of the + // sizes Ackley used, 8, 12, 16 and 20 (section 4) + for n in 3..=12 { + let z = 3 * n / 4; + let ackley = Trap::with_values(1, n, 8.0 * n as f64, 10.0 * n as f64, z).unwrap(); + let bound = (2.0 - 1.0 / (n - z) as f64) / (2.0 - 1.0 / z as f64); + assert_eq!(fully_deceptive(&ackley), 0.8 >= bound, "n = {n}"); + assert_eq!(fully_deceptive(&ackley), n <= 4, "n = {n}"); + } + } + + #[test] + fn invalid_traps_are_errors() { + assert!(Trap::with_values(0, 4, 3.0, 4.0, 3).is_err()); + assert!(Trap::with_values(1, 1, 0.0, 1.0, 1).is_err()); + assert!(Trap::with_values(1, 4, 3.0, 4.0, 0).is_err()); + assert!(Trap::with_values(1, 4, 3.0, 4.0, 4).is_err()); + assert!(Trap::with_values(1, 4, 4.0, 4.0, 3).is_err()); + assert!(Trap::with_values(1, 4, -1.0, 4.0, 3).is_err()); + assert!(Trap::with_values(1, 4, 3.0, f64::INFINITY, 3).is_err()); + assert!(Trap::with_values(1 << 23, 4, 3.0, 4.0, 3).is_err()); + assert!(Trap::with_values(1, 4, 0.0, 4.0, 3).is_ok()); + } + + #[test] + #[should_panic(expected = "Trap takes 1 to 2^24 bits")] + fn traps_of_too_many_bits_panic() { + Trap::new(1 << 23, 4); + } + + #[test] + fn royal_roads_score_their_schemas() { + let r1 = RoyalRoad::r1(); + let r2 = RoyalRoad::r2(); + assert_eq!(r1.optimum().unwrap().value(), 64.0); + assert_eq!(r2.optimum().unwrap().value(), 256.0); + assert_eq!((r1.name(), r2.name()), ("RoyalRoadR1", "RoyalRoadR2")); + // blocks 1 and 3 complete, block 2 one bit short + let mut x = Bits::zeros(64); + for i in (0..8).chain(16..24).chain(9..16) { + x.set(i, true); + } + assert_eq!(r1.evaluate(&x), 16.0); + assert_eq!(r2.evaluate(&x), 16.0); + // the first 32 bits: 4 blocks, 2 pairs, 1 of 32 + let x: Bits = (0..64).map(|i| i < 32).collect(); + assert_eq!(r1.evaluate(&x), 32.0); + assert_eq!(r2.evaluate(&x), 4.0 * 8.0 + 2.0 * 16.0 + 32.0); + // the last 48: 6 blocks, 3 pairs, 1 of 32 + let x: Bits = (0..64).map(|i| i >= 16).collect(); + assert_eq!(r2.evaluate(&x), 6.0 * 8.0 + 3.0 * 16.0 + 32.0); + // other sizes: 16 blocks of 3, levels of 3, 6, 12, 24, 48 bits + let wide = RoyalRoad::hierarchical(16, 3); + assert_eq!(wide.optimum().unwrap().value(), 48.0 * 5.0); + assert_eq!(RoyalRoad::new(5, 7).optimum().unwrap().value(), 35.0); + assert_eq!( + RoyalRoad::hierarchical(1, 4).optimum().unwrap().value(), + 4.0 + ); + } + + #[test] + #[should_panic(expected = "power of 2")] + fn hierarchical_royal_roads_need_a_power_of_two_blocks() { + RoyalRoad::hierarchical(6, 8); + } + + #[test] + #[should_panic(expected = "OneMax takes genomes of 4 bits")] + fn genomes_of_another_length_panic() { + OneMax::new(4).evaluate(&Bits::zeros(5)); + } + + #[test] + fn no_string_beats_the_optimum() { + let mut rng = StreamRng::seed_from_u64(3); + type Score = Box f64>; + let problems: Vec = vec![ + Box::new(|x| OneMax::new(64).evaluate(x)), + Box::new(|x| LeadingOnes::new(64).evaluate(x)), + Box::new(|x| Trap::new(16, 4).evaluate(x)), + Box::new(|x| RoyalRoad::r1().evaluate(x)), + Box::new(|x| RoyalRoad::r2().evaluate(x)), + ]; + let optima = [64.0, 64.0, 64.0, 64.0, 256.0]; + for (problem, optimum) in problems.iter().zip(optima) { + for _ in 0..1_000 { + let x = Binary::new(64).unwrap().random_genome(&mut rng); + assert!(problem(&x) <= optimum); + } + assert_eq!(problem(&Bits::ones(64)), optimum); + } + } +} diff --git a/src/problems/binary/knapsack.rs b/src/problems/binary/knapsack.rs new file mode 100644 index 00000000..f631651a --- /dev/null +++ b/src/problems/binary/knapsack.rs @@ -0,0 +1,832 @@ +//! The 0/1 knapsack problem, with Pisinger's generated instance classes. + +use super::{MAX_BITS, Optimum, Problem, binary, check_length}; +use crate::constraint::at_most; +use crate::engine::FitnessFunction; +use crate::genome::{Binary, Bits}; +use crate::problems::Constraints; +use crate::{Error, Objective, Result, StreamRng}; + +// the stream of the generator that draws an instance from its seed +const KNAPSACK_STREAM: u64 = 4; +// the largest total weight, profit or capacity: integers up to 2^53 are exact as f64 +const MAX_TOTAL: u64 = 1 << 53; +// the most decisions `optimum` records, items × (capacity + 1) bits: 32 MiB +const MAX_DECISIONS: u128 = 1 << 28; + +/// How the items of a spanner instance's spanner set are drawn: as in the instance class of the +/// same name. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub enum SpannerDistribution { + /// Weights and profits drawn independently. + Uncorrelated, + /// Profits within R/10 of the weights. + WeaklyCorrelated, + /// Profits R/10 above the weights. + StronglyCorrelated, +} + +/// A class of generated 0/1 knapsack instances (Pisinger, 2005, section 3 and section 3.3). +/// +/// R is the data range, [`KnapsackGenerator::range`]; "in [x, y]" is an integer drawn uniformly +/// from x to y; R/10 and R/500 are rounded down. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub enum KnapsackClass { + /// Weights and profits in [1, R]: no correlation, generally easy. + Uncorrelated, + /// Weights in [1, R], profits in [w − R/10, w + R/10] and at least 1 (here + /// [max(1, w − R/10), w + R/10]). + WeaklyCorrelated, + /// Weights in [1, R], profits w + R/10: hard, with a large gap between the continuous and the + /// integer optimum. + StronglyCorrelated, + /// Profits in [1, R], weights p + R/10. + InverseStronglyCorrelated, + /// Weights in [1, R], profits in [w + R/10 − R/500, w + R/10 + R/500]. + AlmostStronglyCorrelated, + /// Weights in [1, R], profits equal to them: the best knapsack is the fullest. + SubsetSum, + /// Weights in [100 000, 100 100] and profits in [1, 1000], whatever R. + UncorrelatedSimilarWeights, + /// span(v, m): every item a multiple of one of `v` items, the spanner set, drawn with weights + /// in [1, R] and profits by `distribution`, then scaled down to ⌈2p/m⌉ and ⌈2w/m⌉; each item + /// is a spanner item, drawn uniformly, times a multiplier in [1, m]. The paper uses + /// span(2, 10) ([`KnapsackClass::spanner`]). + Spanner { + /// The size of the spanner set, at least 1. + v: usize, + /// The largest multiplier, at least 1. + m: u64, + /// How the spanner set is drawn. + distribution: SpannerDistribution, + }, + /// mstr(k₁, k₂, d): weights in [1, R], profits w + k₁ for the weights divisible by d and + /// w + k₂ for the others. The paper uses mstr(3R/10, 2R/10, 6) + /// ([`KnapsackClass::multiple_strongly_correlated`]). + MultipleStronglyCorrelated { + /// The profit above the weight of the items whose weight is divisible by d. + k1: u64, + /// The profit above the weight of the others. + k2: u64, + /// The divisor, at least 1. + d: u64, + }, + /// pceil(d): weights in [1, R], profits the weights rounded up to a multiple of d, + /// d⌈w/d⌉. The paper uses pceil(3) ([`KnapsackClass::profit_ceiling`]). + ProfitCeiling { + /// The divisor, at least 1. + d: u64, + }, + /// circle(d): weights in [1, R], profits on an arc of an ellipse, d √(4R² − (w − 2R)²), + /// rounded down (computed exactly, with integers), with d = `numerator` / `denominator`. The + /// paper uses circle(2/3) ([`KnapsackClass::circle`]). + Circle { + /// d's numerator, 1 to 2^16. + numerator: u64, + /// d's denominator, 1 to 2^16. + denominator: u64, + }, +} + +impl KnapsackClass { + /// span(2, 10) with the spanner set drawn by `distribution`, the paper's spanner instances. + pub fn spanner(distribution: SpannerDistribution) -> Self { + Self::Spanner { + v: 2, + m: 10, + distribution, + } + } + + /// mstr(3R/10, 2R/10, 6) for the data range `range`, the paper's. + pub fn multiple_strongly_correlated(range: u64) -> Self { + Self::MultipleStronglyCorrelated { + k1: 3 * range / 10, + k2: 2 * range / 10, + d: 6, + } + } + + /// pceil(3), the paper's. + pub fn profit_ceiling() -> Self { + Self::ProfitCeiling { d: 3 } + } + + /// circle(2/3), the paper's. + pub fn circle() -> Self { + Self::Circle { + numerator: 2, + denominator: 3, + } + } + + // the parameters checked + fn check(&self) -> Result<()> { + let invalid = |reason: &str| { + Err(Error::InvalidSetting { + setting: "class", + reason: reason.to_string(), + }) + }; + match *self { + Self::Spanner { v, m, .. } if v == 0 || m == 0 || v > MAX_BITS || m > 1 << 32 => { + invalid("a spanner set has 1 to 2^24 items and multipliers 1 to 2^32") + } + Self::MultipleStronglyCorrelated { k1, k2, d } + if d == 0 || k1 > 1 << 32 || k2 > 1 << 32 => + { + invalid("mstr(k1, k2, d) needs d of at least 1, and k1 and k2 at most 2^32") + } + Self::ProfitCeiling { d } if d == 0 || d > 1 << 32 => { + invalid("pceil(d) needs d from 1 to 2^32") + } + Self::Circle { + numerator, + denominator, + } if !(1..=1 << 16).contains(&numerator) || !(1..=1 << 16).contains(&denominator) => { + invalid("circle(d) needs d's numerator and denominator from 1 to 2^16") + } + _ => Ok(()), + } + } +} + +/// The settings of a generated [`Knapsack`] instance: its class, number of items, data range, +/// capacity and seed. [`Knapsack::generator`] makes one. +#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +pub struct KnapsackGenerator { + class: KnapsackClass, + items: usize, + range: u64, + instance: u64, + instances: u64, + seed: u64, +} + +impl KnapsackGenerator { + /// The data range R: weights (and, by class, profits) are drawn from 1 to R. 1000 by default, + /// as in the paper's tables 1 to 5 and 9 to 13. + pub fn range(mut self, range: u64) -> Self { + self.range = range; + self + } + + /// The instance number h, from 1 to H: the capacity is ⌊h/(H + 1) Σ w⌋ (eq. 5). 50 by + /// default, about half the total weight. + pub fn instance(mut self, instance: u64) -> Self { + self.instance = instance; + self + } + + /// The number of instances H of a series, 100 by default, as in the paper. + pub fn instances(mut self, instances: u64) -> Self { + self.instances = instances; + self + } + + /// The seed of the items: the same seed gives the same items on every platform. 0 by + /// default. + pub fn seed(mut self, seed: u64) -> Self { + self.seed = seed; + self + } + + /// The instance. + /// + /// # Errors + /// + /// [`Error::InvalidSetting`] if there are no items or more than 2^24, the range isn't + /// between 1 and 2^32, the instance number isn't between 1 and the number of instances (at + /// most 2^32), the class's parameters are out of their bounds, or the total weight or profit + /// is above 2^53. + pub fn generate(&self) -> Result { + let invalid = + |setting: &'static str, reason: String| Err(Error::InvalidSetting { setting, reason }); + if !(1..=MAX_BITS).contains(&self.items) { + return invalid( + "items", + format!("must be between 1 and 2^24, got {}", self.items), + ); + } + if !(1..=1 << 32).contains(&self.range) { + return invalid( + "range", + format!("must be between 1 and 2^32, got {}", self.range), + ); + } + if !(1..=1 << 32).contains(&self.instances) { + return invalid( + "instances", + format!("must be between 1 and 2^32, got {}", self.instances), + ); + } + if !(1..=self.instances).contains(&self.instance) { + return invalid( + "instance", + format!( + "must be between 1 and the instances, {}, got {}", + self.instances, self.instance + ), + ); + } + self.class.check()?; + let (weights, profits) = self.items(); + let total: u128 = weights.iter().map(|&w| u128::from(w)).sum(); + let capacity = total * u128::from(self.instance) / u128::from(self.instances + 1); + let capacity = u64::try_from(capacity).unwrap_or(u64::MAX); + Knapsack::new(weights, profits, capacity) + } + + // the weights and profits, drawn in the order of the items + fn items(&self) -> (Vec, Vec) { + let mut rng = StreamRng::seed_from_u64(self.seed).derive(KNAPSACK_STREAM); + let range = self.range; + let (tenth, fivehundredth) = (range / 10, range / 500); + let mut draw = |low: u64, high: u64| low + rng.below_u64(high - low + 1); + let mut weights = Vec::with_capacity(self.items); + let mut profits = Vec::with_capacity(self.items); + // an item of a simple class, (weight, profit) + let simple = |draw: &mut dyn FnMut(u64, u64) -> u64, class: SpannerDistribution| { + let weight = draw(1, range); + let profit = match class { + SpannerDistribution::Uncorrelated => draw(1, range), + SpannerDistribution::WeaklyCorrelated => { + draw(weight.saturating_sub(tenth).max(1), weight + tenth) + } + SpannerDistribution::StronglyCorrelated => weight + tenth, + }; + (weight, profit) + }; + let spanner = match self.class { + KnapsackClass::Spanner { v, m, distribution } => (0..v) + .map(|_| { + let (weight, profit) = simple(&mut draw, distribution); + ( + weight.saturating_mul(2).div_ceil(m), + profit.saturating_mul(2).div_ceil(m), + ) + }) + .collect(), + _ => Vec::new(), + }; + for _ in 0..self.items { + let (weight, profit) = match self.class { + KnapsackClass::Uncorrelated => simple(&mut draw, SpannerDistribution::Uncorrelated), + KnapsackClass::WeaklyCorrelated => { + simple(&mut draw, SpannerDistribution::WeaklyCorrelated) + } + KnapsackClass::StronglyCorrelated => { + simple(&mut draw, SpannerDistribution::StronglyCorrelated) + } + KnapsackClass::InverseStronglyCorrelated => { + let profit = draw(1, range); + (profit + tenth, profit) + } + KnapsackClass::AlmostStronglyCorrelated => { + let weight = draw(1, range); + let profit = draw( + weight + tenth - fivehundredth, + weight + tenth + fivehundredth, + ); + (weight, profit) + } + KnapsackClass::SubsetSum => { + let weight = draw(1, range); + (weight, weight) + } + KnapsackClass::UncorrelatedSimilarWeights => { + let weight = draw(100_000, 100_100); + (weight, draw(1, 1000)) + } + KnapsackClass::Spanner { v, m, .. } => { + let (weight, profit) = spanner[draw(0, v as u64 - 1) as usize]; + let multiplier = draw(1, m); + (weight * multiplier, profit * multiplier) + } + KnapsackClass::MultipleStronglyCorrelated { k1, k2, d } => { + let weight = draw(1, range); + let k = if weight % d == 0 { k1 } else { k2 }; + (weight, weight + k) + } + KnapsackClass::ProfitCeiling { d } => { + let weight = draw(1, range); + (weight, weight.div_ceil(d) * d) + } + KnapsackClass::Circle { + numerator, + denominator, + } => { + let weight = draw(1, range); + // d² (4R² − (w − 2R)²) = d² w (4R − w), and its square root rounded down + let square = u128::from(weight) * u128::from(4 * range - weight); + let scaled = square * u128::from(numerator * numerator) + / u128::from(denominator * denominator); + (weight, isqrt(scaled) as u64) + } + }; + weights.push(weight); + profits.push(profit); + } + (weights, profits) + } +} + +// the square root of `value`, rounded down +fn isqrt(value: u128) -> u128 { + if value < 2 { + return value; + } + // Newton's method from above + let mut root = 1u128 << (128 - value.leading_zeros()).div_ceil(2); + loop { + let next = (root + value / root) / 2; + if next >= root { + return root; + } + root = next; + } +} + +/// The 0/1 knapsack problem: items with integer weights and profits, and a knapsack of integer +/// capacity; the most profitable selection whose weight fits. +/// +/// A bit per item, 1 to take it. The fitness is `(profit, violation)`, the total profit of the +/// selected items and how far their weight exceeds the capacity, `constraint::at_most(weight, +/// capacity)`, 0 when they fit: under Deb's rules, a selection that fits beats one that doesn't, +/// and overweight ones compare by their excess. +/// +/// [`Knapsack::generator`] draws an instance of one of Pisinger's classes from a seed with +/// genoxide's portable [`StreamRng`], so the same settings give the same items on every +/// platform; [`Knapsack::new`] takes the items of any instance. The weights and profits are +/// drawn item by item, in order (the weight first where both are drawn), each from its interval +/// with Lemire's unbiased method; the paper doesn't give its generator's random numbers, so the +/// classes are Pisinger's and the instances genoxide's. +/// +/// The problem is NP-hard, though only weakly: dynamic programming solves it in O(n c) time +/// (Bellman's recursion), which [`optimum`](Problem::optimum) runs, exactly, when the items times +/// the capacity are at most 2^28, and gives `None` above. Pisinger's classes are hard for +/// branch-and-bound algorithms in different ways: strongly correlated instances have a large gap +/// between the continuous and the integer optimum, subset-sum instances bounds that don't help, +/// and the spanner, multiple strongly correlated, profit ceiling and circle instances bounds that +/// stay loose (section 3.3). +/// +/// Pisinger, D. (2005). Where are the hard knapsack problems? *Computers & Operations Research* +/// 32(9): 2271-2284, the model (1)-(3), the instance classes of section 3 and section 3.3, and +/// the capacity of eq. 5. Martello, S. and Toth, P. (1990). *Knapsack Problems: Algorithms and +/// Computer Implementations*, Wiley, is the general reference. +/// +/// ``` +/// use genoxide::prelude::*; +/// use genoxide::problems::Problem; +/// use genoxide::problems::binary::{Knapsack, KnapsackClass}; +/// +/// let knapsack = Knapsack::generator(KnapsackClass::WeaklyCorrelated, 30).seed(1).generate()?; +/// let optimum = knapsack.optimum().expect("small enough").value(); +/// let ga = Ga::builder(knapsack.representation()) +/// .population_size(100) +/// .select(Tournament::new(3)?) +/// .crossover(UniformCrossover::new()) +/// .mutate(BitFlip::per_gene(1.0 / 30.0)?) +/// .seed(1) +/// .build()?; +/// let outcome = Engine::new(ga, knapsack) +/// .stop_when(Stop::target(optimum).or(Stop::generations(500))) +/// .run()?; +/// assert!(outcome.best_fitness().is_feasible()); +/// assert!(outcome.best_fitness().score().unwrap() >= 0.99 * optimum); +/// # Ok::<(), genoxide::Error>(()) +/// ``` +#[derive(Clone, Debug, PartialEq, Eq, Hash)] +pub struct Knapsack { + weights: Vec, + profits: Vec, + capacity: u64, +} + +impl Knapsack { + /// An instance with these items, a weight and a profit each, and this capacity. + /// + /// # Errors + /// + /// [`Error::InvalidSetting`] if there are no items or more than 2^24, the weights and the + /// profits differ in number, or the total weight, the total profit or the capacity is above + /// 2^53 (so that the fitness is exact). + pub fn new(weights: Vec, profits: Vec, capacity: u64) -> Result { + let invalid = + |setting: &'static str, reason: String| Err(Error::InvalidSetting { setting, reason }); + if !(1..=MAX_BITS).contains(&weights.len()) { + return invalid( + "weights", + format!("must have 1 to 2^24 items, got {}", weights.len()), + ); + } + if profits.len() != weights.len() { + return invalid( + "profits", + format!( + "must have a profit per item, {}, got {}", + weights.len(), + profits.len() + ), + ); + } + let total = |values: &[u64]| values.iter().map(|&v| u128::from(v)).sum::(); + for (setting, sum) in [ + ("weights", total(&weights)), + ("profits", total(&profits)), + ("capacity", u128::from(capacity)), + ] { + if sum > u128::from(MAX_TOTAL) { + return invalid(setting, format!("must add up to at most 2^53, got {sum}")); + } + } + Ok(Self { + weights, + profits, + capacity, + }) + } + + /// The settings of an instance of `class` with `items` items, to generate it: R = 1000, + /// instance 50 of 100 and seed 0 by default. + /// + /// ``` + /// use genoxide::problems::binary::{Knapsack, KnapsackClass}; + /// + /// let knapsack = Knapsack::generator(KnapsackClass::StronglyCorrelated, 50) + /// .range(10_000) + /// .seed(7) + /// .generate()?; + /// // profits 1000 above the weights + /// for (weight, profit) in knapsack.weights().iter().zip(knapsack.profits()) { + /// assert_eq!(profit - weight, 1000); + /// } + /// # Ok::<(), genoxide::Error>(()) + /// ``` + pub fn generator(class: KnapsackClass, items: usize) -> KnapsackGenerator { + KnapsackGenerator { + class, + items, + range: 1000, + instance: 50, + instances: 100, + seed: 0, + } + } + + /// The number of items. + pub fn items(&self) -> usize { + self.weights.len() + } + + /// The weight of each item. + pub fn weights(&self) -> &[u64] { + &self.weights + } + + /// The profit of each item. + pub fn profits(&self) -> &[u64] { + &self.profits + } + + /// The capacity. + pub fn capacity(&self) -> u64 { + self.capacity + } + + /// The total weight and profit of the items `x` selects. + /// + /// # Panics + /// + /// If `x` doesn't have a bit per item. + pub fn totals(&self, x: &Bits) -> (u64, u64) { + check_length("Knapsack", x, self.items()); + let (mut weight, mut profit) = (0, 0); + for (w, &word) in x.as_words().iter().enumerate() { + let mut word = word; + while word != 0 { + let item = w * 64 + word.trailing_zeros() as usize; + weight += self.weights[item]; + profit += self.profits[item]; + word &= word - 1; + } + } + (weight, profit) + } + + // the best selection by Bellman's recursion over the capacities, or None if the decisions + // to record are too many + fn solve(&self) -> Option { + let total: u64 = self.weights.iter().sum(); + let capacity = self.capacity.min(total); + let n = self.items(); + if n as u128 * (u128::from(capacity) + 1) > MAX_DECISIONS { + return None; + } + let capacity = capacity as usize; + let row = (capacity + 1).div_ceil(64); + let mut best = vec![0u64; capacity + 1]; + let mut taken = vec![0u64; n * row]; + for (item, (&weight, &profit)) in self.weights.iter().zip(&self.profits).enumerate() { + let Ok(weight) = usize::try_from(weight) else { + continue; + }; + if weight > capacity { + continue; + } + for c in (weight..=capacity).rev() { + let with = best[c - weight] + profit; + if with > best[c] { + best[c] = with; + taken[item * row + c / 64] |= 1 << (c % 64); + } + } + } + let mut x = Bits::zeros(n); + let mut c = capacity; + for item in (0..n).rev() { + if taken[item * row + c / 64] >> (c % 64) & 1 == 1 { + x.set(item, true); + c -= self.weights[item] as usize; + } + } + Some(x) + } +} + +impl FitnessFunction for Knapsack { + type Output = (f64, f64); + + /// The total profit of the items `x` selects, and how far their weight exceeds the capacity. + /// + /// # Panics + /// + /// If `x` doesn't have a bit per item. + fn evaluate(&self, x: &Bits) -> (f64, f64) { + let (weight, profit) = self.totals(x); + (profit as f64, at_most(weight as f64, self.capacity as f64)) + } +} + +impl Problem for Knapsack { + type Representation = Binary; + + fn name(&self) -> &'static str { + "Knapsack" + } + + fn representation(&self) -> Binary { + binary(self.items()) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + /// The most profitable selection that fits, by dynamic programming; `None` if the items + /// times the capacity (or the total weight, if less) are above 2^28. + fn optimum(&self) -> Option> { + let solution = self.solve()?; + let (_, profit) = self.totals(&solution); + Some(Optimum::proven(profit as f64, vec![solution])) + } + + fn reference(&self) -> &'static str { + "Pisinger, D. (2005). Where are the hard knapsack problems? Computers & Operations \ + Research 32(9): 2271-2284." + } + + fn reference_url(&self) -> Option<&'static str> { + Some("https://doi.org/10.1016/j.cor.2004.03.002") + } + + /// The capacity constraint, `weight − capacity ≤ 0`. + fn constraints(&self, x: &Bits) -> Constraints { + let (weight, _) = self.totals(x); + Constraints::new(vec![weight as f64 - self.capacity as f64], Vec::new()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::genome::Representation; + + fn classes() -> Vec { + vec![ + KnapsackClass::Uncorrelated, + KnapsackClass::WeaklyCorrelated, + KnapsackClass::StronglyCorrelated, + KnapsackClass::InverseStronglyCorrelated, + KnapsackClass::AlmostStronglyCorrelated, + KnapsackClass::SubsetSum, + KnapsackClass::UncorrelatedSimilarWeights, + KnapsackClass::spanner(SpannerDistribution::Uncorrelated), + KnapsackClass::spanner(SpannerDistribution::WeaklyCorrelated), + KnapsackClass::spanner(SpannerDistribution::StronglyCorrelated), + KnapsackClass::multiple_strongly_correlated(1000), + KnapsackClass::profit_ceiling(), + KnapsackClass::circle(), + ] + } + + // the best profit that fits, by trying every selection + fn brute_force(knapsack: &Knapsack) -> u64 { + let n = knapsack.items(); + (0u64..1 << n) + .filter_map(|selection| { + let x: Bits = (0..n).map(|i| selection >> i & 1 == 1).collect(); + let (weight, profit) = knapsack.totals(&x); + (weight <= knapsack.capacity()).then_some(profit) + }) + .max() + .unwrap() + } + + #[test] + fn the_classes_follow_pisinger() { + let r = 1000; + for class in classes() { + let knapsack = Knapsack::generator(class, 500).seed(3).generate().unwrap(); + let pairs = knapsack.weights().iter().zip(knapsack.profits()); + for (&w, &p) in pairs { + let in_range = |v: u64| (1..=r).contains(&v); + let holds = match class { + KnapsackClass::Uncorrelated => in_range(w) && in_range(p), + KnapsackClass::WeaklyCorrelated => { + in_range(w) && p >= 1 && p + 100 >= w && p <= w + 100 + } + KnapsackClass::StronglyCorrelated => in_range(w) && p == w + 100, + KnapsackClass::InverseStronglyCorrelated => in_range(p) && w == p + 100, + KnapsackClass::AlmostStronglyCorrelated => { + in_range(w) && p + 2 >= w + 100 && p <= w + 102 + } + KnapsackClass::SubsetSum => in_range(w) && p == w, + KnapsackClass::UncorrelatedSimilarWeights => { + (100_000..=100_100).contains(&w) && (1..=1000).contains(&p) + } + KnapsackClass::Spanner { .. } => (1..=2000).contains(&w), + KnapsackClass::MultipleStronglyCorrelated { .. } => { + in_range(w) && p == w + if w % 6 == 0 { 300 } else { 200 } + } + KnapsackClass::ProfitCeiling { .. } => { + in_range(w) && p % 3 == 0 && p >= w && p < w + 3 + } + KnapsackClass::Circle { .. } => { + // (3p/2)² ≤ 4R² − (w − 2R)² < (3(p + 1)/2)² + let square = 4 * r * r - (2 * r - w) * (2 * r - w); + in_range(w) && 9 * p * p <= 4 * square && 4 * square < 9 * (p + 1) * (p + 1) + } + }; + assert!(holds, "{class:?}: weight {w}, profit {p}"); + } + // eq. 5: instance 50 of 100 + let total: u64 = knapsack.weights().iter().sum(); + assert_eq!(knapsack.capacity(), total * 50 / 101, "{class:?}"); + } + } + + #[test] + fn spanner_items_are_multiples_of_the_spanner_set() { + fn gcd(a: u64, b: u64) -> u64 { + if b == 0 { a } else { gcd(b, a % b) } + } + for distribution in [ + SpannerDistribution::Uncorrelated, + SpannerDistribution::WeaklyCorrelated, + SpannerDistribution::StronglyCorrelated, + ] { + let class = KnapsackClass::spanner(distribution); + let knapsack = Knapsack::generator(class, 200).seed(5).generate().unwrap(); + // each item is a multiple of one of 2 spanner items: their reduced pairs are at most 2 + let mut reduced: Vec<(u64, u64)> = knapsack + .weights() + .iter() + .zip(knapsack.profits()) + .map(|(&w, &p)| (w / gcd(w, p), p / gcd(w, p))) + .collect(); + reduced.sort_unstable(); + reduced.dedup(); + assert!(reduced.len() <= 2, "{distribution:?}: {reduced:?}"); + // a base weight is at most ⌈2R/m⌉ = 200, times a multiplier of at most 10 + assert!(knapsack.weights().iter().all(|&w| (1..=2000).contains(&w))); + } + } + + // instances are the same on every platform: fixed values, to the bit + #[test] + fn instances_are_reproducible_from_their_seed() { + let knapsack = Knapsack::generator(KnapsackClass::Uncorrelated, 5) + .seed(1) + .generate() + .unwrap(); + assert_eq!(knapsack.weights(), [613, 858, 688, 724, 989]); + assert_eq!(knapsack.profits(), [705, 45, 998, 22, 438]); + assert_eq!(knapsack.capacity(), 1916); + // the circle's profits, rounded down: (2/3) √(898 · 3102) = 1112.67 + let circle = Knapsack::generator(KnapsackClass::circle(), 3) + .seed(2) + .generate() + .unwrap(); + assert_eq!(circle.weights(), [898, 956, 782]); + assert_eq!(circle.profits(), [1112, 1137, 1057]); + assert_eq!(circle.capacity(), 1304); + } + + #[test] + fn the_optimum_is_the_best_selection() { + for (i, class) in classes().into_iter().enumerate() { + for seed in 0..3 { + let knapsack = Knapsack::generator(class, 14) + .range(100) + .instance(1 + 30 * seed) + .seed(i as u64 * 10 + seed) + .generate() + .unwrap(); + let optimum = knapsack.optimum().unwrap(); + assert_eq!( + optimum.value(), + brute_force(&knapsack) as f64, + "{class:?} {seed}" + ); + let fitness = knapsack.evaluate(&optimum.solutions()[0]); + assert_eq!(fitness, (optimum.value(), 0.0)); + } + } + } + + #[test] + fn the_fitness_is_the_profit_and_the_excess_weight() { + let knapsack = Knapsack::new(vec![5, 4, 3], vec![10, 40, 30], 7).unwrap(); + let x = |text: &str| -> Bits { text.chars().map(|bit| bit == '1').collect() }; + assert_eq!(knapsack.evaluate(&x("011")), (70.0, 0.0)); + assert_eq!(knapsack.evaluate(&x("111")), (80.0, 5.0)); + assert_eq!(knapsack.constraints(&x("111")).inequalities(), [5.0]); + assert_eq!(knapsack.constraints(&x("100")).inequalities(), [-2.0]); + assert_eq!(knapsack.optimum().unwrap().value(), 70.0); + // a capacity above the total weight: everything fits + let roomy = Knapsack::new(vec![5, 4], vec![1, 2], 100).unwrap(); + assert_eq!(roomy.optimum().unwrap().value(), 3.0); + // the items of more than one word + let many = Knapsack::new(vec![1; 130], (0..130).collect(), 2).unwrap(); + let optimum = many.optimum().unwrap(); + assert_eq!(optimum.value(), 129.0 + 128.0); + assert_eq!(many.representation().genome_len(), 130); + } + + #[test] + fn integer_square_roots_round_down() { + for value in (0u128..10_000).chain([u64::MAX as u128, 1 << 100, (1 << 100) - 1]) { + let root = isqrt(value); + assert!( + root * root <= value && (root + 1) * (root + 1) > value, + "{value}" + ); + } + } + + #[test] + fn invalid_instances_are_errors() { + let class = KnapsackClass::Uncorrelated; + assert!(Knapsack::generator(class, 0).generate().is_err()); + assert!(Knapsack::generator(class, 10).range(0).generate().is_err()); + assert!( + Knapsack::generator(class, 10) + .instance(0) + .generate() + .is_err() + ); + assert!( + Knapsack::generator(class, 10) + .instance(101) + .generate() + .is_err() + ); + assert!( + Knapsack::generator(class, 10) + .instances(0) + .generate() + .is_err() + ); + let spanner = KnapsackClass::Spanner { + v: 0, + m: 10, + distribution: SpannerDistribution::Uncorrelated, + }; + assert!(Knapsack::generator(spanner, 10).generate().is_err()); + let ceiling = KnapsackClass::ProfitCeiling { d: 0 }; + assert!(Knapsack::generator(ceiling, 10).generate().is_err()); + let circle = KnapsackClass::Circle { + numerator: 2, + denominator: 0, + }; + assert!(Knapsack::generator(circle, 10).generate().is_err()); + assert!(Knapsack::new(vec![], vec![], 1).is_err()); + assert!(Knapsack::new(vec![1, 2], vec![1], 1).is_err()); + assert!(Knapsack::new(vec![1 << 53, 1], vec![1, 1], 1).is_err()); + assert!(Knapsack::new(vec![1], vec![1], (1 << 53) + 1).is_err()); + // too large to solve + let large = Knapsack::new(vec![1 << 40; 2], vec![1; 2], 1 << 41).unwrap(); + assert!(large.optimum().is_none()); + } +} diff --git a/src/problems/binary/nk.rs b/src/problems/binary/nk.rs new file mode 100644 index 00000000..890c0192 --- /dev/null +++ b/src/problems/binary/nk.rs @@ -0,0 +1,550 @@ +//! NK landscapes, generated from a seed. + +use super::{MAX_BITS, Optimum, Problem, binary, check_length}; +use crate::engine::FitnessFunction; +use crate::genome::{Binary, Bits}; +use crate::{Error, Objective, Result, StreamRng}; +use rand::Rng; + +// the stream of the generator that draws a landscape from its seed +const NK_STREAM: u64 = 3; +// the most table entries, N 2^(K+1): 128 MiB of them +const MAX_ENTRIES: usize = 1 << 24; +// the most steps `optimum` takes, exhaustively or by dynamic programming +const MAX_WORK: u128 = 1 << 30; +// a contribution c stands for c / 2^53 +const UNIT: f64 = 1.0 / (1u64 << 53) as f64; + +/// Which K sites bear on each site of an [`NkLandscape`]. +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Hash)] +pub enum Neighborhood { + /// Its flanking sites, on a circle: K/2 on each side for an even K (Kauffman and + /// Weinberger's Table 1), and for an odd K, (K + 1)/2 after it and (K − 1)/2 before it. + Adjacent, + /// K distinct other sites drawn at random, uniformly, for each site (their Table 2). + #[default] + Random, +} + +/// An NK landscape: N bits, each contributing a value that depends on its own bit and on K +/// others, drawn at random, the fitness their mean. +/// +/// Site i contributes `wᵢ`, read from a table of 2^(K+1) values, one for each combination of the +/// bits of i and of the K sites that bear on it, its [`neighbors`](NkLandscape::neighbors). Each +/// value is drawn independently and uniformly from (0, 1), and the fitness is +/// `W = (1/N) Σ wᵢ` (equation 1). K tunes the landscape: at K = 0 the bits are independent and +/// there is a single optimum; as K grows to N − 1 the landscape becomes rugged, with more local +/// optima and lower ones, and at K = N − 1 it's uncorrelated: every one-bit change gives a new +/// random fitness. +/// +/// The neighbors and the tables are drawn from `seed` with genoxide's portable +/// [`StreamRng`]: the same seed gives the same landscape on every platform. A contribution is +/// an odd multiple of 2^−53, `(2m + 1) / 2^53` with m drawn from 52 random bits, so it's in +/// (0, 1); the fitness adds them up exactly, as integers, and divides the sum by 2^53 N once, so +/// it's the same to the bit on every platform, and two strings tie only when their sums are equal. +/// +/// The optimum isn't known in closed form: [`optimum`](Problem::optimum) computes it, by dynamic +/// programming over the circle for [`Adjacent`](Neighborhood::Adjacent) neighborhoods, in +/// O(N 4^K) steps, or by evaluating every string, changed one bit at a time in Gray code order, +/// in O(2^N K) steps, whichever is less work; and gives `None` when both take more than 2^30 +/// steps (N above 25 to 30 with random neighborhoods, depending on K; K above 10 or so with +/// adjacent ones, depending on N). The work is done at each call: up to a few seconds. +/// +/// Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes and +/// its application to maturation of the immune response. *Journal of Theoretical Biology* +/// 141(2): 211-245, the section "The NK model" (two states per site, uniform fitness +/// contributions, equation 1) and Tables 1 and 2 (adjacent and random neighborhoods). +/// +/// ``` +/// use genoxide::prelude::*; +/// use genoxide::problems::Problem; +/// use genoxide::problems::binary::{Neighborhood, NkLandscape}; +/// +/// let landscape = NkLandscape::new(16, 2, Neighborhood::Adjacent, 7)?; +/// let optimum = landscape.optimum().expect("small enough"); +/// let ga = Ga::builder(landscape.representation()) +/// .population_size(50) +/// .select(Tournament::new(2)?) +/// .crossover(UniformCrossover::new()) +/// .mutate(BitFlip::per_gene(1.0 / 16.0)?) +/// .seed(1) +/// .build()?; +/// let outcome = Engine::new(ga, landscape) +/// .stop_when(Stop::target(optimum.value()).or(Stop::generations(500))) +/// .run()?; +/// assert_eq!(outcome.stop_reason(), StopReason::Target); +/// # Ok::<(), genoxide::Error>(()) +/// ``` +#[derive(Clone, Debug, PartialEq, Eq, Hash)] +pub struct NkLandscape { + n: usize, + k: usize, + neighborhood: Neighborhood, + seed: u64, + // the K neighbors of each site, a row per site + neighbors: Vec, + // the 2^(K+1) contributions of each site, a row per site, as odd integers below 2^53 + tables: Vec, +} + +impl NkLandscape { + /// A landscape of `n` bits, each bearing on its contribution with `k` others chosen by + /// `neighborhood`, the neighbors and contributions drawn from `seed`. + /// + /// # Errors + /// + /// [`Error::InvalidSetting`] if `n` is 0 or above 2^24, `k` isn't below `n`, or the tables + /// would have more than 2^24 values in all (N 2^(K+1)). + pub fn new(n: usize, k: usize, neighborhood: Neighborhood, seed: u64) -> Result { + if !(1..=MAX_BITS).contains(&n) { + return Err(Error::InvalidSetting { + setting: "n", + reason: format!("must be between 1 and 2^24, got {n}"), + }); + } + if k >= n { + return Err(Error::InvalidSetting { + setting: "k", + reason: format!("must be below n = {n}, got {k}"), + }); + } + let entries = u32::try_from(k + 1) + .ok() + .and_then(|bits| 1usize.checked_shl(bits)) + .and_then(|table| table.checked_mul(n)); + if entries.is_none_or(|entries| entries > MAX_ENTRIES) { + return Err(Error::InvalidSetting { + setting: "k", + reason: format!( + "the tables of N = {n} sites with K = {k} would have N 2^(K+1) values, more \ + than 2^24" + ), + }); + } + let mut rng = StreamRng::seed_from_u64(seed).derive(NK_STREAM); + let mut neighbors = Vec::with_capacity(n * k); + for site in 0..n { + match neighborhood { + Neighborhood::Adjacent => { + let (before, after) = (k / 2, k - k / 2); + neighbors.extend((1..=before).rev().map(|d| (site + n - d) % n)); + neighbors.extend((1..=after).map(|d| (site + d) % n)); + } + Neighborhood::Random => { + // k of the n − 1 other sites, in ascending order + let others = rng.sample_distinct(k, n - 1); + neighbors.extend(others.into_iter().map(|j| if j < site { j } else { j + 1 })); + } + } + } + let tables = (0..n << (k + 1)) + .map(|_| ((rng.next_u64() >> 12) << 1) | 1) + .collect(); + Ok(Self { + n, + k, + neighborhood, + seed, + neighbors, + tables, + }) + } + + /// The number of bits, N. + pub fn n(&self) -> usize { + self.n + } + + /// The number of other sites that bear on each site, K. + pub fn k(&self) -> usize { + self.k + } + + /// How the neighbors were chosen. + pub fn neighborhood(&self) -> Neighborhood { + self.neighborhood + } + + /// The seed the landscape was drawn from. + pub fn seed(&self) -> u64 { + self.seed + } + + /// The K sites that bear on site `site`, in the order of the bits of its table's index. + /// + /// # Panics + /// + /// If `site` isn't below N. + pub fn neighbors(&self, site: usize) -> &[usize] { + assert!(site < self.n, "site {site} of {} sites", self.n); + &self.neighbors[site * self.k..(site + 1) * self.k] + } + + /// The contribution of site `site` when it and its neighbors have the bits of `index`: the + /// site's own bit is bit K of the index (the highest), and its j-th neighbor's is bit + /// K − 1 − j. + /// + /// # Panics + /// + /// If `site` isn't below N or `index` isn't below 2^(K+1). + pub fn contribution(&self, site: usize, index: usize) -> f64 { + assert!(site < self.n, "site {site} of {} sites", self.n); + assert!(index < 1 << (self.k + 1), "an index has K + 1 bits"); + self.tables[(site << (self.k + 1)) + index] as f64 * UNIT + } + + // the sum of the contributions of `x`, as integers + fn sum(&self, x: &Bits) -> u128 { + let words = x.as_words(); + let bit = |i: usize| (words[i / 64] >> (i % 64) & 1) as usize; + let mut sum = 0; + for site in 0..self.n { + let mut index = bit(site); + for &neighbor in self.neighbors(site) { + index = index << 1 | bit(neighbor); + } + sum += u128::from(self.tables[(site << (self.k + 1)) + index]); + } + sum + } + + // the fitness of a sum of contributions + fn fitness(&self, sum: u128) -> f64 { + sum as f64 * UNIT / self.n as f64 + } + + // the best string, by evaluating every string in Gray code order, a bit flipped at a time + fn exhaustive(&self) -> Bits { + let (n, k) = (self.n, self.k); + // the sites whose index has bit `j`, and where + let mut dependents: Vec> = vec![Vec::new(); n]; + for site in 0..n { + dependents[site].push((site, k)); + for (j, &neighbor) in self.neighbors(site).iter().enumerate() { + dependents[neighbor].push((site, k - 1 - j)); + } + } + let entry = |site: usize, index: usize| u128::from(self.tables[(site << (k + 1)) + index]); + let mut indices = vec![0usize; n]; + let mut sum: u128 = (0..n).map(|site| entry(site, 0)).sum(); + let (mut best, mut best_string, mut string) = (sum, 0u64, 0u64); + for step in 1u64..1 << n { + let flipped = step.trailing_zeros() as usize; + string ^= 1 << flipped; + for &(site, position) in &dependents[flipped] { + sum -= entry(site, indices[site]); + indices[site] ^= 1 << position; + sum += entry(site, indices[site]); + } + if sum > best { + (best, best_string) = (sum, string); + } + } + (0..n).map(|i| best_string >> i & 1 == 1).collect() + } + + // the best string by dynamic programming, for adjacent neighborhoods: the first K bits fixed + // in turn, then the others chosen one by one, the state being the last K bits + fn dynamic(&self) -> Bits { + let (n, k) = (self.n, self.k); + let (before, after) = (k / 2, k - k / 2); + let mut best: Option<(u128, usize, usize)> = None; + for prefix in 0..1usize << k { + let (sums, _) = self.dynamic_from(prefix, false); + for (state, sum) in sums.into_iter().enumerate() { + let Some(sum) = sum else { continue }; + let total = sum + self.wrapped(prefix, state, before, after); + if best.is_none_or(|(best, _, _)| total > best) { + best = Some((total, prefix, state)); + } + } + } + let (_, prefix, state) = best.expect("a string"); + let (_, dropped) = self.dynamic_from(prefix, true); + // back from the last state, each step giving the bit it dropped + let mut x = vec![false; n]; + for (i, bit) in x.iter_mut().take(k).enumerate() { + *bit = prefix >> (k - 1 - i) & 1 == 1; + } + let mut state = state; + for p in (k..n).rev() { + let bit = dropped[(p - k) * (1 << k) + state]; + if k == 0 { + // no state: the window is the bit itself + x[p] = bit; + continue; + } + x[p] = state & 1 == 1; + state = (state >> 1) | (usize::from(bit) << (k - 1)); + } + x.into_iter().collect() + } + + // the best sums of the sites that don't wrap around the circle, for each final state (the + // last K bits, the last one lowest), from `prefix` (the first K bits, the first one highest); + // and, if `record`, the bit dropped at each step and new state, to go back + fn dynamic_from(&self, prefix: usize, record: bool) -> (Vec>, Vec) { + let (n, k) = (self.n, self.k); + let after = k - k / 2; + let states = 1usize << k; + let mask = states - 1; + // a window of K + 1 bits x_{p−K} … x_p, x_p lowest, as the index of site p − after + let index = |window: usize| { + let own = window >> after & 1; + let left = window >> (after + 1); + let right = window & ((1 << after) - 1); + own << k | left << after | right + }; + let mut sums: Vec> = vec![None; states]; + sums[prefix] = Some(0); + let mut dropped = if record { + vec![false; (n - k) * states] + } else { + Vec::new() + }; + let mut next = vec![None; states]; + for p in k..n { + next.fill(None); + let site = p - after; + let table = &self.tables[site << (k + 1)..(site + 1) << (k + 1)]; + for (state, sum) in sums.iter().enumerate() { + let Some(sum) = sum else { continue }; + for bit in 0..2 { + let window = state << 1 | bit; + let total = sum + u128::from(table[index(window)]); + let new = window & mask; + if next[new].is_none_or(|best| total > best) { + next[new] = Some(total); + if record { + dropped[(p - k) * states + new] = window >> k & 1 == 1; + } + } + } + } + std::mem::swap(&mut sums, &mut next); + } + (sums, dropped) + } + + // the contributions of the K sites whose windows wrap around the circle, from the first K + // bits and the last K + fn wrapped(&self, prefix: usize, state: usize, before: usize, after: usize) -> u128 { + let (n, k) = (self.n, self.k); + let bit = |position: usize| { + if position < k { + prefix >> (k - 1 - position) & 1 + } else { + state >> (n - 1 - position) & 1 + } + }; + let mut sum = 0; + for site in (0..before).chain(n - after..n) { + let mut index = bit(site); + for &neighbor in self.neighbors(site) { + index = index << 1 | bit(neighbor); + } + sum += u128::from(self.tables[(site << (k + 1)) + index]); + } + sum + } +} + +impl FitnessFunction for NkLandscape { + type Output = f64; + + /// The mean contribution of the sites of `x`, W. + /// + /// # Panics + /// + /// If `x` doesn't have N bits. + fn evaluate(&self, x: &Bits) -> f64 { + check_length("NkLandscape", x, self.n); + self.fitness(self.sum(x)) + } +} + +impl Problem for NkLandscape { + type Representation = Binary; + + fn name(&self) -> &'static str { + "NkLandscape" + } + + fn representation(&self) -> Binary { + binary(self.n) + } + + fn objective(&self) -> Objective { + Objective::Maximize + } + + /// The global maximum and a string that reaches it (the first found, if several tie), by + /// dynamic programming or exhaustive search, whichever takes fewer steps; `None` if both take + /// more than 2^30. + fn optimum(&self) -> Option> { + let (n, k) = (self.n as u128, self.k as u32); + let exhaustive = (n < 64).then(|| (1u128 << n) * u128::from(k + 1)); + let dynamic = (self.neighborhood == Neighborhood::Adjacent && k < 40) + .then(|| n * (1u128 << (2 * k + 1))); + let solution = match (exhaustive, dynamic) { + (Some(e), Some(d)) if d < e && d <= MAX_WORK => self.dynamic(), + (Some(e), _) if e <= MAX_WORK => self.exhaustive(), + (_, Some(d)) if d <= MAX_WORK => self.dynamic(), + _ => return None, + }; + Some(Optimum::proven(self.evaluate(&solution), vec![solution])) + } + + fn reference(&self) -> &'static str { + "Kauffman, S. A. and Weinberger, E. D. (1989). The NK model of rugged fitness landscapes \ + and its application to maturation of the immune response. Journal of Theoretical \ + Biology 141(2): 211-245." + } + + fn reference_url(&self) -> Option<&'static str> { + Some("https://doi.org/10.1016/S0022-5193(89)80019-0") + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::genome::Representation; + + // the best fitness by evaluating every string directly + fn brute_force(landscape: &NkLandscape) -> f64 { + let n = landscape.n(); + (0u64..1 << n) + .map(|string| { + let x: Bits = (0..n).map(|i| string >> i & 1 == 1).collect(); + landscape.evaluate(&x) + }) + .fold(f64::NEG_INFINITY, f64::max) + } + + #[test] + fn the_fitness_is_the_mean_of_the_tables() { + let landscape = NkLandscape::new(6, 2, Neighborhood::Random, 4).unwrap(); + let x: Bits = "101100".chars().map(|bit| bit == '1').collect(); + let mut sum = 0.0; + for site in 0..6 { + let mut index = usize::from(x.get(site).unwrap()); + for &neighbor in landscape.neighbors(site) { + assert_ne!(neighbor, site); + index = index << 1 | usize::from(x.get(neighbor).unwrap()); + } + let w = landscape.contribution(site, index); + assert!(w > 0.0 && w < 1.0); + sum += w; + } + assert!((landscape.evaluate(&x) - sum / 6.0).abs() < 1e-15); + } + + #[test] + fn adjacent_neighbors_flank_each_site_on_a_circle() { + let landscape = NkLandscape::new(8, 4, Neighborhood::Adjacent, 0).unwrap(); + assert_eq!(landscape.neighbors(0), [6, 7, 1, 2]); + assert_eq!(landscape.neighbors(5), [3, 4, 6, 7]); + let odd = NkLandscape::new(8, 3, Neighborhood::Adjacent, 0).unwrap(); + assert_eq!(odd.neighbors(7), [6, 0, 1]); + let none = NkLandscape::new(3, 0, Neighborhood::Adjacent, 0).unwrap(); + assert!(none.neighbors(1).is_empty()); + } + + #[test] + fn random_neighbors_are_distinct_others() { + let landscape = NkLandscape::new(20, 7, Neighborhood::Random, 9).unwrap(); + for site in 0..20 { + let neighbors = landscape.neighbors(site); + assert_eq!(neighbors.len(), 7); + assert!(!neighbors.contains(&site)); + assert!(neighbors.windows(2).all(|pair| pair[0] < pair[1])); + } + } + + // a landscape is the same on every platform: fixed values, to the bit + #[test] + fn landscapes_are_reproducible_from_their_seed() { + let landscape = NkLandscape::new(10, 3, Neighborhood::Random, 42).unwrap(); + assert_eq!( + landscape, + NkLandscape::new(10, 3, Neighborhood::Random, 42).unwrap() + ); + assert_ne!( + landscape, + NkLandscape::new(10, 3, Neighborhood::Random, 43).unwrap() + ); + assert_eq!(landscape.neighbors(0), [2, 6, 7]); + assert_eq!(landscape.neighbors(9), [4, 5, 8]); + assert_eq!( + landscape.contribution(0, 0).to_bits(), + 0x3fd2_08b8_6307_d686 + ); + assert_eq!( + landscape.contribution(9, 15).to_bits(), + 0x3feb_fb4d_db16_b0c9 + ); + let x: Bits = "1100101101".chars().map(|bit| bit == '1').collect(); + assert_eq!(landscape.evaluate(&x).to_bits(), 0x3fd5_5624_43ce_9d63); + } + + #[test] + fn the_optimum_is_the_best_string() { + for (n, k, neighborhood, seed) in [ + (1, 0, Neighborhood::Random, 0), + (5, 0, Neighborhood::Adjacent, 1), + (8, 1, Neighborhood::Adjacent, 2), + (9, 2, Neighborhood::Adjacent, 3), + (10, 3, Neighborhood::Adjacent, 4), + (7, 6, Neighborhood::Adjacent, 5), + (12, 4, Neighborhood::Random, 6), + (11, 10, Neighborhood::Random, 7), + ] { + let landscape = NkLandscape::new(n, k, neighborhood, seed).unwrap(); + let optimum = landscape.optimum().unwrap(); + assert!(optimum.is_proven()); + assert_eq!(optimum.value(), brute_force(&landscape), "{n} {k}"); + assert_eq!(landscape.evaluate(&optimum.solutions()[0]), optimum.value()); + // dynamic programming and exhaustive search agree + if neighborhood == Neighborhood::Adjacent { + let dynamic = landscape.dynamic(); + let exhaustive = landscape.exhaustive(); + assert_eq!( + landscape.sum(&dynamic), + landscape.sum(&exhaustive), + "{n} {k}" + ); + } + } + } + + #[test] + fn large_adjacent_landscapes_are_solved_by_dynamic_programming() { + let landscape = NkLandscape::new(200, 4, Neighborhood::Adjacent, 1).unwrap(); + let optimum = landscape.optimum().unwrap(); + // no string near it is better + let best = optimum.solutions()[0].clone(); + for i in 0..200 { + let mut x = best.clone(); + x.flip(i); + assert!(landscape.evaluate(&x) <= optimum.value()); + } + let mut rng = StreamRng::seed_from_u64(1); + for _ in 0..1_000 { + let x = Binary::new(200).unwrap().random_genome(&mut rng); + assert!(landscape.evaluate(&x) < optimum.value()); + } + // random neighborhoods of 200 bits are beyond exhaustive search + let random = NkLandscape::new(200, 4, Neighborhood::Random, 1).unwrap(); + assert!(random.optimum().is_none()); + } + + #[test] + fn invalid_landscapes_are_errors() { + assert!(NkLandscape::new(0, 0, Neighborhood::Random, 0).is_err()); + assert!(NkLandscape::new(5, 5, Neighborhood::Random, 0).is_err()); + assert!(NkLandscape::new(100, 30, Neighborhood::Random, 0).is_err()); + assert!(NkLandscape::new((1 << 24) + 1, 0, Neighborhood::Random, 0).is_err()); + assert!(NkLandscape::new(5, 4, Neighborhood::Adjacent, 0).is_ok()); + } +}