From efb8d42c1e11e30a6a04461739dd86fb3b6093da Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 11:04:53 +0200 Subject: [PATCH 01/18] Make a metric one decorator and a board one object A registered metric is now five fields: name, function, family, cost and direction. The first line of its docstring is the question the column answers, written so it reads without the metric's name. Direction None means a diagnostic -- read beside the other columns, never ranked on -- which is what `dpp` becomes: three monotone transforms of one determinant that collapses toward underflow before the models stop differing. Where a metric is computed and how per-design values collapse to one number are no longer properties of the metric. `Board.evaluate` takes `space` (pixel, or a PCA of the reference designs) and `aggregation` (mean or median), so every metric is written once; the spec records the aggregation in each row and the `*_median` columns are retired. The board also scores half of the reference designs against the other half, which is the only honest measurement of what real designs score on a column. Port the math layer from the workshop branch (vendi, geometric-mean DPP, median-sigma, PRDC), fix `register_metric` silently targeting the global registry when handed an empty one, and replace "brief" with "conditions" throughout, including three PR #75 docstrings. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluation/board.py | 195 ++++++++++++++++++++++ engiopt/evaluation/context.py | 32 +++- engiopt/evaluation/evaluator.py | 2 + engiopt/evaluation/metrics/builtin.py | 47 ++++-- engiopt/evaluation/registry.py | 80 +++++++--- engiopt/evaluation/spec.py | 8 +- engiopt/evaluation/submission.py | 2 +- engiopt/metrics.py | 222 +++++++++++++++++++++++++- tests/test_board.py | 83 ++++++++++ tests/test_evaluation.py | 28 ++-- tests/test_metric_registry.py | 62 +++++++ 11 files changed, 703 insertions(+), 58 deletions(-) create mode 100644 engiopt/evaluation/board.py create mode 100644 tests/test_board.py create mode 100644 tests/test_metric_registry.py diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py new file mode 100644 index 00000000..a3c1ccc6 --- /dev/null +++ b/engiopt/evaluation/board.py @@ -0,0 +1,195 @@ +"""Score several models at once and read the result. + +This is the whole user-facing evaluation surface for saved designs:: + + board = Board(problem, reference=REFERENCE, train=TRAIN) + board.evaluate(MODELS, space="pixel", aggregation="mean") + board.explain() + board.rank("mmd") + +Where a metric is computed and how per-design values collapse to one number are +arguments here, not properties of the metric, so every metric is written once. +For a loaded `Generator` -- which is what the physics metrics and `cond_sens` +need -- use `Evaluator.score`. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from dataclasses import field +from typing import Any, Literal, TYPE_CHECKING + +import numpy as np +import pandas as pd + +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.registry import MetricRegistry +from engiopt.evaluation.registry import METRICS + +if TYPE_CHECKING: + from collections.abc import Mapping + +Space = Literal["pixel", "pca"] +"""Where the set-level metrics are computed. Learned latent spaces arrive with the LVAE.""" + +REFERENCE_ROW = "reference (split-half)" +"""Label of the row scoring half of the reference designs against the other half. + +It is what real, correct designs score on every column, measured on this problem +rather than assumed, and it is the row every other row is read against. +""" + + +@dataclass +class Board: + """Several models' designs, scored against one set of reference designs. + + Attributes: + problem: The problem the designs answer; needs `design_space` and + `check_constraints`, which every EngiBench problem has. + reference: The withheld optimal designs the models are scored against. + train: The training designs, for the memorization metrics. Optional. + sigma: Kernel bandwidth for the distribution metrics. + frame: The most recent `evaluate` result. + """ + + problem: Any + reference: Any + train: Any | None = None + sigma: float = 10.0 + registry: MetricRegistry = field(default_factory=lambda: METRICS) + frame: pd.DataFrame = field(default_factory=pd.DataFrame) + + def evaluate( + self, + models: Mapping[str, Any], + *, + metrics: list[str] | None = None, + space: Space = "pixel", + aggregation: Literal["mean", "median"] = "mean", + pca_dims: int = 8, + reference_row: bool = True, + ) -> pd.DataFrame: + """Score every model on every cheap metric. + + Args: + models: Model name -> its generated designs, one array per model. + metrics: Metric names; defaults to every metric that needs no simulator. + space: `pixel` scores the designs themselves. `pca` projects the + designs onto the leading principal components of the reference + designs first, and appends `@pca` to each column; metrics that + need the actual designs (`viol`, `copy_rate`) are skipped there. + aggregation: How metrics with one value per design collapse to one number. + pca_dims: Number of components when `space="pca"`. + reference_row: Also score one random half of the reference designs + against the other half, as the row `REFERENCE_ROW`. + + Returns: + One row per model, one column per metric output. Also kept as `frame`. + """ + specs = self.registry.select(metrics) if metrics else self.registry.select(cost="cheap") + if space != "pixel": + specs = [spec for spec in specs if not spec.pixel_only] + reference = np.asarray(self.reference) + project = _pca_projection(reference, pca_dims) if space == "pca" else (lambda x: x) + + def score(generated: Any, against: Any) -> dict[str, float]: + ctx = EvaluationContext( + problem=self.problem, + problem_id=getattr(self.problem, "problem_id", type(self.problem).__name__), + gen_designs=project(np.asarray(generated)), + ref_designs=project(np.asarray(against)), + sigma=self.sigma, + aggregation=aggregation, + copy_corpus_fn=(lambda: np.asarray(self.train)) if self.train is not None else None, + ) + row: dict[str, float] = {} + for spec in specs: + value = spec.fn(ctx) + row.update(value if isinstance(value, dict) else {spec.name: value}) + return row + + rows = {name: score(designs, reference) for name, designs in models.items()} + if reference_row and len(reference) >= 4: # noqa: PLR2004 - two halves of at least two designs + order = np.random.default_rng(0).permutation(len(reference)) + half = len(reference) // 2 + rows[REFERENCE_ROW] = score(reference[order[:half]], reference[order[half:]]) + + frame = pd.DataFrame.from_dict(rows, orient="index") + if space != "pixel": + frame.columns = [f"{column}@{space}" for column in frame.columns] + self.frame = frame + return frame + + def explain(self) -> pd.DataFrame: + """Read every column of the last evaluation. + + Returns: + One row per column: the question it answers, which way is better, the + model it picks (blank for a diagnostic, `tie` when the best is shared), + and what real designs scored on it. + """ + models = self.frame.drop(index=REFERENCE_ROW, errors="ignore") + rows = {} + for column in self.frame.columns: + spec = self._spec_for(column) + if spec is None: + continue + picks = "" + values = models[column].dropna() + if spec.higher_is_better is not None and not values.empty: + best = values.min() if spec.higher_is_better is False else values.max() + winners = list(values.index[values == best]) + picks = winners[0] if len(winners) == 1 else f"tie: {', '.join(winners)}" + rows[column] = { + "question": spec.description, + "direction": spec.direction, + "picks": picks, + "real designs score": self.frame.loc[REFERENCE_ROW, column] if REFERENCE_ROW in self.frame.index else None, + } + return pd.DataFrame.from_dict(rows, orient="index") + + def rank(self, column: str) -> pd.DataFrame: + """Models ordered best-first on one column. + + Raises: + ValueError: If the column is a diagnostic, which is read but not ranked on. + """ + spec = self._spec_for(column) + if spec is None: + raise KeyError(f"{column!r} is not a metric column on this board.") + if spec.higher_is_better is None: + raise ValueError( + f"{column!r} is a diagnostic: it says whether the other columns mean what they look like. " + f"Read it beside them; do not rank on it." + ) + return self.frame.sort_values(column, ascending=not spec.higher_is_better) + + def _spec_for(self, column: str) -> Any: + base = column.split("@", maxsplit=1)[0] + return next((spec for spec in self.registry.values() if base in spec.columns), None) + + def _repr_html_(self) -> str: + return self.frame._repr_html_() + + def __repr__(self) -> str: + return repr(self.frame) + + +def _pca_projection(reference: np.ndarray, n_components: int) -> Any: + """Projection onto the leading principal components of the reference designs. + + The control that asks whether a *learned* space was needed: if PCA of the + data answers the same questions, any rotation of the same width would do. + """ + flat = np.ascontiguousarray(reference.reshape(len(reference), -1), dtype=np.float64) + mean = flat.mean(axis=0) + _, _, vt = np.linalg.svd(flat - mean, full_matrices=False) + components = np.ascontiguousarray(vt[: min(n_components, len(vt))].T) + + def project(designs: np.ndarray) -> np.ndarray: + centered = np.ascontiguousarray(designs.reshape(len(designs), -1), dtype=np.float64) - mean + # einsum rather than `@`: on small matrices Apple's BLAS raises spurious FP warnings here. + return np.einsum("ij,jk->ik", centered, components) + + return project diff --git a/engiopt/evaluation/context.py b/engiopt/evaluation/context.py index 0525602d..a13dfdea 100644 --- a/engiopt/evaluation/context.py +++ b/engiopt/evaluation/context.py @@ -13,7 +13,7 @@ from dataclasses import dataclass from dataclasses import field from functools import cached_property -from typing import Any, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING from gymnasium import spaces import numpy as np @@ -121,6 +121,8 @@ class EvaluationContext: objective_weight_condition: str | None = None copy_corpus_fn: Callable[[], npt.NDArray[Any]] | None = None copy_tol: float = 0.01 + aggregation: Literal["mean", "median"] = "mean" + """How a metric with one value per design is collapsed into a column; see `reduce`.""" resample_permuted: Callable[[npt.NDArray[Any]], npt.NDArray[Any]] | None = None @property @@ -133,6 +135,30 @@ def is_dict_space(self) -> bool: """Whether the problem uses a `spaces.Dict` design space needing flatten/unflatten.""" return isinstance(self.problem.design_space, spaces.Dict) + def reduce(self, values: Any) -> float: + """Collapse one value per generated design into the column's single number. + + `iog`, `cog`, `fog` and the per-design distances are each a vector with + one entry per generated design. Which single number that vector becomes + is a policy, not a measurement: a mean is pulled by one diverged design, + a median is not, and two boards that chose differently are not comparable. + So the choice is declared once (`EvalSpec.aggregation`), recorded in the + row, and applied here rather than hard-coded metric by metric. + + Rates such as `viol` and `copy_rate` are fractions, not per-design + averages, and do not pass through this. + + Args: + values: One value per generated design. + + Returns: + The mean or median per the declared policy; NaN for no designs. + """ + array = np.asarray(values, dtype=float) + if array.size == 0: + return float("nan") + return float(np.median(array) if self.aggregation == "median" else np.mean(array)) + @cached_property def gen_flat(self) -> npt.NDArray[Any]: """Generated designs flattened to `(n_samples, -1)`.""" @@ -191,7 +217,7 @@ def nearest_corpus_distance(self) -> npt.NDArray[Any] | None: def permuted_designs(self) -> npt.NDArray[Any] | None: """Designs the generator produces when the conditions are shuffled between samples. - Same latent draw, same model, different brief. A model that reads its + Same latent draw, same model, different conditions. A model that reads its conditions produces something different; a model that ignores them produces the identical batch in a different order at best, and an identical batch outright at worst. Comparing the two is what turns "this @@ -352,7 +378,7 @@ def is_infeasible(self, design: Any, conditions: dict[str, Any] | None) -> bool: This is what a new problem gets for free. 2. The volume-fraction budget named by the spec's `volume_condition`, when the problem has one. Missing that target is a design failing to - honor its brief rather than an invalid design, and no EngiBench + honor its conditions rather than an invalid design, and no EngiBench constraint covers it. A problem with neither -- photonics2d has no volume budget -- is scored diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 81aa227d..8cb8ebca 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -178,6 +178,7 @@ def context_for(self, generator: Generator) -> EvaluationContext: objective_weight_condition=self.spec.objective_weight_condition, copy_corpus_fn=self.copy_corpus, copy_tol=self.spec.copy_tol, + aggregation=self.spec.aggregation, resample_permuted=self._permuted_sampler(generator), ) @@ -302,6 +303,7 @@ def _provenance(self, generator: Generator, ctx: EvaluationContext) -> dict[str, "seed": getattr(generator, "seed", None), "spec_version": self.spec.version, "n_samples": ctx.n_samples, + "aggregation": ctx.aggregation, "sample_seconds": ctx.sample_seconds, "checkpoint_repo": getattr(generator, "checkpoint_repo", None), "checkpoint_path": getattr(generator, "checkpoint_path", None), diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 731452e8..756e7fa2 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -32,7 +32,7 @@ family="distribution", cost="cheap", higher_is_better=False, - description="Maximum Mean Discrepancy between generated and reference designs (pixel space).", + description="Taken as a set, how different are the generated designs from the reference optimal designs?", ) def mmd(ctx: EvaluationContext) -> float: """Maximum Mean Discrepancy between generated and reference designs.""" @@ -48,8 +48,8 @@ def mmd(ctx: EvaluationContext) -> float: "dpp", family="diversity", cost="cheap", - higher_is_better=True, - description="Determinantal Point Process diversity of the generated set.", + higher_is_better=None, + description="How spread out are the generated designs, measured as the volume their similarity kernel spans?", ) def dpp(ctx: EvaluationContext) -> float: """Determinantal Point Process diversity of the generated designs.""" @@ -63,11 +63,12 @@ def dpp(ctx: EvaluationContext) -> float: @register_metric( "novelty", + pixel_only=True, family="memorization", cost="cheap", higher_is_better=None, + description="How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?", outputs=("novelty", "copy_rate"), - description="Distance from generated designs to the nearest design the model could have memorized.", ) def novelty(ctx: EvaluationContext) -> dict[str, float]: """How far the generated designs sit from the corpus a model could copy from. @@ -98,28 +99,29 @@ def novelty(ctx: EvaluationContext) -> dict[str, float]: if distances is None: return {"novelty": float("nan"), "copy_rate": float("nan")} return { - "novelty": float(np.mean(distances)), + "novelty": ctx.reduce(distances), "copy_rate": float(np.mean(distances < ctx.copy_tol)), } # ---------------------------------------------------------------------- -# Conditions: does the model actually read the brief it was given? +# Conditions: does the model actually use the conditions it was given? # ---------------------------------------------------------------------- @register_metric( "cond_sens", + pixel_only=True, family="conditions", cost="cheap", higher_is_better=None, - description="How much the generated design changes when the conditions are shuffled.", + description="When the same model is given different conditions, does its output change?", ) def cond_sens(ctx: EvaluationContext) -> float: """Mean per-element RMS change in output when each sample is given another's conditions. Sampled from the same seed, so the latent draw is held fixed and the only - thing that varies is the brief. A model that ignores its conditions returns + thing that varies is the conditions. A model that ignores them returns the identical batch and scores 0; a model that responds to them scores the size of that response. @@ -143,7 +145,7 @@ def cond_sens(ctx: EvaluationContext) -> float: if permuted is None: return float("nan") deltas = np.linalg.norm(ctx.gen_flat - permuted, axis=1) / np.sqrt(ctx.gen_flat.shape[1]) - return float(np.mean(deltas)) + return ctx.reduce(deltas) # ---------------------------------------------------------------------- @@ -153,10 +155,11 @@ def cond_sens(ctx: EvaluationContext) -> float: @register_metric( "viol", + pixel_only=True, family="feasibility", cost="cheap", higher_is_better=False, - description="Fraction of designs violating the problem's constraints or their volume budget.", + description="What fraction of the generated designs violate the problem's constraints?", ) def viol(ctx: EvaluationContext) -> float: """Fraction of infeasible designs; see `EvaluationContext.is_infeasible`. @@ -184,11 +187,11 @@ def viol(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean initial optimality gap: how far generated designs start from the reference optimum.", + description="How much worse than the reference optimum is the generated design, exactly as generated?", ) def iog(ctx: EvaluationContext) -> float: """Mean initial optimality gap, before any re-optimization.""" - return float(np.mean(ctx.optimization.iog)) + return ctx.reduce(ctx.optimization.iog) @register_metric( @@ -196,11 +199,21 @@ def iog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean cumulative optimality gap over re-optimization from each generated design.", + description="Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step?", ) def cog(ctx: EvaluationContext) -> float: - """Mean cumulative optimality gap.""" - return float(np.mean(ctx.optimization.cog)) + """Mean cumulative optimality gap. + + For each generated design, the optimizer is started from that design and + `objective(step) - objective(reference optimum)` is summed over every step it + takes; that sum is the area under the design's optimality-gap curve. The + column is the mean of those sums over the generated designs. + + Because it sums a whole trajectory it mixes two things: how far from optimal + the start was, and how many steps the optimizer needed. `iog` isolates the + first and `fog` the end point; read `cog` beside both. + """ + return ctx.reduce(ctx.optimization.cog) @register_metric( @@ -208,8 +221,8 @@ def cog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean final optimality gap after re-optimization.", + description="Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?", ) def fog(ctx: EvaluationContext) -> float: """Mean final optimality gap.""" - return float(np.mean(ctx.optimization.fog)) + return ctx.reduce(ctx.optimization.fog) diff --git a/engiopt/evaluation/registry.py b/engiopt/evaluation/registry.py index eea3cce9..4da73c2c 100644 --- a/engiopt/evaluation/registry.py +++ b/engiopt/evaluation/registry.py @@ -1,25 +1,31 @@ """Registry of evaluation metrics. -A metric belongs to the *comparison* (generated designs vs a reference set under -a problem), not to a model -- so metrics are registered functions taking an -`EvaluationContext`, rather than methods on a generator. - -Registering a metric is one decorated function:: - - @register_metric("mmd", family="distribution", cost="cheap", higher_is_better=False) - def mmd(ctx: EvaluationContext) -> float: - return engiopt.metrics.mmd(ctx.gen_designs, ctx.ref_designs, sigma=ctx.sigma) - -The declaration is what lets the evaluator keep simulation-free metrics strictly -separate from simulator-backed ones, and what lets the leaderboard know which -direction of each column is "better". +A metric is a function of an `EvaluationContext` -- the generated designs, the +reference designs they are scored against, and what the spec says about how to +read them -- plus a short declaration of what the number means:: + + @register_metric("viol", family="feasibility", cost="cheap", higher_is_better=False) + def viol(ctx: EvaluationContext) -> float: + \"\"\"What fraction of the generated designs violate the problem's constraints?\"\"\" + ... + +The docstring's first line is the metric's description: the question it answers, +written so it makes sense without knowing the metric's name. `higher_is_better` +says which way to rank; `None` means the metric is a diagnostic that is read but +never ranked on. `cost` keeps metrics that touch the simulator separate from the +ones that do not. + +Where a metric is computed -- pixel space, a PCA of the reference designs, a +learned latent space -- and how one value per design collapses to one number are +not properties of the metric. They are chosen when a board is evaluated; see +`engiopt.evaluation.board.Board`. """ from __future__ import annotations from collections.abc import Callable, Iterator, Mapping from dataclasses import dataclass -from typing import Literal, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING if TYPE_CHECKING: from engiopt.evaluation.context import EvaluationContext @@ -41,23 +47,32 @@ def mmd(ctx: EvaluationContext) -> float: @dataclass(frozen=True) class MetricSpec: - """A registered metric and everything the evaluator needs to know about it.""" + """A registered metric and what a reader needs to know to use its column.""" name: str fn: MetricFn family: MetricFamily cost: MetricCost higher_is_better: bool | None - """None for metrics with no intrinsic direction (e.g. a realized volume fraction).""" + """Ranking direction. None: a diagnostic, read beside other columns but never ranked on.""" + description: str = "" + """The question the value answers, in one plain sentence. Defaults to the docstring's first line.""" outputs: tuple[str, ...] = () """Column names produced. Defaults to `(name,)` for single-valued metrics.""" - description: str = "" + pixel_only: bool = False + """True if the metric needs the actual designs -- a constraint check, a copy corpus -- + and so cannot be asked in a PCA or latent space.""" @property def columns(self) -> tuple[str, ...]: """Leaderboard columns this metric fills.""" return self.outputs or (self.name,) + @property + def direction(self) -> str: + """The ranking direction as a reader sees it.""" + return {True: "higher is better", False: "lower is better", None: "diagnostic"}[self.higher_is_better] + class MetricRegistry(Mapping[str, MetricSpec]): """Name-to-`MetricSpec` mapping populated by the `register_metric` decorator.""" @@ -78,7 +93,7 @@ def __len__(self) -> int: return len(self._metrics) def add(self, spec: MetricSpec) -> None: - """Register a metric, rejecting duplicate names and duplicate output columns.""" + """Register a metric, refusing a second definition under the same name or column.""" if spec.name in self._metrics: raise ValueError(f"Metric {spec.name!r} is already registered.") taken = {col: owner.name for owner in self._metrics.values() for col in owner.columns} @@ -94,7 +109,7 @@ def select( cost: MetricCost | None = None, family: MetricFamily | None = None, ) -> list[MetricSpec]: - """Return specs filtered by name, cost, and/or family, in registration order.""" + """Metrics by name, cost, or family -- every metric if nothing is given.""" specs = [self[name] for name in names] if names is not None else list(self._metrics.values()) if cost is not None: specs = [spec for spec in specs if spec.cost == cost] @@ -106,6 +121,20 @@ def columns(self, specs: list[MetricSpec] | None = None) -> list[str]: """All leaderboard columns for the given specs (default: every metric).""" return [col for spec in (specs if specs is not None else self._metrics.values()) for col in spec.columns] + def explain(self) -> Any: + """One row per metric: the question it answers, its family, direction and cost. + + Returns: + A `pandas.DataFrame` indexed by metric name. + """ + import pandas as pd + + rows = { + spec.name: {"question": spec.description, "family": spec.family, "direction": spec.direction, "cost": spec.cost} + for spec in self._metrics.values() + } + return pd.DataFrame.from_dict(rows, orient="index") + METRICS = MetricRegistry() """The global metric registry.""" @@ -117,8 +146,9 @@ def register_metric( family: MetricFamily, cost: MetricCost, higher_is_better: bool | None, - outputs: tuple[str, ...] = (), description: str = "", + outputs: tuple[str, ...] = (), + pixel_only: bool = False, registry: MetricRegistry | None = None, ) -> Callable[[MetricFn], MetricFn]: """Register an evaluation metric. @@ -127,9 +157,10 @@ def register_metric( name: Unique metric name, also the default output column. family: Which engineering question this answers. cost: `cheap` if it never runs a simulator, `expensive` otherwise. - higher_is_better: Ranking direction, or None if the metric is diagnostic. + higher_is_better: Ranking direction, or None for a diagnostic. + description: The question the value answers; defaults to the docstring's first line. outputs: Column names, when the metric returns a dict of several values. - description: One-line explanation, surfaced by `engiopt.evaluate --list-metrics`. + pixel_only: True if the metric needs the actual designs rather than codes in some space. registry: Target registry; defaults to the global one. Returns: @@ -137,15 +168,16 @@ def register_metric( """ def decorator(fn: MetricFn) -> MetricFn: - (registry or METRICS).add( + (METRICS if registry is None else registry).add( MetricSpec( name=name, fn=fn, family=family, cost=cost, higher_is_better=higher_is_better, - outputs=outputs, description=description or (fn.__doc__ or "").strip().split("\n")[0], + outputs=outputs, + pixel_only=pixel_only, ) ) return fn diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index faf6c36d..bba0e4ec 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -18,7 +18,7 @@ import json from pathlib import Path import subprocess -from typing import Any, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING import numpy as np import torch as th @@ -139,6 +139,12 @@ class EvalSpec: condition_seed: int = 1 metrics: tuple[str, ...] = ("mmd", "dpp", "novelty", "cond_sens", "viol", "iog", "cog", "fog") sigma: float = 10.0 + aggregation: Literal["mean", "median"] = "mean" + """How per-design metrics (`iog`, `cog`, `fog`, distances) collapse to one number. + + Changing it changes what every performance column means, so it is part of + the frozen spec and is recorded in every row, not a flag on the run. + """ volfrac_tol: float = 0.01 volume_condition: str | None = None objective_weights: tuple[float, ...] | None = None diff --git a/engiopt/evaluation/submission.py b/engiopt/evaluation/submission.py index 4b6b05ca..b0d99e95 100644 --- a/engiopt/evaluation/submission.py +++ b/engiopt/evaluation/submission.py @@ -105,7 +105,7 @@ def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[s def _declares_conditioning_it_does_not_use(row: dict[str, Any]) -> bool: - """Whether a model claiming to be conditional produced identical output for different briefs. + """Whether a model claiming to be conditional produced identical output for different conditions. No tolerance to choose here, and deliberately so: the comparison holds the latent draw fixed, so a model that genuinely reads its conditions cannot diff --git a/engiopt/metrics.py b/engiopt/metrics.py index de8b56d1..47a46cb3 100644 --- a/engiopt/metrics.py +++ b/engiopt/metrics.py @@ -15,6 +15,13 @@ if TYPE_CHECKING: from engibench import OptiStep +MIN_PRDC_SAMPLES = 2 +"""Below two samples per set there is no neighbourhood to measure.""" + +EIGENVALUE_FLOOR = 1e-12 +"""Eigenvalues below this are treated as zero: a PSD matrix can return +slightly negative ones, and exact zeros would send the entropy to -inf.""" + def mmd(x: np.ndarray, y: np.ndarray, sigma: float = 1.0) -> float: """Compute the Maximum Mean Discrepancy (MMD) between two sets of samples. @@ -60,6 +67,218 @@ def dpp_diversity(x: np.ndarray, sigma: float = 1.0) -> float: return 0.0 # fallback in case of numerical issues +def log_dpp_diversity(x: np.ndarray, sigma: float = 1.0) -> float: + """Log-determinant form of `dpp_diversity`, for spaces where the raw determinant underflows. + + The determinant of an `n x n` kernel matrix is a product of `n` numbers below + one, so at `n = 50` it routinely lands near 1e-11 and can reach the limit of + double precision -- at which point the metric stops distinguishing models and + starts reporting rounding error. `slogdet` measures the same quantity on a + scale that survives. + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The log-determinant. Larger means more diverse. Returns `-inf` for a + singular matrix, which is the honest limit of a degenerate sample set. + """ + x_flat = x.reshape(x.shape[0], -1) + similarity = np.exp(-cdist(x_flat, x_flat, "sqeuclidean") / (2 * sigma**2)) + reg_matrix = similarity + 1e-6 * np.eye(x.shape[0]) + + # On a near-singular kernel matrix NumPy's slogdet warns about intermediate + # overflow while still returning a correct result -- which is precisely the + # case this function exists to handle, so the warnings are noise here. + with np.errstate(divide="ignore", over="ignore", invalid="ignore"): + sign, logabsdet = np.linalg.slogdet(reg_matrix) + + if sign <= 0 or not np.isfinite(logabsdet): + return float("-inf") + return float(logabsdet) + + +def dpp_geometric_mean(x: np.ndarray, sigma: float = 1.0) -> float: + r"""`n`-th root of the DPP determinant: the geometric mean of its eigenvalues. + + The same quantity `dpp_diversity` reports, on the only scale of the three + that is readable *and* comparable across sample sizes. + + - The raw determinant is a product of `n` numbers below one, so it underflows + (1e-11 at n=50, and worse) and distinct models render as identical zeros. + - The log-determinant fixes the underflow but is unbounded below and still + scales with `n`, so a value means nothing without knowing the sample count + it was computed at -- and two papers reporting "log-DPP" at n=50 and n=200 + are not comparable. + - `det(K)^(1/n) = exp(logdet / n)` is the geometric mean of the eigenvalues. + Since `K` has a unit diagonal its eigenvalues sum to `n`, so their + arithmetic mean is exactly 1 and the geometric mean lands in `(0, 1]` by + AM-GM: **1 means a perfectly diverse set (`K = I`), and values approach 0 + as samples collapse onto each other.** That bound holds at every `n`. + + Interpretation is therefore fixed rather than relative: 0.5 is the same + statement about a set of 50 designs as about a set of 500. + + This does **not** fix the pathology `vendi_score` exists to resist -- a + determinant still rewards any perturbation that pushes samples apart, so + adding noise to a collapsed set raises this too. It fixes the *numerical* + failure only, which is a different and more embarrassing one. + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The geometric mean of the kernel eigenvalues, in `(0, 1]`. Returns 0.0 + for a degenerate set, which is the honest limit rather than `-inf`. + """ + n = x.shape[0] + if n == 0: + return 0.0 + + logdet = log_dpp_diversity(x, sigma=sigma) + if not np.isfinite(logdet): + return 0.0 + return float(np.exp(logdet / n)) + + +def vendi_score(x: np.ndarray, sigma: float = 1.0) -> float: + """Effective number of distinct samples in a set. + + The exponential of the von Neumann entropy of the normalized similarity + matrix, which reads directly as a count: `n` identical samples score 1, and + `n` mutually dissimilar ones score `n`. + + Preferred over `dpp_diversity` for diversity. A determinant rewards any + perturbation that makes samples less similar, so adding noise to a collapsed + set *raises* it -- exactly the gaming this measure is meant to resist. The + entropy of the eigenvalue spectrum does not move that way, because noise + spreads eigenvalues without adding modes. + + See Friedman & Dieng (2023), "The Vendi Score". + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The effective sample count, in `[1, n]`. + """ + x_flat = x.reshape(x.shape[0], -1) + n = x_flat.shape[0] + if n == 0: + return 0.0 + + kernel = np.exp(-cdist(x_flat, x_flat, "sqeuclidean") / (2 * sigma**2)) / n + eigenvalues = np.linalg.eigvalsh(kernel) + + # Clip: eigenvalues of a PSD matrix can come back slightly negative, and + # zero eigenvalues contribute nothing to entropy but would produce -inf. + positive = eigenvalues[eigenvalues > EIGENVALUE_FLOOR] + if positive.size == 0: + return 1.0 + return float(np.exp(-np.sum(positive * np.log(positive)))) + + +def compute_median_sigma(x: np.ndarray, y: np.ndarray | None = None) -> float: + """Choose a kernel bandwidth by the median heuristic. + + A fixed bandwidth cannot serve two spaces at once: pixel space and a pruned + latent space differ in scale by orders of magnitude, and a kernel sized for + one saturates in the other. A bandwidth taken from the data adapts to + whichever space it is handed. + + Precisely, this returns `sqrt(median(d^2) / 2)`, which is the **median + distance divided by sqrt(2)** -- the `gamma = 1 / median(d^2)` form of the + median heuristic. Under the kernel `exp(-d^2 / 2 sigma^2)` that puts a pair + at the median distance at `exp(-1)`, not `exp(-0.5)`. Both conventions are + called "the median heuristic"; this is the one in force, it is applied + identically in pixel, PCA and latent space, and every published column was + computed under it. Said explicitly because the previous wording read as + "sigma is the median distance", which it is not. + + Sampling is capped and seeded so the bandwidth is reproducible. + + Args: + x: Samples of shape `(n, d)`. + y: Optional second sample set; distances are then cross-set. + + Returns: + The bandwidth, floored at 1e-6 so a degenerate set cannot divide by zero. + """ + x = x.reshape(x.shape[0], -1) + n_sample = min(500, len(x)) + rng = np.random.default_rng(42) + idx_x = rng.choice(len(x), n_sample, replace=len(x) < n_sample) + + if y is not None: + y = y.reshape(y.shape[0], -1) + idx_y = rng.choice(len(y), n_sample, replace=len(y) < n_sample) + dists = cdist(x[idx_x], y[idx_y], "sqeuclidean") + else: + dists = cdist(x[idx_x], x[idx_x], "sqeuclidean") + dists = dists[np.triu_indices_from(dists, k=1)] + + sigma = np.sqrt(np.median(dists) / 2) if len(dists) > 0 else 1.0 + return max(float(sigma), 1e-6) + + +def compute_prdc(real_features: np.ndarray, fake_features: np.ndarray, nearest_k: int = 5) -> dict[str, float]: + """Compute precision, recall, density, and coverage between two sample sets. + + The four metrics of Naeem et al. 2020, "Reliable Fidelity and Diversity + Metrics for Generative Models" (ICML). They exist because a single + distribution distance conflates two different failures: generating + implausible designs, and generating too few distinct ones. MMD reports one + number for both; these separate them. + + All four lie in `[0, 1]`, higher is better. + + - **precision**: fraction of generated samples inside some real sample's + k-NN ball -- are the generations plausible? + - **recall**: fraction of real samples inside some generated sample's ball + -- is the data manifold covered? + - **density**: smoothed precision, robust to real outliers that would + otherwise inflate it. + - **coverage**: fraction of real samples whose own ball contains a generated + sample; more robust than recall when generations are noisy outliers. + + Args: + real_features: Reference samples, shape `(n_real, d)`. + fake_features: Generated samples, shape `(n_fake, d)`. + nearest_k: Neighbours defining the ball radius, clamped to fit the sets. + + Returns: + Mapping with keys `precision`, `recall`, `density`, `coverage`. All NaN + when either set has fewer than two samples. + """ + real = real_features.reshape(real_features.shape[0], -1) + fake = fake_features.reshape(fake_features.shape[0], -1) + n_real, n_fake = len(real), len(fake) + + if n_real < MIN_PRDC_SAMPLES or n_fake < MIN_PRDC_SAMPLES: + return dict.fromkeys(("precision", "recall", "density", "coverage"), float("nan")) + + k = max(1, min(nearest_k, n_real - 1, n_fake - 1)) + + # Self-distance sits at index 0 of the partition, so index k is the k-th + # non-self neighbour. + real_radii = np.partition(cdist(real, real, "euclidean"), k, axis=1)[:, k] + fake_radii = np.partition(cdist(fake, fake, "euclidean"), k, axis=1)[:, k] + d_rf = cdist(real, fake, "euclidean") + + inside_real = d_rf <= real_radii[:, None] + inside_fake = d_rf <= fake_radii[None, :] + + return { + "precision": float(inside_real.any(axis=0).mean()), + "recall": float(inside_fake.any(axis=1).mean()), + "density": float(inside_real.sum(axis=0).mean()) / k, + "coverage": float((d_rf.min(axis=1) <= real_radii).mean()), + } + + def optimality_gap(opt_history: list[OptiStep], baseline: float) -> list[float]: """Compute the optimality gap of an optimization history. @@ -71,6 +290,3 @@ def optimality_gap(opt_history: list[OptiStep], baseline: float) -> list[float]: list[float]: The optimality gap at each step in opt_history. """ return [opt.obj_values - baseline for opt in opt_history] - - -# Return the failure ratio diff --git a/tests/test_board.py b/tests/test_board.py new file mode 100644 index 00000000..b370f30d --- /dev/null +++ b/tests/test_board.py @@ -0,0 +1,83 @@ +"""`Board`: from saved designs to a readable table in three calls.""" + +from __future__ import annotations + +from typing import Any + +import numpy as np +import pytest + +from engiopt.evaluation.board import Board +from engiopt.evaluation.board import REFERENCE_ROW +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.registry import METRICS + +# Importing the metrics package registers the built-ins. +import engiopt.evaluation.metrics # noqa: F401 # isort: skip + + +@pytest.fixture +def designs(fake_problem: Any) -> tuple[dict[str, np.ndarray], np.ndarray, np.ndarray]: + rng = np.random.default_rng(0) + shape = fake_problem.design_space.shape + train = rng.random((20, *shape)) + ref = rng.random((6, *shape)) + models = {"honest": np.clip(ref + rng.normal(0, 0.01, ref.shape), 0, 1), "copier": ref.copy()} + return models, ref, train + + +def test_evaluate_scores_every_model_on_the_cheap_metrics(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + frame = Board(fake_problem, reference=ref, train=train).evaluate(models) + assert list(frame.index) == ["honest", "copier", REFERENCE_ROW] + assert {"mmd", "copy_rate", "viol"} <= set(frame.columns) + + +def test_explain_names_a_pick_only_for_ranked_columns(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + explained = board.explain() + assert explained.loc["mmd", "picks"] == "copier", "identical sets score the best MMD" + assert explained.loc["copy_rate", "picks"] == "", "a diagnostic picks nobody" + assert explained.loc["mmd", "question"] == METRICS["mmd"].description + assert explained.loc["mmd", "real designs score"] == board.frame.loc[REFERENCE_ROW, "mmd"] + + +def test_the_reference_row_is_measured_and_never_picked(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + assert board.frame.loc[REFERENCE_ROW, "mmd"] > 0.0, "half of the data against the other half is not identical" + assert REFERENCE_ROW not in set(board.explain()["picks"]) + assert REFERENCE_ROW not in board.evaluate(models, reference_row=False).index + + +def test_rank_refuses_a_diagnostic(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + assert board.rank("mmd").index[0] == "copier" + with pytest.raises(ValueError, match="diagnostic"): + board.rank("copy_rate") + + +def test_space_is_an_argument_not_a_metric(fake_problem: Any, designs: Any) -> None: + """The same metric, asked in PCA space, gets a suffixed column; pixel-only metrics are skipped.""" + models, ref, train = designs + frame = Board(fake_problem, reference=ref, train=train).evaluate(models, space="pca", pca_dims=3) + assert "mmd@pca" in frame.columns + assert "viol@pca" not in frame.columns, "a constraint check on PCA codes would be meaningless" + assert frame.loc["copier", "mmd@pca"] == pytest.approx(0.0) + + +def test_aggregation_is_a_policy_on_the_context(fake_problem: Any) -> None: + """One diverged design moves the mean and not the median.""" + values = [0.1, 0.1, 0.1, 100.0] + zeros = np.zeros((4, 8, 10)) + mean_ctx = EvaluationContext(problem=fake_problem, problem_id="f", gen_designs=zeros, ref_designs=zeros) + median_ctx = EvaluationContext( + problem=fake_problem, problem_id="f", gen_designs=zeros, ref_designs=zeros, aggregation="median" + ) + assert mean_ctx.reduce(values) == pytest.approx(25.075) + assert median_ctx.reduce(values) == pytest.approx(0.1) diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index e6e80d68..5dad2313 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -127,11 +127,21 @@ def test_registry_rejects_colliding_output_columns() -> None: """Two metrics writing the same column would silently overwrite each other.""" registry = MetricRegistry() register_metric( - "first", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + "first", + family="diversity", + cost="cheap", + higher_is_better=True, + outputs=("shared",), + registry=registry, )(lambda ctx: {"shared": 1.0}) with pytest.raises(ValueError, match="already emitted"): register_metric( - "second", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + "second", + family="diversity", + cost="cheap", + higher_is_better=True, + outputs=("shared",), + registry=registry, )(lambda ctx: {"shared": 2.0}) @@ -166,7 +176,7 @@ def _board() -> pd.DataFrame: "seed": 1, "spec_version": "v1", "mmd": 0.1, - "dpp": 5.0, + "viol": 0.1, }, { "problem_id": "p", @@ -176,7 +186,7 @@ def _board() -> pd.DataFrame: "seed": 2, "spec_version": "v1", "mmd": 0.3, - "dpp": 7.0, + "viol": 0.1, }, { "problem_id": "p", @@ -186,16 +196,16 @@ def _board() -> pd.DataFrame: "seed": 1, "spec_version": "v1", "mmd": 0.05, - "dpp": 1.0, + "viol": 0.5, }, ] ) def test_rank_respects_each_metric_direction() -> None: - """Lower MMD ranks first; higher DPP ranks first.""" + """Both columns are lower-is-better, and they pick different winners.""" assert rank(_board(), "mmd").iloc[0]["algo_id"] == "b" - assert rank(_board(), "dpp").iloc[0]["algo_id"] == "a" + assert rank(_board(), "viol").iloc[0]["algo_id"] == "a" def test_rank_aggregates_seeds_with_the_median_by_default() -> None: @@ -207,9 +217,9 @@ def test_rank_aggregates_seeds_with_the_median_by_default() -> None: def test_disagreement_surfaces_conflicting_rankings() -> None: """The two metrics crown different winners; the view must show both.""" - table = disagreement(_board(), ["mmd", "dpp"]) + table = disagreement(_board(), ["mmd", "viol"]) assert table.loc[("p", "b", "cfg_b", "someone/engiopt-b", "v1"), "mmd"] == 1 - assert table.loc[("p", "a", "cfg_a", "someone/engiopt-a", "v1"), "dpp"] == 1 + assert table.loc[("p", "a", "cfg_a", "someone/engiopt-a", "v1"), "viol"] == 1 def _two_configs_board() -> pd.DataFrame: diff --git a/tests/test_metric_registry.py b/tests/test_metric_registry.py new file mode 100644 index 00000000..de6c4cc5 --- /dev/null +++ b/tests/test_metric_registry.py @@ -0,0 +1,62 @@ +"""The metric declaration contract: one decorator, one direction, one sentence.""" + +from __future__ import annotations + +import pytest + +from engiopt.evaluation.registry import MetricRegistry +from engiopt.evaluation.registry import METRICS +from engiopt.evaluation.registry import register_metric + + +@pytest.fixture +def registry() -> MetricRegistry: + """A registry of its own, so a test never publishes into the global one.""" + return MetricRegistry() + + +def test_a_custom_registry_is_not_bypassed_when_empty(registry: MetricRegistry) -> None: + """An empty registry is falsy, which `registry or METRICS` silently ignored.""" + before = len(METRICS) + + @register_metric("solo", family="diversity", cost="cheap", higher_is_better=True, registry=registry) + def solo(ctx: object) -> float: + """A metric registered into a registry of its own.""" + return 0.0 + + assert "solo" in registry + assert len(METRICS) == before, "registration leaked into the global registry" + + +def test_the_docstring_is_the_question(registry: MetricRegistry) -> None: + """No separate text to maintain: the first docstring line is what readers see.""" + + @register_metric("share", family="feasibility", cost="cheap", higher_is_better=False, registry=registry) + def share(ctx: object) -> float: + """What fraction of the designs are infeasible? + + Longer explanation lives below the first line. + """ + return 0.0 + + assert registry["share"].description == "What fraction of the designs are infeasible?" + assert registry.explain().loc["share", "direction"] == "lower is better" + + +def test_a_diagnostic_declares_no_direction(registry: MetricRegistry) -> None: + @register_metric("diag", family="conditions", cost="cheap", higher_is_better=None, registry=registry) + def diag(ctx: object) -> float: + """A metric read beside the others, never ranked on.""" + return 0.0 + + assert registry["diag"].direction == "diagnostic" + + +def test_two_metrics_cannot_claim_one_column(registry: MetricRegistry) -> None: + register_metric( + "first", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + )(lambda ctx: {"shared": 1.0}) + with pytest.raises(ValueError, match="already emitted"): + register_metric( + "second", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + )(lambda ctx: {"shared": 2.0}) From 2c21d7bae5bcd80f57bd62bf0cb775e12742355d Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 11:04:56 +0200 Subject: [PATCH 02/18] Add a notebook that walks the metrics suite on a toy problem Ten code cells, one call each: what metrics exist, score five models, read the board, rank, change the aggregation, change the space, register a metric. Two of the models are constructions with known right answers; the one that returns the correct withheld designs under permuted conditions scores a perfect MMD, which is the lesson. Co-Authored-By: Claude Fable 5.1 --- example_metrics_suite.ipynb | 1172 +++++++++++++++++++++++++++++++++++ 1 file changed, 1172 insertions(+) create mode 100644 example_metrics_suite.ipynb diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb new file mode 100644 index 00000000..570d2d64 --- /dev/null +++ b/example_metrics_suite.ipynb @@ -0,0 +1,1172 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "e6135946", + "metadata": {}, + "source": [ + "# Using the metrics suite\n", + "\n", + "You have designs from one or more models and want to know what they are worth. The\n", + "whole workflow is one registry and one object:\n", + "\n", + "| You want to... | You call |\n", + "|---|---|\n", + "| see every metric and what it measures | `METRICS.explain()` |\n", + "| score your models | `Board(problem, reference=..., train=...).evaluate(models)` |\n", + "| read the result | `board.explain()`, `board.rank(\"mmd\")` |\n", + "| change how per-design values are combined | `board.evaluate(models, aggregation=\"median\")` |\n", + "| ask the same questions in another space | `board.evaluate(models, space=\"pca\")` |\n", + "| add a metric of your own | `@register_metric(...)` on a function |\n", + "\n", + "Everything below runs in seconds on a toy problem. With a real EngiBench problem the\n", + "calls are identical." + ] + }, + { + "cell_type": "markdown", + "id": "9f94269b", + "metadata": {}, + "source": [ + "## 1. What metrics exist, and what does each one measure?" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "id": "25da8a4e", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:42.538081Z", + "iopub.status.busy": "2026-09-18T13:05:42.537712Z", + "iopub.status.idle": "2026-09-18T13:05:45.049987Z", + "shell.execute_reply": "2026-09-18T13:05:45.049722Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questionfamilydirectioncost
mmdTaken as a set, how different are the generated designs from the reference optimal designs?distributionlower is bettercheap
dppHow spread out are the generated designs, measured as the volume their similarity kernel spans?diversitydiagnosticcheap
noveltyHow close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?memorizationdiagnosticcheap
cond_sensWhen the same model is given different conditions, does its output change?conditionsdiagnosticcheap
violWhat fraction of the generated designs violate the problem's constraints?feasibilitylower is bettercheap
iogHow much worse than the reference optimum is the generated design, exactly as generated?performancelower is betterexpensive
cogStarting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step?performancelower is betterexpensive
fogStarting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?performancelower is betterexpensive
\n", + "
" + ], + "text/plain": [ + " question \\\n", + "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", + "dpp How spread out are the generated designs, measured as the volume their similarity kernel spans? \n", + "novelty How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", + "cond_sens When the same model is given different conditions, does its output change? \n", + "viol What fraction of the generated designs violate the problem's constraints? \n", + "iog How much worse than the reference optimum is the generated design, exactly as generated? \n", + "cog Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? \n", + "fog Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes? \n", + "\n", + " family direction cost \n", + "mmd distribution lower is better cheap \n", + "dpp diversity diagnostic cheap \n", + "novelty memorization diagnostic cheap \n", + "cond_sens conditions diagnostic cheap \n", + "viol feasibility lower is better cheap \n", + "iog performance lower is better expensive \n", + "cog performance lower is better expensive \n", + "fog performance lower is better expensive " + ] + }, + "execution_count": 1, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import warnings\n", + "\n", + "import pandas as pd\n", + "\n", + "warnings.filterwarnings(\"ignore\", category=UserWarning, module=\"gymnasium\")\n", + "pd.set_option(\"display.max_colwidth\", None)\n", + "\n", + "from engiopt.evaluation.registry import METRICS\n", + "import engiopt.evaluation.metrics # noqa: F401 -- importing registers the built-in metrics\n", + "\n", + "METRICS.explain()" + ] + }, + { + "cell_type": "markdown", + "id": "daed2cfb", + "metadata": {}, + "source": [ + "Each row is a question written so it makes sense without knowing the metric's name,\n", + "plus which way is better. Some say **diagnostic**: those columns tell you whether the\n", + "*other* columns can be trusted — a lookup table posts a perfect `mmd`, and only\n", + "`copy_rate` says so — and they are never ranked on.\n", + "\n", + "`cog` is worth a second look because it is the least obvious: the optimizer is started\n", + "from the generated design, and `objective(step) − objective(reference optimum)` is\n", + "summed over every step it takes. It is the area under the optimality-gap curve, so it\n", + "mixes how far from optimal the start was (`iog`) with how many steps were needed." + ] + }, + { + "cell_type": "markdown", + "id": "89a27d85", + "metadata": {}, + "source": [ + "## 2. A toy problem" + ] + }, + { + "cell_type": "markdown", + "id": "3cf5be3a", + "metadata": {}, + "source": [ + "Every EngiBench dataset has the same structure: **one optimal design per set of\n", + "conditions**, split into training and withheld test conditions. The toy keeps exactly\n", + "that and nothing else — the optimal design for condition `c` is a field with mean `c`.\n", + "With a real problem, `problem` is an EngiBench `Problem` and the two arrays come from\n", + "its dataset splits." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "4e14a02e", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.051482Z", + "iopub.status.busy": "2026-09-18T13:05:45.051342Z", + "iopub.status.idle": "2026-09-18T13:05:45.054205Z", + "shell.execute_reply": "2026-09-18T13:05:45.054017Z" + } + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from gymnasium import spaces\n", + "\n", + "\n", + "class ToyProblem:\n", + " \"\"\"All the cheap metrics need from a problem: a design space and a constraint check.\"\"\"\n", + "\n", + " design_space = spaces.Box(0.0, 1.0, shape=(8, 10), dtype=np.float64)\n", + "\n", + " def check_constraints(self, design, config):\n", + " class Violations:\n", + " violations = [] if self.design_space.contains(design) else [\"outside design space\"]\n", + "\n", + " return Violations()\n", + "\n", + "\n", + "rng = np.random.default_rng(0)\n", + "problem = ToyProblem()\n", + "\n", + "\n", + "def optimal_design(c):\n", + " return np.clip(np.full((8, 10), c) + rng.normal(0, 0.01, (8, 10)), 0, 1)\n", + "\n", + "\n", + "train_conditions = rng.uniform(0.2, 0.8, 64)\n", + "test_conditions = rng.uniform(0.2, 0.8, 16)\n", + "TRAIN = np.stack([optimal_design(c) for c in train_conditions]) # what models are fitted on\n", + "REFERENCE = np.stack([optimal_design(c) for c in test_conditions]) # withheld; what models are scored against" + ] + }, + { + "cell_type": "markdown", + "id": "78758876", + "metadata": {}, + "source": [ + "## 3. Five models" + ] + }, + { + "cell_type": "markdown", + "id": "21d478f4", + "metadata": {}, + "source": [ + "Three you might really train; two are **constructions** whose right score you know in\n", + "advance, so they test the metrics rather than the models.\n", + "\n", + "| Model | What it does |\n", + "|---|---|\n", + "| `honest` | the optimal design for each test condition |\n", + "| `lookup_table` | nearest neighbour over the training set; never sees a withheld design |\n", + "| `unconditional` | real training designs, ignoring the conditions |\n", + "| `permuted_conditions` | the correct withheld designs shuffled across conditions — every design a true optimum, every one for the wrong conditions |\n", + "| `noise` | uniform noise |" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "6bfdaf3f", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.055336Z", + "iopub.status.busy": "2026-09-18T13:05:45.055283Z", + "iopub.status.idle": "2026-09-18T13:05:45.057134Z", + "shell.execute_reply": "2026-09-18T13:05:45.056941Z" + } + }, + "outputs": [], + "source": [ + "def lookup_table(c):\n", + " return TRAIN[np.argmin(np.abs(train_conditions - c))]\n", + "\n", + "\n", + "MODELS = {\n", + " \"honest\": np.stack([optimal_design(c) for c in test_conditions]),\n", + " \"lookup_table\": np.stack([lookup_table(c) for c in test_conditions]),\n", + " \"unconditional\": TRAIN[rng.integers(0, len(TRAIN), len(REFERENCE))],\n", + " \"permuted_conditions\": REFERENCE[rng.permutation(len(REFERENCE))],\n", + " \"noise\": rng.random(REFERENCE.shape),\n", + "}" + ] + }, + { + "cell_type": "markdown", + "id": "8571bcc1", + "metadata": {}, + "source": [ + "## 4. Score them" + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "935e4407", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.058139Z", + "iopub.status.busy": "2026-09-18T13:05:45.058081Z", + "iopub.status.idle": "2026-09-18T13:05:45.063325Z", + "shell.execute_reply": "2026-09-18T13:05:45.063137Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmddppnoveltycopy_ratecond_sensviol
honest0.0000086.710593e-560.0136900.0NaN0.0
lookup_table0.0000196.642568e-630.0000001.0NaN0.0
unconditional0.0026061.044406e-610.0000001.0NaN0.0
permuted_conditions0.0000003.381911e-560.0000001.0NaN0.0
noise0.0060436.005203e-180.2852790.0NaN0.0
reference (split-half)0.0166756.678526e-240.0146030.0NaN0.0
\n", + "
" + ], + "text/plain": [ + " mmd dpp novelty copy_rate \\\n", + "honest 0.000008 6.710593e-56 0.013690 0.0 \n", + "lookup_table 0.000019 6.642568e-63 0.000000 1.0 \n", + "unconditional 0.002606 1.044406e-61 0.000000 1.0 \n", + "permuted_conditions 0.000000 3.381911e-56 0.000000 1.0 \n", + "noise 0.006043 6.005203e-18 0.285279 0.0 \n", + "reference (split-half) 0.016675 6.678526e-24 0.014603 0.0 \n", + "\n", + " cond_sens viol \n", + "honest NaN 0.0 \n", + "lookup_table NaN 0.0 \n", + "unconditional NaN 0.0 \n", + "permuted_conditions NaN 0.0 \n", + "noise NaN 0.0 \n", + "reference (split-half) NaN 0.0 " + ] + }, + "execution_count": 4, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from engiopt.evaluation.board import Board\n", + "\n", + "board = Board(problem, reference=REFERENCE, train=TRAIN)\n", + "board.evaluate(MODELS)" + ] + }, + { + "cell_type": "markdown", + "id": "f8bd45e6", + "metadata": {}, + "source": [ + "One row per model, one column per metric — and a last row that is **not a model**.\n", + "`reference (split-half)` is half of the withheld optimal designs scored against the\n", + "other half: what real, correct designs score on every column, on this problem. An\n", + "`mmd` of 0.0006 means nothing on its own; next to what real designs score, it does.\n", + "\n", + "Ask the board to read itself:" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "b7bb486e", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.064420Z", + "iopub.status.busy": "2026-09-18T13:05:45.064362Z", + "iopub.status.idle": "2026-09-18T13:05:45.067438Z", + "shell.execute_reply": "2026-09-18T13:05:45.067243Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questiondirectionpicksreal designs score
mmdTaken as a set, how different are the generated designs from the reference optimal designs?lower is betterpermuted_conditions1.667532e-02
dppHow spread out are the generated designs, measured as the volume their similarity kernel spans?diagnostic6.678526e-24
noveltyHow close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?diagnostic1.460301e-02
copy_rateHow close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?diagnostic0.000000e+00
cond_sensWhen the same model is given different conditions, does its output change?diagnosticNaN
violWhat fraction of the generated designs violate the problem's constraints?lower is bettertie: honest, lookup_table, unconditional, permuted_conditions, noise0.000000e+00
\n", + "
" + ], + "text/plain": [ + " question \\\n", + "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", + "dpp How spread out are the generated designs, measured as the volume their similarity kernel spans? \n", + "novelty How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", + "copy_rate How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", + "cond_sens When the same model is given different conditions, does its output change? \n", + "viol What fraction of the generated designs violate the problem's constraints? \n", + "\n", + " direction \\\n", + "mmd lower is better \n", + "dpp diagnostic \n", + "novelty diagnostic \n", + "copy_rate diagnostic \n", + "cond_sens diagnostic \n", + "viol lower is better \n", + "\n", + " picks \\\n", + "mmd permuted_conditions \n", + "dpp \n", + "novelty \n", + "copy_rate \n", + "cond_sens \n", + "viol tie: honest, lookup_table, unconditional, permuted_conditions, noise \n", + "\n", + " real designs score \n", + "mmd 1.667532e-02 \n", + "dpp 6.678526e-24 \n", + "novelty 1.460301e-02 \n", + "copy_rate 0.000000e+00 \n", + "cond_sens NaN \n", + "viol 0.000000e+00 " + ] + }, + "execution_count": 5, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "board.explain()" + ] + }, + { + "cell_type": "markdown", + "id": "bf0217c3", + "metadata": {}, + "source": [ + "## 5. What the board is telling you" + ] + }, + { + "cell_type": "markdown", + "id": "ef271490", + "metadata": {}, + "source": [ + "**`mmd` picks `permuted_conditions`.** It is handed the correct withheld optima with the\n", + "conditions scrambled, and the distribution metric rates it *first* — a perfect 0.0,\n", + "better than the honest model and better than real designs against real designs. MMD\n", + "compares two *sets*; it has no access to which design was produced for which\n", + "conditions. Every set-level metric shares this by construction.\n", + "\n", + "**`lookup_table` scores almost as well as `honest` on `mmd`, having never seen a\n", + "withheld design.** Not by cheating: training and test optima come from the same\n", + "distribution, and a distribution metric only compares distributions. A lookup table is a\n", + "legitimately strong baseline; what is not legitimate is reading its `mmd` as evidence\n", + "it learned anything.\n", + "\n", + "**`copy_rate` is 1.0 for all three constructions** and `novelty` is 0.0. One diagnostic\n", + "column separates every construction from the honest model. That is what the\n", + "`memorization` and `conditions` families are for, and why a board that reports only\n", + "distribution metrics cannot be trusted.\n", + "\n", + "**`dpp` reads between 10⁻⁶³ and 10⁻¹⁸.** Those are not diversity measurements; they\n", + "are a determinant collapsing toward underflow. That is why it is diagnostic, and why\n", + "`vendi` — an effective count of distinct designs — replaces it.\n", + "\n", + "**`cond_sens` is `NaN`.** It asks whether the output *changes* when the conditions\n", + "change, which needs the live model re-sampled; a saved array cannot answer it.\n", + "`Board` scores designs; `Evaluator.score(generator)` scores a model and fills it in." + ] + }, + { + "cell_type": "markdown", + "id": "4abf824b", + "metadata": {}, + "source": [ + "## 6. Rank — but only on a column that can be ranked" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "dd93537a", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.068493Z", + "iopub.status.busy": "2026-09-18T13:05:45.068440Z", + "iopub.status.idle": "2026-09-18T13:05:45.071429Z", + "shell.execute_reply": "2026-09-18T13:05:45.071251Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmddppnoveltycopy_ratecond_sensviol
permuted_conditions0.0000003.381911e-560.0000001.0NaN0.0
honest0.0000086.710593e-560.0136900.0NaN0.0
lookup_table0.0000196.642568e-630.0000001.0NaN0.0
unconditional0.0026061.044406e-610.0000001.0NaN0.0
noise0.0060436.005203e-180.2852790.0NaN0.0
reference (split-half)0.0166756.678526e-240.0146030.0NaN0.0
\n", + "
" + ], + "text/plain": [ + " mmd dpp novelty copy_rate \\\n", + "permuted_conditions 0.000000 3.381911e-56 0.000000 1.0 \n", + "honest 0.000008 6.710593e-56 0.013690 0.0 \n", + "lookup_table 0.000019 6.642568e-63 0.000000 1.0 \n", + "unconditional 0.002606 1.044406e-61 0.000000 1.0 \n", + "noise 0.006043 6.005203e-18 0.285279 0.0 \n", + "reference (split-half) 0.016675 6.678526e-24 0.014603 0.0 \n", + "\n", + " cond_sens viol \n", + "permuted_conditions NaN 0.0 \n", + "honest NaN 0.0 \n", + "lookup_table NaN 0.0 \n", + "unconditional NaN 0.0 \n", + "noise NaN 0.0 \n", + "reference (split-half) NaN 0.0 " + ] + }, + "execution_count": 6, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "board.rank(\"mmd\")" + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "id": "6150f040", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.072411Z", + "iopub.status.busy": "2026-09-18T13:05:45.072357Z", + "iopub.status.idle": "2026-09-18T13:05:45.073756Z", + "shell.execute_reply": "2026-09-18T13:05:45.073568Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "'copy_rate' is a diagnostic: it says whether the other columns mean what they look like. Read it beside them; do not rank on it.\n" + ] + } + ], + "source": [ + "try:\n", + " board.rank(\"copy_rate\")\n", + "except ValueError as e:\n", + " print(e)" + ] + }, + { + "cell_type": "markdown", + "id": "1ed1fa6a", + "metadata": {}, + "source": [ + "## 7. Changing how a column is aggregated" + ] + }, + { + "cell_type": "markdown", + "id": "1defe64e", + "metadata": {}, + "source": [ + "`iog`, `cog`, `fog`, `novelty` and `cond_sens` each produce **one value per design**.\n", + "Which single number that becomes is a policy, not a measurement: a mean is dragged by\n", + "one diverged design, a median is not. It is an argument to `evaluate` — and, for a\n", + "published board, a field of the frozen spec recorded in every row — so there are no\n", + "`*_median` columns." + ] + }, + { + "cell_type": "code", + "execution_count": 8, + "id": "36e87484", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.074790Z", + "iopub.status.busy": "2026-09-18T13:05:45.074739Z", + "iopub.status.idle": "2026-09-18T13:05:45.078923Z", + "shell.execute_reply": "2026-09-18T13:05:45.078735Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
novelty (mean)novelty (median)
honest0.0136900.013665
one_diverged0.0311040.013665
reference (split-half)0.0146030.015140
\n", + "
" + ], + "text/plain": [ + " novelty (mean) novelty (median)\n", + "honest 0.013690 0.013665\n", + "one_diverged 0.031104 0.013665\n", + "reference (split-half) 0.014603 0.015140" + ] + }, + "execution_count": 8, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "one_diverged = MODELS[\"honest\"].copy()\n", + "one_diverged[0] = rng.random((8, 10)) # fifteen good designs and one that went wrong\n", + "\n", + "compare = {\"honest\": MODELS[\"honest\"], \"one_diverged\": one_diverged}\n", + "pd.DataFrame({\n", + " \"novelty (mean)\": board.evaluate(compare)[\"novelty\"],\n", + " \"novelty (median)\": board.evaluate(compare, aggregation=\"median\")[\"novelty\"],\n", + "})" + ] + }, + { + "cell_type": "markdown", + "id": "a4afdca0", + "metadata": {}, + "source": [ + "## 8. Asking the same questions in another space" + ] + }, + { + "cell_type": "markdown", + "id": "15759e0d", + "metadata": {}, + "source": [ + "Where a metric is computed is also an argument, not a property of the metric. `pca`\n", + "projects every design onto the leading principal components of the reference designs\n", + "before scoring, and suffixes the columns. Metrics that need the actual designs — a\n", + "constraint check, a copy corpus — are skipped there, because a constraint check on PCA\n", + "coordinates would be meaningless.\n", + "\n", + "This is the control that asks whether a *learned* space is needed at all: if PCA of the\n", + "data answers the same questions, any rotation of the same width would do. Learned\n", + "latent spaces (`space=\"lv\"`) arrive with the LVAE." + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "id": "ef9a97f4", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.079967Z", + "iopub.status.busy": "2026-09-18T13:05:45.079915Z", + "iopub.status.idle": "2026-09-18T13:05:45.082934Z", + "shell.execute_reply": "2026-09-18T13:05:45.082770Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmd@pcadpp@pca
honest1.354329e-063.869144e-72
lookup_table4.033555e-067.335045e-74
unconditional2.592173e-033.266329e-74
permuted_conditions2.220446e-161.128691e-64
noise2.043085e-039.665059e-48
reference (split-half)1.667106e-021.389859e-25
\n", + "
" + ], + "text/plain": [ + " mmd@pca dpp@pca\n", + "honest 1.354329e-06 3.869144e-72\n", + "lookup_table 4.033555e-06 7.335045e-74\n", + "unconditional 2.592173e-03 3.266329e-74\n", + "permuted_conditions 2.220446e-16 1.128691e-64\n", + "noise 2.043085e-03 9.665059e-48\n", + "reference (split-half) 1.667106e-02 1.389859e-25" + ] + }, + "execution_count": 9, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "board.evaluate(MODELS, space=\"pca\", pca_dims=8)" + ] + }, + { + "cell_type": "markdown", + "id": "a7bf45d2", + "metadata": {}, + "source": [ + "## 9. Registering a metric of your own" + ] + }, + { + "cell_type": "markdown", + "id": "086b8d09", + "metadata": {}, + "source": [ + "A metric is a function of an `EvaluationContext`, a family, a cost, and a direction.\n", + "The first line of its docstring is the question readers see. `ctx.reduce` applies\n", + "whatever aggregation the caller chose, so you never hard-code a mean." + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "id": "1adaa8bf", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-18T13:05:45.084068Z", + "iopub.status.busy": "2026-09-18T13:05:45.084017Z", + "iopub.status.idle": "2026-09-18T13:05:45.087304Z", + "shell.execute_reply": "2026-09-18T13:05:45.087133Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questiondirectionpicksreal designs score
mmdTaken as a set, how different are the generated designs from the reference optimal designs?lower is betterpermuted_conditions0.016675
density_ratioHow much material do the generated designs use, relative to the reference optima?diagnostic0.766460
\n", + "
" + ], + "text/plain": [ + " question \\\n", + "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", + "density_ratio How much material do the generated designs use, relative to the reference optima? \n", + "\n", + " direction picks real designs score \n", + "mmd lower is better permuted_conditions 0.016675 \n", + "density_ratio diagnostic 0.766460 " + ] + }, + "execution_count": 10, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from engiopt.evaluation.registry import register_metric\n", + "\n", + "\n", + "@register_metric(\"density_ratio\", family=\"feasibility\", cost=\"cheap\", higher_is_better=None)\n", + "def density_ratio(ctx):\n", + " \"\"\"How much material do the generated designs use, relative to the reference optima?\"\"\"\n", + " return float(ctx.reduce(ctx.gen_flat.mean(axis=1)) / ctx.ref_flat.mean())\n", + "\n", + "\n", + "board.evaluate(MODELS, metrics=[\"mmd\", \"density_ratio\"])\n", + "board.explain()" + ] + }, + { + "cell_type": "markdown", + "id": "3c274682", + "metadata": {}, + "source": [ + "It appears on the next board with its question and direction attached — nothing else\n", + "to wire up. Declared diagnostic here, because more material is not better or worse.\n", + "\n", + "Notice the reference row: two random halves of the *same* real designs do not score\n", + "1.0 against each other. On eight designs a side, that gap is the sampling noise of this\n", + "column on this problem — and it is the scale against which every model's deviation from\n", + "1.0 has to be read. That is what the reference row is for." + ] + }, + { + "cell_type": "markdown", + "id": "853f8773", + "metadata": {}, + "source": [ + "## 10. Not in this notebook" + ] + }, + { + "cell_type": "markdown", + "id": "9500df97", + "metadata": {}, + "source": [ + "- **Physics.** `iog`, `cog`, `fog` need a simulator. With a real problem,\n", + " `Evaluator.score(generator, include_expensive=True)` runs them.\n", + "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec.\n", + "- **Uncertainty.** Confidence intervals beside every value. On sixteen samples, most\n", + " gaps on this board are smaller than their own error bars." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.12.9" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} From 5a5a3171cad48a25c276ae69c9db760c20b8fd15 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 14:25:28 +0200 Subject: [PATCH 03/18] Port the metric suite onto the one-decorator contract Twenty metrics, each one question: the set-level ones (mmd, coverage, vendi, dpp), the per-condition ones (per_condition_distance, volume_error), feasibility, memorization (train_distance, copy_rate), cond_sens, the optimality gaps, four call-budget metrics that price a defect in optimizer calls (calls_to_settle, gap_after_calls, reaches_reference_rate, first_call_gain), and three cost columns. Names say what they measure. `novelty` split into `train_distance` (nearest training design) and `copy_rate` (the submission gate, against everything the model could have copied); `cond_err` became `volume_error`, since it was never about conditions in general; `recovery` became `gap_after_calls`, and `parity_rate` became `reaches_reference_rate`. `calls_to_parity` is gone: a copier reaches parity in zero calls. `dpp` keeps its name and becomes the n-th-root form, since the raw determinant reads 1e-20 on every real board. The context now carries each design's re-optimization trajectory, the generator's parameter count and training time, so the new metrics have what they read. Specs move to v2 with the full list and the aggregation policy; v1 stays committed so published rows can be read under the protocol that produced them, and only the current spec must name live metrics. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluate.py | 6 +- engiopt/evaluation/context.py | 18 ++ engiopt/evaluation/evaluator.py | 12 + engiopt/evaluation/metrics/builtin.py | 374 +++++++++++++++++++++---- engiopt/evaluation/spec.py | 4 +- engiopt/specs/beams2d/v2.json | 53 ++++ engiopt/specs/heatconduction2d/v2.json | 51 ++++ engiopt/specs/photonics2d/v2.json | 55 ++++ engiopt/specs/thermoelastic2d/v2.json | 56 ++++ tests/test_evaluation.py | 28 +- tests/test_leaderboard_integrity.py | 32 +-- tests/test_metrics_suite.py | 192 +++++++++++++ tests/test_specs.py | 12 +- 13 files changed, 813 insertions(+), 80 deletions(-) create mode 100644 engiopt/specs/beams2d/v2.json create mode 100644 engiopt/specs/heatconduction2d/v2.json create mode 100644 engiopt/specs/photonics2d/v2.json create mode 100644 engiopt/specs/thermoelastic2d/v2.json create mode 100644 tests/test_metrics_suite.py diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index bda57745..22eb3cb6 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -197,11 +197,7 @@ def _availability_label(count: int | None) -> str: def _print_metrics() -> None: """Print every registered metric grouped by the question it answers.""" print(f"{len(METRICS)} metrics registered:\n") - for family in sorted({spec.family for spec in METRICS.values()}): - print(f" [{family}]") - for spec in METRICS.select(family=family): - direction = {True: "higher better", False: "lower better", None: "diagnostic"}[spec.higher_is_better] - print(f" {spec.name:<10} {spec.cost:<10} {direction:<14} {spec.description}") + print(METRICS.explain().sort_values(["family", "cost"]).to_string()) def _resolve_generator_names(requested: tuple[str, ...], problem_id: str) -> list[str]: diff --git a/engiopt/evaluation/context.py b/engiopt/evaluation/context.py index a13dfdea..adeb8c6e 100644 --- a/engiopt/evaluation/context.py +++ b/engiopt/evaluation/context.py @@ -64,6 +64,8 @@ class OptimizationResults: shared instance. """ + trajectories: list[npt.NDArray[Any]] = field(default_factory=list) + """Per design, the optimality gap at every optimizer call of its re-optimization.""" iog: list[float] = field(default_factory=list) """Initial optimality gap: how far each generated design starts from the reference optimum.""" cog: list[float] = field(default_factory=list) @@ -122,6 +124,10 @@ class EvaluationContext: copy_corpus_fn: Callable[[], npt.NDArray[Any]] | None = None copy_tol: float = 0.01 aggregation: Literal["mean", "median"] = "mean" + model_params: int | None = None + """Trainable parameter count of the generator, when it is a network.""" + train_minutes: float | None = None + """Wall-clock minutes the generator took to train, when the checkpoint records it.""" """How a metric with one value per design is collapsed into a column; see `reduce`.""" resample_permuted: Callable[[npt.NDArray[Any]], npt.NDArray[Any]] | None = None @@ -172,6 +178,17 @@ def ref_flat(self) -> npt.NDArray[Any]: flattened = [np.asarray(spaces.flatten(self.problem.design_space, design)) for design in self.ref_designs] return np.asarray(flattened) + @cached_property + def train_designs(self) -> npt.NDArray[Any] | None: + """Flattened training designs, or None when the problem has no training split.""" + if self.copy_corpus_fn is None: + return None + train = np.asarray(self.copy_corpus_fn()) + if not len(train): + return None + flat = train.reshape(len(train), -1) + return flat if flat.shape[1] == self.gen_flat.shape[1] else None + @cached_property def copy_corpus(self) -> npt.NDArray[Any] | None: """Flattened designs a generator could have memorized, or None if there are none. @@ -354,6 +371,7 @@ def optimization(self) -> OptimizationResults: results.iog.append(self.scalarize_gap(np.asarray(generated_objective) - np.asarray(reference_optimum), i)) results.cog.append(sum(self.scalarize_gap(step_gap, i) for step_gap in gaps)) results.fog.append(self.scalarize_gap(gaps[-1], i)) + results.trajectories.append(np.asarray([self.scalarize_gap(step_gap, i) for step_gap in gaps], dtype=float)) return results @cached_property diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 8cb8ebca..f1d5966a 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -179,6 +179,8 @@ def context_for(self, generator: Generator) -> EvaluationContext: copy_corpus_fn=self.copy_corpus, copy_tol=self.spec.copy_tol, aggregation=self.spec.aggregation, + model_params=_parameter_count(generator), + train_minutes=getattr(generator, "train_minutes", None), resample_permuted=self._permuted_sampler(generator), ) @@ -356,6 +358,16 @@ def leaderboard( @functools.cache +def _parameter_count(generator: Any) -> int | None: + """Trainable parameters of the generator's network, or None for a model with none.""" + from torch import nn + + modules = [m for m in getattr(generator, "__dict__", {}).values() if isinstance(m, nn.Module)] + if not modules: + return None + return sum(p.numel() for module in modules for p in module.parameters() if p.requires_grad) + + def engibench_version() -> str: """Which EngiBench *ran* this evaluation, however it was installed. diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 756e7fa2..b091974e 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -32,10 +32,9 @@ family="distribution", cost="cheap", higher_is_better=False, - description="Taken as a set, how different are the generated designs from the reference optimal designs?", ) def mmd(ctx: EvaluationContext) -> float: - """Maximum Mean Discrepancy between generated and reference designs.""" + """Taken as a set, how different are the generated designs from the reference optimal designs?""" return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.sigma)) @@ -48,12 +47,22 @@ def mmd(ctx: EvaluationContext) -> float: "dpp", family="diversity", cost="cheap", - higher_is_better=None, - description="How spread out are the generated designs, measured as the volume their similarity kernel spans?", + higher_is_better=True, ) def dpp(ctx: EvaluationContext) -> float: - """Determinantal Point Process diversity of the generated designs.""" - return float(metrics_mod.dpp_diversity(ctx.gen_flat, sigma=ctx.sigma)) + """How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? + + The n-th root of the DPP kernel determinant, i.e. the geometric mean of the + kernel's eigenvalues. The raw determinant that earlier boards published under + this name is a product of n numbers below one and reads 1e-20 on every real + board; the n-th root is the same quantity on a scale that is bounded in + (0, 1] and comparable across sample sizes. Rows from spec v1 carry the raw + form and are not comparable with this column. + + A determinant still rises when a collapsed set is jittered, so it rewards + noise as diversity. `vendi` does not, and is the column to prefer. + """ + return float(metrics_mod.dpp_geometric_mean(ctx.gen_flat, sigma=ctx.sigma)) # ---------------------------------------------------------------------- @@ -62,46 +71,51 @@ def dpp(ctx: EvaluationContext) -> float: @register_metric( - "novelty", + "train_distance", + family="memorization", + cost="cheap", + higher_is_better=None, pixel_only=True, +) +def train_distance(ctx: EvaluationContext) -> float: + """How far is each generated design from the nearest design the model was trained on? + + Per-element RMS distance to the closest training design, so one tolerance + means the same thing on problems of different resolution. Zero means the + model reproduces its training set; large means only that the output is far + from the data, which noise also achieves. Read beside `mmd` and `viol`, + never alone -- which is why it declares no direction. + + NaN when the problem has no training split to compare against. + """ + if ctx.train_designs is None: + return float("nan") + from scipy.spatial.distance import cdist + + nearest = cdist(ctx.gen_flat, ctx.train_designs).min(axis=1) / np.sqrt(ctx.gen_flat.shape[1]) + return ctx.reduce(nearest) + + +@register_metric( + "copy_rate", family="memorization", cost="cheap", higher_is_better=None, - description="How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?", - outputs=("novelty", "copy_rate"), -) -def novelty(ctx: EvaluationContext) -> dict[str, float]: - """How far the generated designs sit from the corpus a model could copy from. - - The evaluation protocol is public: which conditions are scored, and the - dataset-optimal design for each one, can be recomputed by anyone from the - committed spec. A lookup table keyed on the condition vector therefore tops - `mmd`, `iog`, and `fog` -- not by cheating the implementation, but because - those metrics are *defined* as closeness to exactly the designs it returns. - No amount of care in the evaluator changes that; the only defense available - to a public board is to measure retrieval and say so. - - Two columns, because they answer different questions: - - - `novelty` -- mean per-element RMS distance to the nearest corpus design. - Diagnostic, deliberately: it has no good direction. Zero means the model - is a retrieval system, but large means only that the output is far from - the data, which pure noise also achieves. Read it next to `mmd` and - `viol`, never on its own. - - `copy_rate` -- the fraction of designs closer than `copy_tol`, i.e. the - share of this batch that is a reproduction rather than a generation. This - is the one that gates a submission. - - Returns NaN when no corpus is available, rather than claiming novelty that - was never checked. + pixel_only=True, +) +def copy_rate(ctx: EvaluationContext) -> float: + """What fraction of the generated designs are copies of a design the model could have seen? + + A design counts as a copy when it sits within `copy_tol` (per-element RMS) of + any design in the corpus a model could reproduce: the training split *and* + the scored reference optima, because the evaluation protocol is public and a + lookup table keyed on the conditions can return exactly those. This is the + column that gates a submission. NaN when no corpus is available. """ distances = ctx.nearest_corpus_distance if distances is None: - return {"novelty": float("nan"), "copy_rate": float("nan")} - return { - "novelty": ctx.reduce(distances), - "copy_rate": float(np.mean(distances < ctx.copy_tol)), - } + return float("nan") + return float(np.mean(distances < ctx.copy_tol)) # ---------------------------------------------------------------------- @@ -115,10 +129,11 @@ def novelty(ctx: EvaluationContext) -> dict[str, float]: family="conditions", cost="cheap", higher_is_better=None, - description="When the same model is given different conditions, does its output change?", ) def cond_sens(ctx: EvaluationContext) -> float: - """Mean per-element RMS change in output when each sample is given another's conditions. + """When the same model is given different conditions, does its output change? + + Mean per-element RMS change in output when each sample is given another's conditions. Sampled from the same seed, so the latent draw is held fixed and the only thing that varies is the conditions. A model that ignores them returns @@ -159,10 +174,9 @@ def cond_sens(ctx: EvaluationContext) -> float: family="feasibility", cost="cheap", higher_is_better=False, - description="What fraction of the generated designs violate the problem's constraints?", ) def viol(ctx: EvaluationContext) -> float: - """Fraction of infeasible designs; see `EvaluationContext.is_infeasible`. + """What fraction of the generated designs violate the problem's constraints? Defined for every problem: `problem.check_constraints` always applies, and the spec's `volume_condition` adds the volume-fraction budget for problems @@ -187,10 +201,9 @@ def viol(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="How much worse than the reference optimum is the generated design, exactly as generated?", ) def iog(ctx: EvaluationContext) -> float: - """Mean initial optimality gap, before any re-optimization.""" + """How much worse than the reference optimum is the generated design, exactly as generated?""" return ctx.reduce(ctx.optimization.iog) @@ -199,10 +212,9 @@ def iog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step?", ) def cog(ctx: EvaluationContext) -> float: - """Mean cumulative optimality gap. + """Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? For each generated design, the optimizer is started from that design and `objective(step) - objective(reference optimum)` is summed over every step it @@ -221,8 +233,272 @@ def cog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?", ) def fog(ctx: EvaluationContext) -> float: - """Mean final optimality gap.""" + """Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?""" return ctx.reduce(ctx.optimization.fog) + + +# ---------------------------------------------------------------------- +# Conditions: did each design answer the conditions it was given? +# ---------------------------------------------------------------------- + + +@register_metric( + "per_condition_distance", + family="conditions", + cost="cheap", + higher_is_better=False, +) +def per_condition_distance(ctx: EvaluationContext) -> float: + """For each set of conditions, how far is the generated design from the reference optimum for those same conditions? + + Design `i` is compared with reference design `i`, which answers the same + conditions. That is a sharper question than any set-level metric can ask: + not "does the set look right" but "is *this* design what *these* conditions + called for". A model that returns the correct designs in the wrong order + scores a perfect `mmd` and a large value here. Per-element RMS. + """ + if ctx.gen_flat.shape != ctx.ref_flat.shape: + return float("nan") + distances = np.linalg.norm(ctx.gen_flat - ctx.ref_flat, axis=1) / np.sqrt(ctx.gen_flat.shape[1]) + return ctx.reduce(distances) + + +@register_metric( + "volume_error", + family="conditions", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def volume_error(ctx: EvaluationContext) -> float: + """How far is each design's material fraction from the one its conditions requested? + + The volume fraction of a density field is its mean, so this needs nothing + fitted: it is the exact error on the one condition that can be read straight + off the design. `viol` reports how many designs missed the budget by more + than the tolerance; this reports by how much. NaN on problems with no volume + condition. + """ + if ctx.volume_condition is None or ctx.conditions is None: + return float("nan") + requested = np.asarray(ctx.conditions[ctx.volume_condition], dtype=np.float64) + realized = ctx.gen_flat.mean(axis=1) + return ctx.reduce(np.abs(realized - requested)) + + +# ---------------------------------------------------------------------- +# Distribution and diversity: does the set reach the reference, and is it varied? +# ---------------------------------------------------------------------- + +COVERAGE_QUANTILE = 0.05 +"""The radius around each reference design is set from the reference set's own +nearest-neighbor spacing, at this upper quantile, so it adapts to the space.""" + + +@register_metric( + "coverage", + family="distribution", + cost="cheap", + higher_is_better=True, +) +def coverage(ctx: EvaluationContext) -> float: + """What fraction of the reference optima have a generated design nearby? + + Distribution distance can look healthy while whole regions go unvisited, so + this counts reference designs directly: a mode the generator never produces + leaves its neighborhood empty. Bounded in [0, 1]. + """ + from scipy.spatial.distance import cdist + + within_reference = cdist(ctx.ref_flat, ctx.ref_flat) + np.fill_diagonal(within_reference, np.inf) + radius = float(np.quantile(within_reference.min(axis=1), 1.0 - COVERAGE_QUANTILE)) + nearest_generated = cdist(ctx.ref_flat, ctx.gen_flat).min(axis=1) + return float((nearest_generated <= radius).mean()) + + +@register_metric( + "vendi", + family="diversity", + cost="cheap", + higher_is_better=True, +) +def vendi(ctx: EvaluationContext) -> float: + """How many genuinely distinct designs are in the generated set, as an effective count? + + The exponential of the entropy of the similarity kernel's eigenvalues, so it + reads as a count: n identical designs score 1, n mutually dissimilar ones + score n. Unlike a determinant it does not rise when a collapsed set is + jittered, which is why it is the diversity column to prefer. + Friedman & Dieng (2023), The Vendi Score. + """ + return float(metrics_mod.vendi_score(ctx.gen_flat, sigma=ctx.sigma)) + + +# ---------------------------------------------------------------------- +# Performance: what does the optimizer still have to do after the warm start? +# ---------------------------------------------------------------------- +# +# `iog` asks whether the generated design is already good. That is the wrong +# question whenever a defect is cheap to repair, and many are: a design at the +# wrong filter length scale is corrected by the first few updates. The metrics +# below price a defect in the currency an engineer pays, which is optimizer +# calls, and read the same trajectory `iog`, `cog` and `fog` summarize. + +SETTLE_BAND = 0.05 +"""Fraction of a design's own achievable improvement within which it counts as settled.""" + +GAP_AFTER_CALLS = (1, 2, 5, 10) +"""Call budgets to report the remaining gap at: is a defect gone immediately, or not?""" + + +def _calls_to_settle(path: np.ndarray, band: float = SETTLE_BAND) -> float: + """Calls after which `path` stays within `band` of its final value (last exit, not first touch).""" + if path.size < 2: # noqa: PLR2004 - a one-step path has no convergence to measure + return 0.0 + final = float(path[-1]) + reach = max(abs(float(path[0]) - final), abs(final), 1e-12) + outside = np.flatnonzero(np.abs(path - final) > band * reach) + return float(outside[-1] + 1) if outside.size else 0.0 + + +@register_metric( + "calls_to_settle", + family="performance", + cost="expensive", + higher_is_better=False, +) +def calls_to_settle(ctx: EvaluationContext) -> float: + """How many optimizer calls until the objective stays within 5% of where it ends up? + + Settling time in the control-theory sense: last exit from the band, so a + trajectory that dips in and leaves again is not credited. Zero means the + design started already settled. Each call is one simulate plus one + sensitivity evaluation. + """ + paths = ctx.optimization.trajectories + return ctx.reduce([_calls_to_settle(path) for path in paths]) if paths else float("nan") + + +@register_metric( + "gap_after_calls", + family="performance", + cost="expensive", + higher_is_better=False, + outputs=tuple(f"gap_after_{k}_calls" for k in GAP_AFTER_CALLS), +) +def gap_after_calls(ctx: EvaluationContext) -> dict[str, float]: + """How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls? + + A path shorter than the budget has converged early, so its final value is + carried forward: the optimizer would have spent the remaining calls and + changed nothing. + """ + paths = ctx.optimization.trajectories + if not paths: + return {f"gap_after_{k}_calls": float("nan") for k in GAP_AFTER_CALLS} + return { + f"gap_after_{k}_calls": ctx.reduce([float(path[min(k, path.size) - 1]) for path in paths]) for k in GAP_AFTER_CALLS + } + + +@register_metric( + "reaches_reference_rate", + family="performance", + cost="expensive", + higher_is_better=True, +) +def reaches_reference_rate(ctx: EvaluationContext) -> float: + """What fraction of the generated designs, once re-optimized, reach or beat the reference optimum? + + The gap is measured against the reference optimum, so reaching it means the + gap touches zero at some call. A rate rather than a call count, because a + count is censored by designs that never get there, and a copier reaches + parity in zero calls. Bounded in [0, 1]. + """ + paths = ctx.optimization.trajectories + if not paths: + return float("nan") + return float(np.mean([bool(np.any(np.asarray(path) <= 0.0)) for path in paths])) + + +@register_metric( + "first_call_gain", + family="performance", + cost="expensive", + higher_is_better=True, +) +def first_call_gain(ctx: EvaluationContext) -> float: + """What fraction of the whole re-optimization's improvement does the first optimizer call deliver? + + Near 1 means the design's defects clear immediately and the rest is + refinement; near 0 means the warm start bought nothing. Designs with nothing + to gain are skipped, since the fraction is undefined there. + """ + gains = [] + for path in ctx.optimization.trajectories: + if path.size < 2: # noqa: PLR2004 - needs a first step to have a first-step gain + continue + achievable = float(path[0]) - float(path[-1]) + if abs(achievable) < 1e-12: # noqa: PLR2004 - nothing to improve, so no fraction of it exists + continue + gains.append((float(path[0]) - float(path[1])) / achievable) + return ctx.reduce(gains) if gains else float("nan") + + +# ---------------------------------------------------------------------- +# Cost: what does the model cost to run, and what did it cost to make? +# ---------------------------------------------------------------------- + + +@register_metric( + "generation_seconds", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def generation_seconds(ctx: EvaluationContext) -> float: + """How many seconds did the model take to generate this batch of designs? + + Wall-clock, so it compares models within one run on one machine, not across + machines. NaN when the designs did not come from a timed `Generator.sample`. + """ + return float("nan") if ctx.sample_seconds is None else float(ctx.sample_seconds) + + +@register_metric( + "n_parameters", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def n_parameters(ctx: EvaluationContext) -> float: + """How many trainable parameters does the model have? + + The hardware-independent companion to `generation_seconds`. Neither is a + complete account of cost alone: a small diffusion model can be slower than + a large GAN because it samples iteratively. NaN for a model with no network. + """ + return float("nan") if ctx.model_params is None else float(ctx.model_params) + + +@register_metric( + "train_minutes", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def train_minutes(ctx: EvaluationContext) -> float: + """How many minutes did the model take to train? + + Prices the decision to adopt the method rather than a forward pass; it is + where a lookup table and a diffusion model differ by orders of magnitude. + NaN unless the checkpoint records it, which today none do -- that gap is + itself a finding. + """ + return float("nan") if ctx.train_minutes is None else float(ctx.train_minutes) diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index bba0e4ec..79f90acf 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -134,7 +134,7 @@ class EvalSpec: """ problem_id: str - version: str = "v1" + version: str = "v2" n_samples: int = 50 condition_seed: int = 1 metrics: tuple[str, ...] = ("mmd", "dpp", "novelty", "cond_sens", "viol", "iog", "cog", "fog") @@ -184,7 +184,7 @@ def load(cls, reference: str, *, root: Path | None = None) -> EvalSpec: if path.suffix == ".json" and path.exists(): return cls(**json.loads(path.read_text())) problem_id, _, version = reference.partition("/") - version = version or "v1" + version = version or "v2" spec_path = (root or SPEC_ROOT) / problem_id / f"{version}.json" if not spec_path.exists(): raise FileNotFoundError( diff --git a/engiopt/specs/beams2d/v2.json b/engiopt/specs/beams2d/v2.json new file mode 100644 index 00000000..bcd79592 --- /dev/null +++ b/engiopt/specs/beams2d/v2.json @@ -0,0 +1,53 @@ +{ + "condition_digest": "f65e2c02551e7e08", + "condition_seed": 1, + "copy_corpus_size": 512, + "copy_tol": 0.01, + "dataset_id": "IDEALLab/beams_2d_50_100_v0", + "dataset_revision": "ccb64d4f09cf66a5ed9ba2081438d695b9e2438d", + "engibench_version": "0.2.0+0a028c03d02b", + "max_copy_rate": 0.5, + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "copy_rate", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_settle", + "gap_after_calls", + "reaches_reference_rate", + "first_call_gain" + ], + "n_samples": 50, + "notes": "Feasibility = EngiBench constraints plus the volfrac budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "volfrac", + "rmin", + "forcedist", + "overhang_constraint" + ], + "problem_id": "beams2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": 10.0, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volfrac", + "aggregation": "mean" +} diff --git a/engiopt/specs/heatconduction2d/v2.json b/engiopt/specs/heatconduction2d/v2.json new file mode 100644 index 00000000..ac4bffb5 --- /dev/null +++ b/engiopt/specs/heatconduction2d/v2.json @@ -0,0 +1,51 @@ +{ + "condition_digest": "40030d1b2482e003", + "condition_seed": 1, + "copy_corpus_size": 512, + "copy_tol": 0.01, + "dataset_id": "IDEALLab/heat_conduction_2d_v0", + "dataset_revision": "9f07c1e70d244f783e34a71c96db621173871dc1", + "engibench_version": "0.2.0+0a028c03d02b", + "max_copy_rate": 0.5, + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "copy_rate", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_settle", + "gap_after_calls", + "reaches_reference_rate", + "first_call_gain" + ], + "n_samples": 50, + "notes": "Feasibility = EngiBench constraints plus the volume budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "volume", + "length" + ], + "problem_id": "heatconduction2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": 10.0, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volume", + "aggregation": "mean" +} diff --git a/engiopt/specs/photonics2d/v2.json b/engiopt/specs/photonics2d/v2.json new file mode 100644 index 00000000..dee9c339 --- /dev/null +++ b/engiopt/specs/photonics2d/v2.json @@ -0,0 +1,55 @@ +{ + "condition_digest": "6b8330bee5a8718c", + "condition_seed": 1, + "copy_corpus_size": 512, + "copy_tol": 0.01, + "dataset_id": "IDEALLab/photonics_2d_120_120_v1", + "dataset_revision": "be811ff69d67cd8c79148e29f2a68d293ec00955", + "engibench_version": "0.2.0+0a028c03d02b", + "max_copy_rate": 0.5, + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "copy_rate", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_settle", + "gap_after_calls", + "reaches_reference_rate", + "first_call_gain" + ], + "n_samples": 50, + "notes": "No volume budget: feasibility is EngiBench's declared constraints alone. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "lambda1", + "lambda2", + "blur_radius", + "num_elems_x", + "num_elems_y", + "num_optimization_steps" + ], + "problem_id": "photonics2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": 10.0, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": null, + "aggregation": "mean" +} diff --git a/engiopt/specs/thermoelastic2d/v2.json b/engiopt/specs/thermoelastic2d/v2.json new file mode 100644 index 00000000..89120237 --- /dev/null +++ b/engiopt/specs/thermoelastic2d/v2.json @@ -0,0 +1,56 @@ +{ + "condition_digest": "818d93c2a2d5ac25", + "condition_seed": 1, + "copy_corpus_size": 512, + "copy_tol": 0.01, + "dataset_id": "IDEALLab/thermoelastic_2d_v1", + "dataset_revision": "7d5b4a694bd1a8f3f25fd745ac9f669764e29ea5", + "engibench_version": "0.2.0+0a028c03d02b", + "max_copy_rate": 0.5, + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "copy_rate", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_settle", + "gap_after_calls", + "reaches_reference_rate", + "first_call_gain" + ], + "n_samples": 50, + "notes": "Objectives combined per-sample by the `weight` condition. Feasibility = EngiBench constraints plus the volume_fraction_target budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": "weight", + "objective_weights": null, + "problem_conditions": [ + "fixed_elements", + "force_elements_x", + "force_elements_y", + "heatsink_elements", + "volume_fraction_target", + "rmin", + "weight" + ], + "problem_id": "thermoelastic2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": 10.0, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volume_fraction_target", + "aggregation": "mean" +} diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index 5dad2313..7ad993db 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -89,12 +89,34 @@ def test_builtin_metrics_declare_their_cost() -> None: judged by a constraint check rather than by running the optimizer. The two integrity metrics are cheap too, and that matters more than it - sounds: `novelty` and `cond_sens` decide whether a row can be ranked at all, + sounds: `copy_rate` and `cond_sens` decide whether a row can be ranked at all, so a board that could only afford the cheap pass would otherwise have to rank models it had never checked for memorization. """ - assert {spec.name for spec in METRICS.select(cost="cheap")} == {"mmd", "dpp", "viol", "novelty", "cond_sens"} - assert {spec.name for spec in METRICS.select(cost="expensive")} == {"iog", "cog", "fog"} + assert {spec.name for spec in METRICS.select(cost="cheap")} == { + "mmd", + "dpp", + "viol", + "train_distance", + "copy_rate", + "cond_sens", + "per_condition_distance", + "volume_error", + "coverage", + "vendi", + "generation_seconds", + "n_parameters", + "train_minutes", + } + assert {spec.name for spec in METRICS.select(cost="expensive")} == { + "iog", + "cog", + "fog", + "calls_to_settle", + "gap_after_calls", + "reaches_reference_rate", + "first_call_gain", + } def test_cheap_metrics_never_touch_the_solver(fake_problem: Any) -> None: diff --git a/tests/test_leaderboard_integrity.py b/tests/test_leaderboard_integrity.py index 26b73da0..8d8e2d6f 100644 --- a/tests/test_leaderboard_integrity.py +++ b/tests/test_leaderboard_integrity.py @@ -63,10 +63,7 @@ def test_a_generator_returning_the_reference_designs_is_caught(fake_problem: Any ref = rng.random((10, *fake_problem.design_space.shape)) ctx = _context(fake_problem, ref.copy(), ref) - scores = METRICS["novelty"].fn(ctx) - - assert scores["copy_rate"] == 1.0 - assert scores["novelty"] == pytest.approx(0.0, abs=1e-12) + assert METRICS["copy_rate"].fn(ctx) == 1.0 # And it does indeed post a perfect distribution score, which is the point. assert METRICS["mmd"].fn(ctx) == pytest.approx(0.0, abs=1e-9) @@ -77,10 +74,7 @@ def test_a_generator_producing_its_own_designs_is_not_flagged(fake_problem: Any) ref = rng.random((10, *fake_problem.design_space.shape)) ctx = _context(fake_problem, rng.random((10, *fake_problem.design_space.shape)), ref) - scores = METRICS["novelty"].fn(ctx) - - assert scores["copy_rate"] == 0.0 - assert scores["novelty"] > 0 + assert METRICS["copy_rate"].fn(ctx) == 0.0 def test_copying_the_training_split_is_caught_too(fake_problem: Any) -> None: @@ -95,7 +89,8 @@ def test_copying_the_training_split_is_caught_too(fake_problem: Any) -> None: train = rng.random((20, *fake_problem.design_space.shape)) ctx = _context(fake_problem, train[:6].copy(), ref, copy_corpus_fn=lambda: train) - assert METRICS["novelty"].fn(ctx)["copy_rate"] == 1.0 + assert METRICS["copy_rate"].fn(ctx) == 1.0 + assert METRICS["train_distance"].fn(ctx) == pytest.approx(0.0, abs=1e-12) def test_near_copies_count_as_copies(fake_problem: Any) -> None: @@ -108,21 +103,20 @@ def test_near_copies_count_as_copies(fake_problem: Any) -> None: ref = rng.random((8, *fake_problem.design_space.shape)) barely_perturbed = ref + rng.normal(scale=1e-4, size=ref.shape) - scores = METRICS["novelty"].fn(_context(fake_problem, barely_perturbed, ref, copy_tol=0.01)) - - assert scores["copy_rate"] == 1.0 + assert METRICS["copy_rate"].fn(_context(fake_problem, barely_perturbed, ref, copy_tol=0.01)) == 1.0 -def test_novelty_is_diagnostic_and_cannot_be_ranked_on() -> None: - """Ranking on novelty would put pure noise in first place. +def test_memorization_metrics_are_diagnostic_and_cannot_be_ranked_on() -> None: + """Ranking on distance-to-training-data would put pure noise in first place. The metric has no good direction -- zero means retrieval, large means only "unlike the data", which is what a broken model also achieves. Registering a direction would have created a new thing to game while closing an old one. """ - assert METRICS["novelty"].higher_is_better is None + assert METRICS["train_distance"].higher_is_better is None + assert METRICS["copy_rate"].higher_is_better is None with pytest.raises(ValueError, match="diagnostic"): - rank(pd.DataFrame([{"problem_id": "p", "algo_id": "a", "novelty": 0.5}]), "novelty") + rank(pd.DataFrame([{"problem_id": "p", "algo_id": "a", "train_distance": 0.5}]), "train_distance") # ---------------------------------------------------------------------- @@ -404,7 +398,7 @@ def test_the_evaluator_detects_a_lookup_table_end_to_end(fake_problem: Any) -> N through the evaluator rather than against a hand-built context -- a metric that works but is never fed would pass every other test in this file. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "novelty")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "copy_rate")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) @@ -419,7 +413,7 @@ def test_the_evaluator_detects_a_lookup_table_end_to_end(fake_problem: Any) -> N def test_the_evaluator_leaves_an_honest_model_unflagged(fake_problem: Any) -> None: """The same path, for a model that generates rather than retrieves.""" - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "novelty", "cond_sens")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "copy_rate", "cond_sens")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) rng = np.random.default_rng(7) @@ -484,7 +478,7 @@ def test_the_copy_corpus_is_drawn_once_for_a_whole_sweep(fake_problem: Any) -> N Both halves matter: a per-model corpus would be slow, and a corpus that varied between models would make their `copy_rate` values incomparable. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("novelty",)) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("copy_rate",)) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) evaluator = _evaluator(fake_problem, spec, ref, conditions) diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py new file mode 100644 index 00000000..9b889f50 --- /dev/null +++ b/tests/test_metrics_suite.py @@ -0,0 +1,192 @@ +"""The ported metrics: each one on the toy structure every EngiBench dataset shares.""" + +from __future__ import annotations + +from typing import Any + +import numpy as np +import pytest + +from engiopt.evaluation.board import Board +from engiopt.evaluation.board import register_space +from engiopt.evaluation.board import SPACES +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.context import OptimizationResults +from engiopt.evaluation.registry import METRICS +from engiopt.evaluation.spec import EvalSpec + +# Importing the metrics package registers the built-ins. +import engiopt.evaluation.metrics # noqa: F401 # isort: skip + + +def _ctx(problem: Any, gen: np.ndarray, ref: np.ndarray, **kwargs: Any) -> EvaluationContext: + return EvaluationContext(problem=problem, problem_id="fake", gen_designs=gen, ref_designs=ref, **kwargs) + + +@pytest.fixture +def sets(fake_problem: Any) -> tuple[np.ndarray, np.ndarray]: + rng = np.random.default_rng(0) + shape = fake_problem.design_space.shape + return rng.random((12, *shape)), rng.random((12, *shape)) + + +# ---------------------------------------------------------------- diversity + + +def test_vendi_counts_distinct_designs(fake_problem: Any, sets: Any) -> None: + gen, _ = sets + collapsed = np.repeat(gen[:1], 12, axis=0) + assert METRICS["vendi"].fn(_ctx(fake_problem, collapsed, gen)) == pytest.approx(1.0) + spread = METRICS["vendi"].fn(_ctx(fake_problem, gen, gen, sigma=0.05)) + assert spread == pytest.approx(12.0, rel=0.05), "mutually dissimilar designs count as n" + + +def test_dpp_is_on_a_unit_scale_and_prefers_spread(fake_problem: Any, sets: Any) -> None: + gen, _ = sets + collapsed = np.repeat(gen[:1], 12, axis=0) + np.random.default_rng(1).normal(0, 1e-3, gen.shape) + low = METRICS["dpp"].fn(_ctx(fake_problem, collapsed, gen, sigma=1.0)) + high = METRICS["dpp"].fn(_ctx(fake_problem, gen, gen, sigma=1.0)) + assert 0.0 <= low < high <= 1.0 + + +# ------------------------------------------------------------- distribution + + +def test_coverage_is_one_for_the_reference_itself_and_zero_far_away(fake_problem: Any, sets: Any) -> None: + _, ref = sets + assert METRICS["coverage"].fn(_ctx(fake_problem, ref.copy(), ref)) == pytest.approx(1.0) + assert METRICS["coverage"].fn(_ctx(fake_problem, ref + 50.0, ref)) == pytest.approx(0.0) + + +# --------------------------------------------------------------- conditions + + +def test_per_condition_distance_sees_what_mmd_cannot(fake_problem: Any, sets: Any) -> None: + """The correct designs in the wrong order: a perfect set, every one for the wrong conditions.""" + _, ref = sets + permuted = ref[np.random.default_rng(2).permutation(len(ref))] + assert METRICS["mmd"].fn(_ctx(fake_problem, permuted, ref)) == pytest.approx(0.0, abs=1e-12) + assert METRICS["per_condition_distance"].fn(_ctx(fake_problem, permuted, ref)) > 0.1 + assert METRICS["per_condition_distance"].fn(_ctx(fake_problem, ref.copy(), ref)) == pytest.approx(0.0) + + +def test_volume_error_reads_the_material_fraction_off_the_design(fake_problem: Any) -> None: + shape = fake_problem.design_space.shape + gen = np.stack([np.full(shape, 0.3), np.full(shape, 0.5)]) + ctx = _ctx(fake_problem, gen, gen, conditions={"volfrac": np.array([0.3, 0.4])}, volume_condition="volfrac") + assert METRICS["volume_error"].fn(ctx) == pytest.approx(0.05) + assert np.isnan(METRICS["volume_error"].fn(_ctx(fake_problem, gen, gen))) + + +# ------------------------------------------------------------- memorization + + +def test_train_distance_and_copy_rate_ask_different_questions(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + train = np.random.default_rng(3).random((20, *fake_problem.design_space.shape)) + copies_train = _ctx(fake_problem, train[:12], ref, copy_corpus_fn=lambda: train) + assert METRICS["train_distance"].fn(copies_train) == pytest.approx(0.0) + assert METRICS["copy_rate"].fn(copies_train) == pytest.approx(1.0) + copies_reference = _ctx(fake_problem, ref.copy(), ref, copy_corpus_fn=lambda: train) + assert METRICS["train_distance"].fn(copies_reference) > 0.1, "the withheld optima are not training designs" + assert METRICS["copy_rate"].fn(copies_reference) == pytest.approx(1.0), "but they are in the copyable corpus" + assert np.isnan(METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref))) + + +# -------------------------------------------------------------- performance + + +@pytest.fixture +def trajectories(fake_problem: Any, sets: Any) -> EvaluationContext: + """Two re-optimization paths: one reaches the reference optimum, one never does.""" + gen, ref = sets + ctx = _ctx(fake_problem, gen[:2], ref[:2]) + reaches = np.array([5.0, 4.0, 3.0, 0.0, 0.0]) + never = np.array([5.0, 4.0, 3.0, 2.0, 1.0]) + ctx.__dict__["optimization"] = OptimizationResults(trajectories=[reaches, never]) + return ctx + + +def test_calls_to_settle_is_last_exit_from_the_band(trajectories: EvaluationContext) -> None: + # reaches: settles after call 3 (band 0.25 around 0); never: after call 4 (band 0.2 around 1). + assert METRICS["calls_to_settle"].fn(trajectories) == pytest.approx(3.5) + + +def test_gap_after_calls_carries_a_short_path_forward(trajectories: EvaluationContext) -> None: + gaps = METRICS["gap_after_calls"].fn(trajectories) + assert gaps["gap_after_1_calls"] == pytest.approx(5.0) + assert gaps["gap_after_2_calls"] == pytest.approx(4.0) + assert gaps["gap_after_5_calls"] == pytest.approx(0.5) + assert gaps["gap_after_10_calls"] == pytest.approx(0.5), "a five-call path is done at call ten" + + +def test_reaches_reference_rate_is_a_rate_not_a_censored_count(trajectories: EvaluationContext) -> None: + assert METRICS["reaches_reference_rate"].fn(trajectories) == pytest.approx(0.5) + + +def test_first_call_gain_is_the_first_step_over_the_whole_improvement(trajectories: EvaluationContext) -> None: + assert METRICS["first_call_gain"].fn(trajectories) == pytest.approx((0.2 + 0.25) / 2) + + +def test_trajectory_metrics_respect_the_aggregation_policy(trajectories: EvaluationContext) -> None: + trajectories.aggregation = "median" + assert METRICS["calls_to_settle"].fn(trajectories) == pytest.approx(3.5), "median of two equals their mean" + + +# --------------------------------------------------------------------- cost + + +def test_cost_metrics_are_nan_without_a_live_model(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + ctx = _ctx(fake_problem, gen, ref) + assert all(np.isnan(METRICS[name].fn(ctx)) for name in ("generation_seconds", "n_parameters", "train_minutes")) + timed = _ctx(fake_problem, gen, ref, sample_seconds=1.5, model_params=1000, train_minutes=30.0) + assert METRICS["generation_seconds"].fn(timed) == 1.5 + assert METRICS["n_parameters"].fn(timed) == 1000 + assert METRICS["train_minutes"].fn(timed) == 30.0 + + +# -------------------------------------------------------------------- board + + +def test_a_space_is_a_registered_projection(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + register_space("halves", lambda reference, width: lambda designs: designs.reshape(len(designs), -1)[:, ::2]) + frame = Board(fake_problem, reference=ref).evaluate({"m": gen}, space="halves") + assert "mmd@halves" in frame.columns + assert "viol@halves" not in frame.columns + del SPACES["halves"] + with pytest.raises(KeyError, match="Unknown space"): + Board(fake_problem, reference=ref).evaluate({"m": gen}, space="halves") + + +def test_the_reference_row_leaves_per_condition_metrics_blank(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + frame = Board(fake_problem, reference=ref).evaluate({"m": gen}) + assert np.isnan(frame.loc["reference (split-half)", "per_condition_distance"]) + assert not np.isnan(frame.loc["m", "per_condition_distance"]) + + +def test_from_generators_scores_through_an_evaluator() -> None: + class StubEvaluator: + def score(self, generator: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: + return {"mmd": generator, "iog": 1.0 if include_expensive else float("nan")} + + board = Board.from_generators(StubEvaluator(), {"a": 0.2, "b": 0.1}, expensive=True) + assert board.rank("mmd").index[0] == "b" + assert board.frame.loc["a", "iog"] == 1.0 + + +# -------------------------------------------------------------------- specs + + +@pytest.mark.parametrize("problem_id", ["beams2d", "heatconduction2d", "photonics2d", "thermoelastic2d"]) +def test_every_v2_spec_names_only_registered_metrics(problem_id: str) -> None: + spec = EvalSpec.load(f"{problem_id}/v2") + assert spec.version == "v2" + assert spec.aggregation == "mean" + assert set(spec.metrics) <= set(METRICS) + + +def test_the_default_spec_is_v2() -> None: + assert EvalSpec.load("beams2d").version == "v2" diff --git a/tests/test_specs.py b/tests/test_specs.py index 4279d10f..463a1092 100644 --- a/tests/test_specs.py +++ b/tests/test_specs.py @@ -24,6 +24,14 @@ SPEC_PATHS = sorted(SPEC_ROOT.glob("*/*.json")) SPEC_IDS = [f"{path.parent.name}/{path.stem}" for path in SPEC_PATHS] +CURRENT_PATHS = [ + max(SPEC_ROOT.glob(f"{problem}/*.json"), key=lambda p: p.stem) + for problem in sorted({p.parent.name for p in SPEC_PATHS}) +] +CURRENT_IDS = [f"{path.parent.name}/{path.stem}" for path in CURRENT_PATHS] +"""The newest spec per problem. Older versions stay committed so published rows can +be read under the protocol that produced them, and may name metrics since retired.""" + class _FakeConditions: """A stand-in for the sampled-conditions dataset the digest hashes.""" @@ -50,9 +58,9 @@ def test_committed_spec_loads_and_round_trips(path: Any) -> None: assert dataclasses.asdict(spec) == dataclasses.asdict(EvalSpec(**json.loads(path.read_text()))) -@pytest.mark.parametrize("path", SPEC_PATHS, ids=SPEC_IDS) +@pytest.mark.parametrize("path", CURRENT_PATHS, ids=CURRENT_IDS) def test_committed_spec_requests_registered_metrics(path: Any) -> None: - """A spec asking for a metric nobody registered would fail at evaluation time.""" + """A current spec asking for a metric nobody registered would fail at evaluation time.""" for metric in EvalSpec.load(str(path)).metrics: assert metric in METRICS, f"{path.name} requests unregistered metric {metric!r}" From 52246151e7541ca3fb70585de87762f854c163de Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 14:25:30 +0200 Subject: [PATCH 04/18] Let a board choose its space, and load published checkpoints by name `Board.evaluate(space=...)` looks the space up in a registry: `pixel` and `pca` are built in, and a learned latent space registers one projection function. Metrics that need the actual designs declare `pixel_only` and are skipped elsewhere. The kernel bandwidth defaults to the median pairwise distance of the reference designs in the chosen space, so the diversity columns can see anything on a problem the spec never pinned a value for. `Board.from_generators` scores loaded models through an `Evaluator`, which is what cond_sens, the cost columns and the physics need, and `Board.load` pulls published checkpoints by name -- the workshop in one call. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluation/board.py | 139 +++++++++++++++++++++++++++++++----- tests/test_board.py | 13 +++- 2 files changed, 134 insertions(+), 18 deletions(-) diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index a3c1ccc6..93fcfa42 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -15,6 +15,7 @@ from __future__ import annotations +from collections.abc import Callable from dataclasses import dataclass from dataclasses import field from typing import Any, Literal, TYPE_CHECKING @@ -22,6 +23,7 @@ import numpy as np import pandas as pd +from engiopt import metrics as metrics_mod from engiopt.evaluation.context import EvaluationContext from engiopt.evaluation.registry import MetricRegistry from engiopt.evaluation.registry import METRICS @@ -29,8 +31,23 @@ if TYPE_CHECKING: from collections.abc import Mapping -Space = Literal["pixel", "pca"] -"""Where the set-level metrics are computed. Learned latent spaces arrive with the LVAE.""" +SpaceFit = Callable[[np.ndarray, int], Callable[[np.ndarray], np.ndarray]] +"""Given the reference designs and a width, return a function projecting designs into the space.""" + +SPACES: dict[str, SpaceFit] = {} +"""Where the set-level metrics can be computed, by name. `pixel` and `pca` are built in; +a learned latent space registers itself here with `register_space`.""" + + +def register_space(name: str, fit: SpaceFit) -> None: + """Make a space available to `Board.evaluate(space=name)`. + + Args: + name: The space's name; also the column suffix (`mmd@name`). + fit: `fit(reference_designs, width)` returning `project(designs) -> codes`. + """ + SPACES[name] = fit + REFERENCE_ROW = "reference (split-half)" """Label of the row scoring half of the reference designs against the other half. @@ -49,14 +66,17 @@ class Board: `check_constraints`, which every EngiBench problem has. reference: The withheld optimal designs the models are scored against. train: The training designs, for the memorization metrics. Optional. - sigma: Kernel bandwidth for the distribution metrics. + sigma: Kernel bandwidth for the distribution and diversity metrics. None, + the default, uses the median pairwise distance of the reference designs + in whichever space is being scored, so the kernel is neither saturated + nor empty on a problem it has never seen. A published spec pins a value. frame: The most recent `evaluate` result. """ problem: Any reference: Any train: Any | None = None - sigma: float = 10.0 + sigma: float | None = None registry: MetricRegistry = field(default_factory=lambda: METRICS) frame: pd.DataFrame = field(default_factory=pd.DataFrame) @@ -65,9 +85,9 @@ def evaluate( models: Mapping[str, Any], *, metrics: list[str] | None = None, - space: Space = "pixel", + space: str = "pixel", aggregation: Literal["mean", "median"] = "mean", - pca_dims: int = 8, + width: int = 8, reference_row: bool = True, ) -> pd.DataFrame: """Score every model on every cheap metric. @@ -75,12 +95,12 @@ def evaluate( Args: models: Model name -> its generated designs, one array per model. metrics: Metric names; defaults to every metric that needs no simulator. - space: `pixel` scores the designs themselves. `pca` projects the - designs onto the leading principal components of the reference - designs first, and appends `@pca` to each column; metrics that + space: `pixel` scores the designs themselves. Any other name in + `SPACES` (`pca` is built in) projects every design into that + space first and appends `@space` to each column; metrics that need the actual designs (`viol`, `copy_rate`) are skipped there. aggregation: How metrics with one value per design collapse to one number. - pca_dims: Number of components when `space="pca"`. + width: Dimension of the space, for spaces that have one (`pca`). reference_row: Also score one random half of the reference designs against the other half, as the row `REFERENCE_ROW`. @@ -91,29 +111,36 @@ def evaluate( if space != "pixel": specs = [spec for spec in specs if not spec.pixel_only] reference = np.asarray(self.reference) - project = _pca_projection(reference, pca_dims) if space == "pca" else (lambda x: x) + if space not in SPACES: + raise KeyError(f"Unknown space {space!r}. Registered: {', '.join(SPACES)}") + project = SPACES[space](reference, width) + reference_codes = project(reference) + sigma = self.sigma if self.sigma is not None else metrics_mod.compute_median_sigma(reference_codes) - def score(generated: Any, against: Any) -> dict[str, float]: + def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: ctx = EvaluationContext( problem=self.problem, problem_id=getattr(self.problem, "problem_id", type(self.problem).__name__), gen_designs=project(np.asarray(generated)), ref_designs=project(np.asarray(against)), - sigma=self.sigma, + sigma=sigma, aggregation=aggregation, copy_corpus_fn=(lambda: np.asarray(self.train)) if self.train is not None else None, ) row: dict[str, float] = {} - for spec in specs: + for spec in chosen: value = spec.fn(ctx) row.update(value if isinstance(value, dict) else {spec.name: value}) return row - rows = {name: score(designs, reference) for name, designs in models.items()} + rows = {name: score(designs, reference, specs) for name, designs in models.items()} if reference_row and len(reference) >= 4: # noqa: PLR2004 - two halves of at least two designs order = np.random.default_rng(0).permutation(len(reference)) half = len(reference) // 2 - rows[REFERENCE_ROW] = score(reference[order[:half]], reference[order[half:]]) + # Two halves of the reference set answer different conditions, so the + # per-condition metrics have no meaning there and are left blank. + unpaired = [spec for spec in specs if spec.family != "conditions"] + rows[REFERENCE_ROW] = score(reference[order[:half]], reference[order[half:]], unpaired) frame = pd.DataFrame.from_dict(rows, orient="index") if space != "pixel": @@ -121,6 +148,75 @@ def score(generated: Any, against: Any) -> dict[str, float]: self.frame = frame return frame + @classmethod + def from_generators( + cls, + evaluator: Any, + generators: Mapping[str, Any], + *, + expensive: bool = False, + metrics: list[str] | None = None, + ) -> Board: + """Score loaded models through an `Evaluator`, which is what the live-model metrics need. + + `cond_sens` re-samples the model, the cost columns read its network, and + the physics columns run the simulator; none of that is possible from + saved designs. The evaluator supplies the spec, the reference designs + and the conditions, so there is no reference row here -- the spec's own + reference designs already are the comparison. + + Args: + evaluator: An `Evaluator` (anything with `.score(generator, only=, include_expensive=)`). + generators: Label -> loaded `Generator`. + expensive: Whether to run the simulator-backed metrics. + metrics: Metric names; defaults to the spec's list. + + Returns: + A `Board` with one row per generator. + """ + rows = { + label: evaluator.score(generator, only=metrics, include_expensive=expensive) + for label, generator in generators.items() + } + frame = pd.DataFrame.from_dict(rows, orient="index") + board = cls(problem=getattr(evaluator, "problem", None), reference=getattr(evaluator, "resolved", None)) + board.frame = frame + return board + + @classmethod + def load( + cls, + problem_id: str, + models: list[str], + *, + seed: int = 1, + spec: str | None = None, + expensive: bool = False, + ) -> Board: + """Pull published checkpoints by name and score them: the workshop in one call. + + Args: + problem_id: An EngiBench problem id, e.g. `"beams2d"`. + models: Generator names from `BUILTIN_GENERATORS`, e.g. `["cgan_cnn_2d", "vqgan"]`. + seed: Training seed of the checkpoints to load. + spec: Spec reference such as `"beams2d/v2"`; defaults to the problem's current spec. + expensive: Whether to run the simulator-backed metrics. + + Returns: + A `Board` with one row per model, labelled `name/seed`. + """ + from engiopt.evaluation.evaluator import Evaluator + from engiopt.utils.all_generators import BUILTIN_GENERATORS + + evaluator = Evaluator.for_problem(problem_id, spec=spec) + generators = { + f"{name}/seed{seed}": BUILTIN_GENERATORS[name].from_pretrained( + evaluator.problem, problem_id=problem_id, seed=seed + ) + for name in models + } + return cls.from_generators(evaluator, generators, expensive=expensive) + def explain(self) -> pd.DataFrame: """Read every column of the last evaluation. @@ -176,7 +272,12 @@ def __repr__(self) -> str: return repr(self.frame) -def _pca_projection(reference: np.ndarray, n_components: int) -> Any: +def _pixel(reference: np.ndarray, width: int) -> Callable[[np.ndarray], np.ndarray]: # noqa: ARG001 + """The designs as they are.""" + return lambda designs: designs + + +def _pca_projection(reference: np.ndarray, n_components: int) -> Callable[[np.ndarray], np.ndarray]: """Projection onto the leading principal components of the reference designs. The control that asks whether a *learned* space was needed: if PCA of the @@ -193,3 +294,7 @@ def project(designs: np.ndarray) -> np.ndarray: return np.einsum("ij,jk->ik", centered, components) return project + + +register_space("pixel", _pixel) +register_space("pca", _pca_projection) diff --git a/tests/test_board.py b/tests/test_board.py index b370f30d..619de9ec 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -65,7 +65,7 @@ def test_rank_refuses_a_diagnostic(fake_problem: Any, designs: Any) -> None: def test_space_is_an_argument_not_a_metric(fake_problem: Any, designs: Any) -> None: """The same metric, asked in PCA space, gets a suffixed column; pixel-only metrics are skipped.""" models, ref, train = designs - frame = Board(fake_problem, reference=ref, train=train).evaluate(models, space="pca", pca_dims=3) + frame = Board(fake_problem, reference=ref, train=train).evaluate(models, space="pca", width=3) assert "mmd@pca" in frame.columns assert "viol@pca" not in frame.columns, "a constraint check on PCA codes would be meaningless" assert frame.loc["copier", "mmd@pca"] == pytest.approx(0.0) @@ -81,3 +81,14 @@ def test_aggregation_is_a_policy_on_the_context(fake_problem: Any) -> None: ) assert mean_ctx.reduce(values) == pytest.approx(25.075) assert median_ctx.reduce(values) == pytest.approx(0.1) + + +def test_the_default_bandwidth_lets_diversity_metrics_see_anything(fake_problem: Any, designs: Any) -> None: + """At a fixed sigma=10 every design looks identical to every other; the median heuristic does not.""" + models, ref, _ = designs + collapsed = np.repeat(models["honest"][:1], len(ref), axis=0) + frame = Board(fake_problem, reference=ref).evaluate({"collapsed": collapsed, "spread": models["honest"]}) + assert frame.loc["collapsed", "vendi"] == pytest.approx(1.0) + assert frame.loc["spread", "vendi"] > 3.0 + saturated = Board(fake_problem, reference=ref, sigma=10.0).evaluate({"spread": models["honest"]}) + assert saturated.loc["spread", "vendi"] < 1.5, "a pinned, too-wide kernel is still honored" From 1babd6aec1d6e74c0d901bc3dd9521a5570cf708 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 14:25:32 +0200 Subject: [PATCH 05/18] Describe the current metric suite where contributors read about it The docs named `novelty`, listed the old default pass, and pointed at spec v1. They now name the columns that exist, say what --list-metrics prints, and send readers to the notebook for the rest. `Board` is exported from `engiopt.evaluation` as the front door. Co-Authored-By: Claude Fable 5.1 --- CONTRIBUTING_A_MODEL.md | 34 +- LEADERBOARD.md | 22 +- README.md | 2 +- engiopt/evaluation/__init__.py | 2 + example_metrics_suite.ipynb | 760 ++++++++++++++++++++++++--------- 5 files changed, 596 insertions(+), 224 deletions(-) diff --git a/CONTRIBUTING_A_MODEL.md b/CONTRIBUTING_A_MODEL.md index 87c822fa..3c4a77b0 100644 --- a/CONTRIBUTING_A_MODEL.md +++ b/CONTRIBUTING_A_MODEL.md @@ -175,12 +175,16 @@ python -m engiopt.evaluate --list-generators # your model should ap python -m engiopt.evaluate --problem-id beams2d --generators my_model --hf-entity my-hf-username ``` -The default pass runs the cheap metrics: `mmd`, `dpp`, `viol`, and the two -integrity checks (`novelty`, `cond_sens`). Feasibility is among them because it -describes the design as generated, so it is a constraint check rather than a -solver run. `--include-expensive` adds the optimality gaps (`iog`, `cog`, -`fog`), which do run the optimizer and are slow — leave them off while -iterating. +The default pass runs every metric that needs no simulator: the set-level +ones (`mmd`, `coverage`, `vendi`, `dpp`), the per-condition ones +(`per_condition_distance`, `volume_error`), feasibility (`viol`), the two +integrity checks (`copy_rate`, `cond_sens`) with `train_distance`, and the cost +columns. `--include-expensive` adds the columns that re-optimize from each +generated design — the optimality gaps `iog`, `cog`, `fog` and the +call-budget metrics `calls_to_settle`, `gap_after_calls`, +`reaches_reference_rate`, `first_call_gain` — which run the optimizer and are +slow; leave them off while iterating. `python -m engiopt.evaluate +--list-metrics` prints the question each column answers. Watch two columns while you develop: @@ -193,7 +197,7 @@ Watch two columns while you develop: `cond_sens` draws the batch a second time, under shuffled conditions, so it doubles sampling cost. That is nothing for a GAN and noticeable for a diffusion -model — pass `--metrics mmd dpp viol` while iterating if it slows you down, then +model — pass `--metrics mmd viol` while iterating if it slows you down, then run the full list before publishing. ## 5. Publish it @@ -215,13 +219,19 @@ from engiopt.evaluation import register_metric @register_metric("my_metric", family="diversity", cost="cheap", higher_is_better=True) def my_metric(ctx) -> float: - """One line, shown by --list-metrics.""" - return float(...) # ctx.gen_flat, ctx.ref_flat, ctx.conditions, ... + """What question does the number answer? This line is what readers see.""" + return ctx.reduce(...) # one value per design -> the caller's aggregation (mean or median) ``` -Declare `cost="expensive"` if it touches `ctx.optimization` (the simulator or -optimizer). The registry enforces the split, so a cheap run can never -accidentally launch a simulation. +Five declarations: the name, the family (which question it belongs to), the +cost, the direction — `None` for a diagnostic that is read but never ranked on — +and, when the metric needs the actual designs rather than codes in some space +(a constraint check, a copy corpus), `pixel_only=True`. Declare +`cost="expensive"` if it touches `ctx.optimization` (the simulator or +optimizer); the registry enforces the split, so a cheap run can never +accidentally launch a simulation. Where the metric is computed (`space=`) and +how per-design values collapse to one number (`aggregation=`) are chosen when a +board is evaluated, not by the metric. Importing the module is what registers it — which means you can define a metric in a notebook cell and it will appear in the next leaderboard you build. diff --git a/LEADERBOARD.md b/LEADERBOARD.md index 4d076cbf..9638ef02 100644 --- a/LEADERBOARD.md +++ b/LEADERBOARD.md @@ -152,21 +152,27 @@ scored: the whole dataset is public, so a memorizer memorizes all of it. So the board measures retrieval instead: -- **`novelty`** — mean per-element RMS distance from each generated design to the - nearest design the model could have copied: the training split it was fitted - on, plus the reference designs the protocol names. -- **`copy_rate`** — the share of the batch closer than `copy_tol`. This is the - one that flags an entry. +- **`copy_rate`** — the share of the batch within `copy_tol` (per-element RMS) + of any design the model could have copied: the training split it was fitted + on, plus the reference designs the protocol names. This is the one that flags + an entry. +- **`train_distance`** — how far each generated design sits from the nearest + design in the training split. Zero means the model reproduces what it was + trained on. Both are diagnostic, with no ranking direction, and that is not an oversight. -Ranking on novelty would put pure noise in first place — zero means retrieval, +Ranking on `train_distance` would put pure noise in first place — zero means retrieval, but large means only "unlike the data", which a broken model also achieves. Closing one gaming vector by opening another is not progress. The same reasoning applies to `cond_sens`: an unconditional model is a legitimate thing to build, and responding to conditions *wrongly* also moves the output, so a large value is not by itself a good one. -Read them next to `mmd` and `viol`, never on their own. +Read them next to `mmd` and `viol`, never on their own. Every column's question, +direction and cost is one call away — `METRICS.explain()` — and +[`example_metrics_suite.ipynb`](example_metrics_suite.ipynb) walks the whole suite, +including the construction that scores a perfect `mmd` by returning the correct +designs for the wrong conditions. ### What would actually close it @@ -190,7 +196,7 @@ from engiopt.evaluation.leaderboard import disagreement, load_from_hub, rank from engiopt.evaluation.spec import EvalSpec board = load_from_hub("IDEALLab/engiopt-leaderboard") -spec = EvalSpec.load("beams2d/v1") +spec = EvalSpec.load("beams2d/v2") rank(board, "fog", eval_spec=spec) # the public ordering rank(board, "fog", eval_spec=spec, eligible_only=False) # everything, including claims diff --git a/README.md b/README.md index 55db08e8..97ad1238 100644 --- a/README.md +++ b/README.md @@ -166,7 +166,7 @@ python -m engiopt.verify --board ... --verifier ideallab-ci --publish # the o Because the row carries the full address, that audit is not privileged — anyone can run the same command and get the same answer. -**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `novelty` / `copy_rate` catch it (`copy_rate=1.00`), and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. +**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `copy_rate` catches it (`copy_rate=1.00`), and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. The whole suite — what each column measures, which way is better, how to change the space or the aggregation, and how to add a metric — is in [`example_metrics_suite.ipynb`](example_metrics_suite.ipynb). See **[LEADERBOARD.md](LEADERBOARD.md)** for the submission path, the flags, and what would actually close the copying hole. diff --git a/engiopt/evaluation/__init__.py b/engiopt/evaluation/__init__.py index 654d1b6d..3d0f9e44 100644 --- a/engiopt/evaluation/__init__.py +++ b/engiopt/evaluation/__init__.py @@ -12,6 +12,7 @@ property of the comparison, not of the model. """ +from engiopt.evaluation.board import Board from engiopt.evaluation.context import EvaluationContext from engiopt.evaluation.context import OptimizationResults from engiopt.evaluation.evaluator import Evaluator @@ -31,6 +32,7 @@ __all__ = [ "METRICS", + "Board", "EvalSpec", "EvaluationContext", "Evaluator", diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 570d2d64..44a36be3 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -17,6 +17,7 @@ "| read the result | `board.explain()`, `board.rank(\"mmd\")` |\n", "| change how per-design values are combined | `board.evaluate(models, aggregation=\"median\")` |\n", "| ask the same questions in another space | `board.evaluate(models, space=\"pca\")` |\n", + "| score published checkpoints by name | `Board.load(\"beams2d\", [\"cgan_cnn_2d\", \"vqgan\"])` |\n", "| add a metric of your own | `@register_metric(...)` on a function |\n", "\n", "Everything below runs in seconds on a toy problem. With a real EngiBench problem the\n", @@ -37,10 +38,10 @@ "id": "25da8a4e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:42.538081Z", - "iopub.status.busy": "2026-09-18T13:05:42.537712Z", - "iopub.status.idle": "2026-09-18T13:05:45.049987Z", - "shell.execute_reply": "2026-09-18T13:05:45.049722Z" + "iopub.execute_input": "2026-09-25T12:23:44.857863Z", + "iopub.status.busy": "2026-09-25T12:23:44.857775Z", + "iopub.status.idle": "2026-09-25T12:23:47.650176Z", + "shell.execute_reply": "2026-09-25T12:23:47.649910Z" } }, "outputs": [ @@ -81,14 +82,21 @@ " \n", " \n", " dpp\n", - " How spread out are the generated designs, measured as the volume their similarity kernel spans?\n", + " How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set?\n", " diversity\n", + " higher is better\n", + " cheap\n", + " \n", + " \n", + " train_distance\n", + " How far is each generated design from the nearest design the model was trained on?\n", + " memorization\n", " diagnostic\n", " cheap\n", " \n", " \n", - " novelty\n", - " How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?\n", + " copy_rate\n", + " What fraction of the generated designs are copies of a design the model could have seen?\n", " memorization\n", " diagnostic\n", " cheap\n", @@ -128,30 +136,131 @@ " lower is better\n", " expensive\n", " \n", + " \n", + " per_condition_distance\n", + " For each set of conditions, how far is the generated design from the reference optimum for those same conditions?\n", + " conditions\n", + " lower is better\n", + " cheap\n", + " \n", + " \n", + " volume_error\n", + " How far is each design's material fraction from the one its conditions requested?\n", + " conditions\n", + " lower is better\n", + " cheap\n", + " \n", + " \n", + " coverage\n", + " What fraction of the reference optima have a generated design nearby?\n", + " distribution\n", + " higher is better\n", + " cheap\n", + " \n", + " \n", + " vendi\n", + " How many genuinely distinct designs are in the generated set, as an effective count?\n", + " diversity\n", + " higher is better\n", + " cheap\n", + " \n", + " \n", + " calls_to_settle\n", + " How many optimizer calls until the objective stays within 5% of where it ends up?\n", + " performance\n", + " lower is better\n", + " expensive\n", + " \n", + " \n", + " gap_after_calls\n", + " How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls?\n", + " performance\n", + " lower is better\n", + " expensive\n", + " \n", + " \n", + " reaches_reference_rate\n", + " What fraction of the generated designs, once re-optimized, reach or beat the reference optimum?\n", + " performance\n", + " higher is better\n", + " expensive\n", + " \n", + " \n", + " first_call_gain\n", + " What fraction of the whole re-optimization's improvement does the first optimizer call deliver?\n", + " performance\n", + " higher is better\n", + " expensive\n", + " \n", + " \n", + " generation_seconds\n", + " How many seconds did the model take to generate this batch of designs?\n", + " cost\n", + " lower is better\n", + " cheap\n", + " \n", + " \n", + " n_parameters\n", + " How many trainable parameters does the model have?\n", + " cost\n", + " lower is better\n", + " cheap\n", + " \n", + " \n", + " train_minutes\n", + " How many minutes did the model take to train?\n", + " cost\n", + " lower is better\n", + " cheap\n", + " \n", " \n", "\n", "" ], "text/plain": [ - " question \\\n", - "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", - "dpp How spread out are the generated designs, measured as the volume their similarity kernel spans? \n", - "novelty How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", - "cond_sens When the same model is given different conditions, does its output change? \n", - "viol What fraction of the generated designs violate the problem's constraints? \n", - "iog How much worse than the reference optimum is the generated design, exactly as generated? \n", - "cog Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? \n", - "fog Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes? \n", + " question \\\n", + "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", + "dpp How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? \n", + "train_distance How far is each generated design from the nearest design the model was trained on? \n", + "copy_rate What fraction of the generated designs are copies of a design the model could have seen? \n", + "cond_sens When the same model is given different conditions, does its output change? \n", + "viol What fraction of the generated designs violate the problem's constraints? \n", + "iog How much worse than the reference optimum is the generated design, exactly as generated? \n", + "cog Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? \n", + "fog Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes? \n", + "per_condition_distance For each set of conditions, how far is the generated design from the reference optimum for those same conditions? \n", + "volume_error How far is each design's material fraction from the one its conditions requested? \n", + "coverage What fraction of the reference optima have a generated design nearby? \n", + "vendi How many genuinely distinct designs are in the generated set, as an effective count? \n", + "calls_to_settle How many optimizer calls until the objective stays within 5% of where it ends up? \n", + "gap_after_calls How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls? \n", + "reaches_reference_rate What fraction of the generated designs, once re-optimized, reach or beat the reference optimum? \n", + "first_call_gain What fraction of the whole re-optimization's improvement does the first optimizer call deliver? \n", + "generation_seconds How many seconds did the model take to generate this batch of designs? \n", + "n_parameters How many trainable parameters does the model have? \n", + "train_minutes How many minutes did the model take to train? \n", "\n", - " family direction cost \n", - "mmd distribution lower is better cheap \n", - "dpp diversity diagnostic cheap \n", - "novelty memorization diagnostic cheap \n", - "cond_sens conditions diagnostic cheap \n", - "viol feasibility lower is better cheap \n", - "iog performance lower is better expensive \n", - "cog performance lower is better expensive \n", - "fog performance lower is better expensive " + " family direction cost \n", + "mmd distribution lower is better cheap \n", + "dpp diversity higher is better cheap \n", + "train_distance memorization diagnostic cheap \n", + "copy_rate memorization diagnostic cheap \n", + "cond_sens conditions diagnostic cheap \n", + "viol feasibility lower is better cheap \n", + "iog performance lower is better expensive \n", + "cog performance lower is better expensive \n", + "fog performance lower is better expensive \n", + "per_condition_distance conditions lower is better cheap \n", + "volume_error conditions lower is better cheap \n", + "coverage distribution higher is better cheap \n", + "vendi diversity higher is better cheap \n", + "calls_to_settle performance lower is better expensive \n", + "gap_after_calls performance lower is better expensive \n", + "reaches_reference_rate performance higher is better expensive \n", + "first_call_gain performance higher is better expensive \n", + "generation_seconds cost lower is better cheap \n", + "n_parameters cost lower is better cheap \n", + "train_minutes cost lower is better cheap " ] }, "execution_count": 1, @@ -215,10 +324,10 @@ "id": "4e14a02e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.051482Z", - "iopub.status.busy": "2026-09-18T13:05:45.051342Z", - "iopub.status.idle": "2026-09-18T13:05:45.054205Z", - "shell.execute_reply": "2026-09-18T13:05:45.054017Z" + "iopub.execute_input": "2026-09-25T12:23:47.651630Z", + "iopub.status.busy": "2026-09-25T12:23:47.651449Z", + "iopub.status.idle": "2026-09-25T12:23:47.654331Z", + "shell.execute_reply": "2026-09-25T12:23:47.654145Z" } }, "outputs": [], @@ -284,10 +393,10 @@ "id": "6bfdaf3f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.055336Z", - "iopub.status.busy": "2026-09-18T13:05:45.055283Z", - "iopub.status.idle": "2026-09-18T13:05:45.057134Z", - "shell.execute_reply": "2026-09-18T13:05:45.056941Z" + "iopub.execute_input": "2026-09-25T12:23:47.655393Z", + "iopub.status.busy": "2026-09-25T12:23:47.655323Z", + "iopub.status.idle": "2026-09-25T12:23:47.657285Z", + "shell.execute_reply": "2026-09-25T12:23:47.657113Z" } }, "outputs": [], @@ -319,10 +428,10 @@ "id": "935e4407", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.058139Z", - "iopub.status.busy": "2026-09-18T13:05:45.058081Z", - "iopub.status.idle": "2026-09-18T13:05:45.063325Z", - "shell.execute_reply": "2026-09-18T13:05:45.063137Z" + "iopub.execute_input": "2026-09-25T12:23:47.658306Z", + "iopub.status.busy": "2026-09-25T12:23:47.658249Z", + "iopub.status.idle": "2026-09-25T12:23:47.665037Z", + "shell.execute_reply": "2026-09-25T12:23:47.664864Z" } }, "outputs": [ @@ -349,87 +458,152 @@ " \n", " mmd\n", " dpp\n", - " novelty\n", + " train_distance\n", " copy_rate\n", " cond_sens\n", " viol\n", + " per_condition_distance\n", + " volume_error\n", + " coverage\n", + " vendi\n", + " generation_seconds\n", + " n_parameters\n", + " train_minutes\n", " \n", " \n", " \n", " \n", " honest\n", - " 0.000008\n", - " 6.710593e-56\n", - " 0.013690\n", + " 6.366160e-04\n", + " 0.035297\n", + " 0.014761\n", " 0.0\n", " NaN\n", " 0.0\n", + " 0.014059\n", + " NaN\n", + " 1.0000\n", + " 3.254471\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " lookup_table\n", - " 0.000019\n", - " 6.642568e-63\n", + " 1.037788e-03\n", + " 0.004358\n", " 0.000000\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.014891\n", + " NaN\n", + " 1.0000\n", + " 3.241546\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " unconditional\n", - " 0.002606\n", - " 1.044406e-61\n", + " 1.012030e-01\n", + " 0.006445\n", " 0.000000\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.180742\n", + " NaN\n", + " 0.9375\n", + " 2.605583\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " permuted_conditions\n", - " 0.000000\n", - " 3.381911e-56\n", - " 0.000000\n", + " -1.110223e-16\n", + " 0.033897\n", + " 0.014753\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.176872\n", + " NaN\n", + " 1.0000\n", + " 3.248472\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " noise\n", - " 0.006043\n", - " 6.005203e-18\n", - " 0.285279\n", + " 4.239221e-01\n", + " 0.998588\n", + " 0.285783\n", " 0.0\n", " NaN\n", " 0.0\n", + " 0.332793\n", + " NaN\n", + " 0.0000\n", + " 15.976387\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " reference (split-half)\n", - " 0.016675\n", - " 6.678526e-24\n", + " 2.798658e-01\n", + " 0.090922\n", " 0.014603\n", " 0.0\n", " NaN\n", " 0.0\n", + " NaN\n", + " NaN\n", + " 0.5000\n", + " 2.868064\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", "\n", "" ], "text/plain": [ - " mmd dpp novelty copy_rate \\\n", - "honest 0.000008 6.710593e-56 0.013690 0.0 \n", - "lookup_table 0.000019 6.642568e-63 0.000000 1.0 \n", - "unconditional 0.002606 1.044406e-61 0.000000 1.0 \n", - "permuted_conditions 0.000000 3.381911e-56 0.000000 1.0 \n", - "noise 0.006043 6.005203e-18 0.285279 0.0 \n", - "reference (split-half) 0.016675 6.678526e-24 0.014603 0.0 \n", + " mmd dpp train_distance copy_rate \\\n", + "honest 6.366160e-04 0.035297 0.014761 0.0 \n", + "lookup_table 1.037788e-03 0.004358 0.000000 1.0 \n", + "unconditional 1.012030e-01 0.006445 0.000000 1.0 \n", + "permuted_conditions -1.110223e-16 0.033897 0.014753 1.0 \n", + "noise 4.239221e-01 0.998588 0.285783 0.0 \n", + "reference (split-half) 2.798658e-01 0.090922 0.014603 0.0 \n", + "\n", + " cond_sens viol per_condition_distance volume_error \\\n", + "honest NaN 0.0 0.014059 NaN \n", + "lookup_table NaN 0.0 0.014891 NaN \n", + "unconditional NaN 0.0 0.180742 NaN \n", + "permuted_conditions NaN 0.0 0.176872 NaN \n", + "noise NaN 0.0 0.332793 NaN \n", + "reference (split-half) NaN 0.0 NaN NaN \n", + "\n", + " coverage vendi generation_seconds n_parameters \\\n", + "honest 1.0000 3.254471 NaN NaN \n", + "lookup_table 1.0000 3.241546 NaN NaN \n", + "unconditional 0.9375 2.605583 NaN NaN \n", + "permuted_conditions 1.0000 3.248472 NaN NaN \n", + "noise 0.0000 15.976387 NaN NaN \n", + "reference (split-half) 0.5000 2.868064 NaN NaN \n", "\n", - " cond_sens viol \n", - "honest NaN 0.0 \n", - "lookup_table NaN 0.0 \n", - "unconditional NaN 0.0 \n", - "permuted_conditions NaN 0.0 \n", - "noise NaN 0.0 \n", - "reference (split-half) NaN 0.0 " + " train_minutes \n", + "honest NaN \n", + "lookup_table NaN \n", + "unconditional NaN \n", + "permuted_conditions NaN \n", + "noise NaN \n", + "reference (split-half) NaN " ] }, "execution_count": 4, @@ -463,10 +637,10 @@ "id": "b7bb486e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.064420Z", - "iopub.status.busy": "2026-09-18T13:05:45.064362Z", - "iopub.status.idle": "2026-09-18T13:05:45.067438Z", - "shell.execute_reply": "2026-09-18T13:05:45.067243Z" + "iopub.execute_input": "2026-09-25T12:23:47.666091Z", + "iopub.status.busy": "2026-09-25T12:23:47.666032Z", + "iopub.status.idle": "2026-09-25T12:23:47.669867Z", + "shell.execute_reply": "2026-09-25T12:23:47.669687Z" } }, "outputs": [ @@ -503,28 +677,28 @@ " Taken as a set, how different are the generated designs from the reference optimal designs?\n", " lower is better\n", " permuted_conditions\n", - " 1.667532e-02\n", + " 0.279866\n", " \n", " \n", " dpp\n", - " How spread out are the generated designs, measured as the volume their similarity kernel spans?\n", - " diagnostic\n", - " \n", - " 6.678526e-24\n", + " How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set?\n", + " higher is better\n", + " noise\n", + " 0.090922\n", " \n", " \n", - " novelty\n", - " How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?\n", + " train_distance\n", + " How far is each generated design from the nearest design the model was trained on?\n", " diagnostic\n", " \n", - " 1.460301e-02\n", + " 0.014603\n", " \n", " \n", " copy_rate\n", - " How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)?\n", + " What fraction of the generated designs are copies of a design the model could have seen?\n", " diagnostic\n", " \n", - " 0.000000e+00\n", + " 0.000000\n", " \n", " \n", " cond_sens\n", @@ -538,44 +712,121 @@ " What fraction of the generated designs violate the problem's constraints?\n", " lower is better\n", " tie: honest, lookup_table, unconditional, permuted_conditions, noise\n", - " 0.000000e+00\n", + " 0.000000\n", + " \n", + " \n", + " per_condition_distance\n", + " For each set of conditions, how far is the generated design from the reference optimum for those same conditions?\n", + " lower is better\n", + " honest\n", + " NaN\n", + " \n", + " \n", + " volume_error\n", + " How far is each design's material fraction from the one its conditions requested?\n", + " lower is better\n", + " \n", + " NaN\n", + " \n", + " \n", + " coverage\n", + " What fraction of the reference optima have a generated design nearby?\n", + " higher is better\n", + " tie: honest, lookup_table, permuted_conditions\n", + " 0.500000\n", + " \n", + " \n", + " vendi\n", + " How many genuinely distinct designs are in the generated set, as an effective count?\n", + " higher is better\n", + " noise\n", + " 2.868064\n", + " \n", + " \n", + " generation_seconds\n", + " How many seconds did the model take to generate this batch of designs?\n", + " lower is better\n", + " \n", + " NaN\n", + " \n", + " \n", + " n_parameters\n", + " How many trainable parameters does the model have?\n", + " lower is better\n", + " \n", + " NaN\n", + " \n", + " \n", + " train_minutes\n", + " How many minutes did the model take to train?\n", + " lower is better\n", + " \n", + " NaN\n", " \n", " \n", "\n", "" ], "text/plain": [ - " question \\\n", - "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", - "dpp How spread out are the generated designs, measured as the volume their similarity kernel spans? \n", - "novelty How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", - "copy_rate How close is each generated design to the nearest design it could have copied (a training design or a reference optimum)? \n", - "cond_sens When the same model is given different conditions, does its output change? \n", - "viol What fraction of the generated designs violate the problem's constraints? \n", + " question \\\n", + "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", + "dpp How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? \n", + "train_distance How far is each generated design from the nearest design the model was trained on? \n", + "copy_rate What fraction of the generated designs are copies of a design the model could have seen? \n", + "cond_sens When the same model is given different conditions, does its output change? \n", + "viol What fraction of the generated designs violate the problem's constraints? \n", + "per_condition_distance For each set of conditions, how far is the generated design from the reference optimum for those same conditions? \n", + "volume_error How far is each design's material fraction from the one its conditions requested? \n", + "coverage What fraction of the reference optima have a generated design nearby? \n", + "vendi How many genuinely distinct designs are in the generated set, as an effective count? \n", + "generation_seconds How many seconds did the model take to generate this batch of designs? \n", + "n_parameters How many trainable parameters does the model have? \n", + "train_minutes How many minutes did the model take to train? \n", "\n", - " direction \\\n", - "mmd lower is better \n", - "dpp diagnostic \n", - "novelty diagnostic \n", - "copy_rate diagnostic \n", - "cond_sens diagnostic \n", - "viol lower is better \n", + " direction \\\n", + "mmd lower is better \n", + "dpp higher is better \n", + "train_distance diagnostic \n", + "copy_rate diagnostic \n", + "cond_sens diagnostic \n", + "viol lower is better \n", + "per_condition_distance lower is better \n", + "volume_error lower is better \n", + "coverage higher is better \n", + "vendi higher is better \n", + "generation_seconds lower is better \n", + "n_parameters lower is better \n", + "train_minutes lower is better \n", "\n", - " picks \\\n", - "mmd permuted_conditions \n", - "dpp \n", - "novelty \n", - "copy_rate \n", - "cond_sens \n", - "viol tie: honest, lookup_table, unconditional, permuted_conditions, noise \n", + " picks \\\n", + "mmd permuted_conditions \n", + "dpp noise \n", + "train_distance \n", + "copy_rate \n", + "cond_sens \n", + "viol tie: honest, lookup_table, unconditional, permuted_conditions, noise \n", + "per_condition_distance honest \n", + "volume_error \n", + "coverage tie: honest, lookup_table, permuted_conditions \n", + "vendi noise \n", + "generation_seconds \n", + "n_parameters \n", + "train_minutes \n", "\n", - " real designs score \n", - "mmd 1.667532e-02 \n", - "dpp 6.678526e-24 \n", - "novelty 1.460301e-02 \n", - "copy_rate 0.000000e+00 \n", - "cond_sens NaN \n", - "viol 0.000000e+00 " + " real designs score \n", + "mmd 0.279866 \n", + "dpp 0.090922 \n", + "train_distance 0.014603 \n", + "copy_rate 0.000000 \n", + "cond_sens NaN \n", + "viol 0.000000 \n", + "per_condition_distance NaN \n", + "volume_error NaN \n", + "coverage 0.500000 \n", + "vendi 2.868064 \n", + "generation_seconds NaN \n", + "n_parameters NaN \n", + "train_minutes NaN " ] }, "execution_count": 5, @@ -617,9 +868,16 @@ "`memorization` and `conditions` families are for, and why a board that reports only\n", "distribution metrics cannot be trusted.\n", "\n", - "**`dpp` reads between 10⁻⁶³ and 10⁻¹⁸.** Those are not diversity measurements; they\n", - "are a determinant collapsing toward underflow. That is why it is diagnostic, and why\n", - "`vendi` — an effective count of distinct designs — replaces it.\n", + "**`per_condition_distance` is the column that catches `permuted_conditions`.** It\n", + "compares design `i` with the reference optimum for conditions `i`, so the correct\n", + "designs in the wrong order score badly here and perfectly on `mmd`. Read the two\n", + "together.\n", + "\n", + "**`dpp` and `vendi` both measure diversity; prefer `vendi`.** `dpp` is now the n-th\n", + "root of the kernel determinant, on a 0-to-1 scale (the raw determinant older boards\n", + "published reads 10⁻²⁰ on every real board). But a determinant still rises when a\n", + "collapsed set is jittered — it rewards noise as diversity — and `vendi`, an effective\n", + "count of distinct designs, does not.\n", "\n", "**`cond_sens` is `NaN`.** It asks whether the output *changes* when the conditions\n", "change, which needs the live model re-sampled; a saved array cannot answer it.\n", @@ -640,10 +898,10 @@ "id": "dd93537a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.068493Z", - "iopub.status.busy": "2026-09-18T13:05:45.068440Z", - "iopub.status.idle": "2026-09-18T13:05:45.071429Z", - "shell.execute_reply": "2026-09-18T13:05:45.071251Z" + "iopub.execute_input": "2026-09-25T12:23:47.670919Z", + "iopub.status.busy": "2026-09-25T12:23:47.670860Z", + "iopub.status.idle": "2026-09-25T12:23:47.674634Z", + "shell.execute_reply": "2026-09-25T12:23:47.674450Z" } }, "outputs": [ @@ -670,87 +928,152 @@ " \n", " mmd\n", " dpp\n", - " novelty\n", + " train_distance\n", " copy_rate\n", " cond_sens\n", " viol\n", + " per_condition_distance\n", + " volume_error\n", + " coverage\n", + " vendi\n", + " generation_seconds\n", + " n_parameters\n", + " train_minutes\n", " \n", " \n", " \n", " \n", " permuted_conditions\n", - " 0.000000\n", - " 3.381911e-56\n", - " 0.000000\n", + " -1.110223e-16\n", + " 0.033897\n", + " 0.014753\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.176872\n", + " NaN\n", + " 1.0000\n", + " 3.248472\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " honest\n", - " 0.000008\n", - " 6.710593e-56\n", - " 0.013690\n", + " 6.366160e-04\n", + " 0.035297\n", + " 0.014761\n", " 0.0\n", " NaN\n", " 0.0\n", + " 0.014059\n", + " NaN\n", + " 1.0000\n", + " 3.254471\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " lookup_table\n", - " 0.000019\n", - " 6.642568e-63\n", + " 1.037788e-03\n", + " 0.004358\n", " 0.000000\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.014891\n", + " NaN\n", + " 1.0000\n", + " 3.241546\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", " unconditional\n", - " 0.002606\n", - " 1.044406e-61\n", + " 1.012030e-01\n", + " 0.006445\n", " 0.000000\n", " 1.0\n", " NaN\n", " 0.0\n", + " 0.180742\n", + " NaN\n", + " 0.9375\n", + " 2.605583\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", - " noise\n", - " 0.006043\n", - " 6.005203e-18\n", - " 0.285279\n", + " reference (split-half)\n", + " 2.798658e-01\n", + " 0.090922\n", + " 0.014603\n", " 0.0\n", " NaN\n", " 0.0\n", + " NaN\n", + " NaN\n", + " 0.5000\n", + " 2.868064\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", - " reference (split-half)\n", - " 0.016675\n", - " 6.678526e-24\n", - " 0.014603\n", + " noise\n", + " 4.239221e-01\n", + " 0.998588\n", + " 0.285783\n", " 0.0\n", " NaN\n", " 0.0\n", + " 0.332793\n", + " NaN\n", + " 0.0000\n", + " 15.976387\n", + " NaN\n", + " NaN\n", + " NaN\n", " \n", " \n", "\n", "" ], "text/plain": [ - " mmd dpp novelty copy_rate \\\n", - "permuted_conditions 0.000000 3.381911e-56 0.000000 1.0 \n", - "honest 0.000008 6.710593e-56 0.013690 0.0 \n", - "lookup_table 0.000019 6.642568e-63 0.000000 1.0 \n", - "unconditional 0.002606 1.044406e-61 0.000000 1.0 \n", - "noise 0.006043 6.005203e-18 0.285279 0.0 \n", - "reference (split-half) 0.016675 6.678526e-24 0.014603 0.0 \n", + " mmd dpp train_distance copy_rate \\\n", + "permuted_conditions -1.110223e-16 0.033897 0.014753 1.0 \n", + "honest 6.366160e-04 0.035297 0.014761 0.0 \n", + "lookup_table 1.037788e-03 0.004358 0.000000 1.0 \n", + "unconditional 1.012030e-01 0.006445 0.000000 1.0 \n", + "reference (split-half) 2.798658e-01 0.090922 0.014603 0.0 \n", + "noise 4.239221e-01 0.998588 0.285783 0.0 \n", + "\n", + " cond_sens viol per_condition_distance volume_error \\\n", + "permuted_conditions NaN 0.0 0.176872 NaN \n", + "honest NaN 0.0 0.014059 NaN \n", + "lookup_table NaN 0.0 0.014891 NaN \n", + "unconditional NaN 0.0 0.180742 NaN \n", + "reference (split-half) NaN 0.0 NaN NaN \n", + "noise NaN 0.0 0.332793 NaN \n", "\n", - " cond_sens viol \n", - "permuted_conditions NaN 0.0 \n", - "honest NaN 0.0 \n", - "lookup_table NaN 0.0 \n", - "unconditional NaN 0.0 \n", - "noise NaN 0.0 \n", - "reference (split-half) NaN 0.0 " + " coverage vendi generation_seconds n_parameters \\\n", + "permuted_conditions 1.0000 3.248472 NaN NaN \n", + "honest 1.0000 3.254471 NaN NaN \n", + "lookup_table 1.0000 3.241546 NaN NaN \n", + "unconditional 0.9375 2.605583 NaN NaN \n", + "reference (split-half) 0.5000 2.868064 NaN NaN \n", + "noise 0.0000 15.976387 NaN NaN \n", + "\n", + " train_minutes \n", + "permuted_conditions NaN \n", + "honest NaN \n", + "lookup_table NaN \n", + "unconditional NaN \n", + "reference (split-half) NaN \n", + "noise NaN " ] }, "execution_count": 6, @@ -768,10 +1091,10 @@ "id": "6150f040", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.072411Z", - "iopub.status.busy": "2026-09-18T13:05:45.072357Z", - "iopub.status.idle": "2026-09-18T13:05:45.073756Z", - "shell.execute_reply": "2026-09-18T13:05:45.073568Z" + "iopub.execute_input": "2026-09-25T12:23:47.675559Z", + "iopub.status.busy": "2026-09-25T12:23:47.675504Z", + "iopub.status.idle": "2026-09-25T12:23:47.676985Z", + "shell.execute_reply": "2026-09-25T12:23:47.676792Z" } }, "outputs": [ @@ -803,7 +1126,8 @@ "id": "1defe64e", "metadata": {}, "source": [ - "`iog`, `cog`, `fog`, `novelty` and `cond_sens` each produce **one value per design**.\n", + "`iog`, `cog`, `fog`, `train_distance`, `per_condition_distance` and the other per-design\n", + "metrics each produce **one value per design**.\n", "Which single number that becomes is a policy, not a measurement: a mean is dragged by\n", "one diverged design, a median is not. It is an argument to `evaluate` — and, for a\n", "published board, a field of the frozen spec recorded in every row — so there are no\n", @@ -816,10 +1140,10 @@ "id": "36e87484", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.074790Z", - "iopub.status.busy": "2026-09-18T13:05:45.074739Z", - "iopub.status.idle": "2026-09-18T13:05:45.078923Z", - "shell.execute_reply": "2026-09-18T13:05:45.078735Z" + "iopub.execute_input": "2026-09-25T12:23:47.677964Z", + "iopub.status.busy": "2026-09-25T12:23:47.677912Z", + "iopub.status.idle": "2026-09-25T12:23:47.683190Z", + "shell.execute_reply": "2026-09-25T12:23:47.682998Z" } }, "outputs": [ @@ -844,20 +1168,20 @@ " \n", " \n", " \n", - " novelty (mean)\n", - " novelty (median)\n", + " train_distance (mean)\n", + " train_distance (median)\n", " \n", " \n", " \n", " \n", " honest\n", - " 0.013690\n", - " 0.013665\n", + " 0.014761\n", + " 0.014655\n", " \n", " \n", " one_diverged\n", - " 0.031104\n", - " 0.013665\n", + " 0.032199\n", + " 0.015012\n", " \n", " \n", " reference (split-half)\n", @@ -869,10 +1193,10 @@ "" ], "text/plain": [ - " novelty (mean) novelty (median)\n", - "honest 0.013690 0.013665\n", - "one_diverged 0.031104 0.013665\n", - "reference (split-half) 0.014603 0.015140" + " train_distance (mean) train_distance (median)\n", + "honest 0.014761 0.014655\n", + "one_diverged 0.032199 0.015012\n", + "reference (split-half) 0.014603 0.015140" ] }, "execution_count": 8, @@ -886,8 +1210,8 @@ "\n", "compare = {\"honest\": MODELS[\"honest\"], \"one_diverged\": one_diverged}\n", "pd.DataFrame({\n", - " \"novelty (mean)\": board.evaluate(compare)[\"novelty\"],\n", - " \"novelty (median)\": board.evaluate(compare, aggregation=\"median\")[\"novelty\"],\n", + " \"train_distance (mean)\": board.evaluate(compare)[\"train_distance\"],\n", + " \"train_distance (median)\": board.evaluate(compare, aggregation=\"median\")[\"train_distance\"],\n", "})" ] }, @@ -921,10 +1245,10 @@ "id": "ef9a97f4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.079967Z", - "iopub.status.busy": "2026-09-18T13:05:45.079915Z", - "iopub.status.idle": "2026-09-18T13:05:45.082934Z", - "shell.execute_reply": "2026-09-18T13:05:45.082770Z" + "iopub.execute_input": "2026-09-25T12:23:47.684202Z", + "iopub.status.busy": "2026-09-25T12:23:47.684150Z", + "iopub.status.idle": "2026-09-25T12:23:47.687739Z", + "shell.execute_reply": "2026-09-25T12:23:47.687566Z" } }, "outputs": [ @@ -951,51 +1275,80 @@ " \n", " mmd@pca\n", " dpp@pca\n", + " per_condition_distance@pca\n", + " coverage@pca\n", + " vendi@pca\n", " \n", " \n", " \n", " \n", " honest\n", - " 1.354329e-06\n", - " 3.869144e-72\n", + " 0.000124\n", + " 0.007646\n", + " 0.026240\n", + " 1.0000\n", + " 3.168944\n", " \n", " \n", " lookup_table\n", - " 4.033555e-06\n", - " 7.335045e-74\n", + " 0.000206\n", + " 0.001569\n", + " 0.029576\n", + " 1.0000\n", + " 3.182097\n", " \n", " \n", " unconditional\n", - " 2.592173e-03\n", - " 3.266329e-74\n", + " 0.101071\n", + " 0.002117\n", + " 0.569443\n", + " 0.9375\n", + " 2.543748\n", " \n", " \n", " permuted_conditions\n", - " 2.220446e-16\n", - " 1.128691e-64\n", + " 0.000000\n", + " 0.022450\n", + " 0.557238\n", + " 1.0000\n", + " 3.220545\n", " \n", " \n", " noise\n", - " 2.043085e-03\n", - " 9.665059e-48\n", + " 0.181628\n", + " 0.262513\n", + " 0.588434\n", + " 0.2500\n", + " 4.577680\n", " \n", " \n", " reference (split-half)\n", - " 1.667106e-02\n", - " 1.389859e-25\n", + " 0.279983\n", + " 0.071127\n", + " NaN\n", + " 0.5000\n", + " 2.849693\n", " \n", " \n", "\n", "" ], "text/plain": [ - " mmd@pca dpp@pca\n", - "honest 1.354329e-06 3.869144e-72\n", - "lookup_table 4.033555e-06 7.335045e-74\n", - "unconditional 2.592173e-03 3.266329e-74\n", - "permuted_conditions 2.220446e-16 1.128691e-64\n", - "noise 2.043085e-03 9.665059e-48\n", - "reference (split-half) 1.667106e-02 1.389859e-25" + " mmd@pca dpp@pca per_condition_distance@pca \\\n", + "honest 0.000124 0.007646 0.026240 \n", + "lookup_table 0.000206 0.001569 0.029576 \n", + "unconditional 0.101071 0.002117 0.569443 \n", + "permuted_conditions 0.000000 0.022450 0.557238 \n", + "noise 0.181628 0.262513 0.588434 \n", + "reference (split-half) 0.279983 0.071127 NaN \n", + "\n", + " coverage@pca vendi@pca \n", + "honest 1.0000 3.168944 \n", + "lookup_table 1.0000 3.182097 \n", + "unconditional 0.9375 2.543748 \n", + "permuted_conditions 1.0000 3.220545 \n", + "noise 0.2500 4.577680 \n", + "reference (split-half) 0.5000 2.849693 " ] }, "execution_count": 9, @@ -1004,7 +1357,7 @@ } ], "source": [ - "board.evaluate(MODELS, space=\"pca\", pca_dims=8)" + "board.evaluate(MODELS, space=\"pca\", width=8)" ] }, { @@ -1031,10 +1384,10 @@ "id": "1adaa8bf", "metadata": { "execution": { - "iopub.execute_input": "2026-09-18T13:05:45.084068Z", - "iopub.status.busy": "2026-09-18T13:05:45.084017Z", - "iopub.status.idle": "2026-09-18T13:05:45.087304Z", - "shell.execute_reply": "2026-09-18T13:05:45.087133Z" + "iopub.execute_input": "2026-09-25T12:23:47.688753Z", + "iopub.status.busy": "2026-09-25T12:23:47.688698Z", + "iopub.status.idle": "2026-09-25T12:23:47.692135Z", + "shell.execute_reply": "2026-09-25T12:23:47.691977Z" } }, "outputs": [ @@ -1071,7 +1424,7 @@ " Taken as a set, how different are the generated designs from the reference optimal designs?\n", " lower is better\n", " permuted_conditions\n", - " 0.016675\n", + " 0.279866\n", " \n", " \n", " density_ratio\n", @@ -1090,7 +1443,7 @@ "density_ratio How much material do the generated designs use, relative to the reference optima? \n", "\n", " direction picks real designs score \n", - "mmd lower is better permuted_conditions 0.016675 \n", + "mmd lower is better permuted_conditions 0.279866 \n", "density_ratio diagnostic 0.766460 " ] }, @@ -1140,8 +1493,9 @@ "id": "9500df97", "metadata": {}, "source": [ - "- **Physics.** `iog`, `cog`, `fog` need a simulator. With a real problem,\n", - " `Evaluator.score(generator, include_expensive=True)` runs them.\n", + "- **Physics.** `iog`, `cog`, `fog`, `calls_to_settle`, `gap_after_calls`,\n", + " `reaches_reference_rate` and `first_call_gain` need a simulator. With a real problem,\n", + " `Board.load(\"beams2d\", [\"cgan_cnn_2d\", \"vqgan\"], expensive=True)` runs them.\n", "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec.\n", "- **Uncertainty.** Confidence intervals beside every value. On sixteen samples, most\n", " gaps on this board are smaller than their own error bars." From 3c6e6f5c159c7d37ec46587fae77b10cd8202f28 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 14:47:56 +0200 Subject: [PATCH 06/18] Format the notebook's code cells the way CI does `ruff format --check` covers .ipynb files and the notebook had never been through the formatter. Re-executed after formatting so the stored outputs match the source. Co-Authored-By: Claude Fable 5.1 --- example_metrics_suite.ipynb | 94 +++++++++++++++++++------------------ 1 file changed, 48 insertions(+), 46 deletions(-) diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 44a36be3..60ea739a 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -38,10 +38,10 @@ "id": "25da8a4e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:44.857863Z", - "iopub.status.busy": "2026-09-25T12:23:44.857775Z", - "iopub.status.idle": "2026-09-25T12:23:47.650176Z", - "shell.execute_reply": "2026-09-25T12:23:47.649910Z" + "iopub.execute_input": "2026-09-25T12:47:52.176220Z", + "iopub.status.busy": "2026-09-25T12:47:52.175990Z", + "iopub.status.idle": "2026-09-25T12:47:55.539272Z", + "shell.execute_reply": "2026-09-25T12:47:55.539057Z" } }, "outputs": [ @@ -324,10 +324,10 @@ "id": "4e14a02e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.651630Z", - "iopub.status.busy": "2026-09-25T12:23:47.651449Z", - "iopub.status.idle": "2026-09-25T12:23:47.654331Z", - "shell.execute_reply": "2026-09-25T12:23:47.654145Z" + "iopub.execute_input": "2026-09-25T12:47:55.540961Z", + "iopub.status.busy": "2026-09-25T12:47:55.540768Z", + "iopub.status.idle": "2026-09-25T12:47:55.543674Z", + "shell.execute_reply": "2026-09-25T12:47:55.543501Z" } }, "outputs": [], @@ -358,8 +358,8 @@ "\n", "train_conditions = rng.uniform(0.2, 0.8, 64)\n", "test_conditions = rng.uniform(0.2, 0.8, 16)\n", - "TRAIN = np.stack([optimal_design(c) for c in train_conditions]) # what models are fitted on\n", - "REFERENCE = np.stack([optimal_design(c) for c in test_conditions]) # withheld; what models are scored against" + "TRAIN = np.stack([optimal_design(c) for c in train_conditions]) # what models are fitted on\n", + "REFERENCE = np.stack([optimal_design(c) for c in test_conditions]) # withheld; what models are scored against" ] }, { @@ -393,10 +393,10 @@ "id": "6bfdaf3f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.655393Z", - "iopub.status.busy": "2026-09-25T12:23:47.655323Z", - "iopub.status.idle": "2026-09-25T12:23:47.657285Z", - "shell.execute_reply": "2026-09-25T12:23:47.657113Z" + "iopub.execute_input": "2026-09-25T12:47:55.544702Z", + "iopub.status.busy": "2026-09-25T12:47:55.544633Z", + "iopub.status.idle": "2026-09-25T12:47:55.546472Z", + "shell.execute_reply": "2026-09-25T12:47:55.546293Z" } }, "outputs": [], @@ -428,10 +428,10 @@ "id": "935e4407", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.658306Z", - "iopub.status.busy": "2026-09-25T12:23:47.658249Z", - "iopub.status.idle": "2026-09-25T12:23:47.665037Z", - "shell.execute_reply": "2026-09-25T12:23:47.664864Z" + "iopub.execute_input": "2026-09-25T12:47:55.547576Z", + "iopub.status.busy": "2026-09-25T12:47:55.547495Z", + "iopub.status.idle": "2026-09-25T12:47:55.554173Z", + "shell.execute_reply": "2026-09-25T12:47:55.553970Z" } }, "outputs": [ @@ -637,10 +637,10 @@ "id": "b7bb486e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.666091Z", - "iopub.status.busy": "2026-09-25T12:23:47.666032Z", - "iopub.status.idle": "2026-09-25T12:23:47.669867Z", - "shell.execute_reply": "2026-09-25T12:23:47.669687Z" + "iopub.execute_input": "2026-09-25T12:47:55.555221Z", + "iopub.status.busy": "2026-09-25T12:47:55.555159Z", + "iopub.status.idle": "2026-09-25T12:47:55.558994Z", + "shell.execute_reply": "2026-09-25T12:47:55.558784Z" } }, "outputs": [ @@ -898,10 +898,10 @@ "id": "dd93537a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.670919Z", - "iopub.status.busy": "2026-09-25T12:23:47.670860Z", - "iopub.status.idle": "2026-09-25T12:23:47.674634Z", - "shell.execute_reply": "2026-09-25T12:23:47.674450Z" + "iopub.execute_input": "2026-09-25T12:47:55.560043Z", + "iopub.status.busy": "2026-09-25T12:47:55.559988Z", + "iopub.status.idle": "2026-09-25T12:47:55.563565Z", + "shell.execute_reply": "2026-09-25T12:47:55.563385Z" } }, "outputs": [ @@ -1091,10 +1091,10 @@ "id": "6150f040", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.675559Z", - "iopub.status.busy": "2026-09-25T12:23:47.675504Z", - "iopub.status.idle": "2026-09-25T12:23:47.676985Z", - "shell.execute_reply": "2026-09-25T12:23:47.676792Z" + "iopub.execute_input": "2026-09-25T12:47:55.564514Z", + "iopub.status.busy": "2026-09-25T12:47:55.564460Z", + "iopub.status.idle": "2026-09-25T12:47:55.565905Z", + "shell.execute_reply": "2026-09-25T12:47:55.565730Z" } }, "outputs": [ @@ -1140,10 +1140,10 @@ "id": "36e87484", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.677964Z", - "iopub.status.busy": "2026-09-25T12:23:47.677912Z", - "iopub.status.idle": "2026-09-25T12:23:47.683190Z", - "shell.execute_reply": "2026-09-25T12:23:47.682998Z" + "iopub.execute_input": "2026-09-25T12:47:55.567135Z", + "iopub.status.busy": "2026-09-25T12:47:55.567064Z", + "iopub.status.idle": "2026-09-25T12:47:55.572175Z", + "shell.execute_reply": "2026-09-25T12:47:55.571967Z" } }, "outputs": [ @@ -1209,10 +1209,12 @@ "one_diverged[0] = rng.random((8, 10)) # fifteen good designs and one that went wrong\n", "\n", "compare = {\"honest\": MODELS[\"honest\"], \"one_diverged\": one_diverged}\n", - "pd.DataFrame({\n", - " \"train_distance (mean)\": board.evaluate(compare)[\"train_distance\"],\n", - " \"train_distance (median)\": board.evaluate(compare, aggregation=\"median\")[\"train_distance\"],\n", - "})" + "pd.DataFrame(\n", + " {\n", + " \"train_distance (mean)\": board.evaluate(compare)[\"train_distance\"],\n", + " \"train_distance (median)\": board.evaluate(compare, aggregation=\"median\")[\"train_distance\"],\n", + " }\n", + ")" ] }, { @@ -1245,10 +1247,10 @@ "id": "ef9a97f4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.684202Z", - "iopub.status.busy": "2026-09-25T12:23:47.684150Z", - "iopub.status.idle": "2026-09-25T12:23:47.687739Z", - "shell.execute_reply": "2026-09-25T12:23:47.687566Z" + "iopub.execute_input": "2026-09-25T12:47:55.573265Z", + "iopub.status.busy": "2026-09-25T12:47:55.573209Z", + "iopub.status.idle": "2026-09-25T12:47:55.576955Z", + "shell.execute_reply": "2026-09-25T12:47:55.576764Z" } }, "outputs": [ @@ -1384,10 +1386,10 @@ "id": "1adaa8bf", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:23:47.688753Z", - "iopub.status.busy": "2026-09-25T12:23:47.688698Z", - "iopub.status.idle": "2026-09-25T12:23:47.692135Z", - "shell.execute_reply": "2026-09-25T12:23:47.691977Z" + "iopub.execute_input": "2026-09-25T12:47:55.578020Z", + "iopub.status.busy": "2026-09-25T12:47:55.577968Z", + "iopub.status.idle": "2026-09-25T12:47:55.581500Z", + "shell.execute_reply": "2026-09-25T12:47:55.581292Z" } }, "outputs": [ From 32afcc7e33c06acd32f85fa4e1774f7281342679 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Fri, 25 Sep 2026 14:49:10 +0200 Subject: [PATCH 07/18] Put the notebook's imports where the linter expects them CI runs `ruff check .` over notebooks too. The first cell set pandas display options before importing the registry, two cells had imports out of the repo's order, and the toy problem's helpers lacked docstrings. Co-Authored-By: Claude Fable 5.1 --- example_metrics_suite.ipynb | 104 ++++++++++++++++++++---------------- 1 file changed, 59 insertions(+), 45 deletions(-) diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 60ea739a..4971b106 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -38,13 +38,23 @@ "id": "25da8a4e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:52.176220Z", - "iopub.status.busy": "2026-09-25T12:47:52.175990Z", - "iopub.status.idle": "2026-09-25T12:47:55.539272Z", - "shell.execute_reply": "2026-09-25T12:47:55.539057Z" + "iopub.execute_input": "2026-09-25T12:49:06.690009Z", + "iopub.status.busy": "2026-09-25T12:49:06.689892Z", + "iopub.status.idle": "2026-09-25T12:49:09.929012Z", + "shell.execute_reply": "2026-09-25T12:49:09.928790Z" } }, "outputs": [ + { + "name": "stderr", + "output_type": "stream", + "text": [ + "/opt/anaconda3/envs/EngiBench312/lib/python3.12/site-packages/gymnasium/spaces/box.py:235: UserWarning: \u001b[33mWARN: Box low's precision lowered by casting to float32, current low.dtype=float64\u001b[0m\n", + " gym.logger.warn(\n", + "/opt/anaconda3/envs/EngiBench312/lib/python3.12/site-packages/gymnasium/spaces/box.py:305: UserWarning: \u001b[33mWARN: Box high's precision lowered by casting to float32, current high.dtype=float64\u001b[0m\n", + " gym.logger.warn(\n" + ] + }, { "data": { "text/html": [ @@ -273,12 +283,14 @@ "\n", "import pandas as pd\n", "\n", + "from engiopt.evaluation.registry import METRICS\n", + "\n", + "# Importing the metrics package registers the built-in metrics.\n", + "import engiopt.evaluation.metrics # noqa: F401 # isort: skip\n", + "\n", "warnings.filterwarnings(\"ignore\", category=UserWarning, module=\"gymnasium\")\n", "pd.set_option(\"display.max_colwidth\", None)\n", "\n", - "from engiopt.evaluation.registry import METRICS\n", - "import engiopt.evaluation.metrics # noqa: F401 -- importing registers the built-in metrics\n", - "\n", "METRICS.explain()" ] }, @@ -324,16 +336,16 @@ "id": "4e14a02e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.540961Z", - "iopub.status.busy": "2026-09-25T12:47:55.540768Z", - "iopub.status.idle": "2026-09-25T12:47:55.543674Z", - "shell.execute_reply": "2026-09-25T12:47:55.543501Z" + "iopub.execute_input": "2026-09-25T12:49:09.930694Z", + "iopub.status.busy": "2026-09-25T12:49:09.930494Z", + "iopub.status.idle": "2026-09-25T12:49:09.933599Z", + "shell.execute_reply": "2026-09-25T12:49:09.933419Z" } }, "outputs": [], "source": [ - "import numpy as np\n", "from gymnasium import spaces\n", + "import numpy as np\n", "\n", "\n", "class ToyProblem:\n", @@ -341,7 +353,7 @@ "\n", " design_space = spaces.Box(0.0, 1.0, shape=(8, 10), dtype=np.float64)\n", "\n", - " def check_constraints(self, design, config):\n", + " def check_constraints(self, design, config): # noqa: ARG002 - a real Problem reads `config`\n", " class Violations:\n", " violations = [] if self.design_space.contains(design) else [\"outside design space\"]\n", "\n", @@ -353,6 +365,7 @@ "\n", "\n", "def optimal_design(c):\n", + " \"\"\"The one optimal design for condition `c`: a field of mean `c`.\"\"\"\n", " return np.clip(np.full((8, 10), c) + rng.normal(0, 0.01, (8, 10)), 0, 1)\n", "\n", "\n", @@ -393,15 +406,16 @@ "id": "6bfdaf3f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.544702Z", - "iopub.status.busy": "2026-09-25T12:47:55.544633Z", - "iopub.status.idle": "2026-09-25T12:47:55.546472Z", - "shell.execute_reply": "2026-09-25T12:47:55.546293Z" + "iopub.execute_input": "2026-09-25T12:49:09.934763Z", + "iopub.status.busy": "2026-09-25T12:49:09.934679Z", + "iopub.status.idle": "2026-09-25T12:49:09.936706Z", + "shell.execute_reply": "2026-09-25T12:49:09.936503Z" } }, "outputs": [], "source": [ "def lookup_table(c):\n", + " \"\"\"The training design whose condition is nearest to `c`.\"\"\"\n", " return TRAIN[np.argmin(np.abs(train_conditions - c))]\n", "\n", "\n", @@ -428,10 +442,10 @@ "id": "935e4407", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.547576Z", - "iopub.status.busy": "2026-09-25T12:47:55.547495Z", - "iopub.status.idle": "2026-09-25T12:47:55.554173Z", - "shell.execute_reply": "2026-09-25T12:47:55.553970Z" + "iopub.execute_input": "2026-09-25T12:49:09.937824Z", + "iopub.status.busy": "2026-09-25T12:49:09.937741Z", + "iopub.status.idle": "2026-09-25T12:49:09.945283Z", + "shell.execute_reply": "2026-09-25T12:49:09.945049Z" } }, "outputs": [ @@ -637,10 +651,10 @@ "id": "b7bb486e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.555221Z", - "iopub.status.busy": "2026-09-25T12:47:55.555159Z", - "iopub.status.idle": "2026-09-25T12:47:55.558994Z", - "shell.execute_reply": "2026-09-25T12:47:55.558784Z" + "iopub.execute_input": "2026-09-25T12:49:09.946439Z", + "iopub.status.busy": "2026-09-25T12:49:09.946363Z", + "iopub.status.idle": "2026-09-25T12:49:09.950661Z", + "shell.execute_reply": "2026-09-25T12:49:09.950448Z" } }, "outputs": [ @@ -898,10 +912,10 @@ "id": "dd93537a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.560043Z", - "iopub.status.busy": "2026-09-25T12:47:55.559988Z", - "iopub.status.idle": "2026-09-25T12:47:55.563565Z", - "shell.execute_reply": "2026-09-25T12:47:55.563385Z" + "iopub.execute_input": "2026-09-25T12:49:09.951873Z", + "iopub.status.busy": "2026-09-25T12:49:09.951812Z", + "iopub.status.idle": "2026-09-25T12:49:09.955872Z", + "shell.execute_reply": "2026-09-25T12:49:09.955657Z" } }, "outputs": [ @@ -1091,10 +1105,10 @@ "id": "6150f040", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.564514Z", - "iopub.status.busy": "2026-09-25T12:47:55.564460Z", - "iopub.status.idle": "2026-09-25T12:47:55.565905Z", - "shell.execute_reply": "2026-09-25T12:47:55.565730Z" + "iopub.execute_input": "2026-09-25T12:49:09.956919Z", + "iopub.status.busy": "2026-09-25T12:49:09.956858Z", + "iopub.status.idle": "2026-09-25T12:49:09.958462Z", + "shell.execute_reply": "2026-09-25T12:49:09.958281Z" } }, "outputs": [ @@ -1140,10 +1154,10 @@ "id": "36e87484", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.567135Z", - "iopub.status.busy": "2026-09-25T12:47:55.567064Z", - "iopub.status.idle": "2026-09-25T12:47:55.572175Z", - "shell.execute_reply": "2026-09-25T12:47:55.571967Z" + "iopub.execute_input": "2026-09-25T12:49:09.959603Z", + "iopub.status.busy": "2026-09-25T12:49:09.959538Z", + "iopub.status.idle": "2026-09-25T12:49:09.965317Z", + "shell.execute_reply": "2026-09-25T12:49:09.965072Z" } }, "outputs": [ @@ -1247,10 +1261,10 @@ "id": "ef9a97f4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.573265Z", - "iopub.status.busy": "2026-09-25T12:47:55.573209Z", - "iopub.status.idle": "2026-09-25T12:47:55.576955Z", - "shell.execute_reply": "2026-09-25T12:47:55.576764Z" + "iopub.execute_input": "2026-09-25T12:49:09.966454Z", + "iopub.status.busy": "2026-09-25T12:49:09.966396Z", + "iopub.status.idle": "2026-09-25T12:49:09.970315Z", + "shell.execute_reply": "2026-09-25T12:49:09.970122Z" } }, "outputs": [ @@ -1386,10 +1400,10 @@ "id": "1adaa8bf", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:47:55.578020Z", - "iopub.status.busy": "2026-09-25T12:47:55.577968Z", - "iopub.status.idle": "2026-09-25T12:47:55.581500Z", - "shell.execute_reply": "2026-09-25T12:47:55.581292Z" + "iopub.execute_input": "2026-09-25T12:49:09.971456Z", + "iopub.status.busy": "2026-09-25T12:49:09.971392Z", + "iopub.status.idle": "2026-09-25T12:49:09.975471Z", + "shell.execute_reply": "2026-09-25T12:49:09.975235Z" } }, "outputs": [ From 3cc415011d6f46f640b10829455d28e80cf120d7 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Mon, 28 Sep 2026 14:49:21 +0200 Subject: [PATCH 08/18] Say what each metric measures and how to read its scale The first docstring line is the only text a reader gets from explain(), and several read as yes-or-no questions when the column is a quantity with a direction. Each line now names the quantity, says how it is computed in a phrase, and ends with what the two ends of the scale mean. Per-design metrics say "for each design" rather than "average", since the board's aggregation decides which that becomes. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluation/metrics/builtin.py | 40 ++--- example_metrics_suite.ipynb | 224 +++++++++++++------------- 2 files changed, 132 insertions(+), 132 deletions(-) diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index b091974e..325097fb 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -34,7 +34,7 @@ higher_is_better=False, ) def mmd(ctx: EvaluationContext) -> float: - """Taken as a set, how different are the generated designs from the reference optimal designs?""" + """Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.""" return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.sigma)) @@ -50,7 +50,7 @@ def mmd(ctx: EvaluationContext) -> float: higher_is_better=True, ) def dpp(ctx: EvaluationContext) -> float: - """How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? + """Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. The n-th root of the DPP kernel determinant, i.e. the geometric mean of the kernel's eigenvalues. The raw determinant that earlier boards published under @@ -78,7 +78,7 @@ def dpp(ctx: EvaluationContext) -> float: pixel_only=True, ) def train_distance(ctx: EvaluationContext) -> float: - """How far is each generated design from the nearest design the model was trained on? + """Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. Per-element RMS distance to the closest training design, so one tolerance means the same thing on problems of different resolution. Zero means the @@ -104,7 +104,7 @@ def train_distance(ctx: EvaluationContext) -> float: pixel_only=True, ) def copy_rate(ctx: EvaluationContext) -> float: - """What fraction of the generated designs are copies of a design the model could have seen? + """Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. A design counts as a copy when it sits within `copy_tol` (per-element RMS) of any design in the corpus a model could reproduce: the training split *and* @@ -131,7 +131,7 @@ def copy_rate(ctx: EvaluationContext) -> float: higher_is_better=None, ) def cond_sens(ctx: EvaluationContext) -> float: - """When the same model is given different conditions, does its output change? + """How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. Mean per-element RMS change in output when each sample is given another's conditions. @@ -176,7 +176,7 @@ def cond_sens(ctx: EvaluationContext) -> float: higher_is_better=False, ) def viol(ctx: EvaluationContext) -> float: - """What fraction of the generated designs violate the problem's constraints? + """Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. Defined for every problem: `problem.check_constraints` always applies, and the spec's `volume_condition` adds the volume-fraction budget for problems @@ -203,7 +203,7 @@ def viol(ctx: EvaluationContext) -> float: higher_is_better=False, ) def iog(ctx: EvaluationContext) -> float: - """How much worse than the reference optimum is the generated design, exactly as generated?""" + """Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.""" return ctx.reduce(ctx.optimization.iog) @@ -214,7 +214,7 @@ def iog(ctx: EvaluationContext) -> float: higher_is_better=False, ) def cog(ctx: EvaluationContext) -> float: - """Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? + """Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. For each generated design, the optimizer is started from that design and `objective(step) - objective(reference optimum)` is summed over every step it @@ -235,7 +235,7 @@ def cog(ctx: EvaluationContext) -> float: higher_is_better=False, ) def fog(ctx: EvaluationContext) -> float: - """Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?""" + """Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.""" return ctx.reduce(ctx.optimization.fog) @@ -251,7 +251,7 @@ def fog(ctx: EvaluationContext) -> float: higher_is_better=False, ) def per_condition_distance(ctx: EvaluationContext) -> float: - """For each set of conditions, how far is the generated design from the reference optimum for those same conditions? + """Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. Design `i` is compared with reference design `i`, which answers the same conditions. That is a sharper question than any set-level metric can ask: @@ -273,7 +273,7 @@ def per_condition_distance(ctx: EvaluationContext) -> float: pixel_only=True, ) def volume_error(ctx: EvaluationContext) -> float: - """How far is each design's material fraction from the one its conditions requested? + """Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. The volume fraction of a density field is its mean, so this needs nothing fitted: it is the exact error on the one condition that can be read straight @@ -304,7 +304,7 @@ def volume_error(ctx: EvaluationContext) -> float: higher_is_better=True, ) def coverage(ctx: EvaluationContext) -> float: - """What fraction of the reference optima have a generated design nearby? + """Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. Distribution distance can look healthy while whole regions go unvisited, so this counts reference designs directly: a mode the generator never produces @@ -326,7 +326,7 @@ def coverage(ctx: EvaluationContext) -> float: higher_is_better=True, ) def vendi(ctx: EvaluationContext) -> float: - """How many genuinely distinct designs are in the generated set, as an effective count? + """Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. The exponential of the entropy of the similarity kernel's eigenvalues, so it reads as a count: n identical designs score 1, n mutually dissimilar ones @@ -371,7 +371,7 @@ def _calls_to_settle(path: np.ndarray, band: float = SETTLE_BAND) -> float: higher_is_better=False, ) def calls_to_settle(ctx: EvaluationContext) -> float: - """How many optimizer calls until the objective stays within 5% of where it ends up? + """Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. Settling time in the control-theory sense: last exit from the band, so a trajectory that dips in and leaves again is not credited. Zero means the @@ -390,7 +390,7 @@ def calls_to_settle(ctx: EvaluationContext) -> float: outputs=tuple(f"gap_after_{k}_calls" for k in GAP_AFTER_CALLS), ) def gap_after_calls(ctx: EvaluationContext) -> dict[str, float]: - """How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls? + """Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. A path shorter than the budget has converged early, so its final value is carried forward: the optimizer would have spent the remaining calls and @@ -411,7 +411,7 @@ def gap_after_calls(ctx: EvaluationContext) -> dict[str, float]: higher_is_better=True, ) def reaches_reference_rate(ctx: EvaluationContext) -> float: - """What fraction of the generated designs, once re-optimized, reach or beat the reference optimum? + """Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. The gap is measured against the reference optimum, so reaching it means the gap touches zero at some call. A rate rather than a call count, because a @@ -431,7 +431,7 @@ def reaches_reference_rate(ctx: EvaluationContext) -> float: higher_is_better=True, ) def first_call_gain(ctx: EvaluationContext) -> float: - """What fraction of the whole re-optimization's improvement does the first optimizer call deliver? + """Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. Near 1 means the design's defects clear immediately and the rest is refinement; near 0 means the warm start bought nothing. Designs with nothing @@ -461,7 +461,7 @@ def first_call_gain(ctx: EvaluationContext) -> float: pixel_only=True, ) def generation_seconds(ctx: EvaluationContext) -> float: - """How many seconds did the model take to generate this batch of designs? + """Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. Wall-clock, so it compares models within one run on one machine, not across machines. NaN when the designs did not come from a timed `Generator.sample`. @@ -477,7 +477,7 @@ def generation_seconds(ctx: EvaluationContext) -> float: pixel_only=True, ) def n_parameters(ctx: EvaluationContext) -> float: - """How many trainable parameters does the model have? + """Number of trainable parameters in the model. Blank for a model with no network. The hardware-independent companion to `generation_seconds`. Neither is a complete account of cost alone: a small diffusion model can be slower than @@ -494,7 +494,7 @@ def n_parameters(ctx: EvaluationContext) -> float: pixel_only=True, ) def train_minutes(ctx: EvaluationContext) -> float: - """How many minutes did the model take to train? + """Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. Prices the decision to adopt the method rather than a forward pass; it is where a lookup table and a diffusion model differ by orders of magnitude. diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 4971b106..016b6878 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -38,10 +38,10 @@ "id": "25da8a4e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:06.690009Z", - "iopub.status.busy": "2026-09-25T12:49:06.689892Z", - "iopub.status.idle": "2026-09-25T12:49:09.929012Z", - "shell.execute_reply": "2026-09-25T12:49:09.928790Z" + "iopub.execute_input": "2026-09-28T12:49:17.781315Z", + "iopub.status.busy": "2026-09-28T12:49:17.781085Z", + "iopub.status.idle": "2026-09-28T12:49:20.355223Z", + "shell.execute_reply": "2026-09-28T12:49:20.354974Z" } }, "outputs": [ @@ -85,140 +85,140 @@ " \n", " \n", " mmd\n", - " Taken as a set, how different are the generated designs from the reference optimal designs?\n", + " Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.\n", " distribution\n", " lower is better\n", " cheap\n", " \n", " \n", " dpp\n", - " How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set?\n", + " Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed.\n", " diversity\n", " higher is better\n", " cheap\n", " \n", " \n", " train_distance\n", - " How far is each generated design from the nearest design the model was trained on?\n", + " Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.\n", " memorization\n", " diagnostic\n", " cheap\n", " \n", " \n", " copy_rate\n", - " What fraction of the generated designs are copies of a design the model could have seen?\n", + " Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval.\n", " memorization\n", " diagnostic\n", " cheap\n", " \n", " \n", " cond_sens\n", - " When the same model is given different conditions, does its output change?\n", + " How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.\n", " conditions\n", " diagnostic\n", " cheap\n", " \n", " \n", " viol\n", - " What fraction of the generated designs violate the problem's constraints?\n", + " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", " feasibility\n", " lower is better\n", " cheap\n", " \n", " \n", " iog\n", - " How much worse than the reference optimum is the generated design, exactly as generated?\n", + " Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.\n", " performance\n", " lower is better\n", " expensive\n", " \n", " \n", " cog\n", - " Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step?\n", + " Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted.\n", " performance\n", " lower is better\n", " expensive\n", " \n", " \n", " fog\n", - " Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes?\n", + " Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.\n", " performance\n", " lower is better\n", " expensive\n", " \n", " \n", " per_condition_distance\n", - " For each set of conditions, how far is the generated design from the reference optimum for those same conditions?\n", + " Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.\n", " conditions\n", " lower is better\n", " cheap\n", " \n", " \n", " volume_error\n", - " How far is each design's material fraction from the one its conditions requested?\n", + " Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.\n", " conditions\n", " lower is better\n", " cheap\n", " \n", " \n", " coverage\n", - " What fraction of the reference optima have a generated design nearby?\n", + " Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.\n", " distribution\n", " higher is better\n", " cheap\n", " \n", " \n", " vendi\n", - " How many genuinely distinct designs are in the generated set, as an effective count?\n", + " Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.\n", " diversity\n", " higher is better\n", " cheap\n", " \n", " \n", " calls_to_settle\n", - " How many optimizer calls until the objective stays within 5% of where it ends up?\n", + " Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle.\n", " performance\n", " lower is better\n", " expensive\n", " \n", " \n", " gap_after_calls\n", - " How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls?\n", + " Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.\n", " performance\n", " lower is better\n", " expensive\n", " \n", " \n", " reaches_reference_rate\n", - " What fraction of the generated designs, once re-optimized, reach or beat the reference optimum?\n", + " Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there.\n", " performance\n", " higher is better\n", " expensive\n", " \n", " \n", " first_call_gain\n", - " What fraction of the whole re-optimization's improvement does the first optimizer call deliver?\n", + " Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step.\n", " performance\n", " higher is better\n", " expensive\n", " \n", " \n", " generation_seconds\n", - " How many seconds did the model take to generate this batch of designs?\n", + " Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.\n", " cost\n", " lower is better\n", " cheap\n", " \n", " \n", " n_parameters\n", - " How many trainable parameters does the model have?\n", + " Number of trainable parameters in the model. Blank for a model with no network.\n", " cost\n", " lower is better\n", " cheap\n", " \n", " \n", " train_minutes\n", - " How many minutes did the model take to train?\n", + " Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.\n", " cost\n", " lower is better\n", " cheap\n", @@ -228,27 +228,27 @@ "" ], "text/plain": [ - " question \\\n", - "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", - "dpp How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? \n", - "train_distance How far is each generated design from the nearest design the model was trained on? \n", - "copy_rate What fraction of the generated designs are copies of a design the model could have seen? \n", - "cond_sens When the same model is given different conditions, does its output change? \n", - "viol What fraction of the generated designs violate the problem's constraints? \n", - "iog How much worse than the reference optimum is the generated design, exactly as generated? \n", - "cog Starting the optimizer from the generated design, how much worse than optimal is it, summed over every optimizer step? \n", - "fog Starting the optimizer from the generated design, how much worse than optimal is it once the optimizer finishes? \n", - "per_condition_distance For each set of conditions, how far is the generated design from the reference optimum for those same conditions? \n", - "volume_error How far is each design's material fraction from the one its conditions requested? \n", - "coverage What fraction of the reference optima have a generated design nearby? \n", - "vendi How many genuinely distinct designs are in the generated set, as an effective count? \n", - "calls_to_settle How many optimizer calls until the objective stays within 5% of where it ends up? \n", - "gap_after_calls How much optimality gap remains after the optimizer's first 1, 2, 5 and 10 calls? \n", - "reaches_reference_rate What fraction of the generated designs, once re-optimized, reach or beat the reference optimum? \n", - "first_call_gain What fraction of the whole re-optimization's improvement does the first optimizer call deliver? \n", - "generation_seconds How many seconds did the model take to generate this batch of designs? \n", - "n_parameters How many trainable parameters does the model have? \n", - "train_minutes How many minutes did the model take to train? \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "calls_to_settle Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. \n", + "gap_after_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", + "first_call_gain Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", "\n", " family direction cost \n", "mmd distribution lower is better cheap \n", @@ -336,10 +336,10 @@ "id": "4e14a02e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.930694Z", - "iopub.status.busy": "2026-09-25T12:49:09.930494Z", - "iopub.status.idle": "2026-09-25T12:49:09.933599Z", - "shell.execute_reply": "2026-09-25T12:49:09.933419Z" + "iopub.execute_input": "2026-09-28T12:49:20.356600Z", + "iopub.status.busy": "2026-09-28T12:49:20.356456Z", + "iopub.status.idle": "2026-09-28T12:49:20.359306Z", + "shell.execute_reply": "2026-09-28T12:49:20.359107Z" } }, "outputs": [], @@ -406,10 +406,10 @@ "id": "6bfdaf3f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.934763Z", - "iopub.status.busy": "2026-09-25T12:49:09.934679Z", - "iopub.status.idle": "2026-09-25T12:49:09.936706Z", - "shell.execute_reply": "2026-09-25T12:49:09.936503Z" + "iopub.execute_input": "2026-09-28T12:49:20.360352Z", + "iopub.status.busy": "2026-09-28T12:49:20.360299Z", + "iopub.status.idle": "2026-09-28T12:49:20.362328Z", + "shell.execute_reply": "2026-09-28T12:49:20.362136Z" } }, "outputs": [], @@ -442,10 +442,10 @@ "id": "935e4407", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.937824Z", - "iopub.status.busy": "2026-09-25T12:49:09.937741Z", - "iopub.status.idle": "2026-09-25T12:49:09.945283Z", - "shell.execute_reply": "2026-09-25T12:49:09.945049Z" + "iopub.execute_input": "2026-09-28T12:49:20.363310Z", + "iopub.status.busy": "2026-09-28T12:49:20.363257Z", + "iopub.status.idle": "2026-09-28T12:49:20.369633Z", + "shell.execute_reply": "2026-09-28T12:49:20.369431Z" } }, "outputs": [ @@ -651,10 +651,10 @@ "id": "b7bb486e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.946439Z", - "iopub.status.busy": "2026-09-25T12:49:09.946363Z", - "iopub.status.idle": "2026-09-25T12:49:09.950661Z", - "shell.execute_reply": "2026-09-25T12:49:09.950448Z" + "iopub.execute_input": "2026-09-28T12:49:20.370693Z", + "iopub.status.busy": "2026-09-28T12:49:20.370634Z", + "iopub.status.idle": "2026-09-28T12:49:20.374316Z", + "shell.execute_reply": "2026-09-28T12:49:20.374125Z" } }, "outputs": [ @@ -688,91 +688,91 @@ " \n", " \n", " mmd\n", - " Taken as a set, how different are the generated designs from the reference optimal designs?\n", + " Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.\n", " lower is better\n", " permuted_conditions\n", " 0.279866\n", " \n", " \n", " dpp\n", - " How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set?\n", + " Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed.\n", " higher is better\n", " noise\n", " 0.090922\n", " \n", " \n", " train_distance\n", - " How far is each generated design from the nearest design the model was trained on?\n", + " Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.\n", " diagnostic\n", " \n", " 0.014603\n", " \n", " \n", " copy_rate\n", - " What fraction of the generated designs are copies of a design the model could have seen?\n", + " Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval.\n", " diagnostic\n", " \n", " 0.000000\n", " \n", " \n", " cond_sens\n", - " When the same model is given different conditions, does its output change?\n", + " How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.\n", " diagnostic\n", " \n", " NaN\n", " \n", " \n", " viol\n", - " What fraction of the generated designs violate the problem's constraints?\n", + " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", " lower is better\n", " tie: honest, lookup_table, unconditional, permuted_conditions, noise\n", " 0.000000\n", " \n", " \n", " per_condition_distance\n", - " For each set of conditions, how far is the generated design from the reference optimum for those same conditions?\n", + " Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.\n", " lower is better\n", " honest\n", " NaN\n", " \n", " \n", " volume_error\n", - " How far is each design's material fraction from the one its conditions requested?\n", + " Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.\n", " lower is better\n", " \n", " NaN\n", " \n", " \n", " coverage\n", - " What fraction of the reference optima have a generated design nearby?\n", + " Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.\n", " higher is better\n", " tie: honest, lookup_table, permuted_conditions\n", " 0.500000\n", " \n", " \n", " vendi\n", - " How many genuinely distinct designs are in the generated set, as an effective count?\n", + " Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.\n", " higher is better\n", " noise\n", " 2.868064\n", " \n", " \n", " generation_seconds\n", - " How many seconds did the model take to generate this batch of designs?\n", + " Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.\n", " lower is better\n", " \n", " NaN\n", " \n", " \n", " n_parameters\n", - " How many trainable parameters does the model have?\n", + " Number of trainable parameters in the model. Blank for a model with no network.\n", " lower is better\n", " \n", " NaN\n", " \n", " \n", " train_minutes\n", - " How many minutes did the model take to train?\n", + " Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.\n", " lower is better\n", " \n", " NaN\n", @@ -782,20 +782,20 @@ "" ], "text/plain": [ - " question \\\n", - "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", - "dpp How spread out are the generated designs, on a 0-to-1 scale where 1 is a perfectly diverse set? \n", - "train_distance How far is each generated design from the nearest design the model was trained on? \n", - "copy_rate What fraction of the generated designs are copies of a design the model could have seen? \n", - "cond_sens When the same model is given different conditions, does its output change? \n", - "viol What fraction of the generated designs violate the problem's constraints? \n", - "per_condition_distance For each set of conditions, how far is the generated design from the reference optimum for those same conditions? \n", - "volume_error How far is each design's material fraction from the one its conditions requested? \n", - "coverage What fraction of the reference optima have a generated design nearby? \n", - "vendi How many genuinely distinct designs are in the generated set, as an effective count? \n", - "generation_seconds How many seconds did the model take to generate this batch of designs? \n", - "n_parameters How many trainable parameters does the model have? \n", - "train_minutes How many minutes did the model take to train? \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", "\n", " direction \\\n", "mmd lower is better \n", @@ -912,10 +912,10 @@ "id": "dd93537a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.951873Z", - "iopub.status.busy": "2026-09-25T12:49:09.951812Z", - "iopub.status.idle": "2026-09-25T12:49:09.955872Z", - "shell.execute_reply": "2026-09-25T12:49:09.955657Z" + "iopub.execute_input": "2026-09-28T12:49:20.375403Z", + "iopub.status.busy": "2026-09-28T12:49:20.375348Z", + "iopub.status.idle": "2026-09-28T12:49:20.378918Z", + "shell.execute_reply": "2026-09-28T12:49:20.378744Z" } }, "outputs": [ @@ -1105,10 +1105,10 @@ "id": "6150f040", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.956919Z", - "iopub.status.busy": "2026-09-25T12:49:09.956858Z", - "iopub.status.idle": "2026-09-25T12:49:09.958462Z", - "shell.execute_reply": "2026-09-25T12:49:09.958281Z" + "iopub.execute_input": "2026-09-28T12:49:20.379839Z", + "iopub.status.busy": "2026-09-28T12:49:20.379786Z", + "iopub.status.idle": "2026-09-28T12:49:20.381289Z", + "shell.execute_reply": "2026-09-28T12:49:20.381070Z" } }, "outputs": [ @@ -1154,10 +1154,10 @@ "id": "36e87484", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.959603Z", - "iopub.status.busy": "2026-09-25T12:49:09.959538Z", - "iopub.status.idle": "2026-09-25T12:49:09.965317Z", - "shell.execute_reply": "2026-09-25T12:49:09.965072Z" + "iopub.execute_input": "2026-09-28T12:49:20.382346Z", + "iopub.status.busy": "2026-09-28T12:49:20.382293Z", + "iopub.status.idle": "2026-09-28T12:49:20.387136Z", + "shell.execute_reply": "2026-09-28T12:49:20.386952Z" } }, "outputs": [ @@ -1261,10 +1261,10 @@ "id": "ef9a97f4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.966454Z", - "iopub.status.busy": "2026-09-25T12:49:09.966396Z", - "iopub.status.idle": "2026-09-25T12:49:09.970315Z", - "shell.execute_reply": "2026-09-25T12:49:09.970122Z" + "iopub.execute_input": "2026-09-28T12:49:20.388281Z", + "iopub.status.busy": "2026-09-28T12:49:20.388219Z", + "iopub.status.idle": "2026-09-28T12:49:20.392442Z", + "shell.execute_reply": "2026-09-28T12:49:20.392169Z" } }, "outputs": [ @@ -1400,10 +1400,10 @@ "id": "1adaa8bf", "metadata": { "execution": { - "iopub.execute_input": "2026-09-25T12:49:09.971456Z", - "iopub.status.busy": "2026-09-25T12:49:09.971392Z", - "iopub.status.idle": "2026-09-25T12:49:09.975471Z", - "shell.execute_reply": "2026-09-25T12:49:09.975235Z" + "iopub.execute_input": "2026-09-28T12:49:20.393589Z", + "iopub.status.busy": "2026-09-28T12:49:20.393516Z", + "iopub.status.idle": "2026-09-28T12:49:20.397706Z", + "shell.execute_reply": "2026-09-28T12:49:20.397490Z" } }, "outputs": [ @@ -1437,7 +1437,7 @@ " \n", " \n", " mmd\n", - " Taken as a set, how different are the generated designs from the reference optimal designs?\n", + " Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.\n", " lower is better\n", " permuted_conditions\n", " 0.279866\n", @@ -1454,9 +1454,9 @@ "" ], "text/plain": [ - " question \\\n", - "mmd Taken as a set, how different are the generated designs from the reference optimal designs? \n", - "density_ratio How much material do the generated designs use, relative to the reference optima? \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "density_ratio How much material do the generated designs use, relative to the reference optima? \n", "\n", " direction picks real designs score \n", "mmd lower is better permuted_conditions 0.279866 \n", From 62812ca28df385dbbe712175420746eba3934757 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Mon, 28 Sep 2026 17:15:07 +0200 Subject: [PATCH 09/18] Run the metrics notebook on beams2d with published checkpoints The notebook scored a toy problem, which left cond_sens blank, made every copy rate zero or one, and let a reference-copying oracle pass for a model. It now loads four beams2d checkpoints from the Hub, builds three constructions from the dataset, and scores all seven under one spec with the physics columns included. Twelve conditions rather than the published fifty and medians rather than means, both declared in the notebook, so it runs in about seven minutes on a laptop. To score checkpoints and saved designs together, the evaluator gains `context_for_designs`, and `Board.from_evaluator` scores both under the evaluator's spec and keeps the sampled designs so they can be re-scored in another space. `Board.load` takes a config fingerprint per model, since two of the four families have no canonical checkpoint. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluation/board.py | 77 +- engiopt/evaluation/evaluator.py | 31 +- example_metrics_suite.ipynb | 1778 ++++++++++++++++++------------- tests/test_metrics_suite.py | 28 +- 4 files changed, 1138 insertions(+), 776 deletions(-) diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 93fcfa42..6f19d6de 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -15,10 +15,10 @@ from __future__ import annotations -from collections.abc import Callable +from collections.abc import Callable, Mapping from dataclasses import dataclass from dataclasses import field -from typing import Any, Literal, TYPE_CHECKING +from typing import Any, Literal import numpy as np import pandas as pd @@ -28,9 +28,6 @@ from engiopt.evaluation.registry import MetricRegistry from engiopt.evaluation.registry import METRICS -if TYPE_CHECKING: - from collections.abc import Mapping - SpaceFit = Callable[[np.ndarray, int], Callable[[np.ndarray], np.ndarray]] """Given the reference designs and a width, return a function projecting designs into the space.""" @@ -71,6 +68,7 @@ class Board: in whichever space is being scored, so the kernel is neither saturated nor empty on a problem it has never seen. A published spec pins a value. frame: The most recent `evaluate` result. + designs: The designs behind each row of `frame`, when the board sampled them itself. """ problem: Any @@ -79,6 +77,7 @@ class Board: sigma: float | None = None registry: MetricRegistry = field(default_factory=lambda: METRICS) frame: pd.DataFrame = field(default_factory=pd.DataFrame) + designs: dict[str, Any] = field(default_factory=dict) def evaluate( self, @@ -149,73 +148,87 @@ def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: return frame @classmethod - def from_generators( + def from_evaluator( cls, evaluator: Any, - generators: Mapping[str, Any], + models: Mapping[str, Any], *, + designs: Mapping[str, Any] | None = None, expensive: bool = False, metrics: list[str] | None = None, ) -> Board: - """Score loaded models through an `Evaluator`, which is what the live-model metrics need. + """Score loaded models, and optionally saved designs, under one evaluation spec. - `cond_sens` re-samples the model, the cost columns read its network, and - the physics columns run the simulator; none of that is possible from - saved designs. The evaluator supplies the spec, the reference designs - and the conditions, so there is no reference row here -- the spec's own - reference designs already are the comparison. + The evaluator supplies the spec, the reference designs, the conditions and + the copy corpus, so every row is comparable. Live models get every column; + `cond_sens` and the cost columns need a model to re-run or inspect, so they + stay blank for the `designs` rows. The designs each model produced are kept + on `Board.designs`, so they can be re-scored in another space or saved. Args: - evaluator: An `Evaluator` (anything with `.score(generator, only=, include_expensive=)`). - generators: Label -> loaded `Generator`. + evaluator: An `Evaluator` for the problem. + models: Label -> loaded `Generator`. + designs: Label -> one design per spec condition, e.g. a construction + built from the dataset. expensive: Whether to run the simulator-backed metrics. metrics: Metric names; defaults to the spec's list. Returns: - A `Board` with one row per generator. + A `Board` with one row per model and per designs entry, in that order. """ - rows = { - label: evaluator.score(generator, only=metrics, include_expensive=expensive) - for label, generator in generators.items() - } - frame = pd.DataFrame.from_dict(rows, orient="index") - board = cls(problem=getattr(evaluator, "problem", None), reference=getattr(evaluator, "resolved", None)) - board.frame = frame + rows: dict[str, dict[str, float]] = {} + sampled: dict[str, Any] = {} + for label, generator in models.items(): + ctx = evaluator.context_for(generator) + rows[label] = evaluator.score_context(ctx, only=metrics, include_expensive=expensive) + sampled[label] = ctx.gen_designs + for label, batch in (designs or {}).items(): + ctx = evaluator.context_for_designs(batch) + rows[label] = evaluator.score_context(ctx, only=metrics, include_expensive=expensive) + sampled[label] = ctx.gen_designs + board = cls(problem=evaluator.problem, reference=evaluator.resolved.ref_designs, sigma=evaluator.spec.sigma) + board.frame = pd.DataFrame.from_dict(rows, orient="index") + board.designs = sampled return board @classmethod def load( cls, problem_id: str, - models: list[str], + models: list[str] | Mapping[str, str | None], *, + designs: Mapping[str, Any] | None = None, seed: int = 1, - spec: str | None = None, + spec: Any = None, expensive: bool = False, ) -> Board: """Pull published checkpoints by name and score them: the workshop in one call. Args: problem_id: An EngiBench problem id, e.g. `"beams2d"`. - models: Generator names from `BUILTIN_GENERATORS`, e.g. `["cgan_cnn_2d", "vqgan"]`. + models: Generator names, e.g. `["diffusion_2d_cond", "vqgan"]`. A model + whose default configuration was never trained needs its config + fingerprint, given as a mapping: `{"cgan_cnn_2d": "825831f6"}`. + designs: Extra rows of saved designs, one per spec condition; see `from_evaluator`. seed: Training seed of the checkpoints to load. - spec: Spec reference such as `"beams2d/v2"`; defaults to the problem's current spec. + spec: Spec reference such as `"beams2d/v2"`, or an `EvalSpec`; defaults to the current spec. expensive: Whether to run the simulator-backed metrics. Returns: - A `Board` with one row per model, labelled `name/seed`. + A `Board` with one row per model, labelled by name. """ from engiopt.evaluation.evaluator import Evaluator from engiopt.utils.all_generators import BUILTIN_GENERATORS evaluator = Evaluator.for_problem(problem_id, spec=spec) + requested = models if isinstance(models, Mapping) else dict.fromkeys(models) generators = { - f"{name}/seed{seed}": BUILTIN_GENERATORS[name].from_pretrained( - evaluator.problem, problem_id=problem_id, seed=seed + name: BUILTIN_GENERATORS[name].from_pretrained( + evaluator.problem, problem_id=problem_id, seed=seed, config_fingerprint=fingerprint ) - for name in models + for name, fingerprint in requested.items() } - return cls.from_generators(evaluator, generators, expensive=expensive) + return cls.from_evaluator(evaluator, generators, designs=designs, expensive=expensive) def explain(self) -> pd.DataFrame: """Read every column of the last evaluation. diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index f1d5966a..3eb087b3 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -11,6 +11,7 @@ from __future__ import annotations +import dataclasses from dataclasses import dataclass from dataclasses import field import datetime as dt @@ -163,25 +164,43 @@ def context_for(self, generator: Generator) -> EvaluationContext: f"{generator.algo_id!r} supports {generator.design_kinds} design spaces, " f"but {self.problem_id!r} is {kind!r}." ) - designs = self._sample(generator) + ctx = self.context_for_designs(self._sample(generator)) + return dataclasses.replace( + ctx, + sample_seconds=generator.last_sample_seconds, + model_params=_parameter_count(generator), + train_minutes=getattr(generator, "train_minutes", None), + resample_permuted=self._permuted_sampler(generator), + ) + + def context_for_designs(self, designs: npt.NDArray[Any]) -> EvaluationContext: + """Wrap designs that did not come from a generator in the context a generator would get. + + A construction built from the dataset, or a batch saved earlier, is scored + against the same reference designs, conditions, kernel bandwidth and copy + corpus as a live model, so its row is comparable. Only the metrics that + must re-run a model (`cond_sens`) and the cost columns stay blank. + + Args: + designs: One design per spec condition, in the spec's condition order. + + Returns: + An `EvaluationContext` ready for `score_context`. + """ return EvaluationContext( problem=self.problem, problem_id=self.problem_id, - gen_designs=designs, + gen_designs=np.asarray(designs), ref_designs=self.resolved.ref_designs, conditions=self.resolved.conditions, sigma=self.spec.sigma, volfrac_tol=self.spec.volfrac_tol, volume_condition=self.spec.volume_condition, - sample_seconds=generator.last_sample_seconds, objective_weights=self.spec.objective_weights, objective_weight_condition=self.spec.objective_weight_condition, copy_corpus_fn=self.copy_corpus, copy_tol=self.spec.copy_tol, aggregation=self.spec.aggregation, - model_params=_parameter_count(generator), - train_minutes=getattr(generator, "train_minutes", None), - resample_permuted=self._permuted_sampler(generator), ) def _sample(self, generator: Generator, order: npt.NDArray[Any] | None = None) -> npt.NDArray[Any]: diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 016b6878..fe64bdcf 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -2,31 +2,32 @@ "cells": [ { "cell_type": "markdown", - "id": "e6135946", + "id": "cd775ce2", "metadata": {}, "source": [ - "# Using the metrics suite\n", + "# Using the metrics suite, on beams2d\n", "\n", - "You have designs from one or more models and want to know what they are worth. The\n", - "whole workflow is one registry and one object:\n", + "You have trained models and want to know what their designs are worth. This notebook\n", + "runs the whole evaluation surface on a real EngiBench problem, with published\n", + "checkpoints, the physics metrics included:\n", "\n", "| You want to... | You call |\n", "|---|---|\n", "| see every metric and what it measures | `METRICS.explain()` |\n", - "| score your models | `Board(problem, reference=..., train=...).evaluate(models)` |\n", - "| read the result | `board.explain()`, `board.rank(\"mmd\")` |\n", - "| change how per-design values are combined | `board.evaluate(models, aggregation=\"median\")` |\n", - "| ask the same questions in another space | `board.evaluate(models, space=\"pca\")` |\n", - "| score published checkpoints by name | `Board.load(\"beams2d\", [\"cgan_cnn_2d\", \"vqgan\"])` |\n", + "| score published checkpoints by name | `Board.load(\"beams2d\", [\"vqgan\", ...])` |\n", + "| score checkpoints and your own constructions under one spec | `Board.from_evaluator(evaluator, models, designs=...)` |\n", + "| read the result | `board.explain()`, `board.rank(\"fog\")` |\n", + "| ask the same questions in another space | `Board(...).evaluate(board.designs, space=\"pca\")` |\n", "| add a metric of your own | `@register_metric(...)` on a function |\n", "\n", - "Everything below runs in seconds on a toy problem. With a real EngiBench problem the\n", - "calls are identical." + "It downloads four checkpoints and the beams2d dataset from the Hub, samples twelve\n", + "designs per model, and runs the optimizer from each of them. Expect a few minutes on a\n", + "laptop." ] }, { "cell_type": "markdown", - "id": "9f94269b", + "id": "6c31865f", "metadata": {}, "source": [ "## 1. What metrics exist, and what does each one measure?" @@ -35,13 +36,13 @@ { "cell_type": "code", "execution_count": 1, - "id": "25da8a4e", + "id": "22f4b5a3", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:17.781315Z", - "iopub.status.busy": "2026-09-28T12:49:17.781085Z", - "iopub.status.idle": "2026-09-28T12:49:20.355223Z", - "shell.execute_reply": "2026-09-28T12:49:20.354974Z" + "iopub.execute_input": "2026-09-28T15:05:54.929474Z", + "iopub.status.busy": "2026-09-28T15:05:54.928715Z", + "iopub.status.idle": "2026-09-28T15:05:57.569493Z", + "shell.execute_reply": "2026-09-28T15:05:57.569236Z" } }, "outputs": [ @@ -288,167 +289,237 @@ "# Importing the metrics package registers the built-in metrics.\n", "import engiopt.evaluation.metrics # noqa: F401 # isort: skip\n", "\n", - "warnings.filterwarnings(\"ignore\", category=UserWarning, module=\"gymnasium\")\n", + "warnings.filterwarnings(\"ignore\")\n", "pd.set_option(\"display.max_colwidth\", None)\n", + "pd.set_option(\"display.width\", 200)\n", "\n", "METRICS.explain()" ] }, { "cell_type": "markdown", - "id": "daed2cfb", + "id": "0bf1c95d", "metadata": {}, "source": [ - "Each row is a question written so it makes sense without knowing the metric's name,\n", - "plus which way is better. Some say **diagnostic**: those columns tell you whether the\n", - "*other* columns can be trusted — a lookup table posts a perfect `mmd`, and only\n", - "`copy_rate` says so — and they are never ranked on.\n", - "\n", - "`cog` is worth a second look because it is the least obvious: the optimizer is started\n", - "from the generated design, and `objective(step) − objective(reference optimum)` is\n", - "summed over every step it takes. It is the area under the optimality-gap curve, so it\n", - "mixes how far from optimal the start was (`iog`) with how many steps were needed." + "Each row names the quantity, says how it is computed, and says what the two ends of\n", + "its scale mean. The direction column marks some metrics **diagnostic**: those columns\n", + "tell you whether the other columns can be trusted, and are never ranked on." ] }, { "cell_type": "markdown", - "id": "89a27d85", + "id": "556f7636", "metadata": {}, "source": [ - "## 2. A toy problem" + "## 2. The problem, and the evaluation spec" ] }, { "cell_type": "markdown", - "id": "3cf5be3a", + "id": "658d22e5", "metadata": {}, "source": [ - "Every EngiBench dataset has the same structure: **one optimal design per set of\n", - "conditions**, split into training and withheld test conditions. The toy keeps exactly\n", - "that and nothing else — the optimal design for condition `c` is a field with mean `c`.\n", - "With a real problem, `problem` is an EngiBench `Problem` and the two arrays come from\n", - "its dataset splits." + "Every published number on beams2d is computed under one frozen spec. The spec fixes\n", + "which fifty test conditions are scored, the reference optimum for each, the kernel\n", + "bandwidth, the copy tolerance, and how per-design values collapse to one number. Two\n", + "models scored under the same spec are comparable; two scored under different specs\n", + "are not, and every row records which spec produced it.\n", + "\n", + "The dataset has one optimal design per set of conditions. That is what makes a\n", + "lookup table a serious competitor here, and it is why the spec draws its conditions\n", + "from the withheld test split.\n", + "\n", + "This notebook departs from the published spec in two declared ways, so that it runs in\n", + "minutes. It scores twelve conditions rather than fifty, which cuts sampling and\n", + "optimizer time by the same factor, and so it drops the published spec's condition\n", + "digest, which pins the fifty. And it aggregates per-design values with the median rather\n", + "than the mean, because one diverged design drags a mean by orders of magnitude and the\n", + "physics columns are read as medians throughout EngiOpt. Numbers here are therefore not\n", + "comparable with the public leaderboard; the calls are identical." ] }, { "cell_type": "code", "execution_count": 2, - "id": "4e14a02e", + "id": "7f5b40c7", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.356600Z", - "iopub.status.busy": "2026-09-28T12:49:20.356456Z", - "iopub.status.idle": "2026-09-28T12:49:20.359306Z", - "shell.execute_reply": "2026-09-28T12:49:20.359107Z" + "iopub.execute_input": "2026-09-28T15:05:57.571900Z", + "iopub.status.busy": "2026-09-28T15:05:57.571724Z", + "iopub.status.idle": "2026-09-28T15:05:59.122592Z", + "shell.execute_reply": "2026-09-28T15:05:59.122294Z" } }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "conditions scored: 12 | design shape: (50, 100)\n", + "condition keys: ('volfrac', 'rmin', 'forcedist', 'overhang_constraint')\n", + "kernel sigma: 10.0 | copy tolerance: 0.01 | aggregation: median\n" + ] + } + ], "source": [ - "from gymnasium import spaces\n", - "import numpy as np\n", - "\n", - "\n", - "class ToyProblem:\n", - " \"\"\"All the cheap metrics need from a problem: a design space and a constraint check.\"\"\"\n", + "import dataclasses\n", "\n", - " design_space = spaces.Box(0.0, 1.0, shape=(8, 10), dtype=np.float64)\n", + "from engiopt.evaluation.evaluator import Evaluator\n", + "from engiopt.evaluation.spec import EvalSpec\n", "\n", - " def check_constraints(self, design, config): # noqa: ARG002 - a real Problem reads `config`\n", - " class Violations:\n", - " violations = [] if self.design_space.contains(design) else [\"outside design space\"]\n", + "published = EvalSpec.load(\"beams2d/v2\")\n", + "spec = dataclasses.replace(published, n_samples=12, condition_digest=None, aggregation=\"median\")\n", + "evaluator = Evaluator.for_problem(\"beams2d\", spec=spec)\n", "\n", - " return Violations()\n", - "\n", - "\n", - "rng = np.random.default_rng(0)\n", - "problem = ToyProblem()\n", - "\n", - "\n", - "def optimal_design(c):\n", - " \"\"\"The one optimal design for condition `c`: a field of mean `c`.\"\"\"\n", - " return np.clip(np.full((8, 10), c) + rng.normal(0, 0.01, (8, 10)), 0, 1)\n", - "\n", - "\n", - "train_conditions = rng.uniform(0.2, 0.8, 64)\n", - "test_conditions = rng.uniform(0.2, 0.8, 16)\n", - "TRAIN = np.stack([optimal_design(c) for c in train_conditions]) # what models are fitted on\n", - "REFERENCE = np.stack([optimal_design(c) for c in test_conditions]) # withheld; what models are scored against" + "REFERENCE = evaluator.resolved.ref_designs # the withheld optimum for each scored condition\n", + "print(\"conditions scored:\", len(REFERENCE), \"| design shape:\", REFERENCE.shape[1:])\n", + "print(\"condition keys:\", evaluator.resolved.condition_keys)\n", + "print(\"kernel sigma:\", spec.sigma, \"| copy tolerance:\", spec.copy_tol, \"| aggregation:\", spec.aggregation)" ] }, { "cell_type": "markdown", - "id": "78758876", + "id": "ea7d9e5b", "metadata": {}, "source": [ - "## 3. Five models" + "## 3. Three constructions built from the dataset" ] }, { "cell_type": "markdown", - "id": "21d478f4", + "id": "85a51183", "metadata": {}, "source": [ - "Three you might really train; two are **constructions** whose right score you know in\n", - "advance, so they test the metrics rather than the models.\n", + "Beside the trained models, three rows whose right score is known in advance. They are\n", + "not competitors. They are the probes that show what each column can and cannot see.\n", "\n", - "| Model | What it does |\n", - "|---|---|\n", - "| `honest` | the optimal design for each test condition |\n", - "| `lookup_table` | nearest neighbour over the training set; never sees a withheld design |\n", - "| `unconditional` | real training designs, ignoring the conditions |\n", - "| `permuted_conditions` | the correct withheld designs shuffled across conditions — every design a true optimum, every one for the wrong conditions |\n", - "| `noise` | uniform noise |" + "| Row | What it does | What it should reveal |\n", + "|---|---|---|\n", + "| `lookup_table` | for each scored condition, the training design whose conditions are nearest | a model that never generalizes, yet answers each condition plausibly |\n", + "| `unconditional` | training designs drawn at random, ignoring the conditions | a model that ignores its conditions entirely |\n", + "| `permuted_conditions` | the reference optima, shuffled across the conditions | a perfect set of designs, every one for the wrong conditions |" ] }, { "cell_type": "code", "execution_count": 3, - "id": "6bfdaf3f", + "id": "17e539a9", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.360352Z", - "iopub.status.busy": "2026-09-28T12:49:20.360299Z", - "iopub.status.idle": "2026-09-28T12:49:20.362328Z", - "shell.execute_reply": "2026-09-28T12:49:20.362136Z" + "iopub.execute_input": "2026-09-28T15:05:59.124844Z", + "iopub.status.busy": "2026-09-28T15:05:59.124741Z", + "iopub.status.idle": "2026-09-28T15:06:02.348254Z", + "shell.execute_reply": "2026-09-28T15:06:02.347983Z" } }, - "outputs": [], + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "training designs: (3880, 50, 100)\n" + ] + } + ], "source": [ - "def lookup_table(c):\n", - " \"\"\"The training design whose condition is nearest to `c`.\"\"\"\n", - " return TRAIN[np.argmin(np.abs(train_conditions - c))]\n", + "import numpy as np\n", "\n", + "train = evaluator.problem.dataset[\"train\"]\n", + "TRAIN = np.stack([np.asarray(design) for design in train[\"optimal_design\"]])\n", + "train_conditions = np.column_stack([np.asarray(train[key], dtype=float) for key in evaluator.resolved.condition_keys])\n", + "scored_conditions = evaluator.resolved.conditions_tensor.cpu().numpy().astype(float)\n", "\n", - "MODELS = {\n", - " \"honest\": np.stack([optimal_design(c) for c in test_conditions]),\n", - " \"lookup_table\": np.stack([lookup_table(c) for c in test_conditions]),\n", + "# Nearest neighbour in condition space, each condition standardized so none dominates by its units.\n", + "scale = train_conditions.std(axis=0)\n", + "scale[scale == 0] = 1.0\n", + "nearest = np.argmin(np.linalg.norm((train_conditions[None] - scored_conditions[:, None]) / scale, axis=2), axis=1)\n", + "\n", + "rng = np.random.default_rng(0)\n", + "CONSTRUCTIONS = {\n", + " \"lookup_table\": TRAIN[nearest],\n", " \"unconditional\": TRAIN[rng.integers(0, len(TRAIN), len(REFERENCE))],\n", " \"permuted_conditions\": REFERENCE[rng.permutation(len(REFERENCE))],\n", - " \"noise\": rng.random(REFERENCE.shape),\n", - "}" + "}\n", + "print(\"training designs:\", TRAIN.shape)" ] }, { "cell_type": "markdown", - "id": "8571bcc1", + "id": "84c32ab9", "metadata": {}, "source": [ - "## 4. Score them" + "## 4. Score the trained models and the constructions under one spec" + ] + }, + { + "cell_type": "markdown", + "id": "bd9e899f", + "metadata": {}, + "source": [ + "Four published checkpoints at training seed one. Two of them were trained with a\n", + "configuration that is not the script's default, so they are named by the fingerprint\n", + "of that configuration rather than by name alone. Everything else is one call.\n", + "\n", + "The physics columns run the optimizer from each design, so each cell below takes a\n", + "minute or two. The GAN and VQGAN family first, then the diffusion model, which samples\n", + "slowly, then the constructions." ] }, { "cell_type": "code", "execution_count": 4, - "id": "935e4407", + "id": "3634cb40", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.363310Z", - "iopub.status.busy": "2026-09-28T12:49:20.363257Z", - "iopub.status.idle": "2026-09-28T12:49:20.369633Z", - "shell.execute_reply": "2026-09-28T12:49:20.369431Z" + "iopub.execute_input": "2026-09-28T15:06:02.350507Z", + "iopub.status.busy": "2026-09-28T15:06:02.350400Z", + "iopub.status.idle": "2026-09-28T15:08:12.429733Z", + "shell.execute_reply": "2026-09-28T15:08:12.429489Z" } }, "outputs": [ + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "d1af91997b244d52a7da37d0e90ed747", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Fetching 8 files: 0%| | 0/8 [00:00\n", " \n", " mmd\n", + " coverage\n", + " vendi\n", " dpp\n", - " train_distance\n", - " copy_rate\n", - " cond_sens\n", " viol\n", - " per_condition_distance\n", " volume_error\n", - " coverage\n", - " vendi\n", - " generation_seconds\n", - " n_parameters\n", - " train_minutes\n", + " per_condition_distance\n", + " cond_sens\n", + " train_distance\n", + " copy_rate\n", + " ...\n", + " iog\n", + " cog\n", + " fog\n", + " calls_to_settle\n", + " gap_after_1_calls\n", + " gap_after_2_calls\n", + " gap_after_5_calls\n", + " gap_after_10_calls\n", + " reaches_reference_rate\n", + " first_call_gain\n", " \n", " \n", " \n", " \n", - " honest\n", - " 6.366160e-04\n", - " 0.035297\n", - " 0.014761\n", - " 0.0\n", - " NaN\n", + " cgan_cnn_2d\n", + " 0.5941\n", + " 0.50\n", + " 3.4974\n", + " 0.0849\n", + " 0.9167\n", + " 0.0722\n", + " 0.4144\n", + " 0.2407\n", + " 0.2939\n", " 0.0\n", - " 0.014059\n", - " NaN\n", + " ...\n", + " 1.816083e+09\n", + " 1.735349e+09\n", + " 3.0758\n", + " 1.0\n", + " 1.735347e+09\n", + " 1018.6727\n", + " 174.6169\n", + " 56.5027\n", + " 0.1667\n", " 1.0000\n", - " 3.254471\n", - " NaN\n", - " NaN\n", - " NaN\n", " \n", " \n", - " lookup_table\n", - " 1.037788e-03\n", - " 0.004358\n", - " 0.000000\n", - " 1.0\n", - " NaN\n", + " gan_cnn_2d\n", + " 1.0936\n", + " 0.25\n", + " 1.0916\n", + " 0.0006\n", + " 1.0000\n", + " 0.2407\n", + " 0.4447\n", + " 0.0000\n", + " 0.2632\n", " 0.0\n", - " 0.014891\n", - " NaN\n", + " ...\n", + " 8.950722e+09\n", + " 8.689641e+09\n", + " 0.0019\n", + " 1.0\n", + " 8.689632e+09\n", + " 4121.7064\n", + " 189.3875\n", + " 47.1883\n", + " 0.4167\n", " 1.0000\n", - " 3.241546\n", - " NaN\n", - " NaN\n", - " NaN\n", " \n", " \n", - " unconditional\n", - " 1.012030e-01\n", - " 0.006445\n", - " 0.000000\n", - " 1.0\n", - " NaN\n", + " vqgan\n", + " 0.0327\n", + " 1.00\n", + " 9.9069\n", + " 0.7916\n", + " 0.3333\n", + " 0.0058\n", + " 0.0833\n", + " 0.4704\n", + " 0.0531\n", " 0.0\n", - " 0.180742\n", - " NaN\n", - " 0.9375\n", - " 2.605583\n", - " NaN\n", - " NaN\n", - " NaN\n", + " ...\n", + " 2.413100e+00\n", + " 8.195000e-01\n", + " -0.6194\n", + " 32.0\n", + " 2.423100e+00\n", + " 5.6037\n", + " 0.5194\n", + " 0.0135\n", + " 0.6667\n", + " 0.1523\n", " \n", + " \n", + "\n", + "

3 rows × 23 columns

\n", + "" + ], + "text/plain": [ + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", + "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2939 0.0 ... 1.816083e+09 1.735349e+09 3.0758 1.0 \n", + "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2632 0.0 ... 8.950722e+09 8.689641e+09 0.0019 1.0 \n", + "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0531 0.0 ... 2.413100e+00 8.195000e-01 -0.6194 32.0 \n", + "\n", + " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", + "cgan_cnn_2d 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", + "gan_cnn_2d 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", + "vqgan 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", + "\n", + "[3 rows x 23 columns]" + ] + }, + "execution_count": 4, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from engiopt.evaluation.board import Board\n", + "from engiopt.utils.all_generators import BUILTIN_GENERATORS\n", + "\n", + "EXPENSIVE = True # set False to skip the optimizer and get the cheap columns in seconds\n", + "\n", + "\n", + "def load(name, fingerprint=None):\n", + " \"\"\"The published beams2d checkpoint of a model family at training seed one.\"\"\"\n", + " return BUILTIN_GENERATORS[name].from_pretrained(\n", + " evaluator.problem, problem_id=\"beams2d\", seed=1, config_fingerprint=fingerprint\n", + " )\n", + "\n", + "\n", + "fast_models = {\n", + " \"cgan_cnn_2d\": load(\"cgan_cnn_2d\", \"825831f6\"),\n", + " \"gan_cnn_2d\": load(\"gan_cnn_2d\", \"6293adb3\"),\n", + " \"vqgan\": load(\"vqgan\"),\n", + "}\n", + "board = Board.from_evaluator(evaluator, fast_models, expensive=EXPENSIVE)\n", + "board.frame.round(4)" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "5641c92d", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-28T15:08:12.432341Z", + "iopub.status.busy": "2026-09-28T15:08:12.432250Z", + "iopub.status.idle": "2026-09-28T15:11:41.043945Z", + "shell.execute_reply": "2026-09-28T15:11:41.043702Z" + } + }, + "outputs": [ + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "c82612395590444cbea28d7bcbe39a88", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Fetching 3 files: 0%| | 0/3 [00:00\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", - " \n", - " \n", - " \n", - " \n", + " \n", + " \n", " \n", - " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", - " \n", - " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmdcoveragevendidppviolvolume_errorper_condition_distancecond_senstrain_distancecopy_rate...iogcogfogcalls_to_settlegap_after_1_callsgap_after_2_callsgap_after_5_callsgap_after_10_callsreaches_reference_ratefirst_call_gain
permuted_conditions-1.110223e-160.0338970.014753diffusion_2d_cond0.11051.0NaN10.66810.86540.83330.12710.15320.39690.1460.00.176872NaN...-1.72116.0217-0.912817.0-1.712726.39796.19040.31090.91672.9236
\n", + "

1 rows × 23 columns

\n", + "" + ], + "text/plain": [ + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", + "diffusion_2d_cond 0.1105 1.0 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.146 0.0 ... -1.7211 6.0217 -0.9128 17.0 \n", + "\n", + " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", + "diffusion_2d_cond -1.7127 26.3979 6.1904 0.3109 0.9167 2.9236 \n", + "\n", + "[1 rows x 23 columns]" + ] + }, + "execution_count": 5, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "diffusion = Board.from_evaluator(evaluator, {\"diffusion_2d_cond\": load(\"diffusion_2d_cond\")}, expensive=EXPENSIVE)\n", + "board.frame = pd.concat([board.frame, diffusion.frame])\n", + "board.designs.update(diffusion.designs)\n", + "diffusion.frame.round(4)" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "950db794", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-28T15:11:41.046650Z", + "iopub.status.busy": "2026-09-28T15:11:41.046558Z", + "iopub.status.idle": "2026-09-28T15:13:05.685877Z", + "shell.execute_reply": "2026-09-28T15:13:05.685654Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", - " \n", - " \n", - " \n", - " \n", " \n", " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", - " \n", - " \n", - " \n", - " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", " \n", " \n", "
mmdcoveragevendidppviolvolume_errorper_condition_distancecond_senstrain_distancecopy_rate...iogcogfogcalls_to_settlegap_after_1_callsgap_after_2_callsgap_after_5_callsgap_after_10_callsreaches_reference_ratefirst_call_gain
cgan_cnn_2d0.59410.503.49740.08490.91670.07220.41440.24070.29390.00...1.816083e+091.735349e+093.07581.01.735347e+091018.6727174.616956.50270.16671.00003.248472NaNNaNNaN
noise4.239221e-010.9985880.2857830.0NaN0.00.332793NaNgan_cnn_2d1.09360.251.09160.00061.00000.24070.44470.000015.976387NaNNaNNaN0.26320.00...8.950722e+098.689641e+090.00191.08.689632e+094121.7064189.387547.18830.41671.0000
reference (split-half)2.798658e-010.0909220.0146030.0NaN0.0NaNNaN0.50002.868064NaNNaNNaNvqgan0.03271.009.90690.79160.33330.00580.08330.47040.05310.00...2.413100e+008.195000e-01-0.619432.02.423100e+005.60370.51940.01350.66670.1523
diffusion_2d_cond0.11051.0010.66810.86540.83330.12710.15320.39690.14600.00...-1.721100e+006.021700e+00-0.912817.0-1.712700e+0026.39796.19040.31090.91672.9236
lookup_table0.07291.009.71220.30510.75000.01250.0569NaN0.01170.50...9.574600e+007.314000e-01-0.736329.09.590800e+008.28253.33130.36760.58330.3145
unconditional0.16311.0011.17340.92770.91670.05620.3993NaN0.05040.25...4.805870e+032.817597e+041.42876.04.806746e+031994.096366.67044.23960.16670.4913
permuted_conditions0.00001.0010.08890.31620.91670.05620.4231NaN0.05251.00...8.783402e+078.788590e+074.45692.08.782727e+071865.304085.998227.05100.00000.8233
\n", + "

7 rows × 23 columns

\n", "
" ], "text/plain": [ - " mmd dpp train_distance copy_rate \\\n", - "honest 6.366160e-04 0.035297 0.014761 0.0 \n", - "lookup_table 1.037788e-03 0.004358 0.000000 1.0 \n", - "unconditional 1.012030e-01 0.006445 0.000000 1.0 \n", - "permuted_conditions -1.110223e-16 0.033897 0.014753 1.0 \n", - "noise 4.239221e-01 0.998588 0.285783 0.0 \n", - "reference (split-half) 2.798658e-01 0.090922 0.014603 0.0 \n", - "\n", - " cond_sens viol per_condition_distance volume_error \\\n", - "honest NaN 0.0 0.014059 NaN \n", - "lookup_table NaN 0.0 0.014891 NaN \n", - "unconditional NaN 0.0 0.180742 NaN \n", - "permuted_conditions NaN 0.0 0.176872 NaN \n", - "noise NaN 0.0 0.332793 NaN \n", - "reference (split-half) NaN 0.0 NaN NaN \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", + "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2939 0.00 ... 1.816083e+09 1.735349e+09 3.0758 1.0 \n", + "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2632 0.00 ... 8.950722e+09 8.689641e+09 0.0019 1.0 \n", + "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0531 0.00 ... 2.413100e+00 8.195000e-01 -0.6194 32.0 \n", + "diffusion_2d_cond 0.1105 1.00 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.1460 0.00 ... -1.721100e+00 6.021700e+00 -0.9128 17.0 \n", + "lookup_table 0.0729 1.00 9.7122 0.3051 0.7500 0.0125 0.0569 NaN 0.0117 0.50 ... 9.574600e+00 7.314000e-01 -0.7363 29.0 \n", + "unconditional 0.1631 1.00 11.1734 0.9277 0.9167 0.0562 0.3993 NaN 0.0504 0.25 ... 4.805870e+03 2.817597e+04 1.4287 6.0 \n", + "permuted_conditions 0.0000 1.00 10.0889 0.3162 0.9167 0.0562 0.4231 NaN 0.0525 1.00 ... 8.783402e+07 8.788590e+07 4.4569 2.0 \n", "\n", - " coverage vendi generation_seconds n_parameters \\\n", - "honest 1.0000 3.254471 NaN NaN \n", - "lookup_table 1.0000 3.241546 NaN NaN \n", - "unconditional 0.9375 2.605583 NaN NaN \n", - "permuted_conditions 1.0000 3.248472 NaN NaN \n", - "noise 0.0000 15.976387 NaN NaN \n", - "reference (split-half) 0.5000 2.868064 NaN NaN \n", + " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", + "cgan_cnn_2d 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", + "gan_cnn_2d 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", + "vqgan 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", + "diffusion_2d_cond -1.712700e+00 26.3979 6.1904 0.3109 0.9167 2.9236 \n", + "lookup_table 9.590800e+00 8.2825 3.3313 0.3676 0.5833 0.3145 \n", + "unconditional 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 0.4913 \n", + "permuted_conditions 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 0.8233 \n", "\n", - " train_minutes \n", - "honest NaN \n", - "lookup_table NaN \n", - "unconditional NaN \n", - "permuted_conditions NaN \n", - "noise NaN \n", - "reference (split-half) NaN " + "[7 rows x 23 columns]" ] }, - "execution_count": 4, + "execution_count": 6, "metadata": {}, "output_type": "execute_result" } ], "source": [ - "from engiopt.evaluation.board import Board\n", - "\n", - "board = Board(problem, reference=REFERENCE, train=TRAIN)\n", - "board.evaluate(MODELS)" + "probes = Board.from_evaluator(evaluator, {}, designs=CONSTRUCTIONS, expensive=EXPENSIVE)\n", + "board.frame = pd.concat([board.frame, probes.frame])\n", + "board.designs.update(probes.designs)\n", + "board.frame.round(4)" ] }, { "cell_type": "markdown", - "id": "f8bd45e6", + "id": "87dabe65", "metadata": {}, "source": [ - "One row per model, one column per metric — and a last row that is **not a model**.\n", - "`reference (split-half)` is half of the withheld optimal designs scored against the\n", - "other half: what real, correct designs score on every column, on this problem. An\n", - "`mmd` of 0.0006 means nothing on its own; next to what real designs score, it does.\n", - "\n", - "Ask the board to read itself:" + "## 5. Read the board" ] }, { "cell_type": "code", - "execution_count": 5, - "id": "b7bb486e", + "execution_count": 7, + "id": "fa5cb3e9", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.370693Z", - "iopub.status.busy": "2026-09-28T12:49:20.370634Z", - "iopub.status.idle": "2026-09-28T12:49:20.374316Z", - "shell.execute_reply": "2026-09-28T12:49:20.374125Z" + "iopub.execute_input": "2026-09-28T15:13:05.688476Z", + "iopub.status.busy": "2026-09-28T15:13:05.688370Z", + "iopub.status.idle": "2026-09-28T15:13:05.693238Z", + "shell.execute_reply": "2026-09-28T15:13:05.693058Z" } }, "outputs": [ @@ -691,159 +1126,219 @@ " Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.\n", " lower is better\n", " permuted_conditions\n", - " 0.279866\n", + " None\n", + " \n", + " \n", + " coverage\n", + " Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.\n", + " higher is better\n", + " tie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditions\n", + " None\n", + " \n", + " \n", + " vendi\n", + " Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.\n", + " higher is better\n", + " unconditional\n", + " None\n", " \n", " \n", " dpp\n", " Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed.\n", " higher is better\n", - " noise\n", - " 0.090922\n", + " unconditional\n", + " None\n", + " \n", + " \n", + " viol\n", + " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", + " lower is better\n", + " vqgan\n", + " None\n", + " \n", + " \n", + " volume_error\n", + " Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.\n", + " lower is better\n", + " vqgan\n", + " None\n", + " \n", + " \n", + " per_condition_distance\n", + " Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.\n", + " lower is better\n", + " lookup_table\n", + " None\n", + " \n", + " \n", + " cond_sens\n", + " How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.\n", + " diagnostic\n", + " \n", + " None\n", " \n", " \n", " train_distance\n", " Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.\n", " diagnostic\n", " \n", - " 0.014603\n", + " None\n", " \n", " \n", " copy_rate\n", " Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval.\n", " diagnostic\n", " \n", - " 0.000000\n", + " None\n", " \n", " \n", - " cond_sens\n", - " How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.\n", - " diagnostic\n", + " generation_seconds\n", + " Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.\n", + " lower is better\n", + " gan_cnn_2d\n", + " None\n", + " \n", + " \n", + " n_parameters\n", + " Number of trainable parameters in the model. Blank for a model with no network.\n", + " lower is better\n", + " cgan_cnn_2d\n", + " None\n", + " \n", + " \n", + " train_minutes\n", + " Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.\n", + " lower is better\n", " \n", - " NaN\n", + " None\n", " \n", " \n", - " viol\n", - " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", + " iog\n", + " Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.\n", " lower is better\n", - " tie: honest, lookup_table, unconditional, permuted_conditions, noise\n", - " 0.000000\n", + " diffusion_2d_cond\n", + " None\n", " \n", " \n", - " per_condition_distance\n", - " Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.\n", + " cog\n", + " Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted.\n", " lower is better\n", - " honest\n", - " NaN\n", + " lookup_table\n", + " None\n", " \n", " \n", - " volume_error\n", - " Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.\n", + " fog\n", + " Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.\n", " lower is better\n", - " \n", - " NaN\n", + " diffusion_2d_cond\n", + " None\n", " \n", " \n", - " coverage\n", - " Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.\n", - " higher is better\n", - " tie: honest, lookup_table, permuted_conditions\n", - " 0.500000\n", + " calls_to_settle\n", + " Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle.\n", + " lower is better\n", + " tie: cgan_cnn_2d, gan_cnn_2d\n", + " None\n", " \n", " \n", - " vendi\n", - " Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.\n", - " higher is better\n", - " noise\n", - " 2.868064\n", + " gap_after_1_calls\n", + " Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.\n", + " lower is better\n", + " diffusion_2d_cond\n", + " None\n", " \n", " \n", - " generation_seconds\n", - " Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.\n", + " gap_after_2_calls\n", + " Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.\n", " lower is better\n", - " \n", - " NaN\n", + " vqgan\n", + " None\n", " \n", " \n", - " n_parameters\n", - " Number of trainable parameters in the model. Blank for a model with no network.\n", + " gap_after_5_calls\n", + " Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.\n", " lower is better\n", - " \n", - " NaN\n", + " vqgan\n", + " None\n", " \n", " \n", - " train_minutes\n", - " Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.\n", + " gap_after_10_calls\n", + " Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.\n", " lower is better\n", - " \n", - " NaN\n", + " vqgan\n", + " None\n", + " \n", + " \n", + " reaches_reference_rate\n", + " Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there.\n", + " higher is better\n", + " diffusion_2d_cond\n", + " None\n", + " \n", + " \n", + " first_call_gain\n", + " Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step.\n", + " higher is better\n", + " diffusion_2d_cond\n", + " None\n", " \n", " \n", "\n", "" ], "text/plain": [ - " question \\\n", - "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", - "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", - "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", - "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", - "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", - "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", - "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", - "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", - "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", - "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", - "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", - "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", - "\n", - " direction \\\n", - "mmd lower is better \n", - "dpp higher is better \n", - "train_distance diagnostic \n", - "copy_rate diagnostic \n", - "cond_sens diagnostic \n", - "viol lower is better \n", - "per_condition_distance lower is better \n", - "volume_error lower is better \n", - "coverage higher is better \n", - "vendi higher is better \n", - "generation_seconds lower is better \n", - "n_parameters lower is better \n", - "train_minutes lower is better \n", - "\n", - " picks \\\n", - "mmd permuted_conditions \n", - "dpp noise \n", - "train_distance \n", - "copy_rate \n", - "cond_sens \n", - "viol tie: honest, lookup_table, unconditional, permuted_conditions, noise \n", - "per_condition_distance honest \n", - "volume_error \n", - "coverage tie: honest, lookup_table, permuted_conditions \n", - "vendi noise \n", - "generation_seconds \n", - "n_parameters \n", - "train_minutes \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "calls_to_settle Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. \n", + "gap_after_1_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_2_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_5_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_10_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", + "first_call_gain Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. \n", "\n", - " real designs score \n", - "mmd 0.279866 \n", - "dpp 0.090922 \n", - "train_distance 0.014603 \n", - "copy_rate 0.000000 \n", - "cond_sens NaN \n", - "viol 0.000000 \n", - "per_condition_distance NaN \n", - "volume_error NaN \n", - "coverage 0.500000 \n", - "vendi 2.868064 \n", - "generation_seconds NaN \n", - "n_parameters NaN \n", - "train_minutes NaN " + " direction picks real designs score \n", + "mmd lower is better permuted_conditions None \n", + "coverage higher is better tie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditions None \n", + "vendi higher is better unconditional None \n", + "dpp higher is better unconditional None \n", + "viol lower is better vqgan None \n", + "volume_error lower is better vqgan None \n", + "per_condition_distance lower is better lookup_table None \n", + "cond_sens diagnostic None \n", + "train_distance diagnostic None \n", + "copy_rate diagnostic None \n", + "generation_seconds lower is better gan_cnn_2d None \n", + "n_parameters lower is better cgan_cnn_2d None \n", + "train_minutes lower is better None \n", + "iog lower is better diffusion_2d_cond None \n", + "cog lower is better lookup_table None \n", + "fog lower is better diffusion_2d_cond None \n", + "calls_to_settle lower is better tie: cgan_cnn_2d, gan_cnn_2d None \n", + "gap_after_1_calls lower is better diffusion_2d_cond None \n", + "gap_after_2_calls lower is better vqgan None \n", + "gap_after_5_calls lower is better vqgan None \n", + "gap_after_10_calls lower is better vqgan None \n", + "reaches_reference_rate higher is better diffusion_2d_cond None \n", + "first_call_gain higher is better diffusion_2d_cond None " ] }, - "execution_count": 5, + "execution_count": 7, "metadata": {}, "output_type": "execute_result" } @@ -854,68 +1349,78 @@ }, { "cell_type": "markdown", - "id": "bf0217c3", - "metadata": {}, - "source": [ - "## 5. What the board is telling you" - ] - }, - { - "cell_type": "markdown", - "id": "ef271490", + "id": "1a95df7d", "metadata": {}, "source": [ - "**`mmd` picks `permuted_conditions`.** It is handed the correct withheld optima with the\n", - "conditions scrambled, and the distribution metric rates it *first* — a perfect 0.0,\n", - "better than the honest model and better than real designs against real designs. MMD\n", - "compares two *sets*; it has no access to which design was produced for which\n", - "conditions. Every set-level metric shares this by construction.\n", + "Why these seven rows: four trained models and three constructions whose right\n", + "answer is known, so the board shows which columns can tell a good model from a\n", + "construction that games one metric. How: every row was scored under one spec, twelve\n", + "withheld conditions, medians over designs, and the physics columns re-optimized from\n", + "each design. What the numbers say:\n", + "\n", + "**`permuted_conditions` posts an MMD of exactly zero and full coverage. It is the\n", + "reference set.** It also posts the worst final gap on the board, 4.46, and reaches the\n", + "reference optimum from none of its designs. A perfect set of designs, each for the\n", + "wrong conditions, is the best model by the distribution columns and the worst by the\n", + "physics. `per_condition_distance` reads 0.42 for it, in the same tier as the GANs, and\n", + "that is the column that sees the problem before any optimizer runs.\n", "\n", - "**`lookup_table` scores almost as well as `honest` on `mmd`, having never seen a\n", - "withheld design.** Not by cheating: training and test optima come from the same\n", - "distribution, and a distribution metric only compares distributions. A lookup table is a\n", - "legitimately strong baseline; what is not legitimate is reading its `mmd` as evidence\n", - "it learned anything.\n", + "**`lookup_table` never generalizes, half its designs are exact copies, and it beats\n", + "every trained model except diffusion on final gap.** It has the best cumulative gap on\n", + "the board and the smallest per-condition distance, 0.057. That is not a bug. With one\n", + "optimum per condition and a dense training sweep, the nearest training design is a very\n", + "good warm start. `copy_rate` at 0.50 is the only column that says what it is.\n", "\n", - "**`copy_rate` is 1.0 for all three constructions** and `novelty` is 0.0. One diagnostic\n", - "column separates every construction from the honest model. That is what the\n", - "`memorization` and `conditions` families are for, and why a board that reports only\n", - "distribution metrics cannot be trusted.\n", + "**`diffusion_2d_cond` has the best final gap, minus 0.91, and reaches the reference\n", + "optimum from eleven designs in twelve.** It also violates a constraint in ten of twelve\n", + "designs and misses the requested material fraction by the widest margin of the trained\n", + "models, 0.127. Its initial gap is negative, meaning better than the reference optimum\n", + "before any optimization, which is what a design that spends more material than it was\n", + "allowed looks like. Read `iog` beside `viol` and `volume_error`, never alone.\n", "\n", - "**`per_condition_distance` is the column that catches `permuted_conditions`.** It\n", - "compares design `i` with the reference optimum for conditions `i`, so the correct\n", - "designs in the wrong order score badly here and perfectly on `mmd`. Read the two\n", - "together.\n", + "**`vqgan` is the most obedient model.** Fewest violations, smallest volume error, and\n", + "the largest response when its conditions change. Its physics is middling. Obedience\n", + "and optimality are different columns.\n", "\n", - "**`dpp` and `vendi` both measure diversity; prefer `vendi`.** `dpp` is now the n-th\n", - "root of the kernel determinant, on a 0-to-1 scale (the raw determinant older boards\n", - "published reads 10⁻²⁰ on every real board). But a determinant still rises when a\n", - "collapsed set is jittered — it rewards noise as diversity — and `vendi`, an effective\n", - "count of distinct designs, does not.\n", + "**`cgan_cnn_2d` and `gan_cnn_2d` post initial gaps near 10^9 and 10^10.** The\n", + "optimizer repairs them; the final gaps are 3.1 and 0.002. These are designs that start\n", + "absurd and are cheap to fix, which is why `iog` and `fog` disagree about them.\n", + "`gan_cnn_2d` has a `vendi` of 1.09, twelve designs that are effectively one, and a\n", + "`cond_sens` of exactly zero, because it is an unconditional model that was handed\n", + "conditions anyway. That zero is what stops it collecting a conditional model's score.\n", "\n", - "**`cond_sens` is `NaN`.** It asks whether the output *changes* when the conditions\n", - "change, which needs the live model re-sampled; a saved array cannot answer it.\n", - "`Board` scores designs; `Evaluator.score(generator)` scores a model and fills it in." + "**`viol` is high even for real training designs.** On beams2d it includes the volume\n", + "budget at a tolerance of 0.01, and a design made for a neighboring condition misses\n", + "its own budget by more than that. Seven in twelve of the lookup table's designs fail\n", + "on that alone.\n", + "\n", + "Two things this board cannot show. `cond_sens` and the cost columns are blank for the\n", + "constructions, because they have no model to re-run or inspect. And there is no\n", + "reference row here; that row comes from the saved-designs path in section 7, where at\n", + "twelve conditions it is six designs against six, a coarser measurement than the model\n", + "rows. The trajectory columns also assume the gap falls monotonically; for the diffusion\n", + "model it rises after the first call, as the optimizer restores feasibility before it\n", + "improves the objective, and `first_call_gain` reads above one for that reason." ] }, { "cell_type": "markdown", - "id": "4abf824b", + "id": "a0f0269a", "metadata": {}, "source": [ - "## 6. Rank — but only on a column that can be ranked" + "## 6. Rank, but only on a column that can be ranked" ] }, { "cell_type": "code", - "execution_count": 6, - "id": "dd93537a", + "execution_count": 8, + "id": "6f0e620e", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.375403Z", - "iopub.status.busy": "2026-09-28T12:49:20.375348Z", - "iopub.status.idle": "2026-09-28T12:49:20.378918Z", - "shell.execute_reply": "2026-09-28T12:49:20.378744Z" + "iopub.execute_input": "2026-09-28T15:13:05.695701Z", + "iopub.status.busy": "2026-09-28T15:13:05.695623Z", + "iopub.status.idle": "2026-09-28T15:13:05.698788Z", + "shell.execute_reply": "2026-09-28T15:13:05.698554Z" } }, "outputs": [ @@ -940,175 +1445,89 @@ " \n", " \n", " \n", - " mmd\n", - " dpp\n", - " train_distance\n", + " fog\n", " copy_rate\n", - " cond_sens\n", - " viol\n", " per_condition_distance\n", - " volume_error\n", - " coverage\n", - " vendi\n", - " generation_seconds\n", - " n_parameters\n", - " train_minutes\n", " \n", " \n", " \n", " \n", - " permuted_conditions\n", - " -1.110223e-16\n", - " 0.033897\n", - " 0.014753\n", - " 1.0\n", - " NaN\n", - " 0.0\n", - " 0.176872\n", - " NaN\n", - " 1.0000\n", - " 3.248472\n", - " NaN\n", - " NaN\n", - " NaN\n", + " diffusion_2d_cond\n", + " -0.9128\n", + " 0.00\n", + " 0.1532\n", " \n", " \n", - " honest\n", - " 6.366160e-04\n", - " 0.035297\n", - " 0.014761\n", - " 0.0\n", - " NaN\n", - " 0.0\n", - " 0.014059\n", - " NaN\n", - " 1.0000\n", - " 3.254471\n", - " NaN\n", - " NaN\n", - " NaN\n", + " lookup_table\n", + " -0.7363\n", + " 0.50\n", + " 0.0569\n", " \n", " \n", - " lookup_table\n", - " 1.037788e-03\n", - " 0.004358\n", - " 0.000000\n", - " 1.0\n", - " NaN\n", - " 0.0\n", - " 0.014891\n", - " NaN\n", - " 1.0000\n", - " 3.241546\n", - " NaN\n", - " NaN\n", - " NaN\n", + " vqgan\n", + " -0.6194\n", + " 0.00\n", + " 0.0833\n", + " \n", + " \n", + " gan_cnn_2d\n", + " 0.0019\n", + " 0.00\n", + " 0.4447\n", " \n", " \n", " unconditional\n", - " 1.012030e-01\n", - " 0.006445\n", - " 0.000000\n", - " 1.0\n", - " NaN\n", - " 0.0\n", - " 0.180742\n", - " NaN\n", - " 0.9375\n", - " 2.605583\n", - " NaN\n", - " NaN\n", - " NaN\n", + " 1.4287\n", + " 0.25\n", + " 0.3993\n", " \n", " \n", - " reference (split-half)\n", - " 2.798658e-01\n", - " 0.090922\n", - " 0.014603\n", - " 0.0\n", - " NaN\n", - " 0.0\n", - " NaN\n", - " NaN\n", - " 0.5000\n", - " 2.868064\n", - " NaN\n", - " NaN\n", - " NaN\n", + " cgan_cnn_2d\n", + " 3.0758\n", + " 0.00\n", + " 0.4144\n", " \n", " \n", - " noise\n", - " 4.239221e-01\n", - " 0.998588\n", - " 0.285783\n", - " 0.0\n", - " NaN\n", - " 0.0\n", - " 0.332793\n", - " NaN\n", - " 0.0000\n", - " 15.976387\n", - " NaN\n", - " NaN\n", - " NaN\n", + " permuted_conditions\n", + " 4.4569\n", + " 1.00\n", + " 0.4231\n", " \n", " \n", "\n", "" ], "text/plain": [ - " mmd dpp train_distance copy_rate \\\n", - "permuted_conditions -1.110223e-16 0.033897 0.014753 1.0 \n", - "honest 6.366160e-04 0.035297 0.014761 0.0 \n", - "lookup_table 1.037788e-03 0.004358 0.000000 1.0 \n", - "unconditional 1.012030e-01 0.006445 0.000000 1.0 \n", - "reference (split-half) 2.798658e-01 0.090922 0.014603 0.0 \n", - "noise 4.239221e-01 0.998588 0.285783 0.0 \n", - "\n", - " cond_sens viol per_condition_distance volume_error \\\n", - "permuted_conditions NaN 0.0 0.176872 NaN \n", - "honest NaN 0.0 0.014059 NaN \n", - "lookup_table NaN 0.0 0.014891 NaN \n", - "unconditional NaN 0.0 0.180742 NaN \n", - "reference (split-half) NaN 0.0 NaN NaN \n", - "noise NaN 0.0 0.332793 NaN \n", - "\n", - " coverage vendi generation_seconds n_parameters \\\n", - "permuted_conditions 1.0000 3.248472 NaN NaN \n", - "honest 1.0000 3.254471 NaN NaN \n", - "lookup_table 1.0000 3.241546 NaN NaN \n", - "unconditional 0.9375 2.605583 NaN NaN \n", - "reference (split-half) 0.5000 2.868064 NaN NaN \n", - "noise 0.0000 15.976387 NaN NaN \n", - "\n", - " train_minutes \n", - "permuted_conditions NaN \n", - "honest NaN \n", - "lookup_table NaN \n", - "unconditional NaN \n", - "reference (split-half) NaN \n", - "noise NaN " + " fog copy_rate per_condition_distance\n", + "diffusion_2d_cond -0.9128 0.00 0.1532\n", + "lookup_table -0.7363 0.50 0.0569\n", + "vqgan -0.6194 0.00 0.0833\n", + "gan_cnn_2d 0.0019 0.00 0.4447\n", + "unconditional 1.4287 0.25 0.3993\n", + "cgan_cnn_2d 3.0758 0.00 0.4144\n", + "permuted_conditions 4.4569 1.00 0.4231" ] }, - "execution_count": 6, + "execution_count": 8, "metadata": {}, "output_type": "execute_result" } ], "source": [ - "board.rank(\"mmd\")" + "column = \"fog\" if EXPENSIVE else \"mmd\"\n", + "board.rank(column)[[column, \"copy_rate\", \"per_condition_distance\"]].round(4)" ] }, { "cell_type": "code", - "execution_count": 7, - "id": "6150f040", + "execution_count": 9, + "id": "c39d8696", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.379839Z", - "iopub.status.busy": "2026-09-28T12:49:20.379786Z", - "iopub.status.idle": "2026-09-28T12:49:20.381289Z", - "shell.execute_reply": "2026-09-28T12:49:20.381070Z" + "iopub.execute_input": "2026-09-28T15:13:05.701188Z", + "iopub.status.busy": "2026-09-28T15:13:05.701115Z", + "iopub.status.idle": "2026-09-28T15:13:05.702554Z", + "shell.execute_reply": "2026-09-28T15:13:05.702378Z" } }, "outputs": [ @@ -1129,142 +1548,38 @@ }, { "cell_type": "markdown", - "id": "1ed1fa6a", - "metadata": {}, - "source": [ - "## 7. Changing how a column is aggregated" - ] - }, - { - "cell_type": "markdown", - "id": "1defe64e", - "metadata": {}, - "source": [ - "`iog`, `cog`, `fog`, `train_distance`, `per_condition_distance` and the other per-design\n", - "metrics each produce **one value per design**.\n", - "Which single number that becomes is a policy, not a measurement: a mean is dragged by\n", - "one diverged design, a median is not. It is an argument to `evaluate` — and, for a\n", - "published board, a field of the frozen spec recorded in every row — so there are no\n", - "`*_median` columns." - ] - }, - { - "cell_type": "code", - "execution_count": 8, - "id": "36e87484", - "metadata": { - "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.382346Z", - "iopub.status.busy": "2026-09-28T12:49:20.382293Z", - "iopub.status.idle": "2026-09-28T12:49:20.387136Z", - "shell.execute_reply": "2026-09-28T12:49:20.386952Z" - } - }, - "outputs": [ - { - "data": { - "text/html": [ - "
\n", - "\n", - "\n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - " \n", - "
train_distance (mean)train_distance (median)
honest0.0147610.014655
one_diverged0.0321990.015012
reference (split-half)0.0146030.015140
\n", - "
" - ], - "text/plain": [ - " train_distance (mean) train_distance (median)\n", - "honest 0.014761 0.014655\n", - "one_diverged 0.032199 0.015012\n", - "reference (split-half) 0.014603 0.015140" - ] - }, - "execution_count": 8, - "metadata": {}, - "output_type": "execute_result" - } - ], - "source": [ - "one_diverged = MODELS[\"honest\"].copy()\n", - "one_diverged[0] = rng.random((8, 10)) # fifteen good designs and one that went wrong\n", - "\n", - "compare = {\"honest\": MODELS[\"honest\"], \"one_diverged\": one_diverged}\n", - "pd.DataFrame(\n", - " {\n", - " \"train_distance (mean)\": board.evaluate(compare)[\"train_distance\"],\n", - " \"train_distance (median)\": board.evaluate(compare, aggregation=\"median\")[\"train_distance\"],\n", - " }\n", - ")" - ] - }, - { - "cell_type": "markdown", - "id": "a4afdca0", + "id": "fa6f57fa", "metadata": {}, "source": [ - "## 8. Asking the same questions in another space" + "## 7. The same questions in another space" ] }, { "cell_type": "markdown", - "id": "15759e0d", + "id": "0db4a92f", "metadata": {}, "source": [ - "Where a metric is computed is also an argument, not a property of the metric. `pca`\n", - "projects every design onto the leading principal components of the reference designs\n", - "before scoring, and suffixes the columns. Metrics that need the actual designs — a\n", - "constraint check, a copy corpus — are skipped there, because a constraint check on PCA\n", - "coordinates would be meaningless.\n", + "Where a metric is computed is an argument, not a property of the metric. The board\n", + "kept every model's sampled designs, so they can be re-scored after projecting onto\n", + "the leading principal components of the reference designs. Metrics that need the\n", + "actual designs, such as the constraint check and the copy corpus, are skipped there.\n", + "The kernel bandwidth is pinned to the spec's value so the two boards share a scale.\n", "\n", - "This is the control that asks whether a *learned* space is needed at all: if PCA of the\n", - "data answers the same questions, any rotation of the same width would do. Learned\n", - "latent spaces (`space=\"lv\"`) arrive with the LVAE." + "This is the control that asks whether a learned latent space is needed at all. If PCA\n", + "of the data answers the same questions, any rotation of the same width would do. A\n", + "learned space registers itself with `register_space` and appears here by name." ] }, { "cell_type": "code", - "execution_count": 9, - "id": "ef9a97f4", + "execution_count": 10, + "id": "d8783983", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.388281Z", - "iopub.status.busy": "2026-09-28T12:49:20.388219Z", - "iopub.status.idle": "2026-09-28T12:49:20.392442Z", - "shell.execute_reply": "2026-09-28T12:49:20.392169Z" + "iopub.execute_input": "2026-09-28T15:13:05.704903Z", + "iopub.status.busy": "2026-09-28T15:13:05.704818Z", + "iopub.status.idle": "2026-09-28T15:13:05.716554Z", + "shell.execute_reply": "2026-09-28T15:13:05.716385Z" } }, "outputs": [ @@ -1298,112 +1613,123 @@ " \n", " \n", " \n", - " honest\n", - " 0.000124\n", - " 0.007646\n", - " 0.026240\n", + " cgan_cnn_2d\n", + " 0.7543\n", + " 0.0018\n", + " 7.1693\n", + " 1.0000\n", + " 1.4800\n", + " \n", + " \n", + " gan_cnn_2d\n", + " 1.0002\n", + " 0.0000\n", + " 9.2163\n", + " 0.5833\n", + " 1.0032\n", + " \n", + " \n", + " vqgan\n", + " 0.0060\n", + " 0.5743\n", + " 0.4617\n", + " 1.0000\n", + " 8.7158\n", + " \n", + " \n", + " diffusion_2d_cond\n", + " 0.0624\n", + " 0.6964\n", + " 2.0203\n", " 1.0000\n", - " 3.168944\n", + " 9.2823\n", " \n", " \n", " lookup_table\n", - " 0.000206\n", - " 0.001569\n", - " 0.029576\n", + " 0.0614\n", + " 0.2537\n", + " 0.5618\n", " 1.0000\n", - " 3.182097\n", + " 8.3923\n", " \n", " \n", " unconditional\n", - " 0.101071\n", - " 0.002117\n", - " 0.569443\n", - " 0.9375\n", - " 2.543748\n", + " 0.1054\n", + " 0.4121\n", + " 8.6422\n", + " 1.0000\n", + " 6.7416\n", " \n", " \n", " permuted_conditions\n", - " 0.000000\n", - " 0.022450\n", - " 0.557238\n", + " 0.0000\n", + " 0.2804\n", + " 10.5238\n", " 1.0000\n", - " 3.220545\n", - " \n", - " \n", - " noise\n", - " 0.181628\n", - " 0.262513\n", - " 0.588434\n", - " 0.2500\n", - " 4.577680\n", + " 9.3671\n", " \n", " \n", " reference (split-half)\n", - " 0.279983\n", - " 0.071127\n", + " 0.2264\n", + " 0.9268\n", " NaN\n", - " 0.5000\n", - " 2.849693\n", + " 1.0000\n", + " 5.6120\n", " \n", " \n", "\n", "" ], "text/plain": [ - " mmd@pca dpp@pca per_condition_distance@pca \\\n", - "honest 0.000124 0.007646 0.026240 \n", - "lookup_table 0.000206 0.001569 0.029576 \n", - "unconditional 0.101071 0.002117 0.569443 \n", - "permuted_conditions 0.000000 0.022450 0.557238 \n", - "noise 0.181628 0.262513 0.588434 \n", - "reference (split-half) 0.279983 0.071127 NaN \n", - "\n", - " coverage@pca vendi@pca \n", - "honest 1.0000 3.168944 \n", - "lookup_table 1.0000 3.182097 \n", - "unconditional 0.9375 2.543748 \n", - "permuted_conditions 1.0000 3.220545 \n", - "noise 0.2500 4.577680 \n", - "reference (split-half) 0.5000 2.849693 " + " mmd@pca dpp@pca per_condition_distance@pca coverage@pca vendi@pca\n", + "cgan_cnn_2d 0.7543 0.0018 7.1693 1.0000 1.4800\n", + "gan_cnn_2d 1.0002 0.0000 9.2163 0.5833 1.0032\n", + "vqgan 0.0060 0.5743 0.4617 1.0000 8.7158\n", + "diffusion_2d_cond 0.0624 0.6964 2.0203 1.0000 9.2823\n", + "lookup_table 0.0614 0.2537 0.5618 1.0000 8.3923\n", + "unconditional 0.1054 0.4121 8.6422 1.0000 6.7416\n", + "permuted_conditions 0.0000 0.2804 10.5238 1.0000 9.3671\n", + "reference (split-half) 0.2264 0.9268 NaN 1.0000 5.6120" ] }, - "execution_count": 9, + "execution_count": 10, "metadata": {}, "output_type": "execute_result" } ], "source": [ - "board.evaluate(MODELS, space=\"pca\", width=8)" + "pca_board = Board(evaluator.problem, reference=REFERENCE, train=TRAIN, sigma=spec.sigma)\n", + "pca_board.evaluate(board.designs, space=\"pca\", width=8, aggregation=\"median\").round(4)" ] }, { "cell_type": "markdown", - "id": "a7bf45d2", + "id": "deb1252c", "metadata": {}, "source": [ - "## 9. Registering a metric of your own" + "## 8. Registering a metric of your own" ] }, { "cell_type": "markdown", - "id": "086b8d09", + "id": "546a1629", "metadata": {}, "source": [ - "A metric is a function of an `EvaluationContext`, a family, a cost, and a direction.\n", - "The first line of its docstring is the question readers see. `ctx.reduce` applies\n", - "whatever aggregation the caller chose, so you never hard-code a mean." + "A metric is a function of an `EvaluationContext`, a family, a cost and a direction.\n", + "The first line of its docstring is the sentence readers see. `ctx.reduce` applies\n", + "whatever aggregation the spec chose, so a metric never hard-codes a mean." ] }, { "cell_type": "code", - "execution_count": 10, - "id": "1adaa8bf", + "execution_count": 11, + "id": "98b7c485", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T12:49:20.393589Z", - "iopub.status.busy": "2026-09-28T12:49:20.393516Z", - "iopub.status.idle": "2026-09-28T12:49:20.397706Z", - "shell.execute_reply": "2026-09-28T12:49:20.397490Z" + "iopub.execute_input": "2026-09-28T15:13:05.719001Z", + "iopub.status.busy": "2026-09-28T15:13:05.718926Z", + "iopub.status.idle": "2026-09-28T15:13:05.724756Z", + "shell.execute_reply": "2026-09-28T15:13:05.724580Z" } }, "outputs": [ @@ -1436,34 +1762,30 @@ " \n", " \n", " \n", - " mmd\n", - " Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.\n", + " volume_error\n", + " Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.\n", " lower is better\n", - " permuted_conditions\n", - " 0.279866\n", + " vqgan\n", + " None\n", " \n", " \n", " density_ratio\n", - " How much material do the generated designs use, relative to the reference optima?\n", + " Material fraction of each generated design over that of the reference optima. One means the same amount of material.\n", " diagnostic\n", " \n", - " 0.766460\n", + " None\n", " \n", " \n", "\n", "" ], "text/plain": [ - " question \\\n", - "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", - "density_ratio How much material do the generated designs use, relative to the reference optima? \n", - "\n", - " direction picks real designs score \n", - "mmd lower is better permuted_conditions 0.279866 \n", - "density_ratio diagnostic 0.766460 " + " question direction picks real designs score\n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. lower is better vqgan None\n", + "density_ratio Material fraction of each generated design over that of the reference optima. One means the same amount of material. diagnostic None" ] }, - "execution_count": 10, + "execution_count": 11, "metadata": {}, "output_type": "execute_result" } @@ -1474,47 +1796,45 @@ "\n", "@register_metric(\"density_ratio\", family=\"feasibility\", cost=\"cheap\", higher_is_better=None)\n", "def density_ratio(ctx):\n", - " \"\"\"How much material do the generated designs use, relative to the reference optima?\"\"\"\n", + " \"\"\"Material fraction of each generated design over that of the reference optima. One means the same amount of material.\"\"\"\n", " return float(ctx.reduce(ctx.gen_flat.mean(axis=1)) / ctx.ref_flat.mean())\n", "\n", "\n", - "board.evaluate(MODELS, metrics=[\"mmd\", \"density_ratio\"])\n", - "board.explain()" + "Board.from_evaluator(evaluator, {}, designs=board.designs, metrics=[\"volume_error\", \"density_ratio\"]).explain()" ] }, { "cell_type": "markdown", - "id": "3c274682", + "id": "d2742ff7", "metadata": {}, "source": [ - "It appears on the next board with its question and direction attached — nothing else\n", - "to wire up. Declared diagnostic here, because more material is not better or worse.\n", - "\n", - "Notice the reference row: two random halves of the *same* real designs do not score\n", - "1.0 against each other. On eight designs a side, that gap is the sampling noise of this\n", - "column on this problem — and it is the scale against which every model's deviation from\n", - "1.0 has to be read. That is what the reference row is for." + "It appears with its question and direction attached, scored under the same spec as\n", + "everything else. Declared diagnostic, because using more material is not better or\n", + "worse on its own; read it beside `volume_error`, which says whether each design hit\n", + "the fraction its conditions asked for." ] }, { "cell_type": "markdown", - "id": "853f8773", + "id": "19ac14ba", "metadata": {}, "source": [ - "## 10. Not in this notebook" + "## 9. Not in this notebook" ] }, { "cell_type": "markdown", - "id": "9500df97", + "id": "769cb27a", "metadata": {}, "source": [ - "- **Physics.** `iog`, `cog`, `fog`, `calls_to_settle`, `gap_after_calls`,\n", - " `reaches_reference_rate` and `first_call_gain` need a simulator. With a real problem,\n", - " `Board.load(\"beams2d\", [\"cgan_cnn_2d\", \"vqgan\"], expensive=True)` runs them.\n", - "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec.\n", - "- **Uncertainty.** Confidence intervals beside every value. On sixteen samples, most\n", - " gaps on this board are smaller than their own error bars." + "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec. It\n", + " arrives with the LVAE work and plugs into section 7 by name.\n", + "- **Uncertainty.** Every physics column here is a median over fifty designs whose\n", + " per-design values already exist. Error bars from resampling those values, and folds\n", + " of the full test split for the cheap metrics, are the next addition.\n", + "- **Seeds.** Every row is training seed one. The Hub holds eleven seeds of each\n", + " family's default configuration; the spread across them is larger than the sampling\n", + " spread for some families, and it belongs on an honest board beside these numbers." ] } ], @@ -1525,16 +1845,8 @@ "name": "python3" }, "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.12.9" + "pygments_lexer": "ipython3" } }, "nbformat": 4, diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py index 9b889f50..400fd2f5 100644 --- a/tests/test_metrics_suite.py +++ b/tests/test_metrics_suite.py @@ -167,14 +167,32 @@ def test_the_reference_row_leaves_per_condition_metrics_blank(fake_problem: Any, assert not np.isnan(frame.loc["m", "per_condition_distance"]) -def test_from_generators_scores_through_an_evaluator() -> None: +def test_from_evaluator_scores_models_and_saved_designs_under_one_spec() -> None: + class StubContext: + def __init__(self, designs: Any) -> None: + self.gen_designs = designs + class StubEvaluator: - def score(self, generator: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: - return {"mmd": generator, "iog": 1.0 if include_expensive else float("nan")} + problem = None + spec = type("Spec", (), {"sigma": 1.0})() + resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3))})() + + def context_for(self, generator: Any) -> StubContext: + return StubContext(generator) + + def context_for_designs(self, designs: Any) -> StubContext: + return StubContext(designs) + + def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: + return {"mmd": float(np.mean(ctx.gen_designs)), "iog": 1.0 if include_expensive else float("nan")} - board = Board.from_generators(StubEvaluator(), {"a": 0.2, "b": 0.1}, expensive=True) + board = Board.from_evaluator( + StubEvaluator(), {"a": np.full(3, 0.2), "b": np.full(3, 0.1)}, designs={"probe": np.full(3, 0.3)}, expensive=True + ) + assert list(board.frame.index) == ["a", "b", "probe"] assert board.rank("mmd").index[0] == "b" - assert board.frame.loc["a", "iog"] == 1.0 + assert board.frame.loc["probe", "iog"] == 1.0 + assert set(board.designs) == {"a", "b", "probe"} # -------------------------------------------------------------------- specs From 1126e7df87d22418b0219c6d619d54d94652ad16 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Mon, 28 Sep 2026 17:36:45 +0200 Subject: [PATCH 10/18] Measure memorization as a distance, without a tolerance `copy_rate` counted designs within a per-element tolerance of a corpus that held the training split plus the reference optima. No model can see the reference optima, so that half of the corpus caught nothing real, and the tolerance was an arbitrary number in design units deciding who counts as a copier. Both go. `train_distance` now emits two readings of one measurement against the full training split: the per-element distance to the nearest training design, and that distance divided by what the withheld reference designs score, so one means as far from the training data as real unseen designs are and zero means retrieval. The reference designs supply the scale and nothing else. The spec loses its three copy knobs, including the 512-design subsample that let a lookup table through half the time, and the board no longer flags rows for memorization: the columns are read, not enforced. The v1 specs lose the same three keys; their condition digests are untouched. The notebook is re-run on beams2d with the new columns. Co-Authored-By: Claude Fable 5.1 --- CONTRIBUTING_A_MODEL.md | 8 +- LEADERBOARD.md | 15 +- README.md | 2 +- engiopt/evaluate.py | 7 - engiopt/evaluation/board.py | 6 +- engiopt/evaluation/context.py | 71 ++---- engiopt/evaluation/evaluator.py | 35 +-- engiopt/evaluation/leaderboard.py | 4 +- engiopt/evaluation/metrics/builtin.py | 53 ++-- engiopt/evaluation/spec.py | 19 -- engiopt/evaluation/submission.py | 21 +- engiopt/evaluation/verify.py | 2 +- engiopt/specs/beams2d/v1.json | 3 - engiopt/specs/beams2d/v2.json | 4 - engiopt/specs/heatconduction2d/v1.json | 3 - engiopt/specs/heatconduction2d/v2.json | 4 - engiopt/specs/photonics2d/v1.json | 3 - engiopt/specs/photonics2d/v2.json | 4 - engiopt/specs/thermoelastic2d/v1.json | 3 - engiopt/specs/thermoelastic2d/v2.json | 4 - example_metrics_suite.ipynb | 328 ++++++++++++------------- tests/test_board.py | 6 +- tests/test_evaluation.py | 3 +- tests/test_leaderboard_integrity.py | 107 ++++---- tests/test_metrics_suite.py | 19 +- tests/test_specs.py | 8 +- tests/test_verification.py | 15 +- 27 files changed, 297 insertions(+), 460 deletions(-) diff --git a/CONTRIBUTING_A_MODEL.md b/CONTRIBUTING_A_MODEL.md index 3c4a77b0..685e8286 100644 --- a/CONTRIBUTING_A_MODEL.md +++ b/CONTRIBUTING_A_MODEL.md @@ -178,7 +178,7 @@ python -m engiopt.evaluate --problem-id beams2d --generators my_model --hf-entit The default pass runs every metric that needs no simulator: the set-level ones (`mmd`, `coverage`, `vendi`, `dpp`), the per-condition ones (`per_condition_distance`, `volume_error`), feasibility (`viol`), the two -integrity checks (`copy_rate`, `cond_sens`) with `train_distance`, and the cost +integrity checks (`train_distance`, `cond_sens`), and the cost columns. `--include-expensive` adds the columns that re-optimize from each generated design — the optimality gaps `iog`, `cog`, `fog` and the call-budget metrics `calls_to_settle`, `gap_after_calls`, @@ -191,9 +191,9 @@ Watch two columns while you develop: - **`cond_sens`** should be greater than zero if you declared `conditional = True`. Exactly zero means your conditions are not reaching the network, which is a wiring bug far more often than a modeling choice. -- **`copy_rate`** should be near zero. High means your model is reproducing - training designs rather than generating, and the board will publish it without - ranking it. +- **`train_distance_ratio`** should be near one. Near zero means your model is + reproducing training designs rather than generating; the board shows it and + leaves the reading to people. `cond_sens` draws the batch a second time, under shuffled conditions, so it doubles sampling cost. That is nothing for a GAN and noticeable for a diffusion diff --git a/LEADERBOARD.md b/LEADERBOARD.md index 9638ef02..b228e904 100644 --- a/LEADERBOARD.md +++ b/LEADERBOARD.md @@ -132,7 +132,6 @@ to an entry covering every required seed. | Flag | Trips when | |---|---| | `unverified` | No runner has reproduced it yet. | -| `memorized` | `copy_rate` exceeds the spec's `max_copy_rate`. | | `ignores_conditions` | `cond_sens` is exactly zero for a model registered as conditional. | ## The copying problem @@ -152,15 +151,17 @@ scored: the whole dataset is public, so a memorizer memorizes all of it. So the board measures retrieval instead: -- **`copy_rate`** — the share of the batch within `copy_tol` (per-element RMS) - of any design the model could have copied: the training split it was fitted - on, plus the reference designs the protocol names. This is the one that flags - an entry. - **`train_distance`** — how far each generated design sits from the nearest - design in the training split. Zero means the model reproduces what it was - trained on. + design in the training split, per element. Zero means the model reproduces + what it was trained on. +- **`train_distance_ratio`** — the same distance divided by what the withheld + reference designs score against the training split. One means the model's + designs are as far from its training data as real unseen optima are; near zero + means retrieval. There is no tolerance to set: the reference designs supply + the scale and nothing else. Both are diagnostic, with no ranking direction, and that is not an oversight. +No row is flagged for memorization; the columns are read, not enforced. Ranking on `train_distance` would put pure noise in first place — zero means retrieval, but large means only "unlike the data", which a broken model also achieves. Closing one gaming vector by opening another is not progress. The same reasoning diff --git a/README.md b/README.md index 97ad1238..50c8cdf9 100644 --- a/README.md +++ b/README.md @@ -166,7 +166,7 @@ python -m engiopt.verify --board ... --verifier ideallab-ci --publish # the o Because the row carries the full address, that audit is not privileged — anyone can run the same command and get the same answer. -**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `copy_rate` catches it (`copy_rate=1.00`), and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. The whole suite — what each column measures, which way is better, how to change the space or the aggregation, and how to add a metric — is in [`example_metrics_suite.ipynb`](example_metrics_suite.ipynb). +**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `train_distance` reads zero for it, and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. The whole suite — what each column measures, which way is better, how to change the space or the aggregation, and how to add a metric — is in [`example_metrics_suite.ipynb`](example_metrics_suite.ipynb). See **[LEADERBOARD.md](LEADERBOARD.md)** for the submission path, the flags, and what would actually close the copying hole. diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index 22eb3cb6..45cbb2cd 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -32,7 +32,6 @@ from engiopt.evaluation.leaderboard import push_to_hub from engiopt.evaluation.registry import METRICS from engiopt.evaluation.submission import FLAG_IGNORES_CONDITIONS -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.submission import FLAG_UNVERIFIED from engiopt.evaluation.submission import integrity_flags from engiopt.utils.all_generators import BUILTIN_GENERATORS @@ -412,12 +411,6 @@ def _print_integrity_warnings(board: pd.DataFrame) -> None: if not flags: continue label = f"{row.get('algo_id')} seed {row.get('seed')}" - if FLAG_MEMORIZED in flags: - print( - f"\n [{label}] copy_rate={row.get('copy_rate'):.2f}: most of this batch reproduces designs " - "from the dataset rather than generating them. Its distribution and performance scores " - "measure retrieval, and it will be published but not ranked." - ) if FLAG_IGNORES_CONDITIONS in flags: print( f"\n [{label}] cond_sens=0: output did not change at all when the conditions were shuffled, " diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 6f19d6de..31e4cd7a 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -97,7 +97,7 @@ def evaluate( space: `pixel` scores the designs themselves. Any other name in `SPACES` (`pca` is built in) projects every design into that space first and appends `@space` to each column; metrics that - need the actual designs (`viol`, `copy_rate`) are skipped there. + need the actual designs (`viol`, `train_distance`) are skipped there. aggregation: How metrics with one value per design collapse to one number. width: Dimension of the space, for spaces that have one (`pca`). reference_row: Also score one random half of the reference designs @@ -124,7 +124,7 @@ def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: ref_designs=project(np.asarray(against)), sigma=sigma, aggregation=aggregation, - copy_corpus_fn=(lambda: np.asarray(self.train)) if self.train is not None else None, + train_designs_fn=(lambda: np.asarray(self.train)) if self.train is not None else None, ) row: dict[str, float] = {} for spec in chosen: @@ -160,7 +160,7 @@ def from_evaluator( """Score loaded models, and optionally saved designs, under one evaluation spec. The evaluator supplies the spec, the reference designs, the conditions and - the copy corpus, so every row is comparable. Live models get every column; + the training split, so every row is comparable. Live models get every column; `cond_sens` and the cost columns need a model to re-run or inspect, so they stay blank for the `designs` rows. The designs each model produced are kept on `Board.designs`, so they can be re-scored in another space or saved. diff --git a/engiopt/evaluation/context.py b/engiopt/evaluation/context.py index adeb8c6e..46f2c902 100644 --- a/engiopt/evaluation/context.py +++ b/engiopt/evaluation/context.py @@ -97,13 +97,10 @@ class EvaluationContext: objective_weight_condition: Name of a per-sample condition carrying the trade-off instead. thermoelastic2d's `weight` splits the first two objectives as `(w, 1 - w)`; any remaining objectives get zero. - copy_corpus_fn: Returns the designs a model could plausibly have copied - -- the dataset designs it may have trained on, plus the reference - optima it is being scored against. Deferred behind a callable so - that fetching a training split is paid for only when a memorization - metric actually runs, and shared across every model in a sweep. - copy_tol: RMS per-element distance below which a generated design counts - as a copy of something in the corpus. + train_designs_fn: Returns the training split's designs. Deferred behind + a callable so that fetching the split is paid for only when a + memorization metric actually runs, and shared across every model + in a sweep. resample_permuted: Re-runs the generator on a permutation of the same conditions, with the same latent draw. `None` when the comparison is unavailable -- an unconditional problem, or a context built directly @@ -121,8 +118,7 @@ class EvaluationContext: sample_seconds: float | None = None objective_weights: tuple[float, ...] | None = None objective_weight_condition: str | None = None - copy_corpus_fn: Callable[[], npt.NDArray[Any]] | None = None - copy_tol: float = 0.01 + train_designs_fn: Callable[[], npt.NDArray[Any]] | None = None aggregation: Literal["mean", "median"] = "mean" model_params: int | None = None """Trainable parameter count of the generator, when it is a network.""" @@ -151,7 +147,7 @@ def reduce(self, values: Any) -> float: So the choice is declared once (`EvalSpec.aggregation`), recorded in the row, and applied here rather than hard-coded metric by metric. - Rates such as `viol` and `copy_rate` are fractions, not per-design + Rates such as `viol` are fractions, not per-design averages, and do not pass through this. Args: @@ -180,55 +176,32 @@ def ref_flat(self) -> npt.NDArray[Any]: @cached_property def train_designs(self) -> npt.NDArray[Any] | None: - """Flattened training designs, or None when the problem has no training split.""" - if self.copy_corpus_fn is None: + """Flattened training designs, or None when the problem has no training split. + + Fetched once, through `train_designs_fn`, and only when a memorization + metric asks. A split whose flattened width does not match the generated + designs is treated as absent rather than raising: it makes the check + unavailable, which is not a reason to fail an evaluation. + """ + if self.train_designs_fn is None: return None - train = np.asarray(self.copy_corpus_fn()) + train = np.asarray(self.train_designs_fn()) if not len(train): return None flat = train.reshape(len(train), -1) return flat if flat.shape[1] == self.gen_flat.shape[1] else None - @cached_property - def copy_corpus(self) -> npt.NDArray[Any] | None: - """Flattened designs a generator could have memorized, or None if there are none. - - The scored reference designs are always in here, even though they are - also `mmd`'s comparison target. They are the single most attractive - thing to copy precisely *because* the protocol is public and names them, - so a memorization check that omitted them would miss the easiest attack - on this board. `copy_corpus_fn` widens the corpus to the training split - the model was actually fitted on. - - Parts whose flattened width does not match the generated designs are - dropped rather than raising: a mismatched corpus makes the check - unavailable, and that is not a reason to fail an evaluation. - """ - width = self.gen_flat.shape[1] - parts = [part for part in (self.ref_flat,) if part.shape[1] == width] - if self.copy_corpus_fn is not None: - extra = np.asarray(self.copy_corpus_fn()) - if len(extra): - flattened = extra.reshape(len(extra), -1) - if flattened.shape[1] == width: - parts.append(flattened) - return np.concatenate(parts) if parts else None + def nearest_train_distance(self, designs: npt.NDArray[Any]) -> npt.NDArray[Any]: + """Per-design RMS distance from each of `designs` to the closest training design. - @cached_property - def nearest_corpus_distance(self) -> npt.NDArray[Any] | None: - """Per-design RMS distance to the closest design in `copy_corpus`. - - Normalized by the square root of the design dimension so the number is a - *per-element* deviation. That makes one tolerance meaningful across - problems of different resolution, where a raw L2 norm would not be. + Divided by the square root of the design dimension so it reads as a + per-element deviation, the same on a 50 by 100 grid as on an 8 by 10 one. + Requires `train_designs`. """ - corpus = self.copy_corpus - if corpus is None or not len(corpus): - return None from scipy.spatial.distance import cdist - distances = cdist(self.gen_flat, corpus, "euclidean") - return np.asarray(distances.min(axis=1) / np.sqrt(self.gen_flat.shape[1])) + distances = cdist(designs, self.train_designs, "euclidean") + return np.asarray(distances.min(axis=1) / np.sqrt(designs.shape[1])) @cached_property def permuted_designs(self) -> npt.NDArray[Any] | None: diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 3eb087b3..0cbe3f8e 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -198,8 +198,7 @@ def context_for_designs(self, designs: npt.NDArray[Any]) -> EvaluationContext: volume_condition=self.spec.volume_condition, objective_weights=self.spec.objective_weights, objective_weight_condition=self.spec.objective_weight_condition, - copy_corpus_fn=self.copy_corpus, - copy_tol=self.spec.copy_tol, + train_designs_fn=self.train_designs, aggregation=self.spec.aggregation, ) @@ -234,37 +233,29 @@ def _permuted_sampler(self, generator: Generator) -> Callable[[npt.NDArray[Any]] return lambda order: self._sample(generator, order) @functools.cached_property - def copy_corpus(self) -> Callable[[], npt.NDArray[Any]]: - """Designs from the training split that a model on this problem could have memorized. + def train_designs(self) -> Callable[[], npt.NDArray[Any]]: + """The training split's designs, for the memorization metrics. - Built once per evaluator and shared by every model in a sweep, and + Fetched once per evaluator and shared by every model in a sweep, and deferred behind a callable so a run that selects no memorization metric - never pays for the fetch. + never pays for the fetch. The whole split, not a sample of it: a + lookup table over the training data must read as distance zero, and a + subsample would let most of its designs through. """ @functools.cache - def corpus() -> npt.NDArray[Any]: - return self._draw_copy_corpus() + def designs() -> npt.NDArray[Any]: + return self._load_train_designs() - return corpus + return designs - def _draw_copy_corpus(self) -> npt.NDArray[Any]: - """Subsample the training split's optimal designs, deterministically. - - Drawn with the spec's own `condition_seed`, so the corpus a model is - checked against is as reproducible as the conditions it is scored on -- - an audit that drew a different corpus could reach a different verdict. - """ + def _load_train_designs(self) -> npt.NDArray[Any]: + """The training split's optimal designs, or an empty array for a problem without one.""" try: train = self.problem.dataset["train"] - designs = np.asarray(train["optimal_design"]) - # A problem with no training split simply has no wider corpus; the - # reference designs still are one, and they are the case that matters. + return np.asarray(train["optimal_design"]) except (KeyError, TypeError, AttributeError): return np.empty((0, 0)) - size = min(self.spec.copy_corpus_size, len(designs)) - rng = np.random.default_rng(self.spec.condition_seed) - return designs[rng.choice(len(designs), size, replace=False)] def score( self, diff --git a/engiopt/evaluation/leaderboard.py b/engiopt/evaluation/leaderboard.py index 9fdfaa33..af3f38a4 100644 --- a/engiopt/evaluation/leaderboard.py +++ b/engiopt/evaluation/leaderboard.py @@ -302,7 +302,6 @@ def push_to_hub( private: bool = False, commit_message: str | None = None, max_attempts: int = 3, - eval_spec: EvalSpec | None = None, check_admission: bool = True, ) -> pd.DataFrame: """Merge `new_rows` into the published leaderboard and upload the result. @@ -326,7 +325,6 @@ def push_to_hub( private: Create the repo private if it does not exist yet. commit_message: Defaults to a summary of what was added. max_attempts: How many times to retry after losing a race. - eval_spec: Spec the rows were scored under, for the flag thresholds. check_admission: Publish rows unchecked. Only for a verification runner writing back rows it produced itself, which is the one caller whose `verified=True` is not a self-assertion. @@ -344,7 +342,7 @@ def push_to_hub( from engiopt.evaluation.submission import prepare_submission if check_admission: - new_rows = prepare_submission(new_rows, eval_spec) + new_rows = prepare_submission(new_rows) api = HfApi(token=token) api.create_repo(repo_id=repo_id, repo_type="dataset", private=private, exist_ok=True) diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 325097fb..7564e77c 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -76,46 +76,29 @@ def dpp(ctx: EvaluationContext) -> float: cost="cheap", higher_is_better=None, pixel_only=True, + outputs=("train_distance", "train_distance_ratio"), ) -def train_distance(ctx: EvaluationContext) -> float: +def train_distance(ctx: EvaluationContext) -> dict[str, float]: """Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. - Per-element RMS distance to the closest training design, so one tolerance - means the same thing on problems of different resolution. Zero means the - model reproduces its training set; large means only that the output is far - from the data, which noise also achieves. Read beside `mmd` and `viol`, - never alone -- which is why it declares no direction. - - NaN when the problem has no training split to compare against. + Two readings of one measurement. `train_distance` is the distance itself, + per element, so the number means the same thing on problems of different + resolution. `train_distance_ratio` divides it by the same measurement taken + on the withheld reference designs, which is how far real optima the model + never saw sit from the training sweep: one means the model's designs are as + far from its training data as real unseen designs are, near zero means it + reproduces the training set, and well above one means further from the data + than real designs, which noise also achieves. No tolerance anywhere; the + reference designs supply the scale and nothing else. + + Diagnostic, with no direction: zero is retrieval, but large is not good. + Read beside `mmd` and `viol`. NaN when the problem has no training split. """ if ctx.train_designs is None: - return float("nan") - from scipy.spatial.distance import cdist - - nearest = cdist(ctx.gen_flat, ctx.train_designs).min(axis=1) / np.sqrt(ctx.gen_flat.shape[1]) - return ctx.reduce(nearest) - - -@register_metric( - "copy_rate", - family="memorization", - cost="cheap", - higher_is_better=None, - pixel_only=True, -) -def copy_rate(ctx: EvaluationContext) -> float: - """Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. - - A design counts as a copy when it sits within `copy_tol` (per-element RMS) of - any design in the corpus a model could reproduce: the training split *and* - the scored reference optima, because the evaluation protocol is public and a - lookup table keyed on the conditions can return exactly those. This is the - column that gates a submission. NaN when no corpus is available. - """ - distances = ctx.nearest_corpus_distance - if distances is None: - return float("nan") - return float(np.mean(distances < ctx.copy_tol)) + return {"train_distance": float("nan"), "train_distance_ratio": float("nan")} + generated = ctx.reduce(ctx.nearest_train_distance(ctx.gen_flat)) + reference = ctx.reduce(ctx.nearest_train_distance(ctx.ref_flat)) + return {"train_distance": generated, "train_distance_ratio": generated / reference if reference > 0 else float("nan")} # ---------------------------------------------------------------------- diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index 79f90acf..7acafc3e 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -101,16 +101,6 @@ class EvalSpec: submitter running one seed -- it does not stop them running twenty and publishing their best three, which produces a median over a maximum. Fixing *which* seeds removes the choice. - copy_tol: Per-element RMS distance below which a generated design counts - as a copy of a design the model could have memorized. See the - `novelty` metric. - max_copy_rate: The share of copied designs above which an entry is - flagged and left out of the ranking. Set to 1.0 to disable the gate - and report `copy_rate` without acting on it. - copy_corpus_size: How many dataset designs to draw as the memorization - corpus. The scored reference designs are always included on top of - these, since the public spec names them and they are the most - attractive thing to copy. condition_digest: Hash of the drawn indices, condition values, and reference designs. Recomputed at evaluation time and compared, so an upstream dataset change is caught instead of silently shifting every @@ -150,9 +140,6 @@ class EvalSpec: objective_weights: tuple[float, ...] | None = None objective_weight_condition: str | None = None required_seeds: tuple[int, ...] = (1, 2, 3) - copy_tol: float = 0.01 - max_copy_rate: float = 0.5 - copy_corpus_size: int = 512 condition_digest: str | None = None dataset_id: str | None = None dataset_revision: str | None = None @@ -463,9 +450,6 @@ def freeze_spec( objective_weights: tuple[float, ...] | None = None, objective_weight_condition: str | None = None, required_seeds: tuple[int, ...] = EvalSpec.required_seeds, - copy_tol: float = EvalSpec.copy_tol, - max_copy_rate: float = EvalSpec.max_copy_rate, - copy_corpus_size: int = EvalSpec.copy_corpus_size, notes: str = "", ) -> Path: """Draw a problem's test conditions once and commit them as a spec. @@ -499,9 +483,6 @@ def freeze_spec( objective_weights=objective_weights, objective_weight_condition=objective_weight_condition, required_seeds=required_seeds, - copy_tol=copy_tol, - max_copy_rate=max_copy_rate, - copy_corpus_size=copy_corpus_size, notes=notes, ).freeze(problem) path = spec.save() diff --git a/engiopt/evaluation/submission.py b/engiopt/evaluation/submission.py index b0d99e95..daac2b2e 100644 --- a/engiopt/evaluation/submission.py +++ b/engiopt/evaluation/submission.py @@ -20,13 +20,10 @@ from __future__ import annotations import math -from typing import Any, TYPE_CHECKING +from typing import Any import pandas as pd -if TYPE_CHECKING: - from engiopt.evaluation.spec import EvalSpec - CHECKPOINT_ADDRESS_COLUMNS = ("checkpoint_repo", "checkpoint_path", "checkpoint_revision", "checkpoint_hash") """What it takes to fetch the exact weights a row was scored on. @@ -38,16 +35,13 @@ IDENTITY_COLUMNS = ("problem_id", "algo_id", "config_fingerprint", "seed", "spec_version") """What it takes to know which model, and under which protocol, a row describes.""" -FLAG_MEMORIZED = "memorized" -"""The batch is largely reproductions of designs the model could have seen.""" - FLAG_UNVERIFIED = "unverified" """No runner has re-fetched these weights and reproduced these numbers.""" FLAG_IGNORES_CONDITIONS = "ignores_conditions" """Output did not move when the conditions did, on a problem where they vary.""" -DISQUALIFYING_FLAGS = frozenset({FLAG_MEMORIZED, FLAG_IGNORES_CONDITIONS}) +DISQUALIFYING_FLAGS = frozenset({FLAG_IGNORES_CONDITIONS}) """Flags that keep a row out of the ranking while leaving it on the board. `unverified` is not among them only because ranking already requires a @@ -85,7 +79,7 @@ def admission_problems(row: dict[str, Any]) -> list[str]: return problems -def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[str]: +def integrity_flags(row: dict[str, Any]) -> list[str]: """Which integrity checks this row trips. Row-level only. Whether an entry covers the seeds the spec requires is a @@ -93,10 +87,6 @@ def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[s ranking instead -- see `leaderboard.rank`. """ flags: list[str] = [] - copy_rate = as_metric_value(row.get("copy_rate")) - max_copy_rate = spec.max_copy_rate if spec is not None else 0.5 - if copy_rate is not None and copy_rate > max_copy_rate: - flags.append(FLAG_MEMORIZED) if _declares_conditioning_it_does_not_use(row): flags.append(FLAG_IGNORES_CONDITIONS) if not bool(row.get("verified")): @@ -124,12 +114,11 @@ def _declares_conditioning_it_does_not_use(row: dict[str, Any]) -> bool: return generator is not None and generator.conditional -def prepare_submission(frame: pd.DataFrame, spec: EvalSpec | None = None) -> pd.DataFrame: +def prepare_submission(frame: pd.DataFrame) -> pd.DataFrame: """Check a table of new rows and stamp their flags, ready to publish. Args: frame: Rows as produced by `Evaluator.leaderboard`. - spec: The spec they were scored under, for its gate thresholds. Returns: The same rows with `verified` forced False and `flags` filled in. @@ -157,7 +146,7 @@ def prepare_submission(frame: pd.DataFrame, spec: EvalSpec | None = None) -> pd. prepared["verified"] = False prepared["verified_by"] = None prepared["verified_at"] = None - prepared["flags"] = [",".join(integrity_flags(record, spec)) for record in prepared.to_dict("records")] + prepared["flags"] = [",".join(integrity_flags(record)) for record in prepared.to_dict("records")] return prepared diff --git a/engiopt/evaluation/verify.py b/engiopt/evaluation/verify.py index b531a2c8..05294251 100644 --- a/engiopt/evaluation/verify.py +++ b/engiopt/evaluation/verify.py @@ -155,7 +155,7 @@ def verify_row( } ) uncorroborated = _uncorroborated(row, rescored) - rescored["flags"] = ",".join(integrity_flags(rescored, evaluator.spec)) + rescored["flags"] = ",".join(integrity_flags(rescored)) return VerificationResult( key=key, status="verified", diff --git a/engiopt/specs/beams2d/v1.json b/engiopt/specs/beams2d/v1.json index 1e3e5a43..f8926c05 100644 --- a/engiopt/specs/beams2d/v1.json +++ b/engiopt/specs/beams2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "f65e2c02551e7e08", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/beams_2d_50_100_v0", "dataset_revision": "ccb64d4f09cf66a5ed9ba2081438d695b9e2438d", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/beams2d/v2.json b/engiopt/specs/beams2d/v2.json index bcd79592..7278e3d5 100644 --- a/engiopt/specs/beams2d/v2.json +++ b/engiopt/specs/beams2d/v2.json @@ -1,12 +1,9 @@ { "condition_digest": "f65e2c02551e7e08", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/beams_2d_50_100_v0", "dataset_revision": "ccb64d4f09cf66a5ed9ba2081438d695b9e2438d", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "coverage", @@ -17,7 +14,6 @@ "per_condition_distance", "cond_sens", "train_distance", - "copy_rate", "generation_seconds", "n_parameters", "train_minutes", diff --git a/engiopt/specs/heatconduction2d/v1.json b/engiopt/specs/heatconduction2d/v1.json index b2186a46..a0221ecd 100644 --- a/engiopt/specs/heatconduction2d/v1.json +++ b/engiopt/specs/heatconduction2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "40030d1b2482e003", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/heat_conduction_2d_v0", "dataset_revision": "9f07c1e70d244f783e34a71c96db621173871dc1", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/heatconduction2d/v2.json b/engiopt/specs/heatconduction2d/v2.json index ac4bffb5..e423b573 100644 --- a/engiopt/specs/heatconduction2d/v2.json +++ b/engiopt/specs/heatconduction2d/v2.json @@ -1,12 +1,9 @@ { "condition_digest": "40030d1b2482e003", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/heat_conduction_2d_v0", "dataset_revision": "9f07c1e70d244f783e34a71c96db621173871dc1", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "coverage", @@ -17,7 +14,6 @@ "per_condition_distance", "cond_sens", "train_distance", - "copy_rate", "generation_seconds", "n_parameters", "train_minutes", diff --git a/engiopt/specs/photonics2d/v1.json b/engiopt/specs/photonics2d/v1.json index e00b6341..f0171df9 100644 --- a/engiopt/specs/photonics2d/v1.json +++ b/engiopt/specs/photonics2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "6b8330bee5a8718c", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/photonics_2d_120_120_v1", "dataset_revision": "be811ff69d67cd8c79148e29f2a68d293ec00955", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/photonics2d/v2.json b/engiopt/specs/photonics2d/v2.json index dee9c339..cfd317e2 100644 --- a/engiopt/specs/photonics2d/v2.json +++ b/engiopt/specs/photonics2d/v2.json @@ -1,12 +1,9 @@ { "condition_digest": "6b8330bee5a8718c", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/photonics_2d_120_120_v1", "dataset_revision": "be811ff69d67cd8c79148e29f2a68d293ec00955", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "coverage", @@ -17,7 +14,6 @@ "per_condition_distance", "cond_sens", "train_distance", - "copy_rate", "generation_seconds", "n_parameters", "train_minutes", diff --git a/engiopt/specs/thermoelastic2d/v1.json b/engiopt/specs/thermoelastic2d/v1.json index 38c4302f..0271b3d6 100644 --- a/engiopt/specs/thermoelastic2d/v1.json +++ b/engiopt/specs/thermoelastic2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "818d93c2a2d5ac25", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/thermoelastic_2d_v1", "dataset_revision": "7d5b4a694bd1a8f3f25fd745ac9f669764e29ea5", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/thermoelastic2d/v2.json b/engiopt/specs/thermoelastic2d/v2.json index 89120237..30281b66 100644 --- a/engiopt/specs/thermoelastic2d/v2.json +++ b/engiopt/specs/thermoelastic2d/v2.json @@ -1,12 +1,9 @@ { "condition_digest": "818d93c2a2d5ac25", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/thermoelastic_2d_v1", "dataset_revision": "7d5b4a694bd1a8f3f25fd745ac9f669764e29ea5", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "coverage", @@ -17,7 +14,6 @@ "per_condition_distance", "cond_sens", "train_distance", - "copy_rate", "generation_seconds", "n_parameters", "train_minutes", diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index fe64bdcf..3c418e66 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "cd775ce2", + "id": "cfd0f310", "metadata": {}, "source": [ "# Using the metrics suite, on beams2d\n", @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "6c31865f", + "id": "45733beb", "metadata": {}, "source": [ "## 1. What metrics exist, and what does each one measure?" @@ -36,13 +36,13 @@ { "cell_type": "code", "execution_count": 1, - "id": "22f4b5a3", + "id": "74b137ca", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:05:54.929474Z", - "iopub.status.busy": "2026-09-28T15:05:54.928715Z", - "iopub.status.idle": "2026-09-28T15:05:57.569493Z", - "shell.execute_reply": "2026-09-28T15:05:57.569236Z" + "iopub.execute_input": "2026-09-28T15:28:02.465639Z", + "iopub.status.busy": "2026-09-28T15:28:02.465228Z", + "iopub.status.idle": "2026-09-28T15:28:05.075433Z", + "shell.execute_reply": "2026-09-28T15:28:05.075194Z" } }, "outputs": [ @@ -106,13 +106,6 @@ " cheap\n", " \n", " \n", - " copy_rate\n", - " Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval.\n", - " memorization\n", - " diagnostic\n", - " cheap\n", - " \n", - " \n", " cond_sens\n", " How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.\n", " conditions\n", @@ -233,7 +226,6 @@ "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", @@ -255,7 +247,6 @@ "mmd distribution lower is better cheap \n", "dpp diversity higher is better cheap \n", "train_distance memorization diagnostic cheap \n", - "copy_rate memorization diagnostic cheap \n", "cond_sens conditions diagnostic cheap \n", "viol feasibility lower is better cheap \n", "iog performance lower is better expensive \n", @@ -298,7 +289,7 @@ }, { "cell_type": "markdown", - "id": "0bf1c95d", + "id": "8c614e35", "metadata": {}, "source": [ "Each row names the quantity, says how it is computed, and says what the two ends of\n", @@ -308,7 +299,7 @@ }, { "cell_type": "markdown", - "id": "556f7636", + "id": "55b739c4", "metadata": {}, "source": [ "## 2. The problem, and the evaluation spec" @@ -316,12 +307,12 @@ }, { "cell_type": "markdown", - "id": "658d22e5", + "id": "4118d701", "metadata": {}, "source": [ "Every published number on beams2d is computed under one frozen spec. The spec fixes\n", "which fifty test conditions are scored, the reference optimum for each, the kernel\n", - "bandwidth, the copy tolerance, and how per-design values collapse to one number. Two\n", + "bandwidth, and how per-design values collapse to one number. Two\n", "models scored under the same spec are comparable; two scored under different specs\n", "are not, and every row records which spec produced it.\n", "\n", @@ -341,13 +332,13 @@ { "cell_type": "code", "execution_count": 2, - "id": "7f5b40c7", + "id": "31940087", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:05:57.571900Z", - "iopub.status.busy": "2026-09-28T15:05:57.571724Z", - "iopub.status.idle": "2026-09-28T15:05:59.122592Z", - "shell.execute_reply": "2026-09-28T15:05:59.122294Z" + "iopub.execute_input": "2026-09-28T15:28:05.077884Z", + "iopub.status.busy": "2026-09-28T15:28:05.077720Z", + "iopub.status.idle": "2026-09-28T15:28:06.757070Z", + "shell.execute_reply": "2026-09-28T15:28:06.756789Z" } }, "outputs": [ @@ -357,7 +348,7 @@ "text": [ "conditions scored: 12 | design shape: (50, 100)\n", "condition keys: ('volfrac', 'rmin', 'forcedist', 'overhang_constraint')\n", - "kernel sigma: 10.0 | copy tolerance: 0.01 | aggregation: median\n" + "kernel sigma: 10.0 | aggregation: median\n" ] } ], @@ -374,12 +365,12 @@ "REFERENCE = evaluator.resolved.ref_designs # the withheld optimum for each scored condition\n", "print(\"conditions scored:\", len(REFERENCE), \"| design shape:\", REFERENCE.shape[1:])\n", "print(\"condition keys:\", evaluator.resolved.condition_keys)\n", - "print(\"kernel sigma:\", spec.sigma, \"| copy tolerance:\", spec.copy_tol, \"| aggregation:\", spec.aggregation)" + "print(\"kernel sigma:\", spec.sigma, \"| aggregation:\", spec.aggregation)" ] }, { "cell_type": "markdown", - "id": "ea7d9e5b", + "id": "2e1a7840", "metadata": {}, "source": [ "## 3. Three constructions built from the dataset" @@ -387,7 +378,7 @@ }, { "cell_type": "markdown", - "id": "85a51183", + "id": "649c3e65", "metadata": {}, "source": [ "Beside the trained models, three rows whose right score is known in advance. They are\n", @@ -403,13 +394,13 @@ { "cell_type": "code", "execution_count": 3, - "id": "17e539a9", + "id": "c8b7be65", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:05:59.124844Z", - "iopub.status.busy": "2026-09-28T15:05:59.124741Z", - "iopub.status.idle": "2026-09-28T15:06:02.348254Z", - "shell.execute_reply": "2026-09-28T15:06:02.347983Z" + "iopub.execute_input": "2026-09-28T15:28:06.759318Z", + "iopub.status.busy": "2026-09-28T15:28:06.759208Z", + "iopub.status.idle": "2026-09-28T15:28:09.851561Z", + "shell.execute_reply": "2026-09-28T15:28:09.851336Z" } }, "outputs": [ @@ -445,7 +436,7 @@ }, { "cell_type": "markdown", - "id": "84c32ab9", + "id": "c8b5930f", "metadata": {}, "source": [ "## 4. Score the trained models and the constructions under one spec" @@ -453,7 +444,7 @@ }, { "cell_type": "markdown", - "id": "bd9e899f", + "id": "c7faf5b1", "metadata": {}, "source": [ "Four published checkpoints at training seed one. Two of them were trained with a\n", @@ -468,20 +459,20 @@ { "cell_type": "code", "execution_count": 4, - "id": "3634cb40", + "id": "9f2a7786", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:06:02.350507Z", - "iopub.status.busy": "2026-09-28T15:06:02.350400Z", - "iopub.status.idle": "2026-09-28T15:08:12.429733Z", - "shell.execute_reply": "2026-09-28T15:08:12.429489Z" + "iopub.execute_input": "2026-09-28T15:28:09.853741Z", + "iopub.status.busy": "2026-09-28T15:28:09.853658Z", + "iopub.status.idle": "2026-09-28T15:30:18.134672Z", + "shell.execute_reply": "2026-09-28T15:30:18.134417Z" } }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "d1af91997b244d52a7da37d0e90ed747", + "model_id": "5d264478f5ae4104a027686d793af590", "version_major": 2, "version_minor": 0 }, @@ -495,7 +486,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "2412f7988d6547a39efbc08a88ad7310", + "model_id": "8e1f63b226964c988f9b85f514623736", "version_major": 2, "version_minor": 0 }, @@ -509,7 +500,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "9388d935952d4d81aae4962a19911d7f", + "model_id": "acda7f10636c4626b2363090435e8fa2", "version_major": 2, "version_minor": 0 }, @@ -550,7 +541,7 @@ " per_condition_distance\n", " cond_sens\n", " train_distance\n", - " copy_rate\n", + " train_distance_ratio\n", " ...\n", " iog\n", " cog\n", @@ -575,8 +566,8 @@ " 0.0722\n", " 0.4144\n", " 0.2407\n", - " 0.2939\n", - " 0.0\n", + " 0.2902\n", + " 14.2969\n", " ...\n", " 1.816083e+09\n", " 1.735349e+09\n", @@ -599,8 +590,8 @@ " 0.2407\n", " 0.4447\n", " 0.0000\n", - " 0.2632\n", - " 0.0\n", + " 0.2581\n", + " 12.7131\n", " ...\n", " 8.950722e+09\n", " 8.689641e+09\n", @@ -623,8 +614,8 @@ " 0.0058\n", " 0.0833\n", " 0.4704\n", - " 0.0531\n", - " 0.0\n", + " 0.0353\n", + " 1.7389\n", " ...\n", " 2.413100e+00\n", " 8.195000e-01\n", @@ -643,15 +634,15 @@ "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", - "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2939 0.0 ... 1.816083e+09 1.735349e+09 3.0758 1.0 \n", - "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2632 0.0 ... 8.950722e+09 8.689641e+09 0.0019 1.0 \n", - "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0531 0.0 ... 2.413100e+00 8.195000e-01 -0.6194 32.0 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog \\\n", + "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... 1.816083e+09 1.735349e+09 3.0758 \n", + "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... 8.950722e+09 8.689641e+09 0.0019 \n", + "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... 2.413100e+00 8.195000e-01 -0.6194 \n", "\n", - " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", - "cgan_cnn_2d 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", - "gan_cnn_2d 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", - "vqgan 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", + " calls_to_settle gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", + "cgan_cnn_2d 1.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", + "gan_cnn_2d 1.0 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", + "vqgan 32.0 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", "\n", "[3 rows x 23 columns]" ] @@ -687,20 +678,20 @@ { "cell_type": "code", "execution_count": 5, - "id": "5641c92d", + "id": "cfed22a5", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:08:12.432341Z", - "iopub.status.busy": "2026-09-28T15:08:12.432250Z", - "iopub.status.idle": "2026-09-28T15:11:41.043945Z", - "shell.execute_reply": "2026-09-28T15:11:41.043702Z" + "iopub.execute_input": "2026-09-28T15:30:18.137402Z", + "iopub.status.busy": "2026-09-28T15:30:18.137300Z", + "iopub.status.idle": "2026-09-28T15:33:45.992283Z", + "shell.execute_reply": "2026-09-28T15:33:45.992038Z" } }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "c82612395590444cbea28d7bcbe39a88", + "model_id": "970f74188c7d4f7a83de75cec1e6fb39", "version_major": 2, "version_minor": 0 }, @@ -741,7 +732,7 @@ " per_condition_distance\n", " cond_sens\n", " train_distance\n", - " copy_rate\n", + " train_distance_ratio\n", " ...\n", " iog\n", " cog\n", @@ -766,8 +757,8 @@ " 0.1271\n", " 0.1532\n", " 0.3969\n", - " 0.146\n", - " 0.0\n", + " 0.143\n", + " 7.0439\n", " ...\n", " -1.7211\n", " 6.0217\n", @@ -786,8 +777,8 @@ "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", - "diffusion_2d_cond 0.1105 1.0 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.146 0.0 ... -1.7211 6.0217 -0.9128 17.0 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog calls_to_settle \\\n", + "diffusion_2d_cond 0.1105 1.0 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.143 7.0439 ... -1.7211 6.0217 -0.9128 17.0 \n", "\n", " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", "diffusion_2d_cond -1.7127 26.3979 6.1904 0.3109 0.9167 2.9236 \n", @@ -810,13 +801,13 @@ { "cell_type": "code", "execution_count": 6, - "id": "950db794", + "id": "e69473d0", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:11:41.046650Z", - "iopub.status.busy": "2026-09-28T15:11:41.046558Z", - "iopub.status.idle": "2026-09-28T15:13:05.685877Z", - "shell.execute_reply": "2026-09-28T15:13:05.685654Z" + "iopub.execute_input": "2026-09-28T15:33:45.994933Z", + "iopub.status.busy": "2026-09-28T15:33:45.994841Z", + "iopub.status.idle": "2026-09-28T15:35:10.617468Z", + "shell.execute_reply": "2026-09-28T15:35:10.617235Z" } }, "outputs": [ @@ -850,7 +841,7 @@ " per_condition_distance\n", " cond_sens\n", " train_distance\n", - " copy_rate\n", + " train_distance_ratio\n", " ...\n", " iog\n", " cog\n", @@ -875,8 +866,8 @@ " 0.0722\n", " 0.4144\n", " 0.2407\n", - " 0.2939\n", - " 0.00\n", + " 0.2902\n", + " 14.2969\n", " ...\n", " 1.816083e+09\n", " 1.735349e+09\n", @@ -899,8 +890,8 @@ " 0.2407\n", " 0.4447\n", " 0.0000\n", - " 0.2632\n", - " 0.00\n", + " 0.2581\n", + " 12.7131\n", " ...\n", " 8.950722e+09\n", " 8.689641e+09\n", @@ -923,8 +914,8 @@ " 0.0058\n", " 0.0833\n", " 0.4704\n", - " 0.0531\n", - " 0.00\n", + " 0.0353\n", + " 1.7389\n", " ...\n", " 2.413100e+00\n", " 8.195000e-01\n", @@ -947,8 +938,8 @@ " 0.1271\n", " 0.1532\n", " 0.3969\n", - " 0.1460\n", - " 0.00\n", + " 0.1430\n", + " 7.0439\n", " ...\n", " -1.721100e+00\n", " 6.021700e+00\n", @@ -971,8 +962,8 @@ " 0.0125\n", " 0.0569\n", " NaN\n", - " 0.0117\n", - " 0.50\n", + " 0.0000\n", + " 0.0000\n", " ...\n", " 9.574600e+00\n", " 7.314000e-01\n", @@ -995,8 +986,8 @@ " 0.0562\n", " 0.3993\n", " NaN\n", - " 0.0504\n", - " 0.25\n", + " 0.0000\n", + " 0.0000\n", " ...\n", " 4.805870e+03\n", " 2.817597e+04\n", @@ -1019,8 +1010,8 @@ " 0.0562\n", " 0.4231\n", " NaN\n", - " 0.0525\n", - " 1.00\n", + " 0.0203\n", + " 1.0000\n", " ...\n", " 8.783402e+07\n", " 8.788590e+07\n", @@ -1039,23 +1030,23 @@ "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance copy_rate ... iog cog fog calls_to_settle \\\n", - "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2939 0.00 ... 1.816083e+09 1.735349e+09 3.0758 1.0 \n", - "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2632 0.00 ... 8.950722e+09 8.689641e+09 0.0019 1.0 \n", - "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0531 0.00 ... 2.413100e+00 8.195000e-01 -0.6194 32.0 \n", - "diffusion_2d_cond 0.1105 1.00 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.1460 0.00 ... -1.721100e+00 6.021700e+00 -0.9128 17.0 \n", - "lookup_table 0.0729 1.00 9.7122 0.3051 0.7500 0.0125 0.0569 NaN 0.0117 0.50 ... 9.574600e+00 7.314000e-01 -0.7363 29.0 \n", - "unconditional 0.1631 1.00 11.1734 0.9277 0.9167 0.0562 0.3993 NaN 0.0504 0.25 ... 4.805870e+03 2.817597e+04 1.4287 6.0 \n", - "permuted_conditions 0.0000 1.00 10.0889 0.3162 0.9167 0.0562 0.4231 NaN 0.0525 1.00 ... 8.783402e+07 8.788590e+07 4.4569 2.0 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog \\\n", + "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... 1.816083e+09 1.735349e+09 3.0758 \n", + "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... 8.950722e+09 8.689641e+09 0.0019 \n", + "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... 2.413100e+00 8.195000e-01 -0.6194 \n", + "diffusion_2d_cond 0.1105 1.00 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.1430 7.0439 ... -1.721100e+00 6.021700e+00 -0.9128 \n", + "lookup_table 0.0729 1.00 9.7122 0.3051 0.7500 0.0125 0.0569 NaN 0.0000 0.0000 ... 9.574600e+00 7.314000e-01 -0.7363 \n", + "unconditional 0.1631 1.00 11.1734 0.9277 0.9167 0.0562 0.3993 NaN 0.0000 0.0000 ... 4.805870e+03 2.817597e+04 1.4287 \n", + "permuted_conditions 0.0000 1.00 10.0889 0.3162 0.9167 0.0562 0.4231 NaN 0.0203 1.0000 ... 8.783402e+07 8.788590e+07 4.4569 \n", "\n", - " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", - "cgan_cnn_2d 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", - "gan_cnn_2d 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", - "vqgan 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", - "diffusion_2d_cond -1.712700e+00 26.3979 6.1904 0.3109 0.9167 2.9236 \n", - "lookup_table 9.590800e+00 8.2825 3.3313 0.3676 0.5833 0.3145 \n", - "unconditional 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 0.4913 \n", - "permuted_conditions 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 0.8233 \n", + " calls_to_settle gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", + "cgan_cnn_2d 1.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", + "gan_cnn_2d 1.0 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", + "vqgan 32.0 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", + "diffusion_2d_cond 17.0 -1.712700e+00 26.3979 6.1904 0.3109 0.9167 2.9236 \n", + "lookup_table 29.0 9.590800e+00 8.2825 3.3313 0.3676 0.5833 0.3145 \n", + "unconditional 6.0 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 0.4913 \n", + "permuted_conditions 2.0 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 0.8233 \n", "\n", "[7 rows x 23 columns]" ] @@ -1074,7 +1065,7 @@ }, { "cell_type": "markdown", - "id": "87dabe65", + "id": "773f94f8", "metadata": {}, "source": [ "## 5. Read the board" @@ -1083,13 +1074,13 @@ { "cell_type": "code", "execution_count": 7, - "id": "fa5cb3e9", + "id": "cd254904", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:13:05.688476Z", - "iopub.status.busy": "2026-09-28T15:13:05.688370Z", - "iopub.status.idle": "2026-09-28T15:13:05.693238Z", - "shell.execute_reply": "2026-09-28T15:13:05.693058Z" + "iopub.execute_input": "2026-09-28T15:35:10.620119Z", + "iopub.status.busy": "2026-09-28T15:35:10.620036Z", + "iopub.status.idle": "2026-09-28T15:35:10.624691Z", + "shell.execute_reply": "2026-09-28T15:35:10.624518Z" } }, "outputs": [ @@ -1185,8 +1176,8 @@ " None\n", " \n", " \n", - " copy_rate\n", - " Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval.\n", + " train_distance_ratio\n", + " Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.\n", " diagnostic\n", " \n", " None\n", @@ -1297,7 +1288,7 @@ "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "copy_rate Fraction of the generated designs within the copy tolerance of a design the model could have seen. One means pure retrieval. \n", + "train_distance_ratio Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", @@ -1322,7 +1313,7 @@ "per_condition_distance lower is better lookup_table None \n", "cond_sens diagnostic None \n", "train_distance diagnostic None \n", - "copy_rate diagnostic None \n", + "train_distance_ratio diagnostic None \n", "generation_seconds lower is better gan_cnn_2d None \n", "n_parameters lower is better cgan_cnn_2d None \n", "train_minutes lower is better None \n", @@ -1349,7 +1340,7 @@ }, { "cell_type": "markdown", - "id": "1a95df7d", + "id": "6d86e56b", "metadata": {}, "source": [ "Why these seven rows: four trained models and three constructions whose right\n", @@ -1365,11 +1356,16 @@ "physics. `per_condition_distance` reads 0.42 for it, in the same tier as the GANs, and\n", "that is the column that sees the problem before any optimizer runs.\n", "\n", - "**`lookup_table` never generalizes, half its designs are exact copies, and it beats\n", + "**`lookup_table` never generalizes, every design is a training design, and it beats\n", "every trained model except diffusion on final gap.** It has the best cumulative gap on\n", "the board and the smallest per-condition distance, 0.057. That is not a bug. With one\n", "optimum per condition and a dense training sweep, the nearest training design is a very\n", - "good warm start. `copy_rate` at 0.50 is the only column that says what it is.\n", + "good warm start. `train_distance` reads exactly zero for it, and so does the ratio: that\n", + "is the column that says what it is. `permuted_conditions` reads a ratio of exactly one,\n", + "by construction, because it is the reference set and one is what real unseen designs\n", + "read. The trained models read 1.7 for the VQGAN, 7 for diffusion and 13 to 14 for the\n", + "GANs: further from the training data than real designs are, which is what blurred or\n", + "noisy output looks like. Unlike the data is not the same as generalizing.\n", "\n", "**`diffusion_2d_cond` has the best final gap, minus 0.91, and reaches the reference\n", "optimum from eleven designs in twelve.** It also violates a constraint in ten of twelve\n", @@ -1405,7 +1401,7 @@ }, { "cell_type": "markdown", - "id": "a0f0269a", + "id": "7297bed1", "metadata": {}, "source": [ "## 6. Rank, but only on a column that can be ranked" @@ -1414,13 +1410,13 @@ { "cell_type": "code", "execution_count": 8, - "id": "6f0e620e", + "id": "168eb15d", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:13:05.695701Z", - "iopub.status.busy": "2026-09-28T15:13:05.695623Z", - "iopub.status.idle": "2026-09-28T15:13:05.698788Z", - "shell.execute_reply": "2026-09-28T15:13:05.698554Z" + "iopub.execute_input": "2026-09-28T15:35:10.627119Z", + "iopub.status.busy": "2026-09-28T15:35:10.627038Z", + "iopub.status.idle": "2026-09-28T15:35:10.630171Z", + "shell.execute_reply": "2026-09-28T15:35:10.629928Z" } }, "outputs": [ @@ -1446,7 +1442,7 @@ " \n", " \n", " fog\n", - " copy_rate\n", + " train_distance_ratio\n", " per_condition_distance\n", " \n", " \n", @@ -1454,43 +1450,43 @@ " \n", " diffusion_2d_cond\n", " -0.9128\n", - " 0.00\n", + " 7.0439\n", " 0.1532\n", " \n", " \n", " lookup_table\n", " -0.7363\n", - " 0.50\n", + " 0.0000\n", " 0.0569\n", " \n", " \n", " vqgan\n", " -0.6194\n", - " 0.00\n", + " 1.7389\n", " 0.0833\n", " \n", " \n", " gan_cnn_2d\n", " 0.0019\n", - " 0.00\n", + " 12.7131\n", " 0.4447\n", " \n", " \n", " unconditional\n", " 1.4287\n", - " 0.25\n", + " 0.0000\n", " 0.3993\n", " \n", " \n", " cgan_cnn_2d\n", " 3.0758\n", - " 0.00\n", + " 14.2969\n", " 0.4144\n", " \n", " \n", " permuted_conditions\n", " 4.4569\n", - " 1.00\n", + " 1.0000\n", " 0.4231\n", " \n", " \n", @@ -1498,14 +1494,14 @@ "" ], "text/plain": [ - " fog copy_rate per_condition_distance\n", - "diffusion_2d_cond -0.9128 0.00 0.1532\n", - "lookup_table -0.7363 0.50 0.0569\n", - "vqgan -0.6194 0.00 0.0833\n", - "gan_cnn_2d 0.0019 0.00 0.4447\n", - "unconditional 1.4287 0.25 0.3993\n", - "cgan_cnn_2d 3.0758 0.00 0.4144\n", - "permuted_conditions 4.4569 1.00 0.4231" + " fog train_distance_ratio per_condition_distance\n", + "diffusion_2d_cond -0.9128 7.0439 0.1532\n", + "lookup_table -0.7363 0.0000 0.0569\n", + "vqgan -0.6194 1.7389 0.0833\n", + "gan_cnn_2d 0.0019 12.7131 0.4447\n", + "unconditional 1.4287 0.0000 0.3993\n", + "cgan_cnn_2d 3.0758 14.2969 0.4144\n", + "permuted_conditions 4.4569 1.0000 0.4231" ] }, "execution_count": 8, @@ -1515,19 +1511,19 @@ ], "source": [ "column = \"fog\" if EXPENSIVE else \"mmd\"\n", - "board.rank(column)[[column, \"copy_rate\", \"per_condition_distance\"]].round(4)" + "board.rank(column)[[column, \"train_distance_ratio\", \"per_condition_distance\"]].round(4)" ] }, { "cell_type": "code", "execution_count": 9, - "id": "c39d8696", + "id": "5ddab12f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:13:05.701188Z", - "iopub.status.busy": "2026-09-28T15:13:05.701115Z", - "iopub.status.idle": "2026-09-28T15:13:05.702554Z", - "shell.execute_reply": "2026-09-28T15:13:05.702378Z" + "iopub.execute_input": "2026-09-28T15:35:10.632624Z", + "iopub.status.busy": "2026-09-28T15:35:10.632565Z", + "iopub.status.idle": "2026-09-28T15:35:10.634223Z", + "shell.execute_reply": "2026-09-28T15:35:10.634041Z" } }, "outputs": [ @@ -1535,20 +1531,20 @@ "name": "stdout", "output_type": "stream", "text": [ - "'copy_rate' is a diagnostic: it says whether the other columns mean what they look like. Read it beside them; do not rank on it.\n" + "'train_distance_ratio' is a diagnostic: it says whether the other columns mean what they look like. Read it beside them; do not rank on it.\n" ] } ], "source": [ "try:\n", - " board.rank(\"copy_rate\")\n", + " board.rank(\"train_distance_ratio\")\n", "except ValueError as e:\n", " print(e)" ] }, { "cell_type": "markdown", - "id": "fa6f57fa", + "id": "510bdec2", "metadata": {}, "source": [ "## 7. The same questions in another space" @@ -1556,7 +1552,7 @@ }, { "cell_type": "markdown", - "id": "0db4a92f", + "id": "9f2babbc", "metadata": {}, "source": [ "Where a metric is computed is an argument, not a property of the metric. The board\n", @@ -1573,13 +1569,13 @@ { "cell_type": "code", "execution_count": 10, - "id": "d8783983", + "id": "2c5a60a3", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:13:05.704903Z", - "iopub.status.busy": "2026-09-28T15:13:05.704818Z", - "iopub.status.idle": "2026-09-28T15:13:05.716554Z", - "shell.execute_reply": "2026-09-28T15:13:05.716385Z" + "iopub.execute_input": "2026-09-28T15:35:10.636725Z", + "iopub.status.busy": "2026-09-28T15:35:10.636609Z", + "iopub.status.idle": "2026-09-28T15:35:10.646420Z", + "shell.execute_reply": "2026-09-28T15:35:10.646245Z" } }, "outputs": [ @@ -1704,7 +1700,7 @@ }, { "cell_type": "markdown", - "id": "deb1252c", + "id": "69072434", "metadata": {}, "source": [ "## 8. Registering a metric of your own" @@ -1712,7 +1708,7 @@ }, { "cell_type": "markdown", - "id": "546a1629", + "id": "7bc23e01", "metadata": {}, "source": [ "A metric is a function of an `EvaluationContext`, a family, a cost and a direction.\n", @@ -1723,13 +1719,13 @@ { "cell_type": "code", "execution_count": 11, - "id": "98b7c485", + "id": "f6d49da7", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:13:05.719001Z", - "iopub.status.busy": "2026-09-28T15:13:05.718926Z", - "iopub.status.idle": "2026-09-28T15:13:05.724756Z", - "shell.execute_reply": "2026-09-28T15:13:05.724580Z" + "iopub.execute_input": "2026-09-28T15:35:10.648934Z", + "iopub.status.busy": "2026-09-28T15:35:10.648863Z", + "iopub.status.idle": "2026-09-28T15:35:10.654564Z", + "shell.execute_reply": "2026-09-28T15:35:10.654414Z" } }, "outputs": [ @@ -1805,7 +1801,7 @@ }, { "cell_type": "markdown", - "id": "d2742ff7", + "id": "7560f183", "metadata": {}, "source": [ "It appears with its question and direction attached, scored under the same spec as\n", @@ -1816,7 +1812,7 @@ }, { "cell_type": "markdown", - "id": "19ac14ba", + "id": "fe1e1e38", "metadata": {}, "source": [ "## 9. Not in this notebook" @@ -1824,7 +1820,7 @@ }, { "cell_type": "markdown", - "id": "769cb27a", + "id": "27263dd8", "metadata": {}, "source": [ "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec. It\n", diff --git a/tests/test_board.py b/tests/test_board.py index 619de9ec..17d310ba 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -30,7 +30,7 @@ def test_evaluate_scores_every_model_on_the_cheap_metrics(fake_problem: Any, des models, ref, train = designs frame = Board(fake_problem, reference=ref, train=train).evaluate(models) assert list(frame.index) == ["honest", "copier", REFERENCE_ROW] - assert {"mmd", "copy_rate", "viol"} <= set(frame.columns) + assert {"mmd", "train_distance", "viol"} <= set(frame.columns) def test_explain_names_a_pick_only_for_ranked_columns(fake_problem: Any, designs: Any) -> None: @@ -39,7 +39,7 @@ def test_explain_names_a_pick_only_for_ranked_columns(fake_problem: Any, designs board.evaluate(models) explained = board.explain() assert explained.loc["mmd", "picks"] == "copier", "identical sets score the best MMD" - assert explained.loc["copy_rate", "picks"] == "", "a diagnostic picks nobody" + assert explained.loc["train_distance", "picks"] == "", "a diagnostic picks nobody" assert explained.loc["mmd", "question"] == METRICS["mmd"].description assert explained.loc["mmd", "real designs score"] == board.frame.loc[REFERENCE_ROW, "mmd"] @@ -59,7 +59,7 @@ def test_rank_refuses_a_diagnostic(fake_problem: Any, designs: Any) -> None: board.evaluate(models) assert board.rank("mmd").index[0] == "copier" with pytest.raises(ValueError, match="diagnostic"): - board.rank("copy_rate") + board.rank("train_distance") def test_space_is_an_argument_not_a_metric(fake_problem: Any, designs: Any) -> None: diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index 7ad993db..ffed37e2 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -89,7 +89,7 @@ def test_builtin_metrics_declare_their_cost() -> None: judged by a constraint check rather than by running the optimizer. The two integrity metrics are cheap too, and that matters more than it - sounds: `copy_rate` and `cond_sens` decide whether a row can be ranked at all, + sounds: `train_distance` and `cond_sens` decide whether a row can be ranked at all, so a board that could only afford the cheap pass would otherwise have to rank models it had never checked for memorization. """ @@ -98,7 +98,6 @@ def test_builtin_metrics_declare_their_cost() -> None: "dpp", "viol", "train_distance", - "copy_rate", "cond_sens", "per_condition_distance", "volume_error", diff --git a/tests/test_leaderboard_integrity.py b/tests/test_leaderboard_integrity.py index 8d8e2d6f..859212d0 100644 --- a/tests/test_leaderboard_integrity.py +++ b/tests/test_leaderboard_integrity.py @@ -32,7 +32,6 @@ from engiopt.evaluation.spec import EvalSpec from engiopt.evaluation.submission import admission_problems from engiopt.evaluation.submission import FLAG_IGNORES_CONDITIONS -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.submission import FLAG_UNVERIFIED from engiopt.evaluation.submission import integrity_flags from engiopt.evaluation.submission import prepare_submission @@ -52,58 +51,47 @@ def _context(problem: Any, gen: np.ndarray, ref: np.ndarray, **kwargs: Any) -> E def test_a_generator_returning_the_reference_designs_is_caught(fake_problem: Any) -> None: - """The attack this metric exists for. + """A lookup table over the training split reads as distance zero. - A lookup table keyed on the condition vector returns the dataset-optimal - design for each scored condition. That is the *definition* of a perfect - `mmd` and a near-zero `iog`, so no amount of care in those metrics can - distinguish it from a model that learned the problem. This one can. + A lookup table keyed on the condition vector returns a training design for + each scored condition. Its distribution score is as good as the data's, so + no amount of care in those metrics can distinguish it from a model that + learned the problem. The distance to the nearest training design can. """ rng = np.random.default_rng(0) ref = rng.random((10, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, ref.copy(), ref) + train = rng.random((20, *fake_problem.design_space.shape)) + ctx = _context(fake_problem, train[:10].copy(), ref, train_designs_fn=lambda: train) - assert METRICS["copy_rate"].fn(ctx) == 1.0 - # And it does indeed post a perfect distribution score, which is the point. - assert METRICS["mmd"].fn(ctx) == pytest.approx(0.0, abs=1e-9) + scores = METRICS["train_distance"].fn(ctx) + assert scores["train_distance"] == pytest.approx(0.0, abs=1e-12) + assert scores["train_distance_ratio"] == pytest.approx(0.0, abs=1e-9) -def test_a_generator_producing_its_own_designs_is_not_flagged(fake_problem: Any) -> None: - """The check must not fire on a model that merely resembles the data.""" +def test_a_generator_producing_its_own_designs_reads_like_real_data(fake_problem: Any) -> None: + """The ratio's reference point: real unseen designs score about one, and so does a model that generates.""" rng = np.random.default_rng(1) ref = rng.random((10, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, rng.random((10, *fake_problem.design_space.shape)), ref) - - assert METRICS["copy_rate"].fn(ctx) == 0.0 - - -def test_copying_the_training_split_is_caught_too(fake_problem: Any) -> None: - """Not only the scored references are copyable -- the whole public dataset is. - - A model that memorized the training split and emits rows of it under - whatever conditions it is given never touches a reference design, so a check - that looked only at those would report it as perfectly novel. - """ - rng = np.random.default_rng(2) - ref = rng.random((6, *fake_problem.design_space.shape)) train = rng.random((20, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, train[:6].copy(), ref, copy_corpus_fn=lambda: train) + ctx = _context(fake_problem, rng.random((10, *fake_problem.design_space.shape)), ref, train_designs_fn=lambda: train) - assert METRICS["copy_rate"].fn(ctx) == 1.0 - assert METRICS["train_distance"].fn(ctx) == pytest.approx(0.0, abs=1e-12) + assert METRICS["train_distance"].fn(ctx)["train_distance_ratio"] == pytest.approx(1.0, abs=0.25) -def test_near_copies_count_as_copies(fake_problem: Any) -> None: +def test_near_copies_stay_near_zero(fake_problem: Any) -> None: """Adding imperceptible noise to a retrieved design must not launder it. - Otherwise the defense is defeated by one line, and the tolerance is what - decides how much perturbation counts as having generated something. + There is no tolerance to game: the distance simply stays tiny, and the ratio + against real held-out designs stays near zero. """ rng = np.random.default_rng(3) ref = rng.random((8, *fake_problem.design_space.shape)) - barely_perturbed = ref + rng.normal(scale=1e-4, size=ref.shape) + train = rng.random((20, *fake_problem.design_space.shape)) + barely_perturbed = train[:8] + rng.normal(scale=1e-4, size=(8, *fake_problem.design_space.shape)) - assert METRICS["copy_rate"].fn(_context(fake_problem, barely_perturbed, ref, copy_tol=0.01)) == 1.0 + scores = METRICS["train_distance"].fn(_context(fake_problem, barely_perturbed, ref, train_designs_fn=lambda: train)) + assert scores["train_distance"] < 1e-3 + assert scores["train_distance_ratio"] < 0.01 def test_memorization_metrics_are_diagnostic_and_cannot_be_ranked_on() -> None: @@ -114,7 +102,6 @@ def test_memorization_metrics_are_diagnostic_and_cannot_be_ranked_on() -> None: direction would have created a new thing to game while closing an old one. """ assert METRICS["train_distance"].higher_is_better is None - assert METRICS["copy_rate"].higher_is_better is None with pytest.raises(ValueError, match="diagnostic"): rank(pd.DataFrame([{"problem_id": "p", "algo_id": "a", "train_distance": 0.5}]), "train_distance") @@ -243,18 +230,6 @@ def test_every_rejected_row_is_reported_at_once() -> None: assert len(excinfo.value.problems_by_row) == 2 -def test_a_memorizing_row_is_published_but_flagged() -> None: - """Published, because hiding it throws away what the board just learned. - - Flagged, because calling it first place would be the board endorsing the - thing it exists to detect. - """ - prepared = prepare_submission(pd.DataFrame([_publishable(copy_rate=0.9)]), EvalSpec(problem_id="beams2d")) - - assert len(prepared) == 1 - assert FLAG_MEMORIZED in prepared["flags"].iloc[0] - - # ---------------------------------------------------------------------- # Eligibility: what gets ranked, as distinct from what gets published # ---------------------------------------------------------------------- @@ -274,8 +249,8 @@ def test_unverified_rows_are_published_but_never_ranked() -> None: def test_flagged_rows_stay_on_the_board_and_out_of_the_ranking() -> None: - """A retrieval system belongs in the table and not in the ordering.""" - frame = pd.DataFrame([_scored(algo_id="honest"), _scored(algo_id="copier", flags=FLAG_MEMORIZED)]) + """A model that ignores the conditions it declares belongs in the table and not in the ordering.""" + frame = pd.DataFrame([_scored(algo_id="honest"), _scored(algo_id="blind", flags=FLAG_IGNORES_CONDITIONS)]) eligible = eligible_rows(frame) @@ -392,28 +367,28 @@ def _sample(self, conditions: Any, n: int) -> Any: def test_the_evaluator_detects_a_lookup_table_end_to_end(fake_problem: Any) -> None: - """The full path: sample, build the corpus, score, and report `copy_rate`. + """The full path: sample, fetch the training split, score, and report `train_distance`. - This is the claim the whole memorization defense rests on, so it is asserted + This is the claim the memorization column rests on, so it is asserted through the evaluator rather than against a hand-built context -- a metric that works but is never fed would pass every other test in this file. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "copy_rate")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "train_distance")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) + train = np.asarray(fake_problem.dataset["train"]["optimal_design"]) evaluator = _evaluator(fake_problem, spec, ref, conditions) - lookup_table = _generator(fake_problem, lambda _conditions, _n: ref) + lookup_table = _generator(fake_problem, lambda _conditions, n: train[np.arange(n) % len(train)]) row = evaluator.score(lookup_table) - assert row["copy_rate"] == 1.0 - assert row["mmd"] == pytest.approx(0.0, abs=1e-9) - assert FLAG_MEMORIZED in integrity_flags(row, spec) + assert row["train_distance"] == pytest.approx(0.0, abs=1e-12) + assert row["train_distance_ratio"] == pytest.approx(0.0, abs=1e-9) def test_the_evaluator_leaves_an_honest_model_unflagged(fake_problem: Any) -> None: """The same path, for a model that generates rather than retrieves.""" - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "copy_rate", "cond_sens")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "train_distance", "cond_sens")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) rng = np.random.default_rng(7) @@ -429,9 +404,9 @@ def test_the_evaluator_leaves_an_honest_model_unflagged(fake_problem: Any) -> No ) row = evaluator.score(honest) - assert row["copy_rate"] == 0.0 + assert row["train_distance"] > 0.0 assert row["cond_sens"] > 0 - assert not [flag for flag in integrity_flags(row, spec) if flag != FLAG_UNVERIFIED] + assert not [flag for flag in integrity_flags(row) if flag != FLAG_UNVERIFIED] def test_the_evaluator_catches_a_conditional_model_that_ignores_conditions(fake_problem: Any) -> None: @@ -472,26 +447,26 @@ def _pure_noise(_conditions: Any, n: int) -> Any: assert row["cond_sens"] == pytest.approx(0.0, abs=1e-12) -def test_the_copy_corpus_is_drawn_once_for_a_whole_sweep(fake_problem: Any) -> None: - """Every model in a sweep is checked against the same corpus, fetched once. +def test_the_training_split_is_fetched_once_for_a_whole_sweep(fake_problem: Any) -> None: + """Every model in a sweep is checked against the same training designs, fetched once. - Both halves matter: a per-model corpus would be slow, and a corpus that - varied between models would make their `copy_rate` values incomparable. + Both halves matter: a per-model fetch would be slow, and a split that varied + between models would make their `train_distance` values incomparable. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("copy_rate",)) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("train_distance",)) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) evaluator = _evaluator(fake_problem, spec, ref, conditions) draws = 0 - original = evaluator._draw_copy_corpus + original = evaluator._load_train_designs def _counting_draw() -> Any: nonlocal draws draws += 1 return original() - evaluator._draw_copy_corpus = _counting_draw # type: ignore[method-assign] + evaluator._load_train_designs = _counting_draw # type: ignore[method-assign] for _ in range(3): evaluator.score(_generator(fake_problem, lambda _c, n: np.zeros((n, *fake_problem.design_space.shape)))) diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py index 400fd2f5..7893f425 100644 --- a/tests/test_metrics_suite.py +++ b/tests/test_metrics_suite.py @@ -81,16 +81,19 @@ def test_volume_error_reads_the_material_fraction_off_the_design(fake_problem: A # ------------------------------------------------------------- memorization -def test_train_distance_and_copy_rate_ask_different_questions(fake_problem: Any, sets: Any) -> None: +def test_train_distance_reads_zero_for_copies_and_one_for_data_like_designs(fake_problem: Any, sets: Any) -> None: + """The ratio's scale comes from the reference designs, so a real held-out design scores about one.""" gen, ref = sets train = np.random.default_rng(3).random((20, *fake_problem.design_space.shape)) - copies_train = _ctx(fake_problem, train[:12], ref, copy_corpus_fn=lambda: train) - assert METRICS["train_distance"].fn(copies_train) == pytest.approx(0.0) - assert METRICS["copy_rate"].fn(copies_train) == pytest.approx(1.0) - copies_reference = _ctx(fake_problem, ref.copy(), ref, copy_corpus_fn=lambda: train) - assert METRICS["train_distance"].fn(copies_reference) > 0.1, "the withheld optima are not training designs" - assert METRICS["copy_rate"].fn(copies_reference) == pytest.approx(1.0), "but they are in the copyable corpus" - assert np.isnan(METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref))) + copies = METRICS["train_distance"].fn(_ctx(fake_problem, train[:12], ref, train_designs_fn=lambda: train)) + assert copies["train_distance"] == pytest.approx(0.0) + assert copies["train_distance_ratio"] == pytest.approx(0.0) + fresh = METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref, train_designs_fn=lambda: train)) + assert fresh["train_distance"] > 0.1 + assert fresh["train_distance_ratio"] == pytest.approx(1.0, abs=0.25), "as far from training as real unseen designs" + without_split = METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref)) + assert np.isnan(without_split["train_distance"]) + assert np.isnan(without_split["train_distance_ratio"]) # -------------------------------------------------------------- performance diff --git a/tests/test_specs.py b/tests/test_specs.py index 463a1092..b2bf4974 100644 --- a/tests/test_specs.py +++ b/tests/test_specs.py @@ -245,17 +245,11 @@ def _capture(self: EvalSpec, problem: Any) -> EvalSpec: volume_condition="volfrac", volfrac_tol=0.05, required_seeds=(1, 2, 3, 4), - copy_tol=0.02, - max_copy_rate=0.25, - copy_corpus_size=64, ) assert captured["volume_condition"] == "volfrac" assert captured["volfrac_tol"] == 0.05 assert captured["required_seeds"] == (1, 2, 3, 4) - assert captured["copy_tol"] == 0.02 - assert captured["max_copy_rate"] == 0.25 - assert captured["copy_corpus_size"] == 64 def test_freeze_spec_defaults_track_the_dataclass() -> None: @@ -270,5 +264,5 @@ def test_freeze_spec_defaults_track_the_dataclass() -> None: from engiopt.evaluation import spec as spec_mod parameters = inspect.signature(spec_mod.freeze_spec).parameters - for field_name in ("metrics", "volfrac_tol", "required_seeds", "copy_tol", "max_copy_rate", "copy_corpus_size"): + for field_name in ("metrics", "volfrac_tol", "required_seeds"): assert parameters[field_name].default == getattr(EvalSpec, field_name), field_name diff --git a/tests/test_verification.py b/tests/test_verification.py index 8359e233..a289b512 100644 --- a/tests/test_verification.py +++ b/tests/test_verification.py @@ -17,7 +17,6 @@ from engiopt.evaluation import verify as verify_mod from engiopt.evaluation.spec import EvalSpec -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.verify import verify_board from engiopt.evaluation.verify import verify_row @@ -38,7 +37,7 @@ class _FakeEvaluator: def __init__(self, scores: dict[str, Any] | None = None): self.spec = EvalSpec(problem_id="beams2d") - self.scores = scores or {"mmd": 0.04, "copy_rate": 0.0} + self.scores = scores or {"mmd": 0.04, "train_distance": 0.5} self.calls = 0 def score(self, generator: Any, *, include_expensive: bool = False) -> dict[str, Any]: @@ -139,7 +138,7 @@ def test_the_verified_row_carries_the_runners_numbers(loads_fake_generator: None fudging. Re-scoring sidesteps that: the runner's number is the number, and the submitted one becomes a claim reported as corroborated or not. """ - evaluator = _FakeEvaluator({"mmd": 0.42, "copy_rate": 0.0}) + evaluator = _FakeEvaluator({"mmd": 0.42, "train_distance": 0.5}) result = verify_row(_row(mmd=0.001), verifier="ideallab-ci", evaluator=evaluator) @@ -174,16 +173,6 @@ def test_a_reproduced_claim_says_so(loads_fake_generator: None) -> None: assert "corroborated" in result.detail -def test_verification_re_derives_the_flags_rather_than_trusting_them(loads_fake_generator: None) -> None: - """A submitter who cleared their own `memorized` flag must not keep it cleared.""" - evaluator = _FakeEvaluator({"mmd": 0.04, "copy_rate": 0.95}) - - result = verify_row(_row(flags=""), verifier="ci", evaluator=evaluator) - - assert result.row is not None - assert FLAG_MEMORIZED in result.row["flags"] - - # ---------------------------------------------------------------------- # Fetching the package the row actually names # ---------------------------------------------------------------------- From 0af3a34e9f8b807e275de178e5f2998d79a3dd2b Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Mon, 28 Sep 2026 17:55:34 +0200 Subject: [PATCH 11/18] Stop passing the spec to push_to_hub, which no longer reads it The memorization change removed the spec from the publishing path, since no flag reads it any more, and this call site was missed. mypy caught it. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluate.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index 45cbb2cd..dbb81793 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -377,7 +377,7 @@ def main(args: Args) -> int: def _publish(args: Args, evaluator: Evaluator, board: pd.DataFrame) -> None: """Push the results wherever the flags asked, then print the ranking comparison.""" if args.push_to: - merged = push_to_hub(board, args.push_to, eval_spec=evaluator.spec) + merged = push_to_hub(board, args.push_to) print(f"Published {len(board)} row(s) to {args.push_to}; board now holds {len(merged)} rows.") print( "These rows are unverified. They will not be ranked until a runner re-fetches the " From ab2ef942a30930baad3dd62582cb711697ded9fb Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Mon, 28 Sep 2026 17:56:00 +0200 Subject: [PATCH 12/18] Drop the evaluator argument _publish stopped using Left over from the previous fix; ruff flagged it as unused. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluate.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index dbb81793..8a54f97a 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -370,11 +370,11 @@ def main(args: Args) -> int: print(f"\n{board.to_string(index=False)}\n") print(f"Wrote {len(board)} rows to {destination}") - _publish(args, evaluator, board) + _publish(args, board) return 0 -def _publish(args: Args, evaluator: Evaluator, board: pd.DataFrame) -> None: +def _publish(args: Args, board: pd.DataFrame) -> None: """Push the results wherever the flags asked, then print the ranking comparison.""" if args.push_to: merged = push_to_hub(board, args.push_to) From 68b32b990ab2187cb20f9397dc0aabde7695dec6 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Tue, 29 Sep 2026 10:38:28 +0200 Subject: [PATCH 13/18] Measure settling against the optimum, and size the kernel from the training data `calls_to_settle` used a band of five percent of each design's own reach, so a start of 1e9 that dropped to 1e3 in one call read as settled after one call while a VQGAN start of 2.4 took 32. Matthew caught it. The column is now `calls_to_near_optimum`: calls until the gap stays within five percent of the reference optimum's objective for that design's conditions, with one more than the budget for a design that never arrives. The physics pass keeps the reference objective per design to supply the scale. `first_call_gain` is gone; it was a fraction of the guess's own improvement and had the same flaw with no optimum-relative form. The kernel bandwidth behind mmd, vendi and dpp is no longer a pinned 10.0. It defaults to the median pairwise distance of the training designs, in whatever space the context holds, resolved once per context and recorded in every row as `kernel_sigma`. A spec or a Board may still pin a number; the v2 specs pin none. On beams2d the pinned value had every model near ten effective designs; the data-sized kernel separates the collapsed GAN at 1.03 from the rest at five to seven. The two changes share files, so they ship together. Co-Authored-By: Claude Fable 5.1 --- CONTRIBUTING_A_MODEL.md | 4 +- engiopt/evaluation/board.py | 14 +++-- engiopt/evaluation/context.py | 27 +++++++++- engiopt/evaluation/evaluator.py | 2 + engiopt/evaluation/metrics/builtin.py | 75 +++++++++++--------------- engiopt/evaluation/spec.py | 8 +-- engiopt/specs/beams2d/v2.json | 7 ++- engiopt/specs/heatconduction2d/v2.json | 7 ++- engiopt/specs/photonics2d/v2.json | 7 ++- engiopt/specs/thermoelastic2d/v2.json | 7 ++- tests/test_board.py | 2 +- tests/test_evaluation.py | 3 +- tests/test_metrics_suite.py | 41 ++++++++++---- 13 files changed, 116 insertions(+), 88 deletions(-) diff --git a/CONTRIBUTING_A_MODEL.md b/CONTRIBUTING_A_MODEL.md index 685e8286..0f0e78e5 100644 --- a/CONTRIBUTING_A_MODEL.md +++ b/CONTRIBUTING_A_MODEL.md @@ -181,8 +181,8 @@ ones (`mmd`, `coverage`, `vendi`, `dpp`), the per-condition ones integrity checks (`train_distance`, `cond_sens`), and the cost columns. `--include-expensive` adds the columns that re-optimize from each generated design — the optimality gaps `iog`, `cog`, `fog` and the -call-budget metrics `calls_to_settle`, `gap_after_calls`, -`reaches_reference_rate`, `first_call_gain` — which run the optimizer and are +call-budget metrics `calls_to_near_optimum`, `gap_after_calls` and +`reaches_reference_rate` — which run the optimizer and are slow; leave them off while iterating. `python -m engiopt.evaluate --list-metrics` prints the question each column answers. diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 31e4cd7a..26f02f47 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -23,7 +23,6 @@ import numpy as np import pandas as pd -from engiopt import metrics as metrics_mod from engiopt.evaluation.context import EvaluationContext from engiopt.evaluation.registry import MetricRegistry from engiopt.evaluation.registry import METRICS @@ -64,9 +63,10 @@ class Board: reference: The withheld optimal designs the models are scored against. train: The training designs, for the memorization metrics. Optional. sigma: Kernel bandwidth for the distribution and diversity metrics. None, - the default, uses the median pairwise distance of the reference designs - in whichever space is being scored, so the kernel is neither saturated - nor empty on a problem it has never seen. A published spec pins a value. + the default, uses the median pairwise distance of the training designs + in whichever space is being scored, or of the reference designs when no + training split was given, so the kernel is neither saturated nor empty + on a problem it has never seen. frame: The most recent `evaluate` result. designs: The designs behind each row of `frame`, when the board sampled them itself. """ @@ -113,8 +113,6 @@ def evaluate( if space not in SPACES: raise KeyError(f"Unknown space {space!r}. Registered: {', '.join(SPACES)}") project = SPACES[space](reference, width) - reference_codes = project(reference) - sigma = self.sigma if self.sigma is not None else metrics_mod.compute_median_sigma(reference_codes) def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: ctx = EvaluationContext( @@ -122,9 +120,9 @@ def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: problem_id=getattr(self.problem, "problem_id", type(self.problem).__name__), gen_designs=project(np.asarray(generated)), ref_designs=project(np.asarray(against)), - sigma=sigma, + sigma=self.sigma, aggregation=aggregation, - train_designs_fn=(lambda: np.asarray(self.train)) if self.train is not None else None, + train_designs_fn=(lambda: project(np.asarray(self.train))) if self.train is not None else None, ) row: dict[str, float] = {} for spec in chosen: diff --git a/engiopt/evaluation/context.py b/engiopt/evaluation/context.py index 46f2c902..824305d7 100644 --- a/engiopt/evaluation/context.py +++ b/engiopt/evaluation/context.py @@ -66,6 +66,11 @@ class OptimizationResults: trajectories: list[npt.NDArray[Any]] = field(default_factory=list) """Per design, the optimality gap at every optimizer call of its re-optimization.""" + reference_objectives: list[float] = field(default_factory=list) + """Per design, the reference optimum's objective for its conditions, scalarized like the gaps. + + The scale a gap is read against: five percent of this is what "near optimal" means. + """ iog: list[float] = field(default_factory=list) """Initial optimality gap: how far each generated design starts from the reference optimum.""" cog: list[float] = field(default_factory=list) @@ -84,7 +89,8 @@ class EvaluationContext: gen_designs: Generated designs, `(n_samples, *design_shape)`. ref_designs: Reference (dataset-optimal) designs for the same conditions. conditions: The conditions each design was asked to satisfy. - sigma: Gaussian-kernel bandwidth for MMD and DPP. + sigma: Gaussian-kernel bandwidth for the kernel metrics, or None to use + the median pairwise distance of the training designs; see `kernel_sigma`. volfrac_tol: Tolerance used by the volume-fraction violation check. volume_condition: Name of the condition holding the volume-fraction budget a design must hit, when the problem has one. Declared by the @@ -112,7 +118,7 @@ class EvaluationContext: gen_designs: npt.NDArray[Any] ref_designs: npt.NDArray[Any] conditions: Dataset | None = None - sigma: float = 10.0 + sigma: float | None = None volfrac_tol: float = 0.01 volume_condition: str | None = None sample_seconds: float | None = None @@ -191,6 +197,22 @@ def train_designs(self) -> npt.NDArray[Any] | None: flat = train.reshape(len(train), -1) return flat if flat.shape[1] == self.gen_flat.shape[1] else None + @cached_property + def kernel_sigma(self) -> float: + """The bandwidth the kernel metrics use. + + The value the spec pinned, if it pinned one. Otherwise the median + pairwise distance of the training designs, in whatever space this + context holds, so the kernel is neither saturated nor empty on any + problem; the reference designs stand in when there is no training split. + Subsampled with a fixed seed, so the value is reproducible, and recorded + in every published row. + """ + if self.sigma is not None: + return float(self.sigma) + basis = self.train_designs if self.train_designs is not None and len(self.train_designs) > 1 else self.ref_flat + return metrics_mod.compute_median_sigma(basis) + def nearest_train_distance(self, designs: npt.NDArray[Any]) -> npt.NDArray[Any]: """Per-design RMS distance from each of `designs` to the closest training design. @@ -345,6 +367,7 @@ def optimization(self) -> OptimizationResults: results.cog.append(sum(self.scalarize_gap(step_gap, i) for step_gap in gaps)) results.fog.append(self.scalarize_gap(gaps[-1], i)) results.trajectories.append(np.asarray([self.scalarize_gap(step_gap, i) for step_gap in gaps], dtype=float)) + results.reference_objectives.append(self.scalarize_gap(np.asarray(reference_optimum), i)) return results @cached_property diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 0cbe3f8e..bb574cca 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -54,6 +54,7 @@ "spec_version", "n_samples", "sample_seconds", + "kernel_sigma", "checkpoint_repo", "checkpoint_path", "checkpoint_revision", @@ -317,6 +318,7 @@ def _provenance(self, generator: Generator, ctx: EvaluationContext) -> dict[str, "n_samples": ctx.n_samples, "aggregation": ctx.aggregation, "sample_seconds": ctx.sample_seconds, + "kernel_sigma": ctx.kernel_sigma, "checkpoint_repo": getattr(generator, "checkpoint_repo", None), "checkpoint_path": getattr(generator, "checkpoint_path", None), "checkpoint_revision": getattr(generator, "checkpoint_revision", None), diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 7564e77c..579e6f53 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -35,7 +35,7 @@ ) def mmd(ctx: EvaluationContext) -> float: """Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.""" - return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.sigma)) + return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.kernel_sigma)) # ---------------------------------------------------------------------- @@ -62,7 +62,7 @@ def dpp(ctx: EvaluationContext) -> float: A determinant still rises when a collapsed set is jittered, so it rewards noise as diversity. `vendi` does not, and is the column to prefer. """ - return float(metrics_mod.dpp_geometric_mean(ctx.gen_flat, sigma=ctx.sigma)) + return float(metrics_mod.dpp_geometric_mean(ctx.gen_flat, sigma=ctx.kernel_sigma)) # ---------------------------------------------------------------------- @@ -317,7 +317,7 @@ def vendi(ctx: EvaluationContext) -> float: jittered, which is why it is the diversity column to prefer. Friedman & Dieng (2023), The Vendi Score. """ - return float(metrics_mod.vendi_score(ctx.gen_flat, sigma=ctx.sigma)) + return float(metrics_mod.vendi_score(ctx.gen_flat, sigma=ctx.kernel_sigma)) # ---------------------------------------------------------------------- @@ -330,39 +330,48 @@ def vendi(ctx: EvaluationContext) -> float: # below price a defect in the currency an engineer pays, which is optimizer # calls, and read the same trajectory `iog`, `cog` and `fog` summarize. -SETTLE_BAND = 0.05 -"""Fraction of a design's own achievable improvement within which it counts as settled.""" +NEAR_OPTIMUM_FRACTION = 0.05 +"""A gap within this fraction of the reference optimum's objective counts as near optimal.""" GAP_AFTER_CALLS = (1, 2, 5, 10) """Call budgets to report the remaining gap at: is a defect gone immediately, or not?""" -def _calls_to_settle(path: np.ndarray, band: float = SETTLE_BAND) -> float: - """Calls after which `path` stays within `band` of its final value (last exit, not first touch).""" - if path.size < 2: # noqa: PLR2004 - a one-step path has no convergence to measure - return 0.0 - final = float(path[-1]) - reach = max(abs(float(path[0]) - final), abs(final), 1e-12) - outside = np.flatnonzero(np.abs(path - final) > band * reach) +def _calls_to_near_optimum(path: np.ndarray, reference_objective: float) -> float: + """Calls after which the gap stays within five percent of the reference optimum's objective. + + Last exit rather than first touch, so a path that dips into the band and + leaves again is not credited. Zero means the design was near optimal from + its first call. A path still outside the band at its last call never got + there and counts as one more than the budget, so a design that never + arrives ranks worse than one that is merely slow. + """ + band = NEAR_OPTIMUM_FRACTION * max(abs(reference_objective), 1e-12) + if float(path[-1]) > band: + return float(path.size + 1) + outside = np.flatnonzero(path > band) return float(outside[-1] + 1) if outside.size else 0.0 @register_metric( - "calls_to_settle", + "calls_to_near_optimum", family="performance", cost="expensive", higher_is_better=False, ) -def calls_to_settle(ctx: EvaluationContext) -> float: - """Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. +def calls_to_near_optimum(ctx: EvaluationContext) -> float: + """Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. - Settling time in the control-theory sense: last exit from the band, so a - trajectory that dips in and leaves again is not credited. Zero means the - design started already settled. Each call is one simulate plus one - sensitivity evaluation. + The band is set by the optimum for the design's own conditions, not by the + design's starting point, so a design that starts absurdly far away is not + credited with settling merely because its first call removed most of the + absurdity. Each call is one simulate plus one sensitivity evaluation. """ - paths = ctx.optimization.trajectories - return ctx.reduce([_calls_to_settle(path) for path in paths]) if paths else float("nan") + results = ctx.optimization + if not results.trajectories: + return float("nan") + calls = [_calls_to_near_optimum(path, scale) for path, scale in zip(results.trajectories, results.reference_objectives)] + return ctx.reduce(calls) @register_metric( @@ -407,30 +416,6 @@ def reaches_reference_rate(ctx: EvaluationContext) -> float: return float(np.mean([bool(np.any(np.asarray(path) <= 0.0)) for path in paths])) -@register_metric( - "first_call_gain", - family="performance", - cost="expensive", - higher_is_better=True, -) -def first_call_gain(ctx: EvaluationContext) -> float: - """Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. - - Near 1 means the design's defects clear immediately and the rest is - refinement; near 0 means the warm start bought nothing. Designs with nothing - to gain are skipped, since the fraction is undefined there. - """ - gains = [] - for path in ctx.optimization.trajectories: - if path.size < 2: # noqa: PLR2004 - needs a first step to have a first-step gain - continue - achievable = float(path[0]) - float(path[-1]) - if abs(achievable) < 1e-12: # noqa: PLR2004 - nothing to improve, so no fraction of it exists - continue - gains.append((float(path[0]) - float(path[1])) / achievable) - return ctx.reduce(gains) if gains else float("nan") - - # ---------------------------------------------------------------------- # Cost: what does the model cost to run, and what did it cost to make? # ---------------------------------------------------------------------- diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index 7acafc3e..6aa8b910 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -82,7 +82,9 @@ class EvalSpec: n_samples: Number of conditions each model is scored on. condition_seed: Seed used to draw the test conditions. metrics: Metric names to compute, in leaderboard column order. - sigma: Gaussian-kernel bandwidth for MMD and DPP. + sigma: Gaussian-kernel bandwidth for the kernel metrics. None, the + default, means the median pairwise distance of the training designs, + resolved at evaluation time and recorded in every row. volfrac_tol: Tolerance for the volume-fraction violation check. volume_condition: Name of the condition holding the volume-fraction budget a design must hit, e.g. beams2d's `volfrac`. Feasibility is @@ -128,7 +130,7 @@ class EvalSpec: n_samples: int = 50 condition_seed: int = 1 metrics: tuple[str, ...] = ("mmd", "dpp", "novelty", "cond_sens", "viol", "iog", "cog", "fog") - sigma: float = 10.0 + sigma: float | None = None aggregation: Literal["mean", "median"] = "mean" """How per-design metrics (`iog`, `cog`, `fog`, distances) collapse to one number. @@ -444,7 +446,7 @@ def freeze_spec( n_samples: int = 50, condition_seed: int = 1, metrics: tuple[str, ...] = EvalSpec.metrics, - sigma: float = 10.0, + sigma: float | None = None, volume_condition: str | None = None, volfrac_tol: float = EvalSpec.volfrac_tol, objective_weights: tuple[float, ...] | None = None, diff --git a/engiopt/specs/beams2d/v2.json b/engiopt/specs/beams2d/v2.json index 7278e3d5..5f51fbaa 100644 --- a/engiopt/specs/beams2d/v2.json +++ b/engiopt/specs/beams2d/v2.json @@ -20,10 +20,9 @@ "iog", "cog", "fog", - "calls_to_settle", + "calls_to_near_optimum", "gap_after_calls", - "reaches_reference_rate", - "first_call_gain" + "reaches_reference_rate" ], "n_samples": 50, "notes": "Feasibility = EngiBench constraints plus the volfrac budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", @@ -41,7 +40,7 @@ 2, 3 ], - "sigma": 10.0, + "sigma": null, "version": "v2", "volfrac_tol": 0.01, "volume_condition": "volfrac", diff --git a/engiopt/specs/heatconduction2d/v2.json b/engiopt/specs/heatconduction2d/v2.json index e423b573..bdfed215 100644 --- a/engiopt/specs/heatconduction2d/v2.json +++ b/engiopt/specs/heatconduction2d/v2.json @@ -20,10 +20,9 @@ "iog", "cog", "fog", - "calls_to_settle", + "calls_to_near_optimum", "gap_after_calls", - "reaches_reference_rate", - "first_call_gain" + "reaches_reference_rate" ], "n_samples": 50, "notes": "Feasibility = EngiBench constraints plus the volume budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", @@ -39,7 +38,7 @@ 2, 3 ], - "sigma": 10.0, + "sigma": null, "version": "v2", "volfrac_tol": 0.01, "volume_condition": "volume", diff --git a/engiopt/specs/photonics2d/v2.json b/engiopt/specs/photonics2d/v2.json index cfd317e2..a0fe2064 100644 --- a/engiopt/specs/photonics2d/v2.json +++ b/engiopt/specs/photonics2d/v2.json @@ -20,10 +20,9 @@ "iog", "cog", "fog", - "calls_to_settle", + "calls_to_near_optimum", "gap_after_calls", - "reaches_reference_rate", - "first_call_gain" + "reaches_reference_rate" ], "n_samples": 50, "notes": "No volume budget: feasibility is EngiBench's declared constraints alone. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", @@ -43,7 +42,7 @@ 2, 3 ], - "sigma": 10.0, + "sigma": null, "version": "v2", "volfrac_tol": 0.01, "volume_condition": null, diff --git a/engiopt/specs/thermoelastic2d/v2.json b/engiopt/specs/thermoelastic2d/v2.json index 30281b66..977cbd97 100644 --- a/engiopt/specs/thermoelastic2d/v2.json +++ b/engiopt/specs/thermoelastic2d/v2.json @@ -20,10 +20,9 @@ "iog", "cog", "fog", - "calls_to_settle", + "calls_to_near_optimum", "gap_after_calls", - "reaches_reference_rate", - "first_call_gain" + "reaches_reference_rate" ], "n_samples": 50, "notes": "Objectives combined per-sample by the `weight` condition. Feasibility = EngiBench constraints plus the volume_fraction_target budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", @@ -44,7 +43,7 @@ 2, 3 ], - "sigma": 10.0, + "sigma": null, "version": "v2", "volfrac_tol": 0.01, "volume_condition": "volume_fraction_target", diff --git a/tests/test_board.py b/tests/test_board.py index 17d310ba..043e5619 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -84,7 +84,7 @@ def test_aggregation_is_a_policy_on_the_context(fake_problem: Any) -> None: def test_the_default_bandwidth_lets_diversity_metrics_see_anything(fake_problem: Any, designs: Any) -> None: - """At a fixed sigma=10 every design looks identical to every other; the median heuristic does not.""" + """At a fixed sigma=10 every design looks identical to every other; the training-data median does not.""" models, ref, _ = designs collapsed = np.repeat(models["honest"][:1], len(ref), axis=0) frame = Board(fake_problem, reference=ref).evaluate({"collapsed": collapsed, "spread": models["honest"]}) diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index ffed37e2..17dafcb9 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -111,10 +111,9 @@ def test_builtin_metrics_declare_their_cost() -> None: "iog", "cog", "fog", - "calls_to_settle", + "calls_to_near_optimum", "gap_after_calls", "reaches_reference_rate", - "first_call_gain", } diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py index 7893f425..8de0d3d6 100644 --- a/tests/test_metrics_suite.py +++ b/tests/test_metrics_suite.py @@ -96,6 +96,23 @@ def test_train_distance_reads_zero_for_copies_and_one_for_data_like_designs(fake assert np.isnan(without_split["train_distance_ratio"]) +def test_the_bandwidth_defaults_to_the_training_designs_median(fake_problem: Any, sets: Any) -> None: + """No number anywhere: the kernel scale comes from the training split, or the reference set without one.""" + from engiopt import metrics as metrics_mod + + gen, ref = sets + train = np.random.default_rng(5).random((30, *fake_problem.design_space.shape)) + with_train = _ctx(fake_problem, gen, ref, train_designs_fn=lambda: train) + assert with_train.kernel_sigma == pytest.approx(metrics_mod.compute_median_sigma(train.reshape(30, -1))) + without = _ctx(fake_problem, gen, ref) + assert without.kernel_sigma == pytest.approx(metrics_mod.compute_median_sigma(ref.reshape(len(ref), -1))) + assert _ctx(fake_problem, gen, ref, sigma=3.0).kernel_sigma == 3.0, "a pinned value wins" + + +def test_the_v2_specs_pin_no_bandwidth() -> None: + assert EvalSpec.load("beams2d/v2").sigma is None + + # -------------------------------------------------------------- performance @@ -106,13 +123,23 @@ def trajectories(fake_problem: Any, sets: Any) -> EvaluationContext: ctx = _ctx(fake_problem, gen[:2], ref[:2]) reaches = np.array([5.0, 4.0, 3.0, 0.0, 0.0]) never = np.array([5.0, 4.0, 3.0, 2.0, 1.0]) - ctx.__dict__["optimization"] = OptimizationResults(trajectories=[reaches, never]) + # Reference objective 10 for both, so "near optimal" means a gap of 0.5 or less. + ctx.__dict__["optimization"] = OptimizationResults(trajectories=[reaches, never], reference_objectives=[10.0, 10.0]) return ctx -def test_calls_to_settle_is_last_exit_from_the_band(trajectories: EvaluationContext) -> None: - # reaches: settles after call 3 (band 0.25 around 0); never: after call 4 (band 0.2 around 1). - assert METRICS["calls_to_settle"].fn(trajectories) == pytest.approx(3.5) +def test_calls_to_near_optimum_is_set_by_the_optimum_not_the_start(trajectories: EvaluationContext) -> None: + # reaches: last outside the 0.5 band at call 3; never: still outside at its last call, so budget + 1 = 6. + assert METRICS["calls_to_near_optimum"].fn(trajectories) == pytest.approx((3 + 6) / 2) + + +def test_an_absurd_start_is_not_credited_with_settling(fake_problem: Any, sets: Any) -> None: + """A gap of 1e9 that drops to 1e3 in one call has not arrived; the old relative band said it had.""" + gen, ref = sets + ctx = _ctx(fake_problem, gen[:1], ref[:1]) + path = np.array([1e9, 1e3, 1e2, 10.0, 3.0]) + ctx.__dict__["optimization"] = OptimizationResults(trajectories=[path], reference_objectives=[10.0]) + assert METRICS["calls_to_near_optimum"].fn(ctx) == pytest.approx(6.0), "never within 0.5 of the optimum" def test_gap_after_calls_carries_a_short_path_forward(trajectories: EvaluationContext) -> None: @@ -127,13 +154,9 @@ def test_reaches_reference_rate_is_a_rate_not_a_censored_count(trajectories: Eva assert METRICS["reaches_reference_rate"].fn(trajectories) == pytest.approx(0.5) -def test_first_call_gain_is_the_first_step_over_the_whole_improvement(trajectories: EvaluationContext) -> None: - assert METRICS["first_call_gain"].fn(trajectories) == pytest.approx((0.2 + 0.25) / 2) - - def test_trajectory_metrics_respect_the_aggregation_policy(trajectories: EvaluationContext) -> None: trajectories.aggregation = "median" - assert METRICS["calls_to_settle"].fn(trajectories) == pytest.approx(3.5), "median of two equals their mean" + assert METRICS["calls_to_near_optimum"].fn(trajectories) == pytest.approx(4.5), "median of two equals their mean" # --------------------------------------------------------------------- cost From 59c198dd5f29c8b0ece2a6e563148d181fcd2c06 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Tue, 29 Sep 2026 10:38:30 +0200 Subject: [PATCH 14/18] Explain the evaluation context before asking the reader to write a metric The register-a-metric cell used attributes nobody had introduced. It now names what the context holds, with shapes, before the example, and the example reads each step in a comment. Re-run with the new settle column and the data-sized kernel; the interpretation reflects both. Co-Authored-By: Claude Fable 5.1 --- example_metrics_suite.ipynb | 590 ++++++++++++++++++------------------ 1 file changed, 291 insertions(+), 299 deletions(-) diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 3c418e66..7aae5ba6 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "cfd0f310", + "id": "e32d8ced", "metadata": {}, "source": [ "# Using the metrics suite, on beams2d\n", @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "45733beb", + "id": "ceeb955f", "metadata": {}, "source": [ "## 1. What metrics exist, and what does each one measure?" @@ -36,13 +36,13 @@ { "cell_type": "code", "execution_count": 1, - "id": "74b137ca", + "id": "dac2ff45", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:28:02.465639Z", - "iopub.status.busy": "2026-09-28T15:28:02.465228Z", - "iopub.status.idle": "2026-09-28T15:28:05.075433Z", - "shell.execute_reply": "2026-09-28T15:28:05.075194Z" + "iopub.execute_input": "2026-09-28T16:13:40.116290Z", + "iopub.status.busy": "2026-09-28T16:13:40.115944Z", + "iopub.status.idle": "2026-09-28T16:13:42.773684Z", + "shell.execute_reply": "2026-09-28T16:13:42.773430Z" } }, "outputs": [ @@ -169,8 +169,8 @@ " cheap\n", " \n", " \n", - " calls_to_settle\n", - " Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle.\n", + " calls_to_near_optimum\n", + " Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there.\n", " performance\n", " lower is better\n", " expensive\n", @@ -190,13 +190,6 @@ " expensive\n", " \n", " \n", - " first_call_gain\n", - " Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step.\n", - " performance\n", - " higher is better\n", - " expensive\n", - " \n", - " \n", " generation_seconds\n", " Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.\n", " cost\n", @@ -222,26 +215,25 @@ "" ], "text/plain": [ - " question \\\n", - "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", - "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", - "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", - "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", - "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", - "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", - "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", - "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", - "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", - "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", - "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", - "calls_to_settle Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. \n", - "gap_after_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", - "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", - "first_call_gain Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. \n", - "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", - "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", - "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "calls_to_near_optimum Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. \n", + "gap_after_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", "\n", " family direction cost \n", "mmd distribution lower is better cheap \n", @@ -256,10 +248,9 @@ "volume_error conditions lower is better cheap \n", "coverage distribution higher is better cheap \n", "vendi diversity higher is better cheap \n", - "calls_to_settle performance lower is better expensive \n", + "calls_to_near_optimum performance lower is better expensive \n", "gap_after_calls performance lower is better expensive \n", "reaches_reference_rate performance higher is better expensive \n", - "first_call_gain performance higher is better expensive \n", "generation_seconds cost lower is better cheap \n", "n_parameters cost lower is better cheap \n", "train_minutes cost lower is better cheap " @@ -289,7 +280,7 @@ }, { "cell_type": "markdown", - "id": "8c614e35", + "id": "8ee178ea", "metadata": {}, "source": [ "Each row names the quantity, says how it is computed, and says what the two ends of\n", @@ -299,7 +290,7 @@ }, { "cell_type": "markdown", - "id": "55b739c4", + "id": "a597ddfa", "metadata": {}, "source": [ "## 2. The problem, and the evaluation spec" @@ -307,12 +298,14 @@ }, { "cell_type": "markdown", - "id": "4118d701", + "id": "fda28edc", "metadata": {}, "source": [ "Every published number on beams2d is computed under one frozen spec. The spec fixes\n", - "which fifty test conditions are scored, the reference optimum for each, the kernel\n", - "bandwidth, and how per-design values collapse to one number. Two\n", + "which fifty test conditions are scored, the reference optimum for each, and how\n", + "per-design values collapse to one number. The kernel bandwidth behind `mmd`, `vendi`\n", + "and `dpp` is the median pairwise distance of the training designs unless the spec pins\n", + "a value, and the resolved number is recorded in every row. Two\n", "models scored under the same spec are comparable; two scored under different specs\n", "are not, and every row records which spec produced it.\n", "\n", @@ -332,13 +325,13 @@ { "cell_type": "code", "execution_count": 2, - "id": "31940087", + "id": "93f920cc", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:28:05.077884Z", - "iopub.status.busy": "2026-09-28T15:28:05.077720Z", - "iopub.status.idle": "2026-09-28T15:28:06.757070Z", - "shell.execute_reply": "2026-09-28T15:28:06.756789Z" + "iopub.execute_input": "2026-09-28T16:13:42.776190Z", + "iopub.status.busy": "2026-09-28T16:13:42.775973Z", + "iopub.status.idle": "2026-09-28T16:13:44.372558Z", + "shell.execute_reply": "2026-09-28T16:13:44.372303Z" } }, "outputs": [ @@ -348,7 +341,7 @@ "text": [ "conditions scored: 12 | design shape: (50, 100)\n", "condition keys: ('volfrac', 'rmin', 'forcedist', 'overhang_constraint')\n", - "kernel sigma: 10.0 | aggregation: median\n" + "kernel bandwidth: median distance of the training designs | aggregation: median\n" ] } ], @@ -365,12 +358,12 @@ "REFERENCE = evaluator.resolved.ref_designs # the withheld optimum for each scored condition\n", "print(\"conditions scored:\", len(REFERENCE), \"| design shape:\", REFERENCE.shape[1:])\n", "print(\"condition keys:\", evaluator.resolved.condition_keys)\n", - "print(\"kernel sigma:\", spec.sigma, \"| aggregation:\", spec.aggregation)" + "print(\"kernel bandwidth:\", spec.sigma or \"median distance of the training designs\", \"| aggregation:\", spec.aggregation)" ] }, { "cell_type": "markdown", - "id": "2e1a7840", + "id": "8f79d611", "metadata": {}, "source": [ "## 3. Three constructions built from the dataset" @@ -378,7 +371,7 @@ }, { "cell_type": "markdown", - "id": "649c3e65", + "id": "7df43936", "metadata": {}, "source": [ "Beside the trained models, three rows whose right score is known in advance. They are\n", @@ -394,13 +387,13 @@ { "cell_type": "code", "execution_count": 3, - "id": "c8b7be65", + "id": "a7e667de", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:28:06.759318Z", - "iopub.status.busy": "2026-09-28T15:28:06.759208Z", - "iopub.status.idle": "2026-09-28T15:28:09.851561Z", - "shell.execute_reply": "2026-09-28T15:28:09.851336Z" + "iopub.execute_input": "2026-09-28T16:13:44.374835Z", + "iopub.status.busy": "2026-09-28T16:13:44.374724Z", + "iopub.status.idle": "2026-09-28T16:13:47.582309Z", + "shell.execute_reply": "2026-09-28T16:13:47.582076Z" } }, "outputs": [ @@ -436,7 +429,7 @@ }, { "cell_type": "markdown", - "id": "c8b5930f", + "id": "81a3e8fa", "metadata": {}, "source": [ "## 4. Score the trained models and the constructions under one spec" @@ -444,7 +437,7 @@ }, { "cell_type": "markdown", - "id": "c7faf5b1", + "id": "9b9b4673", "metadata": {}, "source": [ "Four published checkpoints at training seed one. Two of them were trained with a\n", @@ -459,20 +452,20 @@ { "cell_type": "code", "execution_count": 4, - "id": "9f2a7786", + "id": "776e8be7", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:28:09.853741Z", - "iopub.status.busy": "2026-09-28T15:28:09.853658Z", - "iopub.status.idle": "2026-09-28T15:30:18.134672Z", - "shell.execute_reply": "2026-09-28T15:30:18.134417Z" + "iopub.execute_input": "2026-09-28T16:13:47.584517Z", + "iopub.status.busy": "2026-09-28T16:13:47.584417Z", + "iopub.status.idle": "2026-09-28T17:10:49.830356Z", + "shell.execute_reply": "2026-09-28T17:10:49.830118Z" } }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "5d264478f5ae4104a027686d793af590", + "model_id": "d59b6c22ef814f8ab1b59ba34ec2b954", "version_major": 2, "version_minor": 0 }, @@ -486,7 +479,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "8e1f63b226964c988f9b85f514623736", + "model_id": "7c456c29bcad4acca9ca51dd441d0acb", "version_major": 2, "version_minor": 0 }, @@ -500,7 +493,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "acda7f10636c4626b2363090435e8fa2", + "model_id": "f5323a232e7c433f88c3b34a89fa2eb5", "version_major": 2, "version_minor": 0 }, @@ -543,25 +536,25 @@ " train_distance\n", " train_distance_ratio\n", " ...\n", + " train_minutes\n", " iog\n", " cog\n", " fog\n", - " calls_to_settle\n", + " calls_to_near_optimum\n", " gap_after_1_calls\n", " gap_after_2_calls\n", " gap_after_5_calls\n", " gap_after_10_calls\n", " reaches_reference_rate\n", - " first_call_gain\n", " \n", " \n", " \n", " \n", " cgan_cnn_2d\n", - " 0.5941\n", + " 0.5337\n", " 0.50\n", - " 3.4974\n", - " 0.0849\n", + " 2.0504\n", + " 0.0266\n", " 0.9167\n", " 0.0722\n", " 0.4144\n", @@ -569,23 +562,23 @@ " 0.2902\n", " 14.2969\n", " ...\n", + " NaN\n", " 1.816083e+09\n", " 1.735349e+09\n", " 3.0758\n", - " 1.0\n", + " 78.0\n", " 1.735347e+09\n", " 1018.6727\n", " 174.6169\n", " 56.5027\n", " 0.1667\n", - " 1.0000\n", " \n", " \n", " gan_cnn_2d\n", - " 1.0936\n", + " 0.8228\n", " 0.25\n", - " 1.0916\n", - " 0.0006\n", + " 1.0290\n", + " 0.0002\n", " 1.0000\n", " 0.2407\n", " 0.4447\n", @@ -593,23 +586,23 @@ " 0.2581\n", " 12.7131\n", " ...\n", + " NaN\n", " 8.950722e+09\n", " 8.689641e+09\n", " 0.0019\n", - " 1.0\n", + " 17.5\n", " 8.689632e+09\n", " 4121.7064\n", " 189.3875\n", " 47.1883\n", " 0.4167\n", - " 1.0000\n", " \n", " \n", " vqgan\n", - " 0.0327\n", + " 0.0096\n", " 1.00\n", - " 9.9069\n", - " 0.7916\n", + " 5.8642\n", + " 0.4346\n", " 0.3333\n", " 0.0058\n", " 0.0833\n", @@ -617,34 +610,34 @@ " 0.0353\n", " 1.7389\n", " ...\n", + " NaN\n", " 2.413100e+00\n", " 8.195000e-01\n", " -0.6194\n", - " 32.0\n", + " 1.5\n", " 2.423100e+00\n", " 5.6037\n", " 0.5194\n", " 0.0135\n", " 0.6667\n", - " 0.1523\n", " \n", " \n", "\n", - "

3 rows × 23 columns

\n", + "

3 rows × 22 columns

\n", "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog \\\n", - "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... 1.816083e+09 1.735349e+09 3.0758 \n", - "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... 8.950722e+09 8.689641e+09 0.0019 \n", - "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... 2.413100e+00 8.195000e-01 -0.6194 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog fog \\\n", + "cgan_cnn_2d 0.5337 0.50 2.0504 0.0266 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... NaN 1.816083e+09 1.735349e+09 3.0758 \n", + "gan_cnn_2d 0.8228 0.25 1.0290 0.0002 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... NaN 8.950722e+09 8.689641e+09 0.0019 \n", + "vqgan 0.0096 1.00 5.8642 0.4346 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... NaN 2.413100e+00 8.195000e-01 -0.6194 \n", "\n", - " calls_to_settle gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", - "cgan_cnn_2d 1.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", - "gan_cnn_2d 1.0 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", - "vqgan 32.0 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", + " calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "cgan_cnn_2d 78.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 \n", + "gan_cnn_2d 17.5 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 \n", + "vqgan 1.5 2.423100e+00 5.6037 0.5194 0.0135 0.6667 \n", "\n", - "[3 rows x 23 columns]" + "[3 rows x 22 columns]" ] }, "execution_count": 4, @@ -678,30 +671,16 @@ { "cell_type": "code", "execution_count": 5, - "id": "cfed22a5", + "id": "13578188", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:30:18.137402Z", - "iopub.status.busy": "2026-09-28T15:30:18.137300Z", - "iopub.status.idle": "2026-09-28T15:33:45.992283Z", - "shell.execute_reply": "2026-09-28T15:33:45.992038Z" + "iopub.execute_input": "2026-09-28T17:10:49.832935Z", + "iopub.status.busy": "2026-09-28T17:10:49.832839Z", + "iopub.status.idle": "2026-09-28T20:47:28.874800Z", + "shell.execute_reply": "2026-09-28T20:47:28.874563Z" } }, "outputs": [ - { - "data": { - "application/vnd.jupyter.widget-view+json": { - "model_id": "970f74188c7d4f7a83de75cec1e6fb39", - "version_major": 2, - "version_minor": 0 - }, - "text/plain": [ - "Fetching 3 files: 0%| | 0/3 [00:00train_distance\n", " train_distance_ratio\n", " ...\n", + " train_minutes\n", " iog\n", " cog\n", " fog\n", - " calls_to_settle\n", + " calls_to_near_optimum\n", " gap_after_1_calls\n", " gap_after_2_calls\n", " gap_after_5_calls\n", " gap_after_10_calls\n", " reaches_reference_rate\n", - " first_call_gain\n", " \n", " \n", " \n", " \n", " diffusion_2d_cond\n", - " 0.1105\n", + " 0.1041\n", " 1.0\n", - " 10.6681\n", - " 0.8654\n", + " 6.0826\n", + " 0.4784\n", " 0.8333\n", " 0.1271\n", " 0.1532\n", @@ -760,30 +739,30 @@ " 0.143\n", " 7.0439\n", " ...\n", + " NaN\n", " -1.7211\n", " 6.0217\n", " -0.9128\n", - " 17.0\n", + " 4.0\n", " -1.7127\n", " 26.3979\n", " 6.1904\n", " 0.3109\n", " 0.9167\n", - " 2.9236\n", " \n", " \n", "\n", - "

1 rows × 23 columns

\n", + "

1 rows × 22 columns

\n", "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog calls_to_settle \\\n", - "diffusion_2d_cond 0.1105 1.0 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.143 7.0439 ... -1.7211 6.0217 -0.9128 17.0 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog fog \\\n", + "diffusion_2d_cond 0.1041 1.0 6.0826 0.4784 0.8333 0.1271 0.1532 0.3969 0.143 7.0439 ... NaN -1.7211 6.0217 -0.9128 \n", "\n", - " gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", - "diffusion_2d_cond -1.7127 26.3979 6.1904 0.3109 0.9167 2.9236 \n", + " calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "diffusion_2d_cond 4.0 -1.7127 26.3979 6.1904 0.3109 0.9167 \n", "\n", - "[1 rows x 23 columns]" + "[1 rows x 22 columns]" ] }, "execution_count": 5, @@ -801,13 +780,13 @@ { "cell_type": "code", "execution_count": 6, - "id": "e69473d0", + "id": "f59948e7", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:33:45.994933Z", - "iopub.status.busy": "2026-09-28T15:33:45.994841Z", - "iopub.status.idle": "2026-09-28T15:35:10.617468Z", - "shell.execute_reply": "2026-09-28T15:35:10.617235Z" + "iopub.execute_input": "2026-09-28T20:47:28.878519Z", + "iopub.status.busy": "2026-09-28T20:47:28.878419Z", + "iopub.status.idle": "2026-09-28T20:48:54.464586Z", + "shell.execute_reply": "2026-09-28T20:48:54.464265Z" } }, "outputs": [ @@ -843,25 +822,25 @@ " train_distance\n", " train_distance_ratio\n", " ...\n", + " train_minutes\n", " iog\n", " cog\n", " fog\n", - " calls_to_settle\n", + " calls_to_near_optimum\n", " gap_after_1_calls\n", " gap_after_2_calls\n", " gap_after_5_calls\n", " gap_after_10_calls\n", " reaches_reference_rate\n", - " first_call_gain\n", " \n", " \n", " \n", " \n", " cgan_cnn_2d\n", - " 0.5941\n", + " 0.5337\n", " 0.50\n", - " 3.4974\n", - " 0.0849\n", + " 2.0504\n", + " 0.0266\n", " 0.9167\n", " 0.0722\n", " 0.4144\n", @@ -869,23 +848,23 @@ " 0.2902\n", " 14.2969\n", " ...\n", + " NaN\n", " 1.816083e+09\n", " 1.735349e+09\n", " 3.0758\n", - " 1.0\n", + " 78.0\n", " 1.735347e+09\n", " 1018.6727\n", " 174.6169\n", " 56.5027\n", " 0.1667\n", - " 1.0000\n", " \n", " \n", " gan_cnn_2d\n", - " 1.0936\n", + " 0.8228\n", " 0.25\n", - " 1.0916\n", - " 0.0006\n", + " 1.0290\n", + " 0.0002\n", " 1.0000\n", " 0.2407\n", " 0.4447\n", @@ -893,23 +872,23 @@ " 0.2581\n", " 12.7131\n", " ...\n", + " NaN\n", " 8.950722e+09\n", " 8.689641e+09\n", " 0.0019\n", - " 1.0\n", + " 17.5\n", " 8.689632e+09\n", " 4121.7064\n", " 189.3875\n", " 47.1883\n", " 0.4167\n", - " 1.0000\n", " \n", " \n", " vqgan\n", - " 0.0327\n", + " 0.0096\n", " 1.00\n", - " 9.9069\n", - " 0.7916\n", + " 5.8642\n", + " 0.4346\n", " 0.3333\n", " 0.0058\n", " 0.0833\n", @@ -917,23 +896,23 @@ " 0.0353\n", " 1.7389\n", " ...\n", + " NaN\n", " 2.413100e+00\n", " 8.195000e-01\n", " -0.6194\n", - " 32.0\n", + " 1.5\n", " 2.423100e+00\n", " 5.6037\n", " 0.5194\n", " 0.0135\n", " 0.6667\n", - " 0.1523\n", " \n", " \n", " diffusion_2d_cond\n", - " 0.1105\n", + " 0.1041\n", " 1.00\n", - " 10.6681\n", - " 0.8654\n", + " 6.0826\n", + " 0.4784\n", " 0.8333\n", " 0.1271\n", " 0.1532\n", @@ -941,23 +920,23 @@ " 0.1430\n", " 7.0439\n", " ...\n", + " NaN\n", " -1.721100e+00\n", " 6.021700e+00\n", " -0.9128\n", - " 17.0\n", + " 4.0\n", " -1.712700e+00\n", " 26.3979\n", " 6.1904\n", " 0.3109\n", " 0.9167\n", - " 2.9236\n", " \n", " \n", " lookup_table\n", - " 0.0729\n", + " 0.0395\n", " 1.00\n", - " 9.7122\n", - " 0.3051\n", + " 5.3611\n", + " 0.1796\n", " 0.7500\n", " 0.0125\n", " 0.0569\n", @@ -965,23 +944,23 @@ " 0.0000\n", " 0.0000\n", " ...\n", + " NaN\n", " 9.574600e+00\n", " 7.314000e-01\n", " -0.7363\n", - " 29.0\n", + " 3.5\n", " 9.590800e+00\n", " 8.2825\n", " 3.3313\n", " 0.3676\n", " 0.5833\n", - " 0.3145\n", " \n", " \n", " unconditional\n", - " 0.1631\n", + " 0.0902\n", " 1.00\n", - " 11.1734\n", - " 0.9277\n", + " 7.0133\n", + " 0.5779\n", " 0.9167\n", " 0.0562\n", " 0.3993\n", @@ -989,23 +968,23 @@ " 0.0000\n", " 0.0000\n", " ...\n", + " NaN\n", " 4.805870e+03\n", " 2.817597e+04\n", " 1.4287\n", - " 6.0\n", + " 101.0\n", " 4.806746e+03\n", " 1994.0963\n", " 66.6704\n", " 4.2396\n", " 0.1667\n", - " 0.4913\n", " \n", " \n", " permuted_conditions\n", " 0.0000\n", " 1.00\n", - " 10.0889\n", - " 0.3162\n", + " 6.1870\n", + " 0.2039\n", " 0.9167\n", " 0.0562\n", " 0.4231\n", @@ -1013,42 +992,42 @@ " 0.0203\n", " 1.0000\n", " ...\n", + " NaN\n", " 8.783402e+07\n", " 8.788590e+07\n", " 4.4569\n", - " 2.0\n", + " 101.0\n", " 8.782727e+07\n", " 1865.3040\n", " 85.9982\n", " 27.0510\n", " 0.0000\n", - " 0.8233\n", " \n", " \n", "\n", - "

7 rows × 23 columns

\n", + "

7 rows × 22 columns

\n", "" ], "text/plain": [ - " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... iog cog fog \\\n", - "cgan_cnn_2d 0.5941 0.50 3.4974 0.0849 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... 1.816083e+09 1.735349e+09 3.0758 \n", - "gan_cnn_2d 1.0936 0.25 1.0916 0.0006 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... 8.950722e+09 8.689641e+09 0.0019 \n", - "vqgan 0.0327 1.00 9.9069 0.7916 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... 2.413100e+00 8.195000e-01 -0.6194 \n", - "diffusion_2d_cond 0.1105 1.00 10.6681 0.8654 0.8333 0.1271 0.1532 0.3969 0.1430 7.0439 ... -1.721100e+00 6.021700e+00 -0.9128 \n", - "lookup_table 0.0729 1.00 9.7122 0.3051 0.7500 0.0125 0.0569 NaN 0.0000 0.0000 ... 9.574600e+00 7.314000e-01 -0.7363 \n", - "unconditional 0.1631 1.00 11.1734 0.9277 0.9167 0.0562 0.3993 NaN 0.0000 0.0000 ... 4.805870e+03 2.817597e+04 1.4287 \n", - "permuted_conditions 0.0000 1.00 10.0889 0.3162 0.9167 0.0562 0.4231 NaN 0.0203 1.0000 ... 8.783402e+07 8.788590e+07 4.4569 \n", + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog \\\n", + "cgan_cnn_2d 0.5337 0.50 2.0504 0.0266 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... NaN 1.816083e+09 1.735349e+09 \n", + "gan_cnn_2d 0.8228 0.25 1.0290 0.0002 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... NaN 8.950722e+09 8.689641e+09 \n", + "vqgan 0.0096 1.00 5.8642 0.4346 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... NaN 2.413100e+00 8.195000e-01 \n", + "diffusion_2d_cond 0.1041 1.00 6.0826 0.4784 0.8333 0.1271 0.1532 0.3969 0.1430 7.0439 ... NaN -1.721100e+00 6.021700e+00 \n", + "lookup_table 0.0395 1.00 5.3611 0.1796 0.7500 0.0125 0.0569 NaN 0.0000 0.0000 ... NaN 9.574600e+00 7.314000e-01 \n", + "unconditional 0.0902 1.00 7.0133 0.5779 0.9167 0.0562 0.3993 NaN 0.0000 0.0000 ... NaN 4.805870e+03 2.817597e+04 \n", + "permuted_conditions 0.0000 1.00 6.1870 0.2039 0.9167 0.0562 0.4231 NaN 0.0203 1.0000 ... NaN 8.783402e+07 8.788590e+07 \n", "\n", - " calls_to_settle gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate first_call_gain \n", - "cgan_cnn_2d 1.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 1.0000 \n", - "gan_cnn_2d 1.0 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 1.0000 \n", - "vqgan 32.0 2.423100e+00 5.6037 0.5194 0.0135 0.6667 0.1523 \n", - "diffusion_2d_cond 17.0 -1.712700e+00 26.3979 6.1904 0.3109 0.9167 2.9236 \n", - "lookup_table 29.0 9.590800e+00 8.2825 3.3313 0.3676 0.5833 0.3145 \n", - "unconditional 6.0 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 0.4913 \n", - "permuted_conditions 2.0 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 0.8233 \n", + " fog calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "cgan_cnn_2d 3.0758 78.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 \n", + "gan_cnn_2d 0.0019 17.5 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 \n", + "vqgan -0.6194 1.5 2.423100e+00 5.6037 0.5194 0.0135 0.6667 \n", + "diffusion_2d_cond -0.9128 4.0 -1.712700e+00 26.3979 6.1904 0.3109 0.9167 \n", + "lookup_table -0.7363 3.5 9.590800e+00 8.2825 3.3313 0.3676 0.5833 \n", + "unconditional 1.4287 101.0 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 \n", + "permuted_conditions 4.4569 101.0 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 \n", "\n", - "[7 rows x 23 columns]" + "[7 rows x 22 columns]" ] }, "execution_count": 6, @@ -1065,7 +1044,7 @@ }, { "cell_type": "markdown", - "id": "773f94f8", + "id": "4f440867", "metadata": {}, "source": [ "## 5. Read the board" @@ -1074,13 +1053,13 @@ { "cell_type": "code", "execution_count": 7, - "id": "cd254904", + "id": "6fb01bb4", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:35:10.620119Z", - "iopub.status.busy": "2026-09-28T15:35:10.620036Z", - "iopub.status.idle": "2026-09-28T15:35:10.624691Z", - "shell.execute_reply": "2026-09-28T15:35:10.624518Z" + "iopub.execute_input": "2026-09-28T20:48:54.467228Z", + "iopub.status.busy": "2026-09-28T20:48:54.467146Z", + "iopub.status.idle": "2026-09-28T20:48:54.471885Z", + "shell.execute_reply": "2026-09-28T20:48:54.471687Z" } }, "outputs": [ @@ -1225,10 +1204,10 @@ " None\n", " \n", " \n", - " calls_to_settle\n", - " Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle.\n", + " calls_to_near_optimum\n", + " Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there.\n", " lower is better\n", - " tie: cgan_cnn_2d, gan_cnn_2d\n", + " vqgan\n", " None\n", " \n", " \n", @@ -1266,42 +1245,34 @@ " diffusion_2d_cond\n", " None\n", " \n", - " \n", - " first_call_gain\n", - " Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step.\n", - " higher is better\n", - " diffusion_2d_cond\n", - " None\n", - " \n", " \n", "\n", "" ], "text/plain": [ - " question \\\n", - "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", - "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", - "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", - "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", - "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", - "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", - "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", - "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", - "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "train_distance_ratio Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", - "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", - "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", - "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", - "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", - "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", - "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", - "calls_to_settle Optimizer calls from each generated design until the objective stays within five percent of its final value. Fewer means a faster settle. \n", - "gap_after_1_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", - "gap_after_2_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", - "gap_after_5_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", - "gap_after_10_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", - "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", - "first_call_gain Share of the total re-optimization improvement delivered by the first optimizer call. Near one means the defects clear in a single step. \n", + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "train_distance_ratio Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "calls_to_near_optimum Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. \n", + "gap_after_1_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_2_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_5_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_10_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", "\n", " direction picks real designs score \n", "mmd lower is better permuted_conditions None \n", @@ -1320,13 +1291,12 @@ "iog lower is better diffusion_2d_cond None \n", "cog lower is better lookup_table None \n", "fog lower is better diffusion_2d_cond None \n", - "calls_to_settle lower is better tie: cgan_cnn_2d, gan_cnn_2d None \n", + "calls_to_near_optimum lower is better vqgan None \n", "gap_after_1_calls lower is better diffusion_2d_cond None \n", "gap_after_2_calls lower is better vqgan None \n", "gap_after_5_calls lower is better vqgan None \n", "gap_after_10_calls lower is better vqgan None \n", - "reaches_reference_rate higher is better diffusion_2d_cond None \n", - "first_call_gain higher is better diffusion_2d_cond None " + "reaches_reference_rate higher is better diffusion_2d_cond None " ] }, "execution_count": 7, @@ -1340,7 +1310,7 @@ }, { "cell_type": "markdown", - "id": "6d86e56b", + "id": "773639bf", "metadata": {}, "source": [ "Why these seven rows: four trained models and three constructions whose right\n", @@ -1380,7 +1350,13 @@ "\n", "**`cgan_cnn_2d` and `gan_cnn_2d` post initial gaps near 10^9 and 10^10.** The\n", "optimizer repairs them; the final gaps are 3.1 and 0.002. These are designs that start\n", - "absurd and are cheap to fix, which is why `iog` and `fog` disagree about them.\n", + "absurd and are cheap to fix, which is why `iog` and `fog` disagree about them. Cheap\n", + "in the end, but not fast: they need 78 and 17.5 optimizer calls to get within five\n", + "percent of the optimum, against 1.5 for the VQGAN, 4 for diffusion and 3.5 for the\n", + "lookup table. The unconditional and permuted rows read 101, one more than the budget,\n", + "because they never get there. That column is measured against the optimum for each\n", + "design's conditions, not against where the design started, so an absurd start does not\n", + "read as a fast one.\n", "`gan_cnn_2d` has a `vendi` of 1.09, twelve designs that are effectively one, and a\n", "`cond_sens` of exactly zero, because it is an unconditional model that was handed\n", "conditions anyway. That zero is what stops it collecting a conditional model's score.\n", @@ -1394,14 +1370,14 @@ "constructions, because they have no model to re-run or inspect. And there is no\n", "reference row here; that row comes from the saved-designs path in section 7, where at\n", "twelve conditions it is six designs against six, a coarser measurement than the model\n", - "rows. The trajectory columns also assume the gap falls monotonically; for the diffusion\n", - "model it rises after the first call, as the optimizer restores feasibility before it\n", - "improves the objective, and `first_call_gain` reads above one for that reason." + "rows. The gap path is not always monotone: for the diffusion model it rises after the\n", + "first call, as the optimizer restores feasibility before it improves the objective,\n", + "which is why its gap after two calls is larger than after one." ] }, { "cell_type": "markdown", - "id": "7297bed1", + "id": "5c0f9320", "metadata": {}, "source": [ "## 6. Rank, but only on a column that can be ranked" @@ -1410,13 +1386,13 @@ { "cell_type": "code", "execution_count": 8, - "id": "168eb15d", + "id": "6de7f90a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:35:10.627119Z", - "iopub.status.busy": "2026-09-28T15:35:10.627038Z", - "iopub.status.idle": "2026-09-28T15:35:10.630171Z", - "shell.execute_reply": "2026-09-28T15:35:10.629928Z" + "iopub.execute_input": "2026-09-28T20:48:54.474424Z", + "iopub.status.busy": "2026-09-28T20:48:54.474342Z", + "iopub.status.idle": "2026-09-28T20:48:54.477351Z", + "shell.execute_reply": "2026-09-28T20:48:54.477162Z" } }, "outputs": [ @@ -1517,13 +1493,13 @@ { "cell_type": "code", "execution_count": 9, - "id": "5ddab12f", + "id": "9ea49f63", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:35:10.632624Z", - "iopub.status.busy": "2026-09-28T15:35:10.632565Z", - "iopub.status.idle": "2026-09-28T15:35:10.634223Z", - "shell.execute_reply": "2026-09-28T15:35:10.634041Z" + "iopub.execute_input": "2026-09-28T20:48:54.479760Z", + "iopub.status.busy": "2026-09-28T20:48:54.479696Z", + "iopub.status.idle": "2026-09-28T20:48:54.481228Z", + "shell.execute_reply": "2026-09-28T20:48:54.481028Z" } }, "outputs": [ @@ -1544,7 +1520,7 @@ }, { "cell_type": "markdown", - "id": "510bdec2", + "id": "dd3092d7", "metadata": {}, "source": [ "## 7. The same questions in another space" @@ -1552,14 +1528,15 @@ }, { "cell_type": "markdown", - "id": "9f2babbc", + "id": "b3c40d42", "metadata": {}, "source": [ "Where a metric is computed is an argument, not a property of the metric. The board\n", "kept every model's sampled designs, so they can be re-scored after projecting onto\n", "the leading principal components of the reference designs. Metrics that need the\n", - "actual designs, such as the constraint check and the copy corpus, are skipped there.\n", - "The kernel bandwidth is pinned to the spec's value so the two boards share a scale.\n", + "actual designs, such as the constraint check and the training-distance column, are\n", + "skipped there. The kernel bandwidth is the training designs' median distance in the\n", + "projected space, found the same way it was in pixel space.\n", "\n", "This is the control that asks whether a learned latent space is needed at all. If PCA\n", "of the data answers the same questions, any rotation of the same width would do. A\n", @@ -1569,13 +1546,13 @@ { "cell_type": "code", "execution_count": 10, - "id": "2c5a60a3", + "id": "40bc912f", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:35:10.636725Z", - "iopub.status.busy": "2026-09-28T15:35:10.636609Z", - "iopub.status.idle": "2026-09-28T15:35:10.646420Z", - "shell.execute_reply": "2026-09-28T15:35:10.646245Z" + "iopub.execute_input": "2026-09-28T20:48:54.483878Z", + "iopub.status.busy": "2026-09-28T20:48:54.483810Z", + "iopub.status.idle": "2026-09-28T20:48:55.193824Z", + "shell.execute_reply": "2026-09-28T20:48:55.193585Z" } }, "outputs": [ @@ -1610,67 +1587,67 @@ " \n", " \n", " cgan_cnn_2d\n", - " 0.7543\n", - " 0.0018\n", + " 0.4708\n", + " 0.0008\n", " 7.1693\n", " 1.0000\n", - " 1.4800\n", + " 1.2525\n", " \n", " \n", " gan_cnn_2d\n", - " 1.0002\n", + " 0.7570\n", " 0.0000\n", " 9.2163\n", " 0.5833\n", - " 1.0032\n", + " 1.0015\n", " \n", " \n", " vqgan\n", - " 0.0060\n", - " 0.5743\n", + " 0.0030\n", + " 0.3758\n", " 0.4617\n", " 1.0000\n", - " 8.7158\n", + " 6.7750\n", " \n", " \n", " diffusion_2d_cond\n", - " 0.0624\n", - " 0.6964\n", + " 0.0548\n", + " 0.4440\n", " 2.0203\n", " 1.0000\n", - " 9.2823\n", + " 6.5482\n", " \n", " \n", " lookup_table\n", - " 0.0614\n", - " 0.2537\n", + " 0.0450\n", + " 0.1659\n", " 0.5618\n", " 1.0000\n", - " 8.3923\n", + " 5.8951\n", " \n", " \n", " unconditional\n", - " 0.1054\n", - " 0.4121\n", + " 0.0688\n", + " 0.2280\n", " 8.6422\n", " 1.0000\n", - " 6.7416\n", + " 4.8833\n", " \n", " \n", " permuted_conditions\n", " 0.0000\n", - " 0.2804\n", + " 0.2091\n", " 10.5238\n", " 1.0000\n", - " 9.3671\n", + " 7.4443\n", " \n", " \n", " reference (split-half)\n", - " 0.2264\n", - " 0.9268\n", + " 0.1696\n", + " 0.7890\n", " NaN\n", " 1.0000\n", - " 5.6120\n", + " 4.9480\n", " \n", " \n", "\n", @@ -1678,14 +1655,14 @@ ], "text/plain": [ " mmd@pca dpp@pca per_condition_distance@pca coverage@pca vendi@pca\n", - "cgan_cnn_2d 0.7543 0.0018 7.1693 1.0000 1.4800\n", - "gan_cnn_2d 1.0002 0.0000 9.2163 0.5833 1.0032\n", - "vqgan 0.0060 0.5743 0.4617 1.0000 8.7158\n", - "diffusion_2d_cond 0.0624 0.6964 2.0203 1.0000 9.2823\n", - "lookup_table 0.0614 0.2537 0.5618 1.0000 8.3923\n", - "unconditional 0.1054 0.4121 8.6422 1.0000 6.7416\n", - "permuted_conditions 0.0000 0.2804 10.5238 1.0000 9.3671\n", - "reference (split-half) 0.2264 0.9268 NaN 1.0000 5.6120" + "cgan_cnn_2d 0.4708 0.0008 7.1693 1.0000 1.2525\n", + "gan_cnn_2d 0.7570 0.0000 9.2163 0.5833 1.0015\n", + "vqgan 0.0030 0.3758 0.4617 1.0000 6.7750\n", + "diffusion_2d_cond 0.0548 0.4440 2.0203 1.0000 6.5482\n", + "lookup_table 0.0450 0.1659 0.5618 1.0000 5.8951\n", + "unconditional 0.0688 0.2280 8.6422 1.0000 4.8833\n", + "permuted_conditions 0.0000 0.2091 10.5238 1.0000 7.4443\n", + "reference (split-half) 0.1696 0.7890 NaN 1.0000 4.9480" ] }, "execution_count": 10, @@ -1694,13 +1671,13 @@ } ], "source": [ - "pca_board = Board(evaluator.problem, reference=REFERENCE, train=TRAIN, sigma=spec.sigma)\n", + "pca_board = Board(evaluator.problem, reference=REFERENCE, train=TRAIN)\n", "pca_board.evaluate(board.designs, space=\"pca\", width=8, aggregation=\"median\").round(4)" ] }, { "cell_type": "markdown", - "id": "69072434", + "id": "03463c3e", "metadata": {}, "source": [ "## 8. Registering a metric of your own" @@ -1708,24 +1685,37 @@ }, { "cell_type": "markdown", - "id": "7bc23e01", + "id": "bad3937b", "metadata": {}, "source": [ - "A metric is a function of an `EvaluationContext`, a family, a cost and a direction.\n", - "The first line of its docstring is the sentence readers see. `ctx.reduce` applies\n", - "whatever aggregation the spec chose, so a metric never hard-codes a mean." + "A metric is a plain function. The evaluator calls it once per model and hands it one\n", + "object, the evaluation context, which holds everything that model's evaluation knows:\n", + "\n", + "| On the context | What it is |\n", + "|---|---|\n", + "| `gen_designs` | the designs the model produced, one per scored condition, as an array of shape `(n, 50, 100)` here |\n", + "| `ref_designs` | the reference optimum for each of those conditions, same shape |\n", + "| `conditions` | the scored conditions themselves, one row per design |\n", + "| `train_designs` | the training split, flattened, for anything that asks \"has the model seen this?\" |\n", + "| `optimization` | the physics results, one gap path per design; touching it runs the optimizer, so declare `cost=\"expensive\"` |\n", + "| `reduce(values)` | collapses one value per design to one number the way the spec asked, median here, mean by default |\n", + "\n", + "The decorator adds the four declarations a reader needs: the name, which family of\n", + "question it belongs to, whether it needs the simulator, and which direction is better,\n", + "with `None` for a column that is read but never ranked on. The first line of the\n", + "docstring is the sentence `explain()` shows. Nothing else has to be wired up." ] }, { "cell_type": "code", "execution_count": 11, - "id": "f6d49da7", + "id": "adec143c", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T15:35:10.648934Z", - "iopub.status.busy": "2026-09-28T15:35:10.648863Z", - "iopub.status.idle": "2026-09-28T15:35:10.654564Z", - "shell.execute_reply": "2026-09-28T15:35:10.654414Z" + "iopub.execute_input": "2026-09-28T20:48:55.198263Z", + "iopub.status.busy": "2026-09-28T20:48:55.198168Z", + "iopub.status.idle": "2026-09-28T20:48:55.204228Z", + "shell.execute_reply": "2026-09-28T20:48:55.204032Z" } }, "outputs": [ @@ -1791,9 +1781,11 @@ "\n", "\n", "@register_metric(\"density_ratio\", family=\"feasibility\", cost=\"cheap\", higher_is_better=None)\n", - "def density_ratio(ctx):\n", + "def density_ratio(context):\n", " \"\"\"Material fraction of each generated design over that of the reference optima. One means the same amount of material.\"\"\"\n", - " return float(ctx.reduce(ctx.gen_flat.mean(axis=1)) / ctx.ref_flat.mean())\n", + " generated = context.gen_designs.reshape(len(context.gen_designs), -1).mean(axis=1) # one fraction per generated design\n", + " reference = context.ref_designs.reshape(len(context.ref_designs), -1).mean(axis=1) # one per reference optimum\n", + " return context.reduce(generated) / context.reduce(reference) # each collapsed the way the spec asked, then compared\n", "\n", "\n", "Board.from_evaluator(evaluator, {}, designs=board.designs, metrics=[\"volume_error\", \"density_ratio\"]).explain()" @@ -1801,7 +1793,7 @@ }, { "cell_type": "markdown", - "id": "7560f183", + "id": "f07bc0f9", "metadata": {}, "source": [ "It appears with its question and direction attached, scored under the same spec as\n", @@ -1812,7 +1804,7 @@ }, { "cell_type": "markdown", - "id": "fe1e1e38", + "id": "39f2d3f8", "metadata": {}, "source": [ "## 9. Not in this notebook" @@ -1820,7 +1812,7 @@ }, { "cell_type": "markdown", - "id": "27263dd8", + "id": "4e3e0524", "metadata": {}, "source": [ "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec. It\n", From cc5f16e662d085a33b34ae6878fe05caa406ba69 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Tue, 29 Sep 2026 11:19:14 +0200 Subject: [PATCH 15/18] Address the first review: one default spec, conditions for saved designs, and a board that keeps its registry Six findings from Soheyl on the current head, each reproduced from the code before changing it. The default spec disagreed with itself: `Evaluator.for_problem` and the CLI loaded v1 when given no spec, `EvalSpec.load` chose v2, and a bare `EvalSpec(problem_id=...)` still listed `novelty`. All three now resolve to v2, and the dataclass default names only registered metrics. The saved-designs path never carried the designs' conditions, so on beams2d `viol` reached the problem's constraint check without the volume fraction it reads and raised. `Board` now takes the conditions the reference designs answer and passes them through; `viol` reports blank rather than raising when none were given, and a test fixture that relied on the raise-free fake now supplies conditions. `rank` no longer lists the split-half reference row, which `explain` already excluded, and the row's docstring and `explain` column say what it is: a scale reference at half the sample size, not a number to beat. A board built from an evaluator keeps that evaluator's metric registry, so a custom metric it scored is readable and rankable. The parameter count is no longer memoized by generator object, which had kept every scored network alive for the life of the process. Co-Authored-By: Claude Fable 5.1 --- engiopt/evaluation/board.py | 31 +++++++++++++++++----- engiopt/evaluation/evaluator.py | 3 +-- engiopt/evaluation/metrics/builtin.py | 4 ++- engiopt/evaluation/spec.py | 21 ++++++++++++++- tests/test_board.py | 3 ++- tests/test_conditions_and_gaps.py | 3 +++ tests/test_metrics_suite.py | 38 ++++++++++++++++++++++++++- 7 files changed, 90 insertions(+), 13 deletions(-) diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 26f02f47..92b4ac0e 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -46,10 +46,12 @@ def register_space(name: str, fit: SpaceFit) -> None: REFERENCE_ROW = "reference (split-half)" -"""Label of the row scoring half of the reference designs against the other half. +"""Label of the row scoring one random half of the reference designs against the other. -It is what real, correct designs score on every column, measured on this problem -rather than assumed, and it is the row every other row is read against. +A scale reference, not a competitor: it shows where real, correct designs land on +each column on this problem. It is half the sample size of a model row, and the +set-level metrics move with sample size, so read it as the order of magnitude a +good model should reach rather than as a number to beat. Never ranked. """ @@ -61,6 +63,10 @@ class Board: problem: The problem the designs answer; needs `design_space` and `check_constraints`, which every EngiBench problem has. reference: The withheld optimal designs the models are scored against. + conditions: The conditions each reference design answers, one row per + design, as anything indexable by condition name and by row. Needed + by `viol` on problems whose constraint check reads the conditions, + and by `volume_error`; without them those columns are blank. train: The training designs, for the memorization metrics. Optional. sigma: Kernel bandwidth for the distribution and diversity metrics. None, the default, uses the median pairwise distance of the training designs @@ -73,6 +79,7 @@ class Board: problem: Any reference: Any + conditions: Any | None = None train: Any | None = None sigma: float | None = None registry: MetricRegistry = field(default_factory=lambda: METRICS) @@ -120,6 +127,7 @@ def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: problem_id=getattr(self.problem, "problem_id", type(self.problem).__name__), gen_designs=project(np.asarray(generated)), ref_designs=project(np.asarray(against)), + conditions=self.conditions if against is reference else None, sigma=self.sigma, aggregation=aggregation, train_designs_fn=(lambda: project(np.asarray(self.train))) if self.train is not None else None, @@ -184,7 +192,13 @@ def from_evaluator( ctx = evaluator.context_for_designs(batch) rows[label] = evaluator.score_context(ctx, only=metrics, include_expensive=expensive) sampled[label] = ctx.gen_designs - board = cls(problem=evaluator.problem, reference=evaluator.resolved.ref_designs, sigma=evaluator.spec.sigma) + board = cls( + problem=evaluator.problem, + reference=evaluator.resolved.ref_designs, + conditions=evaluator.resolved.conditions, + sigma=evaluator.spec.sigma, + registry=evaluator.registry, + ) board.frame = pd.DataFrame.from_dict(rows, orient="index") board.designs = sampled return board @@ -234,7 +248,7 @@ def explain(self) -> pd.DataFrame: Returns: One row per column: the question it answers, which way is better, the model it picks (blank for a diagnostic, `tie` when the best is shared), - and what real designs scored on it. + and the split-half reference value when the board carries that row. """ models = self.frame.drop(index=REFERENCE_ROW, errors="ignore") rows = {} @@ -252,7 +266,9 @@ def explain(self) -> pd.DataFrame: "question": spec.description, "direction": spec.direction, "picks": picks, - "real designs score": self.frame.loc[REFERENCE_ROW, column] if REFERENCE_ROW in self.frame.index else None, + "split-half reference": self.frame.loc[REFERENCE_ROW, column] + if REFERENCE_ROW in self.frame.index + else None, } return pd.DataFrame.from_dict(rows, orient="index") @@ -270,7 +286,8 @@ def rank(self, column: str) -> pd.DataFrame: f"{column!r} is a diagnostic: it says whether the other columns mean what they look like. " f"Read it beside them; do not rank on it." ) - return self.frame.sort_values(column, ascending=not spec.higher_is_better) + models = self.frame.drop(index=REFERENCE_ROW, errors="ignore") + return models.sort_values(column, ascending=not spec.higher_is_better) def _spec_for(self, column: str) -> Any: base = column.split("@", maxsplit=1)[0] diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index bb574cca..927f8090 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -135,7 +135,7 @@ def for_problem( from engibench.utils.all_problems import BUILTIN_PROBLEMS problem = BUILTIN_PROBLEMS[problem_id]() - eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or f"{problem_id}/v1") + eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or problem_id) if eval_spec.problem_id != problem_id: raise ValueError(f"Spec is for {eval_spec.problem_id!r}, not {problem_id!r}.") device = device or pick_device() @@ -369,7 +369,6 @@ def leaderboard( return order_columns(pd.DataFrame(rows)) -@functools.cache def _parameter_count(generator: Any) -> int | None: """Trainable parameters of the generator's network, or None for a model with none.""" from torch import nn diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 579e6f53..bb1c6fa2 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -159,7 +159,7 @@ def cond_sens(ctx: EvaluationContext) -> float: higher_is_better=False, ) def viol(ctx: EvaluationContext) -> float: - """Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. + """Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. Defined for every problem: `problem.check_constraints` always applies, and the spec's `volume_condition` adds the volume-fraction budget for problems @@ -170,6 +170,8 @@ def viol(ctx: EvaluationContext) -> float: the optimizer refuses to start from an invalid design, the case where the answer matters most. """ + if ctx.conditions is None: + return float("nan") values = ctx.feasibility return float(np.mean(values)) if values else float("nan") diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index 6aa8b910..4af8e720 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -129,7 +129,26 @@ class EvalSpec: version: str = "v2" n_samples: int = 50 condition_seed: int = 1 - metrics: tuple[str, ...] = ("mmd", "dpp", "novelty", "cond_sens", "viol", "iog", "cog", "fog") + metrics: tuple[str, ...] = ( + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate", + ) sigma: float | None = None aggregation: Literal["mean", "median"] = "mean" """How per-design metrics (`iog`, `cog`, `fog`, distances) collapse to one number. diff --git a/tests/test_board.py b/tests/test_board.py index 043e5619..cd3ac5cc 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -41,7 +41,7 @@ def test_explain_names_a_pick_only_for_ranked_columns(fake_problem: Any, designs assert explained.loc["mmd", "picks"] == "copier", "identical sets score the best MMD" assert explained.loc["train_distance", "picks"] == "", "a diagnostic picks nobody" assert explained.loc["mmd", "question"] == METRICS["mmd"].description - assert explained.loc["mmd", "real designs score"] == board.frame.loc[REFERENCE_ROW, "mmd"] + assert explained.loc["mmd", "split-half reference"] == board.frame.loc[REFERENCE_ROW, "mmd"] def test_the_reference_row_is_measured_and_never_picked(fake_problem: Any, designs: Any) -> None: @@ -50,6 +50,7 @@ def test_the_reference_row_is_measured_and_never_picked(fake_problem: Any, desig board.evaluate(models) assert board.frame.loc[REFERENCE_ROW, "mmd"] > 0.0, "half of the data against the other half is not identical" assert REFERENCE_ROW not in set(board.explain()["picks"]) + assert REFERENCE_ROW not in board.rank("mmd").index, "a scale reference is never ranked" assert REFERENCE_ROW not in board.evaluate(models, reference_row=False).index diff --git a/tests/test_conditions_and_gaps.py b/tests/test_conditions_and_gaps.py index 05d93526..63b165b8 100644 --- a/tests/test_conditions_and_gaps.py +++ b/tests/test_conditions_and_gaps.py @@ -17,6 +17,7 @@ from engiopt.evaluation.registry import METRICS from engiopt.transforms import get_image_condition_keys from engiopt.transforms import get_scalar_condition_keys +from tests.conftest import FakeDataset from tests.conftest import FakeViolations # Importing the metrics package registers the built-ins. @@ -283,6 +284,8 @@ def test_signs_are_per_objective_for_mixed_directions() -> None: def _feasibility_context(problem: Any, design_value: float, **kwargs: Any) -> EvaluationContext: designs = np.full((1, *problem.design_space.shape), design_value) + # A feasibility check reads the design's conditions, so a context without them reports nothing. + kwargs.setdefault("conditions", FakeDataset({"volfrac": [design_value]})) return EvaluationContext( problem=problem, problem_id="fake", diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py index 8de0d3d6..603b03f7 100644 --- a/tests/test_metrics_suite.py +++ b/tests/test_metrics_suite.py @@ -200,8 +200,9 @@ def __init__(self, designs: Any) -> None: class StubEvaluator: problem = None + registry = METRICS spec = type("Spec", (), {"sigma": 1.0})() - resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3))})() + resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() def context_for(self, generator: Any) -> StubContext: return StubContext(generator) @@ -221,6 +222,41 @@ def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict assert set(board.designs) == {"a", "b", "probe"} +def test_from_evaluator_keeps_the_evaluators_registry() -> None: + """A custom metric the evaluator scored must be readable and rankable on the returned board.""" + from engiopt.evaluation.registry import MetricRegistry + from engiopt.evaluation.registry import register_metric + + custom = MetricRegistry() + register_metric("mine", family="diversity", cost="cheap", higher_is_better=True, registry=custom)(lambda ctx: 0.0) + + class StubEvaluator: + problem = None + registry = custom + spec = type("Spec", (), {"sigma": None})() + resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() + + def context_for(self, generator: Any) -> Any: + return type("Ctx", (), {"gen_designs": generator})() + + def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: + return {"mine": float(np.mean(ctx.gen_designs))} + + board = Board.from_evaluator(StubEvaluator(), {"a": np.full(3, 0.2), "b": np.full(3, 0.4)}) + assert board.rank("mine").index[0] == "b" + assert "mine" in board.explain().index + + +def test_viol_is_blank_without_the_designs_conditions(fake_problem: Any, sets: Any) -> None: + """A constraint check that reads the conditions cannot run on bare designs; blank beats raising.""" + gen, ref = sets + assert np.isnan(METRICS["viol"].fn(_ctx(fake_problem, gen, ref))) + + +def test_the_default_spec_metrics_are_all_registered() -> None: + assert set(EvalSpec(problem_id="beams2d").metrics) <= set(METRICS) + + # -------------------------------------------------------------------- specs From 6a1ad7718f9779c6287965d1f85d1118ef414961 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Tue, 29 Sep 2026 16:58:22 +0200 Subject: [PATCH 16/18] Re-run the notebook with the renamed explain column The stored outputs still showed the old column name for the split-half row. Same numbers as before; the physics is unchanged. Co-Authored-By: Claude Fable 5.1 --- example_metrics_suite.ipynb | 232 +++++++++++++++++++----------------- 1 file changed, 123 insertions(+), 109 deletions(-) diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb index 7aae5ba6..2237acfe 100644 --- a/example_metrics_suite.ipynb +++ b/example_metrics_suite.ipynb @@ -2,7 +2,7 @@ "cells": [ { "cell_type": "markdown", - "id": "e32d8ced", + "id": "0c0d6c21", "metadata": {}, "source": [ "# Using the metrics suite, on beams2d\n", @@ -27,7 +27,7 @@ }, { "cell_type": "markdown", - "id": "ceeb955f", + "id": "e8b6c327", "metadata": {}, "source": [ "## 1. What metrics exist, and what does each one measure?" @@ -36,13 +36,13 @@ { "cell_type": "code", "execution_count": 1, - "id": "dac2ff45", + "id": "95381ed9", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T16:13:40.116290Z", - "iopub.status.busy": "2026-09-28T16:13:40.115944Z", - "iopub.status.idle": "2026-09-28T16:13:42.773684Z", - "shell.execute_reply": "2026-09-28T16:13:42.773430Z" + "iopub.execute_input": "2026-09-29T09:19:46.072636Z", + "iopub.status.busy": "2026-09-29T09:19:46.072216Z", + "iopub.status.idle": "2026-09-29T09:19:48.724305Z", + "shell.execute_reply": "2026-09-29T09:19:48.724074Z" } }, "outputs": [ @@ -114,7 +114,7 @@ " \n", " \n", " viol\n", - " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", + " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied.\n", " feasibility\n", " lower is better\n", " cheap\n", @@ -220,7 +220,7 @@ "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", - "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. \n", "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", @@ -280,7 +280,7 @@ }, { "cell_type": "markdown", - "id": "8ee178ea", + "id": "30340537", "metadata": {}, "source": [ "Each row names the quantity, says how it is computed, and says what the two ends of\n", @@ -290,7 +290,7 @@ }, { "cell_type": "markdown", - "id": "a597ddfa", + "id": "810b2c11", "metadata": {}, "source": [ "## 2. The problem, and the evaluation spec" @@ -298,7 +298,7 @@ }, { "cell_type": "markdown", - "id": "fda28edc", + "id": "a5e804dc", "metadata": {}, "source": [ "Every published number on beams2d is computed under one frozen spec. The spec fixes\n", @@ -325,13 +325,13 @@ { "cell_type": "code", "execution_count": 2, - "id": "93f920cc", + "id": "e6a82efe", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T16:13:42.776190Z", - "iopub.status.busy": "2026-09-28T16:13:42.775973Z", - "iopub.status.idle": "2026-09-28T16:13:44.372558Z", - "shell.execute_reply": "2026-09-28T16:13:44.372303Z" + "iopub.execute_input": "2026-09-29T09:19:48.726734Z", + "iopub.status.busy": "2026-09-29T09:19:48.726545Z", + "iopub.status.idle": "2026-09-29T09:19:50.397378Z", + "shell.execute_reply": "2026-09-29T09:19:50.397093Z" } }, "outputs": [ @@ -363,7 +363,7 @@ }, { "cell_type": "markdown", - "id": "8f79d611", + "id": "a42d95f7", "metadata": {}, "source": [ "## 3. Three constructions built from the dataset" @@ -371,7 +371,7 @@ }, { "cell_type": "markdown", - "id": "7df43936", + "id": "9cd57fca", "metadata": {}, "source": [ "Beside the trained models, three rows whose right score is known in advance. They are\n", @@ -387,13 +387,13 @@ { "cell_type": "code", "execution_count": 3, - "id": "a7e667de", + "id": "ed671c87", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T16:13:44.374835Z", - "iopub.status.busy": "2026-09-28T16:13:44.374724Z", - "iopub.status.idle": "2026-09-28T16:13:47.582309Z", - "shell.execute_reply": "2026-09-28T16:13:47.582076Z" + "iopub.execute_input": "2026-09-29T09:19:50.399757Z", + "iopub.status.busy": "2026-09-29T09:19:50.399657Z", + "iopub.status.idle": "2026-09-29T09:19:53.339577Z", + "shell.execute_reply": "2026-09-29T09:19:53.339355Z" } }, "outputs": [ @@ -429,7 +429,7 @@ }, { "cell_type": "markdown", - "id": "81a3e8fa", + "id": "d7550774", "metadata": {}, "source": [ "## 4. Score the trained models and the constructions under one spec" @@ -437,7 +437,7 @@ }, { "cell_type": "markdown", - "id": "9b9b4673", + "id": "cdaf60e3", "metadata": {}, "source": [ "Four published checkpoints at training seed one. Two of them were trained with a\n", @@ -452,20 +452,20 @@ { "cell_type": "code", "execution_count": 4, - "id": "776e8be7", + "id": "13d92b40", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T16:13:47.584517Z", - "iopub.status.busy": "2026-09-28T16:13:47.584417Z", - "iopub.status.idle": "2026-09-28T17:10:49.830356Z", - "shell.execute_reply": "2026-09-28T17:10:49.830118Z" + "iopub.execute_input": "2026-09-29T09:19:53.341784Z", + "iopub.status.busy": "2026-09-29T09:19:53.341697Z", + "iopub.status.idle": "2026-09-29T09:22:29.216672Z", + "shell.execute_reply": "2026-09-29T09:22:29.216442Z" } }, "outputs": [ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "d59b6c22ef814f8ab1b59ba34ec2b954", + "model_id": "fdffa82ec9bd43a49684f69d954dbf18", "version_major": 2, "version_minor": 0 }, @@ -479,7 +479,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "7c456c29bcad4acca9ca51dd441d0acb", + "model_id": "0f4834022f2e499196f1fc6813a55962", "version_major": 2, "version_minor": 0 }, @@ -493,7 +493,7 @@ { "data": { "application/vnd.jupyter.widget-view+json": { - "model_id": "f5323a232e7c433f88c3b34a89fa2eb5", + "model_id": "7c805e3c39594c6aa8c1cc697f73be7d", "version_major": 2, "version_minor": 0 }, @@ -671,16 +671,30 @@ { "cell_type": "code", "execution_count": 5, - "id": "13578188", + "id": "1c9be14a", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T17:10:49.832935Z", - "iopub.status.busy": "2026-09-28T17:10:49.832839Z", - "iopub.status.idle": "2026-09-28T20:47:28.874800Z", - "shell.execute_reply": "2026-09-28T20:47:28.874563Z" + "iopub.execute_input": "2026-09-29T09:22:29.219834Z", + "iopub.status.busy": "2026-09-29T09:22:29.219741Z", + "iopub.status.idle": "2026-09-29T09:27:42.199916Z", + "shell.execute_reply": "2026-09-29T09:27:42.199651Z" } }, "outputs": [ + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "d116c8f9594549a9b8684d6ff7d865a6", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Fetching 3 files: 0%| | 0/3 [00:00question\n", " direction\n", " picks\n", - " real designs score\n", + " split-half reference\n", " \n", " \n", " \n", @@ -1121,7 +1135,7 @@ " \n", " \n", " viol\n", - " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible.\n", + " Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied.\n", " lower is better\n", " vqgan\n", " None\n", @@ -1255,7 +1269,7 @@ "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", - "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. \n", "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", @@ -1274,29 +1288,29 @@ "gap_after_10_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", "\n", - " direction picks real designs score \n", - "mmd lower is better permuted_conditions None \n", - "coverage higher is better tie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditions None \n", - "vendi higher is better unconditional None \n", - "dpp higher is better unconditional None \n", - "viol lower is better vqgan None \n", - "volume_error lower is better vqgan None \n", - "per_condition_distance lower is better lookup_table None \n", - "cond_sens diagnostic None \n", - "train_distance diagnostic None \n", - "train_distance_ratio diagnostic None \n", - "generation_seconds lower is better gan_cnn_2d None \n", - "n_parameters lower is better cgan_cnn_2d None \n", - "train_minutes lower is better None \n", - "iog lower is better diffusion_2d_cond None \n", - "cog lower is better lookup_table None \n", - "fog lower is better diffusion_2d_cond None \n", - "calls_to_near_optimum lower is better vqgan None \n", - "gap_after_1_calls lower is better diffusion_2d_cond None \n", - "gap_after_2_calls lower is better vqgan None \n", - "gap_after_5_calls lower is better vqgan None \n", - "gap_after_10_calls lower is better vqgan None \n", - "reaches_reference_rate higher is better diffusion_2d_cond None " + " direction picks split-half reference \n", + "mmd lower is better permuted_conditions None \n", + "coverage higher is better tie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditions None \n", + "vendi higher is better unconditional None \n", + "dpp higher is better unconditional None \n", + "viol lower is better vqgan None \n", + "volume_error lower is better vqgan None \n", + "per_condition_distance lower is better lookup_table None \n", + "cond_sens diagnostic None \n", + "train_distance diagnostic None \n", + "train_distance_ratio diagnostic None \n", + "generation_seconds lower is better gan_cnn_2d None \n", + "n_parameters lower is better cgan_cnn_2d None \n", + "train_minutes lower is better None \n", + "iog lower is better diffusion_2d_cond None \n", + "cog lower is better lookup_table None \n", + "fog lower is better diffusion_2d_cond None \n", + "calls_to_near_optimum lower is better vqgan None \n", + "gap_after_1_calls lower is better diffusion_2d_cond None \n", + "gap_after_2_calls lower is better vqgan None \n", + "gap_after_5_calls lower is better vqgan None \n", + "gap_after_10_calls lower is better vqgan None \n", + "reaches_reference_rate higher is better diffusion_2d_cond None " ] }, "execution_count": 7, @@ -1310,7 +1324,7 @@ }, { "cell_type": "markdown", - "id": "773639bf", + "id": "ab758e29", "metadata": {}, "source": [ "Why these seven rows: four trained models and three constructions whose right\n", @@ -1377,7 +1391,7 @@ }, { "cell_type": "markdown", - "id": "5c0f9320", + "id": "86fa6b15", "metadata": {}, "source": [ "## 6. Rank, but only on a column that can be ranked" @@ -1386,13 +1400,13 @@ { "cell_type": "code", "execution_count": 8, - "id": "6de7f90a", + "id": "556ad06c", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T20:48:54.474424Z", - "iopub.status.busy": "2026-09-28T20:48:54.474342Z", - "iopub.status.idle": "2026-09-28T20:48:54.477351Z", - "shell.execute_reply": "2026-09-28T20:48:54.477162Z" + "iopub.execute_input": "2026-09-29T09:29:08.300718Z", + "iopub.status.busy": "2026-09-29T09:29:08.300641Z", + "iopub.status.idle": "2026-09-29T09:29:08.303791Z", + "shell.execute_reply": "2026-09-29T09:29:08.303584Z" } }, "outputs": [ @@ -1493,13 +1507,13 @@ { "cell_type": "code", "execution_count": 9, - "id": "9ea49f63", + "id": "31513288", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T20:48:54.479760Z", - "iopub.status.busy": "2026-09-28T20:48:54.479696Z", - "iopub.status.idle": "2026-09-28T20:48:54.481228Z", - "shell.execute_reply": "2026-09-28T20:48:54.481028Z" + "iopub.execute_input": "2026-09-29T09:29:08.306192Z", + "iopub.status.busy": "2026-09-29T09:29:08.306129Z", + "iopub.status.idle": "2026-09-29T09:29:08.307671Z", + "shell.execute_reply": "2026-09-29T09:29:08.307506Z" } }, "outputs": [ @@ -1520,7 +1534,7 @@ }, { "cell_type": "markdown", - "id": "dd3092d7", + "id": "a53a253a", "metadata": {}, "source": [ "## 7. The same questions in another space" @@ -1528,7 +1542,7 @@ }, { "cell_type": "markdown", - "id": "b3c40d42", + "id": "f08ba1a9", "metadata": {}, "source": [ "Where a metric is computed is an argument, not a property of the metric. The board\n", @@ -1546,13 +1560,13 @@ { "cell_type": "code", "execution_count": 10, - "id": "40bc912f", + "id": "3ae7f494", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T20:48:54.483878Z", - "iopub.status.busy": "2026-09-28T20:48:54.483810Z", - "iopub.status.idle": "2026-09-28T20:48:55.193824Z", - "shell.execute_reply": "2026-09-28T20:48:55.193585Z" + "iopub.execute_input": "2026-09-29T09:29:08.309937Z", + "iopub.status.busy": "2026-09-29T09:29:08.309870Z", + "iopub.status.idle": "2026-09-29T09:29:09.062834Z", + "shell.execute_reply": "2026-09-29T09:29:09.062604Z" } }, "outputs": [ @@ -1677,7 +1691,7 @@ }, { "cell_type": "markdown", - "id": "03463c3e", + "id": "3391d356", "metadata": {}, "source": [ "## 8. Registering a metric of your own" @@ -1685,7 +1699,7 @@ }, { "cell_type": "markdown", - "id": "bad3937b", + "id": "5ecea0fe", "metadata": {}, "source": [ "A metric is a plain function. The evaluator calls it once per model and hands it one\n", @@ -1709,13 +1723,13 @@ { "cell_type": "code", "execution_count": 11, - "id": "adec143c", + "id": "277a5953", "metadata": { "execution": { - "iopub.execute_input": "2026-09-28T20:48:55.198263Z", - "iopub.status.busy": "2026-09-28T20:48:55.198168Z", - "iopub.status.idle": "2026-09-28T20:48:55.204228Z", - "shell.execute_reply": "2026-09-28T20:48:55.204032Z" + "iopub.execute_input": "2026-09-29T09:29:09.065670Z", + "iopub.status.busy": "2026-09-29T09:29:09.065580Z", + "iopub.status.idle": "2026-09-29T09:29:09.071959Z", + "shell.execute_reply": "2026-09-29T09:29:09.071778Z" } }, "outputs": [ @@ -1743,7 +1757,7 @@ " question\n", " direction\n", " picks\n", - " real designs score\n", + " split-half reference\n", " \n", " \n", " \n", @@ -1766,9 +1780,9 @@ "" ], "text/plain": [ - " question direction picks real designs score\n", - "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. lower is better vqgan None\n", - "density_ratio Material fraction of each generated design over that of the reference optima. One means the same amount of material. diagnostic None" + " question direction picks split-half reference\n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. lower is better vqgan None\n", + "density_ratio Material fraction of each generated design over that of the reference optima. One means the same amount of material. diagnostic None" ] }, "execution_count": 11, @@ -1793,7 +1807,7 @@ }, { "cell_type": "markdown", - "id": "f07bc0f9", + "id": "02e4c560", "metadata": {}, "source": [ "It appears with its question and direction attached, scored under the same spec as\n", @@ -1804,7 +1818,7 @@ }, { "cell_type": "markdown", - "id": "39f2d3f8", + "id": "5f536feb", "metadata": {}, "source": [ "## 9. Not in this notebook" @@ -1812,7 +1826,7 @@ }, { "cell_type": "markdown", - "id": "4e3e0524", + "id": "f452d664", "metadata": {}, "source": [ "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec. It\n", From 481fd39df8a8fd698939fa6088449a8dbb57912b Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Wed, 30 Sep 2026 15:13:51 +0200 Subject: [PATCH 17/18] Address the second review: one CLI default, a volume budget the board knows, one bandwidth per board Three fixes, one per comment on the second review round. The CLI still built its own default spec as problem_id/v1 after the library default moved to v2, so a bare run selected a metric list the registry no longer serves. The CLI now passes --spec through and EvalSpec.load owns the only default, with a regression test pinning the no-flag path. Board never told its contexts which condition is the volume budget, so volume_error returned NaN on the saved-design path even when targets were supplied. Board gains an explicit volume_condition field, declared rather than guessed, and from_evaluator carries it over from the spec. When no bandwidth was pinned, each row inferred kernel_sigma from its own reference set, so the split-half row was scored under a different bandwidth as well as a different sample size; a half landing in one cluster hit the 1e-6 floor. evaluate now resolves the bandwidth once, from the training designs or the full reference set in the scored space, and every row shares it. Co-Authored-By: Claude Fable 5 --- engiopt/evaluate.py | 6 ++--- engiopt/evaluation/__init__.py | 2 +- engiopt/evaluation/board.py | 23 +++++++++++++++-- engiopt/evaluation/evaluator.py | 4 +-- tests/test_board.py | 33 +++++++++++++++++++++++++ tests/test_evaluate_default_spec.py | 38 +++++++++++++++++++++++++++++ tests/test_metrics_suite.py | 5 ++-- 7 files changed, 101 insertions(+), 10 deletions(-) create mode 100644 tests/test_evaluate_default_spec.py diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index 8a54f97a..00f89a25 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -64,7 +64,8 @@ class Args: cgan_cnn_2d:023dd1fb gan_cnn_2d:06d9a9a1` asks for exactly the two that exist.""" spec: str | None = None - """Eval spec reference, e.g. `beams2d/v1`. Defaults to `/v1`.""" + """Eval spec reference, e.g. `beams2d/v2`. Omitted, the problem's current spec loads; + `EvalSpec.load` owns that default, so the CLI cannot drift from the library.""" metrics: tuple[str, ...] = () """Metric names; defaults to the spec's list.""" include_expensive: bool = False @@ -341,8 +342,7 @@ def main(args: Args) -> int: if args.list_generators or args.list_metrics: return 0 - spec = args.spec or f"{args.problem_id}/v1" - evaluator = Evaluator.for_problem(args.problem_id, spec=spec) + evaluator = Evaluator.for_problem(args.problem_id, spec=args.spec) print(f"Problem {args.problem_id} | spec {evaluator.spec.version} | n={evaluator.spec.n_samples}") generators = _load_generators(args, evaluator) diff --git a/engiopt/evaluation/__init__.py b/engiopt/evaluation/__init__.py index 3d0f9e44..3ea24c30 100644 --- a/engiopt/evaluation/__init__.py +++ b/engiopt/evaluation/__init__.py @@ -3,7 +3,7 @@ from engiopt.evaluation import Evaluator from engiopt.utils.all_generators import BUILTIN_GENERATORS - ev = Evaluator.for_problem("beams2d", spec="beams2d/v1") + ev = Evaluator.for_problem("beams2d", spec="beams2d/v2") gen = BUILTIN_GENERATORS["cgan_cnn_2d"].from_pretrained(ev.problem, problem_id="beams2d", seed=1) ev.score(gen) diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 92b4ac0e..4ebf16b3 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -23,6 +23,7 @@ import numpy as np import pandas as pd +from engiopt import metrics as metrics_mod from engiopt.evaluation.context import EvaluationContext from engiopt.evaluation.registry import MetricRegistry from engiopt.evaluation.registry import METRICS @@ -67,12 +68,18 @@ class Board: design, as anything indexable by condition name and by row. Needed by `viol` on problems whose constraint check reads the conditions, and by `volume_error`; without them those columns are blank. + volume_condition: Name of the condition holding the volume-fraction + budget, e.g. beams2d's `volfrac`. Declared rather than guessed, as + on `EvalSpec`; without it `volume_error` is blank even when the + conditions carry a budget. train: The training designs, for the memorization metrics. Optional. sigma: Kernel bandwidth for the distribution and diversity metrics. None, the default, uses the median pairwise distance of the training designs in whichever space is being scored, or of the reference designs when no training split was given, so the kernel is neither saturated nor empty - on a problem it has never seen. + on a problem it has never seen. Resolved once per `evaluate`, so every + row, the split-half reference row included, is scored under the same + bandwidth. frame: The most recent `evaluate` result. designs: The designs behind each row of `frame`, when the board sampled them itself. """ @@ -80,6 +87,7 @@ class Board: problem: Any reference: Any conditions: Any | None = None + volume_condition: str | None = None train: Any | None = None sigma: float | None = None registry: MetricRegistry = field(default_factory=lambda: METRICS) @@ -121,6 +129,15 @@ def evaluate( raise KeyError(f"Unknown space {space!r}. Registered: {', '.join(SPACES)}") project = SPACES[space](reference, width) + # One bandwidth for the whole board. Each context would otherwise infer + # its own from whatever reference set it holds, and the split-half row + # holds half the designs, so its kernel columns would differ from the + # model rows in bandwidth as well as in sample size. + sigma = self.sigma + if sigma is None: + basis = project(np.asarray(self.train if self.train is not None else reference)) + sigma = metrics_mod.compute_median_sigma(basis.reshape(len(basis), -1)) + def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: ctx = EvaluationContext( problem=self.problem, @@ -128,7 +145,8 @@ def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: gen_designs=project(np.asarray(generated)), ref_designs=project(np.asarray(against)), conditions=self.conditions if against is reference else None, - sigma=self.sigma, + volume_condition=self.volume_condition, + sigma=sigma, aggregation=aggregation, train_designs_fn=(lambda: project(np.asarray(self.train))) if self.train is not None else None, ) @@ -196,6 +214,7 @@ def from_evaluator( problem=evaluator.problem, reference=evaluator.resolved.ref_designs, conditions=evaluator.resolved.conditions, + volume_condition=evaluator.spec.volume_condition, sigma=evaluator.spec.sigma, registry=evaluator.registry, ) diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 927f8090..3dc7acf3 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -4,7 +4,7 @@ they loaded and called their model -- which is now the `Generator` contract -- so everything else lives here once:: - ev = Evaluator.for_problem("beams2d", spec="beams2d/v1") + ev = Evaluator.for_problem("beams2d", spec="beams2d/v2") row = ev.score(generator) # cheap metrics only board = ev.leaderboard(zoo) # a DataFrame, one row per model """ @@ -128,7 +128,7 @@ def for_problem( Args: problem_id: EngiBench problem registry key. spec: `"/"`, an `EvalSpec`, or None to load - `"/v1"`. + the problem's current spec version. device: Torch device; auto-selected when omitted. registry: Metric registry override, useful in tests. """ diff --git a/tests/test_board.py b/tests/test_board.py index cd3ac5cc..14f3a7bf 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -84,6 +84,39 @@ def test_aggregation_is_a_policy_on_the_context(fake_problem: Any) -> None: assert median_ctx.reduce(values) == pytest.approx(0.1) +def test_volume_error_reaches_saved_designs(fake_problem: Any, designs: Any) -> None: + """A board told which condition is the volume budget scores it; one not told leaves it blank.""" + models, ref, _ = designs + conditions = {"volfrac": [float(design.mean()) for design in ref]} + told = Board(fake_problem, reference=ref, conditions=conditions, volume_condition="volfrac") + frame = told.evaluate({"exact": ref.copy()}, metrics=["volume_error"]) + assert frame.loc["exact", "volume_error"] == pytest.approx(0.0), "designs at their own budgets have no error" + untold = Board(fake_problem, reference=ref, conditions=conditions) + assert np.isnan(untold.evaluate(models, metrics=["volume_error"]).loc["honest", "volume_error"]) + + +def test_every_row_shares_one_inferred_bandwidth(fake_problem: Any, designs: Any, monkeypatch: pytest.MonkeyPatch) -> None: + """The split-half row must not size its kernel from its own half. + + Left to each context, a half landing in one cluster of the reference set + infers a degenerate bandwidth (the 1e-6 floor), so its kernel columns would + differ from the model rows in bandwidth as well as in sample size. + """ + from engiopt import metrics as metrics_mod + + models, ref, _ = designs + calls: list[int] = [] + real = metrics_mod.compute_median_sigma + + def spy(x: Any, y: Any = None) -> float: + calls.append(len(x)) + return real(x, y) + + monkeypatch.setattr(metrics_mod, "compute_median_sigma", spy) + Board(fake_problem, reference=ref).evaluate(models) + assert calls == [len(ref)], "one bandwidth, sized from the full reference set, for every row" + + def test_the_default_bandwidth_lets_diversity_metrics_see_anything(fake_problem: Any, designs: Any) -> None: """At a fixed sigma=10 every design looks identical to every other; the training-data median does not.""" models, ref, _ = designs diff --git a/tests/test_evaluate_default_spec.py b/tests/test_evaluate_default_spec.py new file mode 100644 index 00000000..d3992b40 --- /dev/null +++ b/tests/test_evaluate_default_spec.py @@ -0,0 +1,38 @@ +"""The CLI must not carry its own default spec version. + +The library default moved to v2 while `evaluate.py` still built +`f"{problem_id}/v1"`, so a bare `python -m engiopt.evaluate` selected a spec +whose metric list no longer matched the registry (`KeyError: 'novelty'`). +The default lives in `EvalSpec.load` alone; the CLI passes `--spec` through. +""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from engiopt import evaluate +from engiopt.evaluation.spec import EvalSpec + + +class _StopAtSpecError(Exception): + """Raised by the fake to stop `main` before it loads anything real.""" + + +def test_the_cli_without_a_spec_defers_to_the_library_default(monkeypatch: pytest.MonkeyPatch) -> None: + seen: dict[str, Any] = {} + + def fake_for_problem(problem_id: str, *, spec: Any = None, **_: Any) -> Any: + seen["spec"] = spec + raise _StopAtSpecError + + monkeypatch.setattr(evaluate.Evaluator, "for_problem", fake_for_problem) + with pytest.raises(_StopAtSpecError): + evaluate.main(evaluate.Args()) + assert seen["spec"] is None, "no --spec means the library default, not a version the CLI invents" + + +def test_a_bare_problem_reference_loads_the_current_spec_version() -> None: + """`EvalSpec.load("beams2d")` is what the CLI's no-flag path now resolves.""" + assert EvalSpec.load("beams2d").version == EvalSpec.version, "a bare reference follows the dataclass default" diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py index 603b03f7..a68f88f4 100644 --- a/tests/test_metrics_suite.py +++ b/tests/test_metrics_suite.py @@ -201,7 +201,7 @@ def __init__(self, designs: Any) -> None: class StubEvaluator: problem = None registry = METRICS - spec = type("Spec", (), {"sigma": 1.0})() + spec = type("Spec", (), {"sigma": 1.0, "volume_condition": "volfrac"})() resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() def context_for(self, generator: Any) -> StubContext: @@ -220,6 +220,7 @@ def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict assert board.rank("mmd").index[0] == "b" assert board.frame.loc["probe", "iog"] == 1.0 assert set(board.designs) == {"a", "b", "probe"} + assert board.volume_condition == "volfrac", "re-scoring the kept designs must still see the volume budget" def test_from_evaluator_keeps_the_evaluators_registry() -> None: @@ -233,7 +234,7 @@ def test_from_evaluator_keeps_the_evaluators_registry() -> None: class StubEvaluator: problem = None registry = custom - spec = type("Spec", (), {"sigma": None})() + spec = type("Spec", (), {"sigma": None, "volume_condition": None})() resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() def context_for(self, generator: Any) -> Any: From 876e09422451d3a71da4bf6b0f02660bbb4ecfd7 Mon Sep 17 00:00:00 2001 From: Matthew Keeler Date: Thu, 1 Oct 2026 10:40:23 +0200 Subject: [PATCH 18/18] Let the board take the test rows themselves, and the evaluator a chosen slice The reviewer showed that a plain dict of condition columns feeds volume_error and crashes viol, because the two metrics read conditions differently: one by column name, one by row. The deeper problem is that the briefs already sit beside the optimal designs in the dataset, and slicing the designs into a bare array is what loses them. Three pieces, smallest surface first. ConditionTable wraps a dict of columns or a DataFrame so both access patterns work, and a conditions object whose length disagrees with the reference is refused, since a short answer key silently grades design i against the wrong brief. Board(reference=rows) accepts a dataset slice directly, splitting it into designs and briefs so the two cannot fall out of alignment. Evaluator.for_rows swaps the spec's seeded draw for rows the user chose, keeping the rest of the spec in force; generators are sampled under exactly those conditions, the old evaluate-script flow. The digest is cleared on that path, because it certifies the frozen draw and a custom slice is not leaderboard material. Board.load grows a rows argument that routes through it. Co-Authored-By: Claude Fable 5 --- engiopt/evaluation/board.py | 81 +++++++++++++++++++++++++++++++-- engiopt/evaluation/evaluator.py | 58 +++++++++++++++++++++++ tests/conftest.py | 4 ++ tests/test_board.py | 39 ++++++++++++++++ tests/test_evaluator_rows.py | 59 ++++++++++++++++++++++++ 5 files changed, 236 insertions(+), 5 deletions(-) create mode 100644 tests/test_evaluator_rows.py diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py index 4ebf16b3..f701c019 100644 --- a/engiopt/evaluation/board.py +++ b/engiopt/evaluation/board.py @@ -46,6 +46,47 @@ def register_space(name: str, fit: SpaceFit) -> None: SPACES[name] = fit +DESIGN_COLUMN = "optimal_design" +"""The dataset column holding each row's optimal design, EngiBench's convention. +A dataset slice passed as `Board(reference=...)` is split on it: this column +becomes the reference designs, the remaining columns become the conditions.""" + + +class ConditionTable: + """A dict of condition columns, readable by column name and by row. + + Metrics ask two different questions of the conditions. `volume_error` wants + a whole column at once (`table["volfrac"]`); `viol` wants one design's full + brief (`table[0]`). A HuggingFace dataset answers both, a plain dict only + the first, so the board wraps dicts here rather than asking every notebook + to build a dataset. + """ + + def __init__(self, columns: Mapping[str, Any]): + self.columns = dict(columns) + + def __getitem__(self, key: Any) -> Any: + if isinstance(key, str): + return self.columns[key] + return {name: column[key] for name, column in self.columns.items()} + + def __len__(self) -> int: + return len(next(iter(self.columns.values()))) + + +def _as_condition_table(conditions: Any) -> Any: + """Give loose condition inputs the row access the metrics need. + + A dict of columns or a DataFrame is wrapped; anything else, such as a + HuggingFace dataset or None, already behaves and passes through. + """ + if isinstance(conditions, pd.DataFrame): + conditions = {name: conditions[name].tolist() for name in conditions.columns} + if isinstance(conditions, Mapping): + return ConditionTable(conditions) + return conditions + + REFERENCE_ROW = "reference (split-half)" """Label of the row scoring one random half of the reference designs against the other. @@ -63,11 +104,17 @@ class Board: Attributes: problem: The problem the designs answer; needs `design_space` and `check_constraints`, which every EngiBench problem has. - reference: The withheld optimal designs the models are scored against. + reference: The withheld optimal designs the models are scored against, + as an array of designs, or as the test rows themselves: a dataset + slice carrying an `optimal_design` column. The slice is split on + construction, designs out of that column and conditions out of the + rest, so design `i` and its brief cannot fall out of alignment. conditions: The conditions each reference design answers, one row per - design, as anything indexable by condition name and by row. Needed - by `viol` on problems whose constraint check reads the conditions, - and by `volume_error`; without them those columns are blank. + design: a dict of columns, a DataFrame, or anything already + indexable by condition name and by row, such as a dataset slice. + Needed by `viol` on problems whose constraint check reads the + conditions, and by `volume_error`; without them those columns are + blank. Derived automatically when `reference` is a dataset slice. volume_condition: Name of the condition holding the volume-fraction budget, e.g. beams2d's `volfrac`. Declared rather than guessed, as on `EvalSpec`; without it `volume_error` is blank even when the @@ -94,6 +141,23 @@ class Board: frame: pd.DataFrame = field(default_factory=pd.DataFrame) designs: dict[str, Any] = field(default_factory=dict) + def __post_init__(self) -> None: + """Split a dataset-slice reference, wrap loose conditions, check alignment.""" + if hasattr(self.reference, "column_names") and DESIGN_COLUMN in self.reference.column_names: + rows = self.reference + self.reference = np.asarray(rows[DESIGN_COLUMN]) + if self.conditions is None: + briefs = {name: rows[name] for name in rows.column_names if name != DESIGN_COLUMN} + self.conditions = ConditionTable(briefs) + self.conditions = _as_condition_table(self.conditions) + if self.conditions is not None and hasattr(self.conditions, "__len__"): + n_designs = len(np.asarray(self.reference)) + if len(self.conditions) != n_designs: + raise ValueError( + f"{len(self.conditions)} condition rows for {n_designs} reference designs. " + f"Conditions do not select designs; row i must be the brief design i answers." + ) + def evaluate( self, models: Mapping[str, Any], @@ -231,6 +295,7 @@ def load( designs: Mapping[str, Any] | None = None, seed: int = 1, spec: Any = None, + rows: Any = None, expensive: bool = False, ) -> Board: """Pull published checkpoints by name and score them: the workshop in one call. @@ -243,6 +308,9 @@ def load( designs: Extra rows of saved designs, one per spec condition; see `from_evaluator`. seed: Training seed of the checkpoints to load. spec: Spec reference such as `"beams2d/v2"`, or an `EvalSpec`; defaults to the current spec. + rows: A dataset slice to score against instead of the spec's drawn + conditions; see `Evaluator.for_rows`. None, the default, keeps + the spec's seeded 50-row draw from the test split. expensive: Whether to run the simulator-backed metrics. Returns: @@ -251,7 +319,10 @@ def load( from engiopt.evaluation.evaluator import Evaluator from engiopt.utils.all_generators import BUILTIN_GENERATORS - evaluator = Evaluator.for_problem(problem_id, spec=spec) + if rows is not None: + evaluator = Evaluator.for_rows(problem_id, rows, spec=spec) + else: + evaluator = Evaluator.for_problem(problem_id, spec=spec) requested = models if isinstance(models, Mapping) else dict.fromkeys(models) generators = { name: BUILTIN_GENERATORS[name].from_pretrained( diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 3dc7acf3..41d58c19 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -148,6 +148,64 @@ def for_problem( registry=registry or METRICS, ) + @classmethod + def for_rows( + cls, + problem_id: str, + rows: Any, + *, + spec: str | EvalSpec | None = None, + device: th.device | None = None, + registry: MetricRegistry | None = None, + ) -> Evaluator: + """Build an evaluator whose reference is a dataset slice you chose. + + The spec's own draw is the leaderboard contract: a seeded, digest-checked + sample of the test split. This constructor swaps only that draw for the + rows you pass, a region of condition space, a harder subset, your own + split, and keeps everything else the spec declares: the metric list, the + aggregation, the volume condition, the bandwidth policy. Generators are + then sampled under exactly these rows' conditions and scored against + these rows' optimal designs, one for one. + + Numbers from a custom slice are not leaderboard rows; the frozen draw is + what published numbers mean, so the digest is cleared rather than lied to. + + Args: + problem_id: EngiBench problem registry key. + rows: A dataset slice with an `optimal_design` column and the + problem's condition columns, e.g. `problem.dataset["test"].select(...)`. + spec: `"/"`, an `EvalSpec`, or None to load + the problem's current spec version. + device: Torch device; auto-selected when omitted. + registry: Metric registry override, useful in tests. + """ + from engibench.utils.all_problems import BUILTIN_PROBLEMS + import torch as th + + from engiopt.transforms import get_scalar_condition_keys + + problem = BUILTIN_PROBLEMS[problem_id]() + eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or problem_id) + if eval_spec.problem_id != problem_id: + raise ValueError(f"Spec is for {eval_spec.problem_id!r}, not {problem_id!r}.") + device = device or pick_device() + problem.reset(seed=eval_spec.condition_seed) + + available = [key for key in problem.conditions_keys if key in rows.column_names] + scalar_keys = get_scalar_condition_keys(problem, rows) + conditions = rows.select_columns(available) + tensor = th.tensor([conditions[key] for key in scalar_keys], dtype=th.float32, device=device).T + resolved = ResolvedSpec( + spec=dataclasses.replace(eval_spec, n_samples=len(rows), condition_digest=None), + conditions_tensor=tensor, + conditions=conditions, + ref_designs=np.asarray(rows["optimal_design"]), + indices=np.arange(len(rows)), + condition_keys=tuple(scalar_keys), + ) + return cls(problem=problem, problem_id=problem_id, resolved=resolved, device=device, registry=registry or METRICS) + @property def spec(self) -> EvalSpec: """The frozen evaluation contract in force.""" diff --git a/tests/conftest.py b/tests/conftest.py index 9afbcb6e..ab57faf3 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -131,6 +131,10 @@ def select(self, indices: Any) -> FakeDataset: order = list(indices) return FakeDataset({name: [values[i] for i in order] for name, values in self._columns.items()}) + def select_columns(self, names: Any) -> FakeDataset: + """Keep only the named columns, as `datasets.Dataset.select_columns` does.""" + return FakeDataset({name: self._columns[name] for name in names}) + @pytest.fixture def mixed_condition_dataset() -> FakeDataset: diff --git a/tests/test_board.py b/tests/test_board.py index 14f3a7bf..13dba630 100644 --- a/tests/test_board.py +++ b/tests/test_board.py @@ -95,6 +95,45 @@ def test_volume_error_reaches_saved_designs(fake_problem: Any, designs: Any) -> assert np.isnan(untold.evaluate(models, metrics=["volume_error"]).loc["honest", "volume_error"]) +def test_dict_conditions_survive_a_full_evaluate(fake_problem: Any, designs: Any) -> None: + """A plain dict of columns must feed row-reading metrics too. + + `viol` asks for one design's brief at a time (`conditions[0]`), which a bare + dict answers with `KeyError: 0`. The board wraps the dict, so the same input + that fed `volume_error` survives a default evaluate. + """ + models, ref, _ = designs + conditions = {"volfrac": [float(design.mean()) for design in ref]} + board = Board(fake_problem, reference=ref, conditions=conditions, volume_condition="volfrac") + frame = board.evaluate(models) + assert not np.isnan(frame.loc["honest", "viol"]), "the row-access path must work, not just the column path" + assert not np.isnan(frame.loc["honest", "volume_error"]) + + +def test_a_dataset_slice_is_designs_and_briefs_at_once(fake_problem: Any) -> None: + """Passing the test rows keeps design i and brief i aligned by construction. + + The briefs already sit beside the optimal designs in the dataset; slicing + the designs into an array is what loses them. Handing the board the rows + themselves means nothing is lost and nothing can be misaligned. + """ + rows = fake_problem.dataset["train"] + board = Board(fake_problem, reference=rows, volume_condition="volfrac") + assert board.reference.shape == (3, 8, 10), "the design column became the reference array" + assert board.conditions is not None + assert board.conditions["volfrac"] == [0.5, 0.5, 0.5], "the remaining columns became the briefs" + frame = board.evaluate({"exact": board.reference.copy()}, metrics=["volume_error"]) + # The dataset's designs are flat fields at 0.1, 0.2 and 0.3 against a 0.5 budget. + assert frame.loc["exact", "volume_error"] == pytest.approx(np.mean([0.4, 0.3, 0.2])) + + +def test_misaligned_conditions_are_refused(fake_problem: Any, designs: Any) -> None: + """Conditions are an answer key, not a filter; a short one is wrong labels, not a subset.""" + _, ref, _ = designs + with pytest.raises(ValueError, match="do not select"): + Board(fake_problem, reference=ref, conditions={"volfrac": [0.2, 0.3]}) + + def test_every_row_shares_one_inferred_bandwidth(fake_problem: Any, designs: Any, monkeypatch: pytest.MonkeyPatch) -> None: """The split-half row must not size its kernel from its own half. diff --git a/tests/test_evaluator_rows.py b/tests/test_evaluator_rows.py new file mode 100644 index 00000000..9db81ee4 --- /dev/null +++ b/tests/test_evaluator_rows.py @@ -0,0 +1,59 @@ +"""`Evaluator.for_rows`: score against a dataset slice you chose. + +The spec's own draw stays the leaderboard contract. This path swaps only the +draw: your rows' conditions drive the sampling, your rows' optimal designs are +the reference, and everything else the spec declares stays in force. +""" + +from __future__ import annotations + +import numpy as np +import pytest + +from engiopt.evaluation.evaluator import Evaluator +from engiopt.evaluation.spec import EvalSpec +from tests.conftest import FakeDataset +from tests.conftest import FakeProblem + + +@pytest.fixture +def rows() -> FakeDataset: + """Two hand-picked test rows with distinct briefs and known optima.""" + shape = (8, 10) + return FakeDataset( + { + "volfrac": [0.2, 0.3], + "rmin": [2.0, 2.0], + "optimal_design": [np.full(shape, 0.2), np.full(shape, 0.3)], + } + ) + + +@pytest.fixture +def evaluator(rows: FakeDataset, monkeypatch: pytest.MonkeyPatch) -> Evaluator: + from engibench.utils.all_problems import BUILTIN_PROBLEMS + + monkeypatch.setitem(BUILTIN_PROBLEMS, "fake2d", FakeProblem) + spec = EvalSpec(problem_id="fake2d", volume_condition="volfrac") + return Evaluator.for_rows("fake2d", rows, spec=spec) + + +def test_the_rows_become_the_reference_and_the_briefs(evaluator: Evaluator, rows: FakeDataset) -> None: + assert evaluator.resolved.ref_designs.shape == (2, 8, 10) + assert evaluator.resolved.ref_designs[0].mean() == pytest.approx(0.2) + assert evaluator.resolved.conditions["volfrac"] == [0.2, 0.3] + assert evaluator.resolved.conditions_tensor.shape == (2, 2), "one row per design, one column per scalar condition" + assert evaluator.spec.n_samples == len(rows), "generators sample one design per chosen row" + + +def test_a_custom_slice_is_not_the_frozen_contract(evaluator: Evaluator) -> None: + """The digest certifies the spec's own draw; a custom draw must not inherit it.""" + assert evaluator.spec.condition_digest is None + assert evaluator.spec.volume_condition == "volfrac", "everything but the draw survives from the spec" + + +def test_designs_are_scored_against_the_chosen_rows(evaluator: Evaluator) -> None: + """A batch equal to the rows' own optima has zero volume error under their briefs.""" + ctx = evaluator.context_for_designs(evaluator.resolved.ref_designs.copy()) + row = evaluator.score_context(ctx, only=["volume_error"], include_expensive=False) + assert row["volume_error"] == pytest.approx(0.0)