diff --git a/CONTRIBUTING_A_MODEL.md b/CONTRIBUTING_A_MODEL.md index 87c822fa..0f0e78e5 100644 --- a/CONTRIBUTING_A_MODEL.md +++ b/CONTRIBUTING_A_MODEL.md @@ -175,25 +175,29 @@ python -m engiopt.evaluate --list-generators # your model should ap python -m engiopt.evaluate --problem-id beams2d --generators my_model --hf-entity my-hf-username ``` -The default pass runs the cheap metrics: `mmd`, `dpp`, `viol`, and the two -integrity checks (`novelty`, `cond_sens`). Feasibility is among them because it -describes the design as generated, so it is a constraint check rather than a -solver run. `--include-expensive` adds the optimality gaps (`iog`, `cog`, -`fog`), which do run the optimizer and are slow — leave them off while -iterating. +The default pass runs every metric that needs no simulator: the set-level +ones (`mmd`, `coverage`, `vendi`, `dpp`), the per-condition ones +(`per_condition_distance`, `volume_error`), feasibility (`viol`), the two +integrity checks (`train_distance`, `cond_sens`), and the cost +columns. `--include-expensive` adds the columns that re-optimize from each +generated design — the optimality gaps `iog`, `cog`, `fog` and the +call-budget metrics `calls_to_near_optimum`, `gap_after_calls` and +`reaches_reference_rate` — which run the optimizer and are +slow; leave them off while iterating. `python -m engiopt.evaluate +--list-metrics` prints the question each column answers. Watch two columns while you develop: - **`cond_sens`** should be greater than zero if you declared `conditional = True`. Exactly zero means your conditions are not reaching the network, which is a wiring bug far more often than a modeling choice. -- **`copy_rate`** should be near zero. High means your model is reproducing - training designs rather than generating, and the board will publish it without - ranking it. +- **`train_distance_ratio`** should be near one. Near zero means your model is + reproducing training designs rather than generating; the board shows it and + leaves the reading to people. `cond_sens` draws the batch a second time, under shuffled conditions, so it doubles sampling cost. That is nothing for a GAN and noticeable for a diffusion -model — pass `--metrics mmd dpp viol` while iterating if it slows you down, then +model — pass `--metrics mmd viol` while iterating if it slows you down, then run the full list before publishing. ## 5. Publish it @@ -215,13 +219,19 @@ from engiopt.evaluation import register_metric @register_metric("my_metric", family="diversity", cost="cheap", higher_is_better=True) def my_metric(ctx) -> float: - """One line, shown by --list-metrics.""" - return float(...) # ctx.gen_flat, ctx.ref_flat, ctx.conditions, ... + """What question does the number answer? This line is what readers see.""" + return ctx.reduce(...) # one value per design -> the caller's aggregation (mean or median) ``` -Declare `cost="expensive"` if it touches `ctx.optimization` (the simulator or -optimizer). The registry enforces the split, so a cheap run can never -accidentally launch a simulation. +Five declarations: the name, the family (which question it belongs to), the +cost, the direction — `None` for a diagnostic that is read but never ranked on — +and, when the metric needs the actual designs rather than codes in some space +(a constraint check, a copy corpus), `pixel_only=True`. Declare +`cost="expensive"` if it touches `ctx.optimization` (the simulator or +optimizer); the registry enforces the split, so a cheap run can never +accidentally launch a simulation. Where the metric is computed (`space=`) and +how per-design values collapse to one number (`aggregation=`) are chosen when a +board is evaluated, not by the metric. Importing the module is what registers it — which means you can define a metric in a notebook cell and it will appear in the next leaderboard you build. diff --git a/LEADERBOARD.md b/LEADERBOARD.md index 4d076cbf..b228e904 100644 --- a/LEADERBOARD.md +++ b/LEADERBOARD.md @@ -132,7 +132,6 @@ to an entry covering every required seed. | Flag | Trips when | |---|---| | `unverified` | No runner has reproduced it yet. | -| `memorized` | `copy_rate` exceeds the spec's `max_copy_rate`. | | `ignores_conditions` | `cond_sens` is exactly zero for a model registered as conditional. | ## The copying problem @@ -152,21 +151,29 @@ scored: the whole dataset is public, so a memorizer memorizes all of it. So the board measures retrieval instead: -- **`novelty`** — mean per-element RMS distance from each generated design to the - nearest design the model could have copied: the training split it was fitted - on, plus the reference designs the protocol names. -- **`copy_rate`** — the share of the batch closer than `copy_tol`. This is the - one that flags an entry. +- **`train_distance`** — how far each generated design sits from the nearest + design in the training split, per element. Zero means the model reproduces + what it was trained on. +- **`train_distance_ratio`** — the same distance divided by what the withheld + reference designs score against the training split. One means the model's + designs are as far from its training data as real unseen optima are; near zero + means retrieval. There is no tolerance to set: the reference designs supply + the scale and nothing else. Both are diagnostic, with no ranking direction, and that is not an oversight. -Ranking on novelty would put pure noise in first place — zero means retrieval, +No row is flagged for memorization; the columns are read, not enforced. +Ranking on `train_distance` would put pure noise in first place — zero means retrieval, but large means only "unlike the data", which a broken model also achieves. Closing one gaming vector by opening another is not progress. The same reasoning applies to `cond_sens`: an unconditional model is a legitimate thing to build, and responding to conditions *wrongly* also moves the output, so a large value is not by itself a good one. -Read them next to `mmd` and `viol`, never on their own. +Read them next to `mmd` and `viol`, never on their own. Every column's question, +direction and cost is one call away — `METRICS.explain()` — and +[`example_metrics_suite.ipynb`](example_metrics_suite.ipynb) walks the whole suite, +including the construction that scores a perfect `mmd` by returning the correct +designs for the wrong conditions. ### What would actually close it @@ -190,7 +197,7 @@ from engiopt.evaluation.leaderboard import disagreement, load_from_hub, rank from engiopt.evaluation.spec import EvalSpec board = load_from_hub("IDEALLab/engiopt-leaderboard") -spec = EvalSpec.load("beams2d/v1") +spec = EvalSpec.load("beams2d/v2") rank(board, "fog", eval_spec=spec) # the public ordering rank(board, "fog", eval_spec=spec, eligible_only=False) # everything, including claims diff --git a/README.md b/README.md index 55db08e8..50c8cdf9 100644 --- a/README.md +++ b/README.md @@ -166,7 +166,7 @@ python -m engiopt.verify --board ... --verifier ideallab-ci --publish # the o Because the row carries the full address, that audit is not privileged — anyone can run the same command and get the same answer. -**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `novelty` / `copy_rate` catch it (`copy_rate=1.00`), and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. +**Two integrity metrics decide whether a score means what it looks like.** The evaluation protocol is public, so a lookup table keyed on the condition vector returns the dataset-optimal designs and posts a perfect `mmd` and a zero `viol` — measured on beams2d, it beats a trained cGAN on every headline metric. `train_distance` reads zero for it, and `cond_sens` catches a model that ignores the conditions it claims to use. Both are diagnostic rather than ranked, because ranking on them would just reward the opposite extreme. Flagged rows are published and left out of the ordering. The whole suite — what each column measures, which way is better, how to change the space or the aggregation, and how to add a metric — is in [`example_metrics_suite.ipynb`](example_metrics_suite.ipynb). See **[LEADERBOARD.md](LEADERBOARD.md)** for the submission path, the flags, and what would actually close the copying hole. diff --git a/engiopt/evaluate.py b/engiopt/evaluate.py index bda57745..00f89a25 100644 --- a/engiopt/evaluate.py +++ b/engiopt/evaluate.py @@ -32,7 +32,6 @@ from engiopt.evaluation.leaderboard import push_to_hub from engiopt.evaluation.registry import METRICS from engiopt.evaluation.submission import FLAG_IGNORES_CONDITIONS -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.submission import FLAG_UNVERIFIED from engiopt.evaluation.submission import integrity_flags from engiopt.utils.all_generators import BUILTIN_GENERATORS @@ -65,7 +64,8 @@ class Args: cgan_cnn_2d:023dd1fb gan_cnn_2d:06d9a9a1` asks for exactly the two that exist.""" spec: str | None = None - """Eval spec reference, e.g. `beams2d/v1`. Defaults to `/v1`.""" + """Eval spec reference, e.g. `beams2d/v2`. Omitted, the problem's current spec loads; + `EvalSpec.load` owns that default, so the CLI cannot drift from the library.""" metrics: tuple[str, ...] = () """Metric names; defaults to the spec's list.""" include_expensive: bool = False @@ -197,11 +197,7 @@ def _availability_label(count: int | None) -> str: def _print_metrics() -> None: """Print every registered metric grouped by the question it answers.""" print(f"{len(METRICS)} metrics registered:\n") - for family in sorted({spec.family for spec in METRICS.values()}): - print(f" [{family}]") - for spec in METRICS.select(family=family): - direction = {True: "higher better", False: "lower better", None: "diagnostic"}[spec.higher_is_better] - print(f" {spec.name:<10} {spec.cost:<10} {direction:<14} {spec.description}") + print(METRICS.explain().sort_values(["family", "cost"]).to_string()) def _resolve_generator_names(requested: tuple[str, ...], problem_id: str) -> list[str]: @@ -346,8 +342,7 @@ def main(args: Args) -> int: if args.list_generators or args.list_metrics: return 0 - spec = args.spec or f"{args.problem_id}/v1" - evaluator = Evaluator.for_problem(args.problem_id, spec=spec) + evaluator = Evaluator.for_problem(args.problem_id, spec=args.spec) print(f"Problem {args.problem_id} | spec {evaluator.spec.version} | n={evaluator.spec.n_samples}") generators = _load_generators(args, evaluator) @@ -375,14 +370,14 @@ def main(args: Args) -> int: print(f"\n{board.to_string(index=False)}\n") print(f"Wrote {len(board)} rows to {destination}") - _publish(args, evaluator, board) + _publish(args, board) return 0 -def _publish(args: Args, evaluator: Evaluator, board: pd.DataFrame) -> None: +def _publish(args: Args, board: pd.DataFrame) -> None: """Push the results wherever the flags asked, then print the ranking comparison.""" if args.push_to: - merged = push_to_hub(board, args.push_to, eval_spec=evaluator.spec) + merged = push_to_hub(board, args.push_to) print(f"Published {len(board)} row(s) to {args.push_to}; board now holds {len(merged)} rows.") print( "These rows are unverified. They will not be ranked until a runner re-fetches the " @@ -416,12 +411,6 @@ def _print_integrity_warnings(board: pd.DataFrame) -> None: if not flags: continue label = f"{row.get('algo_id')} seed {row.get('seed')}" - if FLAG_MEMORIZED in flags: - print( - f"\n [{label}] copy_rate={row.get('copy_rate'):.2f}: most of this batch reproduces designs " - "from the dataset rather than generating them. Its distribution and performance scores " - "measure retrieval, and it will be published but not ranked." - ) if FLAG_IGNORES_CONDITIONS in flags: print( f"\n [{label}] cond_sens=0: output did not change at all when the conditions were shuffled, " diff --git a/engiopt/evaluation/__init__.py b/engiopt/evaluation/__init__.py index 654d1b6d..3ea24c30 100644 --- a/engiopt/evaluation/__init__.py +++ b/engiopt/evaluation/__init__.py @@ -3,7 +3,7 @@ from engiopt.evaluation import Evaluator from engiopt.utils.all_generators import BUILTIN_GENERATORS - ev = Evaluator.for_problem("beams2d", spec="beams2d/v1") + ev = Evaluator.for_problem("beams2d", spec="beams2d/v2") gen = BUILTIN_GENERATORS["cgan_cnn_2d"].from_pretrained(ev.problem, problem_id="beams2d", seed=1) ev.score(gen) @@ -12,6 +12,7 @@ property of the comparison, not of the model. """ +from engiopt.evaluation.board import Board from engiopt.evaluation.context import EvaluationContext from engiopt.evaluation.context import OptimizationResults from engiopt.evaluation.evaluator import Evaluator @@ -31,6 +32,7 @@ __all__ = [ "METRICS", + "Board", "EvalSpec", "EvaluationContext", "Evaluator", diff --git a/engiopt/evaluation/board.py b/engiopt/evaluation/board.py new file mode 100644 index 00000000..f701c019 --- /dev/null +++ b/engiopt/evaluation/board.py @@ -0,0 +1,418 @@ +"""Score several models at once and read the result. + +This is the whole user-facing evaluation surface for saved designs:: + + board = Board(problem, reference=REFERENCE, train=TRAIN) + board.evaluate(MODELS, space="pixel", aggregation="mean") + board.explain() + board.rank("mmd") + +Where a metric is computed and how per-design values collapse to one number are +arguments here, not properties of the metric, so every metric is written once. +For a loaded `Generator` -- which is what the physics metrics and `cond_sens` +need -- use `Evaluator.score`. +""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from dataclasses import field +from typing import Any, Literal + +import numpy as np +import pandas as pd + +from engiopt import metrics as metrics_mod +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.registry import MetricRegistry +from engiopt.evaluation.registry import METRICS + +SpaceFit = Callable[[np.ndarray, int], Callable[[np.ndarray], np.ndarray]] +"""Given the reference designs and a width, return a function projecting designs into the space.""" + +SPACES: dict[str, SpaceFit] = {} +"""Where the set-level metrics can be computed, by name. `pixel` and `pca` are built in; +a learned latent space registers itself here with `register_space`.""" + + +def register_space(name: str, fit: SpaceFit) -> None: + """Make a space available to `Board.evaluate(space=name)`. + + Args: + name: The space's name; also the column suffix (`mmd@name`). + fit: `fit(reference_designs, width)` returning `project(designs) -> codes`. + """ + SPACES[name] = fit + + +DESIGN_COLUMN = "optimal_design" +"""The dataset column holding each row's optimal design, EngiBench's convention. +A dataset slice passed as `Board(reference=...)` is split on it: this column +becomes the reference designs, the remaining columns become the conditions.""" + + +class ConditionTable: + """A dict of condition columns, readable by column name and by row. + + Metrics ask two different questions of the conditions. `volume_error` wants + a whole column at once (`table["volfrac"]`); `viol` wants one design's full + brief (`table[0]`). A HuggingFace dataset answers both, a plain dict only + the first, so the board wraps dicts here rather than asking every notebook + to build a dataset. + """ + + def __init__(self, columns: Mapping[str, Any]): + self.columns = dict(columns) + + def __getitem__(self, key: Any) -> Any: + if isinstance(key, str): + return self.columns[key] + return {name: column[key] for name, column in self.columns.items()} + + def __len__(self) -> int: + return len(next(iter(self.columns.values()))) + + +def _as_condition_table(conditions: Any) -> Any: + """Give loose condition inputs the row access the metrics need. + + A dict of columns or a DataFrame is wrapped; anything else, such as a + HuggingFace dataset or None, already behaves and passes through. + """ + if isinstance(conditions, pd.DataFrame): + conditions = {name: conditions[name].tolist() for name in conditions.columns} + if isinstance(conditions, Mapping): + return ConditionTable(conditions) + return conditions + + +REFERENCE_ROW = "reference (split-half)" +"""Label of the row scoring one random half of the reference designs against the other. + +A scale reference, not a competitor: it shows where real, correct designs land on +each column on this problem. It is half the sample size of a model row, and the +set-level metrics move with sample size, so read it as the order of magnitude a +good model should reach rather than as a number to beat. Never ranked. +""" + + +@dataclass +class Board: + """Several models' designs, scored against one set of reference designs. + + Attributes: + problem: The problem the designs answer; needs `design_space` and + `check_constraints`, which every EngiBench problem has. + reference: The withheld optimal designs the models are scored against, + as an array of designs, or as the test rows themselves: a dataset + slice carrying an `optimal_design` column. The slice is split on + construction, designs out of that column and conditions out of the + rest, so design `i` and its brief cannot fall out of alignment. + conditions: The conditions each reference design answers, one row per + design: a dict of columns, a DataFrame, or anything already + indexable by condition name and by row, such as a dataset slice. + Needed by `viol` on problems whose constraint check reads the + conditions, and by `volume_error`; without them those columns are + blank. Derived automatically when `reference` is a dataset slice. + volume_condition: Name of the condition holding the volume-fraction + budget, e.g. beams2d's `volfrac`. Declared rather than guessed, as + on `EvalSpec`; without it `volume_error` is blank even when the + conditions carry a budget. + train: The training designs, for the memorization metrics. Optional. + sigma: Kernel bandwidth for the distribution and diversity metrics. None, + the default, uses the median pairwise distance of the training designs + in whichever space is being scored, or of the reference designs when no + training split was given, so the kernel is neither saturated nor empty + on a problem it has never seen. Resolved once per `evaluate`, so every + row, the split-half reference row included, is scored under the same + bandwidth. + frame: The most recent `evaluate` result. + designs: The designs behind each row of `frame`, when the board sampled them itself. + """ + + problem: Any + reference: Any + conditions: Any | None = None + volume_condition: str | None = None + train: Any | None = None + sigma: float | None = None + registry: MetricRegistry = field(default_factory=lambda: METRICS) + frame: pd.DataFrame = field(default_factory=pd.DataFrame) + designs: dict[str, Any] = field(default_factory=dict) + + def __post_init__(self) -> None: + """Split a dataset-slice reference, wrap loose conditions, check alignment.""" + if hasattr(self.reference, "column_names") and DESIGN_COLUMN in self.reference.column_names: + rows = self.reference + self.reference = np.asarray(rows[DESIGN_COLUMN]) + if self.conditions is None: + briefs = {name: rows[name] for name in rows.column_names if name != DESIGN_COLUMN} + self.conditions = ConditionTable(briefs) + self.conditions = _as_condition_table(self.conditions) + if self.conditions is not None and hasattr(self.conditions, "__len__"): + n_designs = len(np.asarray(self.reference)) + if len(self.conditions) != n_designs: + raise ValueError( + f"{len(self.conditions)} condition rows for {n_designs} reference designs. " + f"Conditions do not select designs; row i must be the brief design i answers." + ) + + def evaluate( + self, + models: Mapping[str, Any], + *, + metrics: list[str] | None = None, + space: str = "pixel", + aggregation: Literal["mean", "median"] = "mean", + width: int = 8, + reference_row: bool = True, + ) -> pd.DataFrame: + """Score every model on every cheap metric. + + Args: + models: Model name -> its generated designs, one array per model. + metrics: Metric names; defaults to every metric that needs no simulator. + space: `pixel` scores the designs themselves. Any other name in + `SPACES` (`pca` is built in) projects every design into that + space first and appends `@space` to each column; metrics that + need the actual designs (`viol`, `train_distance`) are skipped there. + aggregation: How metrics with one value per design collapse to one number. + width: Dimension of the space, for spaces that have one (`pca`). + reference_row: Also score one random half of the reference designs + against the other half, as the row `REFERENCE_ROW`. + + Returns: + One row per model, one column per metric output. Also kept as `frame`. + """ + specs = self.registry.select(metrics) if metrics else self.registry.select(cost="cheap") + if space != "pixel": + specs = [spec for spec in specs if not spec.pixel_only] + reference = np.asarray(self.reference) + if space not in SPACES: + raise KeyError(f"Unknown space {space!r}. Registered: {', '.join(SPACES)}") + project = SPACES[space](reference, width) + + # One bandwidth for the whole board. Each context would otherwise infer + # its own from whatever reference set it holds, and the split-half row + # holds half the designs, so its kernel columns would differ from the + # model rows in bandwidth as well as in sample size. + sigma = self.sigma + if sigma is None: + basis = project(np.asarray(self.train if self.train is not None else reference)) + sigma = metrics_mod.compute_median_sigma(basis.reshape(len(basis), -1)) + + def score(generated: Any, against: Any, chosen: list[Any]) -> dict[str, float]: + ctx = EvaluationContext( + problem=self.problem, + problem_id=getattr(self.problem, "problem_id", type(self.problem).__name__), + gen_designs=project(np.asarray(generated)), + ref_designs=project(np.asarray(against)), + conditions=self.conditions if against is reference else None, + volume_condition=self.volume_condition, + sigma=sigma, + aggregation=aggregation, + train_designs_fn=(lambda: project(np.asarray(self.train))) if self.train is not None else None, + ) + row: dict[str, float] = {} + for spec in chosen: + value = spec.fn(ctx) + row.update(value if isinstance(value, dict) else {spec.name: value}) + return row + + rows = {name: score(designs, reference, specs) for name, designs in models.items()} + if reference_row and len(reference) >= 4: # noqa: PLR2004 - two halves of at least two designs + order = np.random.default_rng(0).permutation(len(reference)) + half = len(reference) // 2 + # Two halves of the reference set answer different conditions, so the + # per-condition metrics have no meaning there and are left blank. + unpaired = [spec for spec in specs if spec.family != "conditions"] + rows[REFERENCE_ROW] = score(reference[order[:half]], reference[order[half:]], unpaired) + + frame = pd.DataFrame.from_dict(rows, orient="index") + if space != "pixel": + frame.columns = [f"{column}@{space}" for column in frame.columns] + self.frame = frame + return frame + + @classmethod + def from_evaluator( + cls, + evaluator: Any, + models: Mapping[str, Any], + *, + designs: Mapping[str, Any] | None = None, + expensive: bool = False, + metrics: list[str] | None = None, + ) -> Board: + """Score loaded models, and optionally saved designs, under one evaluation spec. + + The evaluator supplies the spec, the reference designs, the conditions and + the training split, so every row is comparable. Live models get every column; + `cond_sens` and the cost columns need a model to re-run or inspect, so they + stay blank for the `designs` rows. The designs each model produced are kept + on `Board.designs`, so they can be re-scored in another space or saved. + + Args: + evaluator: An `Evaluator` for the problem. + models: Label -> loaded `Generator`. + designs: Label -> one design per spec condition, e.g. a construction + built from the dataset. + expensive: Whether to run the simulator-backed metrics. + metrics: Metric names; defaults to the spec's list. + + Returns: + A `Board` with one row per model and per designs entry, in that order. + """ + rows: dict[str, dict[str, float]] = {} + sampled: dict[str, Any] = {} + for label, generator in models.items(): + ctx = evaluator.context_for(generator) + rows[label] = evaluator.score_context(ctx, only=metrics, include_expensive=expensive) + sampled[label] = ctx.gen_designs + for label, batch in (designs or {}).items(): + ctx = evaluator.context_for_designs(batch) + rows[label] = evaluator.score_context(ctx, only=metrics, include_expensive=expensive) + sampled[label] = ctx.gen_designs + board = cls( + problem=evaluator.problem, + reference=evaluator.resolved.ref_designs, + conditions=evaluator.resolved.conditions, + volume_condition=evaluator.spec.volume_condition, + sigma=evaluator.spec.sigma, + registry=evaluator.registry, + ) + board.frame = pd.DataFrame.from_dict(rows, orient="index") + board.designs = sampled + return board + + @classmethod + def load( + cls, + problem_id: str, + models: list[str] | Mapping[str, str | None], + *, + designs: Mapping[str, Any] | None = None, + seed: int = 1, + spec: Any = None, + rows: Any = None, + expensive: bool = False, + ) -> Board: + """Pull published checkpoints by name and score them: the workshop in one call. + + Args: + problem_id: An EngiBench problem id, e.g. `"beams2d"`. + models: Generator names, e.g. `["diffusion_2d_cond", "vqgan"]`. A model + whose default configuration was never trained needs its config + fingerprint, given as a mapping: `{"cgan_cnn_2d": "825831f6"}`. + designs: Extra rows of saved designs, one per spec condition; see `from_evaluator`. + seed: Training seed of the checkpoints to load. + spec: Spec reference such as `"beams2d/v2"`, or an `EvalSpec`; defaults to the current spec. + rows: A dataset slice to score against instead of the spec's drawn + conditions; see `Evaluator.for_rows`. None, the default, keeps + the spec's seeded 50-row draw from the test split. + expensive: Whether to run the simulator-backed metrics. + + Returns: + A `Board` with one row per model, labelled by name. + """ + from engiopt.evaluation.evaluator import Evaluator + from engiopt.utils.all_generators import BUILTIN_GENERATORS + + if rows is not None: + evaluator = Evaluator.for_rows(problem_id, rows, spec=spec) + else: + evaluator = Evaluator.for_problem(problem_id, spec=spec) + requested = models if isinstance(models, Mapping) else dict.fromkeys(models) + generators = { + name: BUILTIN_GENERATORS[name].from_pretrained( + evaluator.problem, problem_id=problem_id, seed=seed, config_fingerprint=fingerprint + ) + for name, fingerprint in requested.items() + } + return cls.from_evaluator(evaluator, generators, designs=designs, expensive=expensive) + + def explain(self) -> pd.DataFrame: + """Read every column of the last evaluation. + + Returns: + One row per column: the question it answers, which way is better, the + model it picks (blank for a diagnostic, `tie` when the best is shared), + and the split-half reference value when the board carries that row. + """ + models = self.frame.drop(index=REFERENCE_ROW, errors="ignore") + rows = {} + for column in self.frame.columns: + spec = self._spec_for(column) + if spec is None: + continue + picks = "" + values = models[column].dropna() + if spec.higher_is_better is not None and not values.empty: + best = values.min() if spec.higher_is_better is False else values.max() + winners = list(values.index[values == best]) + picks = winners[0] if len(winners) == 1 else f"tie: {', '.join(winners)}" + rows[column] = { + "question": spec.description, + "direction": spec.direction, + "picks": picks, + "split-half reference": self.frame.loc[REFERENCE_ROW, column] + if REFERENCE_ROW in self.frame.index + else None, + } + return pd.DataFrame.from_dict(rows, orient="index") + + def rank(self, column: str) -> pd.DataFrame: + """Models ordered best-first on one column. + + Raises: + ValueError: If the column is a diagnostic, which is read but not ranked on. + """ + spec = self._spec_for(column) + if spec is None: + raise KeyError(f"{column!r} is not a metric column on this board.") + if spec.higher_is_better is None: + raise ValueError( + f"{column!r} is a diagnostic: it says whether the other columns mean what they look like. " + f"Read it beside them; do not rank on it." + ) + models = self.frame.drop(index=REFERENCE_ROW, errors="ignore") + return models.sort_values(column, ascending=not spec.higher_is_better) + + def _spec_for(self, column: str) -> Any: + base = column.split("@", maxsplit=1)[0] + return next((spec for spec in self.registry.values() if base in spec.columns), None) + + def _repr_html_(self) -> str: + return self.frame._repr_html_() + + def __repr__(self) -> str: + return repr(self.frame) + + +def _pixel(reference: np.ndarray, width: int) -> Callable[[np.ndarray], np.ndarray]: # noqa: ARG001 + """The designs as they are.""" + return lambda designs: designs + + +def _pca_projection(reference: np.ndarray, n_components: int) -> Callable[[np.ndarray], np.ndarray]: + """Projection onto the leading principal components of the reference designs. + + The control that asks whether a *learned* space was needed: if PCA of the + data answers the same questions, any rotation of the same width would do. + """ + flat = np.ascontiguousarray(reference.reshape(len(reference), -1), dtype=np.float64) + mean = flat.mean(axis=0) + _, _, vt = np.linalg.svd(flat - mean, full_matrices=False) + components = np.ascontiguousarray(vt[: min(n_components, len(vt))].T) + + def project(designs: np.ndarray) -> np.ndarray: + centered = np.ascontiguousarray(designs.reshape(len(designs), -1), dtype=np.float64) - mean + # einsum rather than `@`: on small matrices Apple's BLAS raises spurious FP warnings here. + return np.einsum("ij,jk->ik", centered, components) + + return project + + +register_space("pixel", _pixel) +register_space("pca", _pca_projection) diff --git a/engiopt/evaluation/context.py b/engiopt/evaluation/context.py index 0525602d..824305d7 100644 --- a/engiopt/evaluation/context.py +++ b/engiopt/evaluation/context.py @@ -13,7 +13,7 @@ from dataclasses import dataclass from dataclasses import field from functools import cached_property -from typing import Any, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING from gymnasium import spaces import numpy as np @@ -64,6 +64,13 @@ class OptimizationResults: shared instance. """ + trajectories: list[npt.NDArray[Any]] = field(default_factory=list) + """Per design, the optimality gap at every optimizer call of its re-optimization.""" + reference_objectives: list[float] = field(default_factory=list) + """Per design, the reference optimum's objective for its conditions, scalarized like the gaps. + + The scale a gap is read against: five percent of this is what "near optimal" means. + """ iog: list[float] = field(default_factory=list) """Initial optimality gap: how far each generated design starts from the reference optimum.""" cog: list[float] = field(default_factory=list) @@ -82,7 +89,8 @@ class EvaluationContext: gen_designs: Generated designs, `(n_samples, *design_shape)`. ref_designs: Reference (dataset-optimal) designs for the same conditions. conditions: The conditions each design was asked to satisfy. - sigma: Gaussian-kernel bandwidth for MMD and DPP. + sigma: Gaussian-kernel bandwidth for the kernel metrics, or None to use + the median pairwise distance of the training designs; see `kernel_sigma`. volfrac_tol: Tolerance used by the volume-fraction violation check. volume_condition: Name of the condition holding the volume-fraction budget a design must hit, when the problem has one. Declared by the @@ -95,13 +103,10 @@ class EvaluationContext: objective_weight_condition: Name of a per-sample condition carrying the trade-off instead. thermoelastic2d's `weight` splits the first two objectives as `(w, 1 - w)`; any remaining objectives get zero. - copy_corpus_fn: Returns the designs a model could plausibly have copied - -- the dataset designs it may have trained on, plus the reference - optima it is being scored against. Deferred behind a callable so - that fetching a training split is paid for only when a memorization - metric actually runs, and shared across every model in a sweep. - copy_tol: RMS per-element distance below which a generated design counts - as a copy of something in the corpus. + train_designs_fn: Returns the training split's designs. Deferred behind + a callable so that fetching the split is paid for only when a + memorization metric actually runs, and shared across every model + in a sweep. resample_permuted: Re-runs the generator on a permutation of the same conditions, with the same latent draw. `None` when the comparison is unavailable -- an unconditional problem, or a context built directly @@ -113,14 +118,19 @@ class EvaluationContext: gen_designs: npt.NDArray[Any] ref_designs: npt.NDArray[Any] conditions: Dataset | None = None - sigma: float = 10.0 + sigma: float | None = None volfrac_tol: float = 0.01 volume_condition: str | None = None sample_seconds: float | None = None objective_weights: tuple[float, ...] | None = None objective_weight_condition: str | None = None - copy_corpus_fn: Callable[[], npt.NDArray[Any]] | None = None - copy_tol: float = 0.01 + train_designs_fn: Callable[[], npt.NDArray[Any]] | None = None + aggregation: Literal["mean", "median"] = "mean" + model_params: int | None = None + """Trainable parameter count of the generator, when it is a network.""" + train_minutes: float | None = None + """Wall-clock minutes the generator took to train, when the checkpoint records it.""" + """How a metric with one value per design is collapsed into a column; see `reduce`.""" resample_permuted: Callable[[npt.NDArray[Any]], npt.NDArray[Any]] | None = None @property @@ -133,6 +143,30 @@ def is_dict_space(self) -> bool: """Whether the problem uses a `spaces.Dict` design space needing flatten/unflatten.""" return isinstance(self.problem.design_space, spaces.Dict) + def reduce(self, values: Any) -> float: + """Collapse one value per generated design into the column's single number. + + `iog`, `cog`, `fog` and the per-design distances are each a vector with + one entry per generated design. Which single number that vector becomes + is a policy, not a measurement: a mean is pulled by one diverged design, + a median is not, and two boards that chose differently are not comparable. + So the choice is declared once (`EvalSpec.aggregation`), recorded in the + row, and applied here rather than hard-coded metric by metric. + + Rates such as `viol` are fractions, not per-design + averages, and do not pass through this. + + Args: + values: One value per generated design. + + Returns: + The mean or median per the declared policy; NaN for no designs. + """ + array = np.asarray(values, dtype=float) + if array.size == 0: + return float("nan") + return float(np.median(array) if self.aggregation == "median" else np.mean(array)) + @cached_property def gen_flat(self) -> npt.NDArray[Any]: """Generated designs flattened to `(n_samples, -1)`.""" @@ -147,51 +181,55 @@ def ref_flat(self) -> npt.NDArray[Any]: return np.asarray(flattened) @cached_property - def copy_corpus(self) -> npt.NDArray[Any] | None: - """Flattened designs a generator could have memorized, or None if there are none. - - The scored reference designs are always in here, even though they are - also `mmd`'s comparison target. They are the single most attractive - thing to copy precisely *because* the protocol is public and names them, - so a memorization check that omitted them would miss the easiest attack - on this board. `copy_corpus_fn` widens the corpus to the training split - the model was actually fitted on. - - Parts whose flattened width does not match the generated designs are - dropped rather than raising: a mismatched corpus makes the check - unavailable, and that is not a reason to fail an evaluation. + def train_designs(self) -> npt.NDArray[Any] | None: + """Flattened training designs, or None when the problem has no training split. + + Fetched once, through `train_designs_fn`, and only when a memorization + metric asks. A split whose flattened width does not match the generated + designs is treated as absent rather than raising: it makes the check + unavailable, which is not a reason to fail an evaluation. """ - width = self.gen_flat.shape[1] - parts = [part for part in (self.ref_flat,) if part.shape[1] == width] - if self.copy_corpus_fn is not None: - extra = np.asarray(self.copy_corpus_fn()) - if len(extra): - flattened = extra.reshape(len(extra), -1) - if flattened.shape[1] == width: - parts.append(flattened) - return np.concatenate(parts) if parts else None + if self.train_designs_fn is None: + return None + train = np.asarray(self.train_designs_fn()) + if not len(train): + return None + flat = train.reshape(len(train), -1) + return flat if flat.shape[1] == self.gen_flat.shape[1] else None @cached_property - def nearest_corpus_distance(self) -> npt.NDArray[Any] | None: - """Per-design RMS distance to the closest design in `copy_corpus`. + def kernel_sigma(self) -> float: + """The bandwidth the kernel metrics use. + + The value the spec pinned, if it pinned one. Otherwise the median + pairwise distance of the training designs, in whatever space this + context holds, so the kernel is neither saturated nor empty on any + problem; the reference designs stand in when there is no training split. + Subsampled with a fixed seed, so the value is reproducible, and recorded + in every published row. + """ + if self.sigma is not None: + return float(self.sigma) + basis = self.train_designs if self.train_designs is not None and len(self.train_designs) > 1 else self.ref_flat + return metrics_mod.compute_median_sigma(basis) + + def nearest_train_distance(self, designs: npt.NDArray[Any]) -> npt.NDArray[Any]: + """Per-design RMS distance from each of `designs` to the closest training design. - Normalized by the square root of the design dimension so the number is a - *per-element* deviation. That makes one tolerance meaningful across - problems of different resolution, where a raw L2 norm would not be. + Divided by the square root of the design dimension so it reads as a + per-element deviation, the same on a 50 by 100 grid as on an 8 by 10 one. + Requires `train_designs`. """ - corpus = self.copy_corpus - if corpus is None or not len(corpus): - return None from scipy.spatial.distance import cdist - distances = cdist(self.gen_flat, corpus, "euclidean") - return np.asarray(distances.min(axis=1) / np.sqrt(self.gen_flat.shape[1])) + distances = cdist(designs, self.train_designs, "euclidean") + return np.asarray(distances.min(axis=1) / np.sqrt(designs.shape[1])) @cached_property def permuted_designs(self) -> npt.NDArray[Any] | None: """Designs the generator produces when the conditions are shuffled between samples. - Same latent draw, same model, different brief. A model that reads its + Same latent draw, same model, different conditions. A model that reads its conditions produces something different; a model that ignores them produces the identical batch in a different order at best, and an identical batch outright at worst. Comparing the two is what turns "this @@ -328,6 +366,8 @@ def optimization(self) -> OptimizationResults: results.iog.append(self.scalarize_gap(np.asarray(generated_objective) - np.asarray(reference_optimum), i)) results.cog.append(sum(self.scalarize_gap(step_gap, i) for step_gap in gaps)) results.fog.append(self.scalarize_gap(gaps[-1], i)) + results.trajectories.append(np.asarray([self.scalarize_gap(step_gap, i) for step_gap in gaps], dtype=float)) + results.reference_objectives.append(self.scalarize_gap(np.asarray(reference_optimum), i)) return results @cached_property @@ -352,7 +392,7 @@ def is_infeasible(self, design: Any, conditions: dict[str, Any] | None) -> bool: This is what a new problem gets for free. 2. The volume-fraction budget named by the spec's `volume_condition`, when the problem has one. Missing that target is a design failing to - honor its brief rather than an invalid design, and no EngiBench + honor its conditions rather than an invalid design, and no EngiBench constraint covers it. A problem with neither -- photonics2d has no volume budget -- is scored diff --git a/engiopt/evaluation/evaluator.py b/engiopt/evaluation/evaluator.py index 81aa227d..41d58c19 100644 --- a/engiopt/evaluation/evaluator.py +++ b/engiopt/evaluation/evaluator.py @@ -4,13 +4,14 @@ they loaded and called their model -- which is now the `Generator` contract -- so everything else lives here once:: - ev = Evaluator.for_problem("beams2d", spec="beams2d/v1") + ev = Evaluator.for_problem("beams2d", spec="beams2d/v2") row = ev.score(generator) # cheap metrics only board = ev.leaderboard(zoo) # a DataFrame, one row per model """ from __future__ import annotations +import dataclasses from dataclasses import dataclass from dataclasses import field import datetime as dt @@ -53,6 +54,7 @@ "spec_version", "n_samples", "sample_seconds", + "kernel_sigma", "checkpoint_repo", "checkpoint_path", "checkpoint_revision", @@ -126,14 +128,14 @@ def for_problem( Args: problem_id: EngiBench problem registry key. spec: `"/"`, an `EvalSpec`, or None to load - `"/v1"`. + the problem's current spec version. device: Torch device; auto-selected when omitted. registry: Metric registry override, useful in tests. """ from engibench.utils.all_problems import BUILTIN_PROBLEMS problem = BUILTIN_PROBLEMS[problem_id]() - eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or f"{problem_id}/v1") + eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or problem_id) if eval_spec.problem_id != problem_id: raise ValueError(f"Spec is for {eval_spec.problem_id!r}, not {problem_id!r}.") device = device or pick_device() @@ -146,6 +148,64 @@ def for_problem( registry=registry or METRICS, ) + @classmethod + def for_rows( + cls, + problem_id: str, + rows: Any, + *, + spec: str | EvalSpec | None = None, + device: th.device | None = None, + registry: MetricRegistry | None = None, + ) -> Evaluator: + """Build an evaluator whose reference is a dataset slice you chose. + + The spec's own draw is the leaderboard contract: a seeded, digest-checked + sample of the test split. This constructor swaps only that draw for the + rows you pass, a region of condition space, a harder subset, your own + split, and keeps everything else the spec declares: the metric list, the + aggregation, the volume condition, the bandwidth policy. Generators are + then sampled under exactly these rows' conditions and scored against + these rows' optimal designs, one for one. + + Numbers from a custom slice are not leaderboard rows; the frozen draw is + what published numbers mean, so the digest is cleared rather than lied to. + + Args: + problem_id: EngiBench problem registry key. + rows: A dataset slice with an `optimal_design` column and the + problem's condition columns, e.g. `problem.dataset["test"].select(...)`. + spec: `"/"`, an `EvalSpec`, or None to load + the problem's current spec version. + device: Torch device; auto-selected when omitted. + registry: Metric registry override, useful in tests. + """ + from engibench.utils.all_problems import BUILTIN_PROBLEMS + import torch as th + + from engiopt.transforms import get_scalar_condition_keys + + problem = BUILTIN_PROBLEMS[problem_id]() + eval_spec = spec if isinstance(spec, EvalSpec) else EvalSpec.load(spec or problem_id) + if eval_spec.problem_id != problem_id: + raise ValueError(f"Spec is for {eval_spec.problem_id!r}, not {problem_id!r}.") + device = device or pick_device() + problem.reset(seed=eval_spec.condition_seed) + + available = [key for key in problem.conditions_keys if key in rows.column_names] + scalar_keys = get_scalar_condition_keys(problem, rows) + conditions = rows.select_columns(available) + tensor = th.tensor([conditions[key] for key in scalar_keys], dtype=th.float32, device=device).T + resolved = ResolvedSpec( + spec=dataclasses.replace(eval_spec, n_samples=len(rows), condition_digest=None), + conditions_tensor=tensor, + conditions=conditions, + ref_designs=np.asarray(rows["optimal_design"]), + indices=np.arange(len(rows)), + condition_keys=tuple(scalar_keys), + ) + return cls(problem=problem, problem_id=problem_id, resolved=resolved, device=device, registry=registry or METRICS) + @property def spec(self) -> EvalSpec: """The frozen evaluation contract in force.""" @@ -163,22 +223,42 @@ def context_for(self, generator: Generator) -> EvaluationContext: f"{generator.algo_id!r} supports {generator.design_kinds} design spaces, " f"but {self.problem_id!r} is {kind!r}." ) - designs = self._sample(generator) + ctx = self.context_for_designs(self._sample(generator)) + return dataclasses.replace( + ctx, + sample_seconds=generator.last_sample_seconds, + model_params=_parameter_count(generator), + train_minutes=getattr(generator, "train_minutes", None), + resample_permuted=self._permuted_sampler(generator), + ) + + def context_for_designs(self, designs: npt.NDArray[Any]) -> EvaluationContext: + """Wrap designs that did not come from a generator in the context a generator would get. + + A construction built from the dataset, or a batch saved earlier, is scored + against the same reference designs, conditions, kernel bandwidth and copy + corpus as a live model, so its row is comparable. Only the metrics that + must re-run a model (`cond_sens`) and the cost columns stay blank. + + Args: + designs: One design per spec condition, in the spec's condition order. + + Returns: + An `EvaluationContext` ready for `score_context`. + """ return EvaluationContext( problem=self.problem, problem_id=self.problem_id, - gen_designs=designs, + gen_designs=np.asarray(designs), ref_designs=self.resolved.ref_designs, conditions=self.resolved.conditions, sigma=self.spec.sigma, volfrac_tol=self.spec.volfrac_tol, volume_condition=self.spec.volume_condition, - sample_seconds=generator.last_sample_seconds, objective_weights=self.spec.objective_weights, objective_weight_condition=self.spec.objective_weight_condition, - copy_corpus_fn=self.copy_corpus, - copy_tol=self.spec.copy_tol, - resample_permuted=self._permuted_sampler(generator), + train_designs_fn=self.train_designs, + aggregation=self.spec.aggregation, ) def _sample(self, generator: Generator, order: npt.NDArray[Any] | None = None) -> npt.NDArray[Any]: @@ -212,37 +292,29 @@ def _permuted_sampler(self, generator: Generator) -> Callable[[npt.NDArray[Any]] return lambda order: self._sample(generator, order) @functools.cached_property - def copy_corpus(self) -> Callable[[], npt.NDArray[Any]]: - """Designs from the training split that a model on this problem could have memorized. + def train_designs(self) -> Callable[[], npt.NDArray[Any]]: + """The training split's designs, for the memorization metrics. - Built once per evaluator and shared by every model in a sweep, and + Fetched once per evaluator and shared by every model in a sweep, and deferred behind a callable so a run that selects no memorization metric - never pays for the fetch. + never pays for the fetch. The whole split, not a sample of it: a + lookup table over the training data must read as distance zero, and a + subsample would let most of its designs through. """ @functools.cache - def corpus() -> npt.NDArray[Any]: - return self._draw_copy_corpus() - - return corpus + def designs() -> npt.NDArray[Any]: + return self._load_train_designs() - def _draw_copy_corpus(self) -> npt.NDArray[Any]: - """Subsample the training split's optimal designs, deterministically. + return designs - Drawn with the spec's own `condition_seed`, so the corpus a model is - checked against is as reproducible as the conditions it is scored on -- - an audit that drew a different corpus could reach a different verdict. - """ + def _load_train_designs(self) -> npt.NDArray[Any]: + """The training split's optimal designs, or an empty array for a problem without one.""" try: train = self.problem.dataset["train"] - designs = np.asarray(train["optimal_design"]) - # A problem with no training split simply has no wider corpus; the - # reference designs still are one, and they are the case that matters. + return np.asarray(train["optimal_design"]) except (KeyError, TypeError, AttributeError): return np.empty((0, 0)) - size = min(self.spec.copy_corpus_size, len(designs)) - rng = np.random.default_rng(self.spec.condition_seed) - return designs[rng.choice(len(designs), size, replace=False)] def score( self, @@ -302,7 +374,9 @@ def _provenance(self, generator: Generator, ctx: EvaluationContext) -> dict[str, "seed": getattr(generator, "seed", None), "spec_version": self.spec.version, "n_samples": ctx.n_samples, + "aggregation": ctx.aggregation, "sample_seconds": ctx.sample_seconds, + "kernel_sigma": ctx.kernel_sigma, "checkpoint_repo": getattr(generator, "checkpoint_repo", None), "checkpoint_path": getattr(generator, "checkpoint_path", None), "checkpoint_revision": getattr(generator, "checkpoint_revision", None), @@ -353,7 +427,16 @@ def leaderboard( return order_columns(pd.DataFrame(rows)) -@functools.cache +def _parameter_count(generator: Any) -> int | None: + """Trainable parameters of the generator's network, or None for a model with none.""" + from torch import nn + + modules = [m for m in getattr(generator, "__dict__", {}).values() if isinstance(m, nn.Module)] + if not modules: + return None + return sum(p.numel() for module in modules for p in module.parameters() if p.requires_grad) + + def engibench_version() -> str: """Which EngiBench *ran* this evaluation, however it was installed. diff --git a/engiopt/evaluation/leaderboard.py b/engiopt/evaluation/leaderboard.py index 9fdfaa33..af3f38a4 100644 --- a/engiopt/evaluation/leaderboard.py +++ b/engiopt/evaluation/leaderboard.py @@ -302,7 +302,6 @@ def push_to_hub( private: bool = False, commit_message: str | None = None, max_attempts: int = 3, - eval_spec: EvalSpec | None = None, check_admission: bool = True, ) -> pd.DataFrame: """Merge `new_rows` into the published leaderboard and upload the result. @@ -326,7 +325,6 @@ def push_to_hub( private: Create the repo private if it does not exist yet. commit_message: Defaults to a summary of what was added. max_attempts: How many times to retry after losing a race. - eval_spec: Spec the rows were scored under, for the flag thresholds. check_admission: Publish rows unchecked. Only for a verification runner writing back rows it produced itself, which is the one caller whose `verified=True` is not a self-assertion. @@ -344,7 +342,7 @@ def push_to_hub( from engiopt.evaluation.submission import prepare_submission if check_admission: - new_rows = prepare_submission(new_rows, eval_spec) + new_rows = prepare_submission(new_rows) api = HfApi(token=token) api.create_repo(repo_id=repo_id, repo_type="dataset", private=private, exist_ok=True) diff --git a/engiopt/evaluation/metrics/builtin.py b/engiopt/evaluation/metrics/builtin.py index 731452e8..bb1c6fa2 100644 --- a/engiopt/evaluation/metrics/builtin.py +++ b/engiopt/evaluation/metrics/builtin.py @@ -32,11 +32,10 @@ family="distribution", cost="cheap", higher_is_better=False, - description="Maximum Mean Discrepancy between generated and reference designs (pixel space).", ) def mmd(ctx: EvaluationContext) -> float: - """Maximum Mean Discrepancy between generated and reference designs.""" - return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.sigma)) + """Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.""" + return float(metrics_mod.mmd(ctx.gen_flat, ctx.ref_flat, sigma=ctx.kernel_sigma)) # ---------------------------------------------------------------------- @@ -49,11 +48,21 @@ def mmd(ctx: EvaluationContext) -> float: family="diversity", cost="cheap", higher_is_better=True, - description="Determinantal Point Process diversity of the generated set.", ) def dpp(ctx: EvaluationContext) -> float: - """Determinantal Point Process diversity of the generated designs.""" - return float(metrics_mod.dpp_diversity(ctx.gen_flat, sigma=ctx.sigma)) + """Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. + + The n-th root of the DPP kernel determinant, i.e. the geometric mean of the + kernel's eigenvalues. The raw determinant that earlier boards published under + this name is a product of n numbers below one and reads 1e-20 on every real + board; the n-th root is the same quantity on a scale that is bounded in + (0, 1] and comparable across sample sizes. Rows from spec v1 carry the raw + form and are not comparable with this column. + + A determinant still rises when a collapsed set is jittered, so it rewards + noise as diversity. `vendi` does not, and is the column to prefer. + """ + return float(metrics_mod.dpp_geometric_mean(ctx.gen_flat, sigma=ctx.kernel_sigma)) # ---------------------------------------------------------------------- @@ -62,64 +71,55 @@ def dpp(ctx: EvaluationContext) -> float: @register_metric( - "novelty", + "train_distance", family="memorization", cost="cheap", higher_is_better=None, - outputs=("novelty", "copy_rate"), - description="Distance from generated designs to the nearest design the model could have memorized.", -) -def novelty(ctx: EvaluationContext) -> dict[str, float]: - """How far the generated designs sit from the corpus a model could copy from. - - The evaluation protocol is public: which conditions are scored, and the - dataset-optimal design for each one, can be recomputed by anyone from the - committed spec. A lookup table keyed on the condition vector therefore tops - `mmd`, `iog`, and `fog` -- not by cheating the implementation, but because - those metrics are *defined* as closeness to exactly the designs it returns. - No amount of care in the evaluator changes that; the only defense available - to a public board is to measure retrieval and say so. - - Two columns, because they answer different questions: - - - `novelty` -- mean per-element RMS distance to the nearest corpus design. - Diagnostic, deliberately: it has no good direction. Zero means the model - is a retrieval system, but large means only that the output is far from - the data, which pure noise also achieves. Read it next to `mmd` and - `viol`, never on its own. - - `copy_rate` -- the fraction of designs closer than `copy_tol`, i.e. the - share of this batch that is a reproduction rather than a generation. This - is the one that gates a submission. - - Returns NaN when no corpus is available, rather than claiming novelty that - was never checked. - """ - distances = ctx.nearest_corpus_distance - if distances is None: - return {"novelty": float("nan"), "copy_rate": float("nan")} - return { - "novelty": float(np.mean(distances)), - "copy_rate": float(np.mean(distances < ctx.copy_tol)), - } + pixel_only=True, + outputs=("train_distance", "train_distance_ratio"), +) +def train_distance(ctx: EvaluationContext) -> dict[str, float]: + """Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. + + Two readings of one measurement. `train_distance` is the distance itself, + per element, so the number means the same thing on problems of different + resolution. `train_distance_ratio` divides it by the same measurement taken + on the withheld reference designs, which is how far real optima the model + never saw sit from the training sweep: one means the model's designs are as + far from its training data as real unseen designs are, near zero means it + reproduces the training set, and well above one means further from the data + than real designs, which noise also achieves. No tolerance anywhere; the + reference designs supply the scale and nothing else. + + Diagnostic, with no direction: zero is retrieval, but large is not good. + Read beside `mmd` and `viol`. NaN when the problem has no training split. + """ + if ctx.train_designs is None: + return {"train_distance": float("nan"), "train_distance_ratio": float("nan")} + generated = ctx.reduce(ctx.nearest_train_distance(ctx.gen_flat)) + reference = ctx.reduce(ctx.nearest_train_distance(ctx.ref_flat)) + return {"train_distance": generated, "train_distance_ratio": generated / reference if reference > 0 else float("nan")} # ---------------------------------------------------------------------- -# Conditions: does the model actually read the brief it was given? +# Conditions: does the model actually use the conditions it was given? # ---------------------------------------------------------------------- @register_metric( "cond_sens", + pixel_only=True, family="conditions", cost="cheap", higher_is_better=None, - description="How much the generated design changes when the conditions are shuffled.", ) def cond_sens(ctx: EvaluationContext) -> float: - """Mean per-element RMS change in output when each sample is given another's conditions. + """How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. + + Mean per-element RMS change in output when each sample is given another's conditions. Sampled from the same seed, so the latent draw is held fixed and the only - thing that varies is the brief. A model that ignores its conditions returns + thing that varies is the conditions. A model that ignores them returns the identical batch and scores 0; a model that responds to them scores the size of that response. @@ -143,7 +143,7 @@ def cond_sens(ctx: EvaluationContext) -> float: if permuted is None: return float("nan") deltas = np.linalg.norm(ctx.gen_flat - permuted, axis=1) / np.sqrt(ctx.gen_flat.shape[1]) - return float(np.mean(deltas)) + return ctx.reduce(deltas) # ---------------------------------------------------------------------- @@ -153,13 +153,13 @@ def cond_sens(ctx: EvaluationContext) -> float: @register_metric( "viol", + pixel_only=True, family="feasibility", cost="cheap", higher_is_better=False, - description="Fraction of designs violating the problem's constraints or their volume budget.", ) def viol(ctx: EvaluationContext) -> float: - """Fraction of infeasible designs; see `EvaluationContext.is_infeasible`. + """Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. Defined for every problem: `problem.check_constraints` always applies, and the spec's `volume_condition` adds the volume-fraction budget for problems @@ -170,6 +170,8 @@ def viol(ctx: EvaluationContext) -> float: the optimizer refuses to start from an invalid design, the case where the answer matters most. """ + if ctx.conditions is None: + return float("nan") values = ctx.feasibility return float(np.mean(values)) if values else float("nan") @@ -184,11 +186,10 @@ def viol(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean initial optimality gap: how far generated designs start from the reference optimum.", ) def iog(ctx: EvaluationContext) -> float: - """Mean initial optimality gap, before any re-optimization.""" - return float(np.mean(ctx.optimization.iog)) + """Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.""" + return ctx.reduce(ctx.optimization.iog) @register_metric( @@ -196,11 +197,20 @@ def iog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean cumulative optimality gap over re-optimization from each generated design.", ) def cog(ctx: EvaluationContext) -> float: - """Mean cumulative optimality gap.""" - return float(np.mean(ctx.optimization.cog)) + """Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. + + For each generated design, the optimizer is started from that design and + `objective(step) - objective(reference optimum)` is summed over every step it + takes; that sum is the area under the design's optimality-gap curve. The + column is the mean of those sums over the generated designs. + + Because it sums a whole trajectory it mixes two things: how far from optimal + the start was, and how many steps the optimizer needed. `iog` isolates the + first and `fog` the end point; read `cog` beside both. + """ + return ctx.reduce(ctx.optimization.cog) @register_metric( @@ -208,8 +218,257 @@ def cog(ctx: EvaluationContext) -> float: family="performance", cost="expensive", higher_is_better=False, - description="Mean final optimality gap after re-optimization.", ) def fog(ctx: EvaluationContext) -> float: - """Mean final optimality gap.""" - return float(np.mean(ctx.optimization.fog)) + """Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.""" + return ctx.reduce(ctx.optimization.fog) + + +# ---------------------------------------------------------------------- +# Conditions: did each design answer the conditions it was given? +# ---------------------------------------------------------------------- + + +@register_metric( + "per_condition_distance", + family="conditions", + cost="cheap", + higher_is_better=False, +) +def per_condition_distance(ctx: EvaluationContext) -> float: + """Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. + + Design `i` is compared with reference design `i`, which answers the same + conditions. That is a sharper question than any set-level metric can ask: + not "does the set look right" but "is *this* design what *these* conditions + called for". A model that returns the correct designs in the wrong order + scores a perfect `mmd` and a large value here. Per-element RMS. + """ + if ctx.gen_flat.shape != ctx.ref_flat.shape: + return float("nan") + distances = np.linalg.norm(ctx.gen_flat - ctx.ref_flat, axis=1) / np.sqrt(ctx.gen_flat.shape[1]) + return ctx.reduce(distances) + + +@register_metric( + "volume_error", + family="conditions", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def volume_error(ctx: EvaluationContext) -> float: + """Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. + + The volume fraction of a density field is its mean, so this needs nothing + fitted: it is the exact error on the one condition that can be read straight + off the design. `viol` reports how many designs missed the budget by more + than the tolerance; this reports by how much. NaN on problems with no volume + condition. + """ + if ctx.volume_condition is None or ctx.conditions is None: + return float("nan") + requested = np.asarray(ctx.conditions[ctx.volume_condition], dtype=np.float64) + realized = ctx.gen_flat.mean(axis=1) + return ctx.reduce(np.abs(realized - requested)) + + +# ---------------------------------------------------------------------- +# Distribution and diversity: does the set reach the reference, and is it varied? +# ---------------------------------------------------------------------- + +COVERAGE_QUANTILE = 0.05 +"""The radius around each reference design is set from the reference set's own +nearest-neighbor spacing, at this upper quantile, so it adapts to the space.""" + + +@register_metric( + "coverage", + family="distribution", + cost="cheap", + higher_is_better=True, +) +def coverage(ctx: EvaluationContext) -> float: + """Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. + + Distribution distance can look healthy while whole regions go unvisited, so + this counts reference designs directly: a mode the generator never produces + leaves its neighborhood empty. Bounded in [0, 1]. + """ + from scipy.spatial.distance import cdist + + within_reference = cdist(ctx.ref_flat, ctx.ref_flat) + np.fill_diagonal(within_reference, np.inf) + radius = float(np.quantile(within_reference.min(axis=1), 1.0 - COVERAGE_QUANTILE)) + nearest_generated = cdist(ctx.ref_flat, ctx.gen_flat).min(axis=1) + return float((nearest_generated <= radius).mean()) + + +@register_metric( + "vendi", + family="diversity", + cost="cheap", + higher_is_better=True, +) +def vendi(ctx: EvaluationContext) -> float: + """Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. + + The exponential of the entropy of the similarity kernel's eigenvalues, so it + reads as a count: n identical designs score 1, n mutually dissimilar ones + score n. Unlike a determinant it does not rise when a collapsed set is + jittered, which is why it is the diversity column to prefer. + Friedman & Dieng (2023), The Vendi Score. + """ + return float(metrics_mod.vendi_score(ctx.gen_flat, sigma=ctx.kernel_sigma)) + + +# ---------------------------------------------------------------------- +# Performance: what does the optimizer still have to do after the warm start? +# ---------------------------------------------------------------------- +# +# `iog` asks whether the generated design is already good. That is the wrong +# question whenever a defect is cheap to repair, and many are: a design at the +# wrong filter length scale is corrected by the first few updates. The metrics +# below price a defect in the currency an engineer pays, which is optimizer +# calls, and read the same trajectory `iog`, `cog` and `fog` summarize. + +NEAR_OPTIMUM_FRACTION = 0.05 +"""A gap within this fraction of the reference optimum's objective counts as near optimal.""" + +GAP_AFTER_CALLS = (1, 2, 5, 10) +"""Call budgets to report the remaining gap at: is a defect gone immediately, or not?""" + + +def _calls_to_near_optimum(path: np.ndarray, reference_objective: float) -> float: + """Calls after which the gap stays within five percent of the reference optimum's objective. + + Last exit rather than first touch, so a path that dips into the band and + leaves again is not credited. Zero means the design was near optimal from + its first call. A path still outside the band at its last call never got + there and counts as one more than the budget, so a design that never + arrives ranks worse than one that is merely slow. + """ + band = NEAR_OPTIMUM_FRACTION * max(abs(reference_objective), 1e-12) + if float(path[-1]) > band: + return float(path.size + 1) + outside = np.flatnonzero(path > band) + return float(outside[-1] + 1) if outside.size else 0.0 + + +@register_metric( + "calls_to_near_optimum", + family="performance", + cost="expensive", + higher_is_better=False, +) +def calls_to_near_optimum(ctx: EvaluationContext) -> float: + """Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. + + The band is set by the optimum for the design's own conditions, not by the + design's starting point, so a design that starts absurdly far away is not + credited with settling merely because its first call removed most of the + absurdity. Each call is one simulate plus one sensitivity evaluation. + """ + results = ctx.optimization + if not results.trajectories: + return float("nan") + calls = [_calls_to_near_optimum(path, scale) for path, scale in zip(results.trajectories, results.reference_objectives)] + return ctx.reduce(calls) + + +@register_metric( + "gap_after_calls", + family="performance", + cost="expensive", + higher_is_better=False, + outputs=tuple(f"gap_after_{k}_calls" for k in GAP_AFTER_CALLS), +) +def gap_after_calls(ctx: EvaluationContext) -> dict[str, float]: + """Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. + + A path shorter than the budget has converged early, so its final value is + carried forward: the optimizer would have spent the remaining calls and + changed nothing. + """ + paths = ctx.optimization.trajectories + if not paths: + return {f"gap_after_{k}_calls": float("nan") for k in GAP_AFTER_CALLS} + return { + f"gap_after_{k}_calls": ctx.reduce([float(path[min(k, path.size) - 1]) for path in paths]) for k in GAP_AFTER_CALLS + } + + +@register_metric( + "reaches_reference_rate", + family="performance", + cost="expensive", + higher_is_better=True, +) +def reaches_reference_rate(ctx: EvaluationContext) -> float: + """Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. + + The gap is measured against the reference optimum, so reaching it means the + gap touches zero at some call. A rate rather than a call count, because a + count is censored by designs that never get there, and a copier reaches + parity in zero calls. Bounded in [0, 1]. + """ + paths = ctx.optimization.trajectories + if not paths: + return float("nan") + return float(np.mean([bool(np.any(np.asarray(path) <= 0.0)) for path in paths])) + + +# ---------------------------------------------------------------------- +# Cost: what does the model cost to run, and what did it cost to make? +# ---------------------------------------------------------------------- + + +@register_metric( + "generation_seconds", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def generation_seconds(ctx: EvaluationContext) -> float: + """Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. + + Wall-clock, so it compares models within one run on one machine, not across + machines. NaN when the designs did not come from a timed `Generator.sample`. + """ + return float("nan") if ctx.sample_seconds is None else float(ctx.sample_seconds) + + +@register_metric( + "n_parameters", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def n_parameters(ctx: EvaluationContext) -> float: + """Number of trainable parameters in the model. Blank for a model with no network. + + The hardware-independent companion to `generation_seconds`. Neither is a + complete account of cost alone: a small diffusion model can be slower than + a large GAN because it samples iteratively. NaN for a model with no network. + """ + return float("nan") if ctx.model_params is None else float(ctx.model_params) + + +@register_metric( + "train_minutes", + family="cost", + cost="cheap", + higher_is_better=False, + pixel_only=True, +) +def train_minutes(ctx: EvaluationContext) -> float: + """Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. + + Prices the decision to adopt the method rather than a forward pass; it is + where a lookup table and a diffusion model differ by orders of magnitude. + NaN unless the checkpoint records it, which today none do -- that gap is + itself a finding. + """ + return float("nan") if ctx.train_minutes is None else float(ctx.train_minutes) diff --git a/engiopt/evaluation/registry.py b/engiopt/evaluation/registry.py index eea3cce9..4da73c2c 100644 --- a/engiopt/evaluation/registry.py +++ b/engiopt/evaluation/registry.py @@ -1,25 +1,31 @@ """Registry of evaluation metrics. -A metric belongs to the *comparison* (generated designs vs a reference set under -a problem), not to a model -- so metrics are registered functions taking an -`EvaluationContext`, rather than methods on a generator. - -Registering a metric is one decorated function:: - - @register_metric("mmd", family="distribution", cost="cheap", higher_is_better=False) - def mmd(ctx: EvaluationContext) -> float: - return engiopt.metrics.mmd(ctx.gen_designs, ctx.ref_designs, sigma=ctx.sigma) - -The declaration is what lets the evaluator keep simulation-free metrics strictly -separate from simulator-backed ones, and what lets the leaderboard know which -direction of each column is "better". +A metric is a function of an `EvaluationContext` -- the generated designs, the +reference designs they are scored against, and what the spec says about how to +read them -- plus a short declaration of what the number means:: + + @register_metric("viol", family="feasibility", cost="cheap", higher_is_better=False) + def viol(ctx: EvaluationContext) -> float: + \"\"\"What fraction of the generated designs violate the problem's constraints?\"\"\" + ... + +The docstring's first line is the metric's description: the question it answers, +written so it makes sense without knowing the metric's name. `higher_is_better` +says which way to rank; `None` means the metric is a diagnostic that is read but +never ranked on. `cost` keeps metrics that touch the simulator separate from the +ones that do not. + +Where a metric is computed -- pixel space, a PCA of the reference designs, a +learned latent space -- and how one value per design collapses to one number are +not properties of the metric. They are chosen when a board is evaluated; see +`engiopt.evaluation.board.Board`. """ from __future__ import annotations from collections.abc import Callable, Iterator, Mapping from dataclasses import dataclass -from typing import Literal, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING if TYPE_CHECKING: from engiopt.evaluation.context import EvaluationContext @@ -41,23 +47,32 @@ def mmd(ctx: EvaluationContext) -> float: @dataclass(frozen=True) class MetricSpec: - """A registered metric and everything the evaluator needs to know about it.""" + """A registered metric and what a reader needs to know to use its column.""" name: str fn: MetricFn family: MetricFamily cost: MetricCost higher_is_better: bool | None - """None for metrics with no intrinsic direction (e.g. a realized volume fraction).""" + """Ranking direction. None: a diagnostic, read beside other columns but never ranked on.""" + description: str = "" + """The question the value answers, in one plain sentence. Defaults to the docstring's first line.""" outputs: tuple[str, ...] = () """Column names produced. Defaults to `(name,)` for single-valued metrics.""" - description: str = "" + pixel_only: bool = False + """True if the metric needs the actual designs -- a constraint check, a copy corpus -- + and so cannot be asked in a PCA or latent space.""" @property def columns(self) -> tuple[str, ...]: """Leaderboard columns this metric fills.""" return self.outputs or (self.name,) + @property + def direction(self) -> str: + """The ranking direction as a reader sees it.""" + return {True: "higher is better", False: "lower is better", None: "diagnostic"}[self.higher_is_better] + class MetricRegistry(Mapping[str, MetricSpec]): """Name-to-`MetricSpec` mapping populated by the `register_metric` decorator.""" @@ -78,7 +93,7 @@ def __len__(self) -> int: return len(self._metrics) def add(self, spec: MetricSpec) -> None: - """Register a metric, rejecting duplicate names and duplicate output columns.""" + """Register a metric, refusing a second definition under the same name or column.""" if spec.name in self._metrics: raise ValueError(f"Metric {spec.name!r} is already registered.") taken = {col: owner.name for owner in self._metrics.values() for col in owner.columns} @@ -94,7 +109,7 @@ def select( cost: MetricCost | None = None, family: MetricFamily | None = None, ) -> list[MetricSpec]: - """Return specs filtered by name, cost, and/or family, in registration order.""" + """Metrics by name, cost, or family -- every metric if nothing is given.""" specs = [self[name] for name in names] if names is not None else list(self._metrics.values()) if cost is not None: specs = [spec for spec in specs if spec.cost == cost] @@ -106,6 +121,20 @@ def columns(self, specs: list[MetricSpec] | None = None) -> list[str]: """All leaderboard columns for the given specs (default: every metric).""" return [col for spec in (specs if specs is not None else self._metrics.values()) for col in spec.columns] + def explain(self) -> Any: + """One row per metric: the question it answers, its family, direction and cost. + + Returns: + A `pandas.DataFrame` indexed by metric name. + """ + import pandas as pd + + rows = { + spec.name: {"question": spec.description, "family": spec.family, "direction": spec.direction, "cost": spec.cost} + for spec in self._metrics.values() + } + return pd.DataFrame.from_dict(rows, orient="index") + METRICS = MetricRegistry() """The global metric registry.""" @@ -117,8 +146,9 @@ def register_metric( family: MetricFamily, cost: MetricCost, higher_is_better: bool | None, - outputs: tuple[str, ...] = (), description: str = "", + outputs: tuple[str, ...] = (), + pixel_only: bool = False, registry: MetricRegistry | None = None, ) -> Callable[[MetricFn], MetricFn]: """Register an evaluation metric. @@ -127,9 +157,10 @@ def register_metric( name: Unique metric name, also the default output column. family: Which engineering question this answers. cost: `cheap` if it never runs a simulator, `expensive` otherwise. - higher_is_better: Ranking direction, or None if the metric is diagnostic. + higher_is_better: Ranking direction, or None for a diagnostic. + description: The question the value answers; defaults to the docstring's first line. outputs: Column names, when the metric returns a dict of several values. - description: One-line explanation, surfaced by `engiopt.evaluate --list-metrics`. + pixel_only: True if the metric needs the actual designs rather than codes in some space. registry: Target registry; defaults to the global one. Returns: @@ -137,15 +168,16 @@ def register_metric( """ def decorator(fn: MetricFn) -> MetricFn: - (registry or METRICS).add( + (METRICS if registry is None else registry).add( MetricSpec( name=name, fn=fn, family=family, cost=cost, higher_is_better=higher_is_better, - outputs=outputs, description=description or (fn.__doc__ or "").strip().split("\n")[0], + outputs=outputs, + pixel_only=pixel_only, ) ) return fn diff --git a/engiopt/evaluation/spec.py b/engiopt/evaluation/spec.py index faf6c36d..4af8e720 100644 --- a/engiopt/evaluation/spec.py +++ b/engiopt/evaluation/spec.py @@ -18,7 +18,7 @@ import json from pathlib import Path import subprocess -from typing import Any, TYPE_CHECKING +from typing import Any, Literal, TYPE_CHECKING import numpy as np import torch as th @@ -82,7 +82,9 @@ class EvalSpec: n_samples: Number of conditions each model is scored on. condition_seed: Seed used to draw the test conditions. metrics: Metric names to compute, in leaderboard column order. - sigma: Gaussian-kernel bandwidth for MMD and DPP. + sigma: Gaussian-kernel bandwidth for the kernel metrics. None, the + default, means the median pairwise distance of the training designs, + resolved at evaluation time and recorded in every row. volfrac_tol: Tolerance for the volume-fraction violation check. volume_condition: Name of the condition holding the volume-fraction budget a design must hit, e.g. beams2d's `volfrac`. Feasibility is @@ -101,16 +103,6 @@ class EvalSpec: submitter running one seed -- it does not stop them running twenty and publishing their best three, which produces a median over a maximum. Fixing *which* seeds removes the choice. - copy_tol: Per-element RMS distance below which a generated design counts - as a copy of a design the model could have memorized. See the - `novelty` metric. - max_copy_rate: The share of copied designs above which an entry is - flagged and left out of the ranking. Set to 1.0 to disable the gate - and report `copy_rate` without acting on it. - copy_corpus_size: How many dataset designs to draw as the memorization - corpus. The scored reference designs are always included on top of - these, since the public spec names them and they are the most - attractive thing to copy. condition_digest: Hash of the drawn indices, condition values, and reference designs. Recomputed at evaluation time and compared, so an upstream dataset change is caught instead of silently shifting every @@ -134,19 +126,41 @@ class EvalSpec: """ problem_id: str - version: str = "v1" + version: str = "v2" n_samples: int = 50 condition_seed: int = 1 - metrics: tuple[str, ...] = ("mmd", "dpp", "novelty", "cond_sens", "viol", "iog", "cog", "fog") - sigma: float = 10.0 + metrics: tuple[str, ...] = ( + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate", + ) + sigma: float | None = None + aggregation: Literal["mean", "median"] = "mean" + """How per-design metrics (`iog`, `cog`, `fog`, distances) collapse to one number. + + Changing it changes what every performance column means, so it is part of + the frozen spec and is recorded in every row, not a flag on the run. + """ volfrac_tol: float = 0.01 volume_condition: str | None = None objective_weights: tuple[float, ...] | None = None objective_weight_condition: str | None = None required_seeds: tuple[int, ...] = (1, 2, 3) - copy_tol: float = 0.01 - max_copy_rate: float = 0.5 - copy_corpus_size: int = 512 condition_digest: str | None = None dataset_id: str | None = None dataset_revision: str | None = None @@ -178,7 +192,7 @@ def load(cls, reference: str, *, root: Path | None = None) -> EvalSpec: if path.suffix == ".json" and path.exists(): return cls(**json.loads(path.read_text())) problem_id, _, version = reference.partition("/") - version = version or "v1" + version = version or "v2" spec_path = (root or SPEC_ROOT) / problem_id / f"{version}.json" if not spec_path.exists(): raise FileNotFoundError( @@ -451,15 +465,12 @@ def freeze_spec( n_samples: int = 50, condition_seed: int = 1, metrics: tuple[str, ...] = EvalSpec.metrics, - sigma: float = 10.0, + sigma: float | None = None, volume_condition: str | None = None, volfrac_tol: float = EvalSpec.volfrac_tol, objective_weights: tuple[float, ...] | None = None, objective_weight_condition: str | None = None, required_seeds: tuple[int, ...] = EvalSpec.required_seeds, - copy_tol: float = EvalSpec.copy_tol, - max_copy_rate: float = EvalSpec.max_copy_rate, - copy_corpus_size: int = EvalSpec.copy_corpus_size, notes: str = "", ) -> Path: """Draw a problem's test conditions once and commit them as a spec. @@ -493,9 +504,6 @@ def freeze_spec( objective_weights=objective_weights, objective_weight_condition=objective_weight_condition, required_seeds=required_seeds, - copy_tol=copy_tol, - max_copy_rate=max_copy_rate, - copy_corpus_size=copy_corpus_size, notes=notes, ).freeze(problem) path = spec.save() diff --git a/engiopt/evaluation/submission.py b/engiopt/evaluation/submission.py index 4b6b05ca..daac2b2e 100644 --- a/engiopt/evaluation/submission.py +++ b/engiopt/evaluation/submission.py @@ -20,13 +20,10 @@ from __future__ import annotations import math -from typing import Any, TYPE_CHECKING +from typing import Any import pandas as pd -if TYPE_CHECKING: - from engiopt.evaluation.spec import EvalSpec - CHECKPOINT_ADDRESS_COLUMNS = ("checkpoint_repo", "checkpoint_path", "checkpoint_revision", "checkpoint_hash") """What it takes to fetch the exact weights a row was scored on. @@ -38,16 +35,13 @@ IDENTITY_COLUMNS = ("problem_id", "algo_id", "config_fingerprint", "seed", "spec_version") """What it takes to know which model, and under which protocol, a row describes.""" -FLAG_MEMORIZED = "memorized" -"""The batch is largely reproductions of designs the model could have seen.""" - FLAG_UNVERIFIED = "unverified" """No runner has re-fetched these weights and reproduced these numbers.""" FLAG_IGNORES_CONDITIONS = "ignores_conditions" """Output did not move when the conditions did, on a problem where they vary.""" -DISQUALIFYING_FLAGS = frozenset({FLAG_MEMORIZED, FLAG_IGNORES_CONDITIONS}) +DISQUALIFYING_FLAGS = frozenset({FLAG_IGNORES_CONDITIONS}) """Flags that keep a row out of the ranking while leaving it on the board. `unverified` is not among them only because ranking already requires a @@ -85,7 +79,7 @@ def admission_problems(row: dict[str, Any]) -> list[str]: return problems -def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[str]: +def integrity_flags(row: dict[str, Any]) -> list[str]: """Which integrity checks this row trips. Row-level only. Whether an entry covers the seeds the spec requires is a @@ -93,10 +87,6 @@ def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[s ranking instead -- see `leaderboard.rank`. """ flags: list[str] = [] - copy_rate = as_metric_value(row.get("copy_rate")) - max_copy_rate = spec.max_copy_rate if spec is not None else 0.5 - if copy_rate is not None and copy_rate > max_copy_rate: - flags.append(FLAG_MEMORIZED) if _declares_conditioning_it_does_not_use(row): flags.append(FLAG_IGNORES_CONDITIONS) if not bool(row.get("verified")): @@ -105,7 +95,7 @@ def integrity_flags(row: dict[str, Any], spec: EvalSpec | None = None) -> list[s def _declares_conditioning_it_does_not_use(row: dict[str, Any]) -> bool: - """Whether a model claiming to be conditional produced identical output for different briefs. + """Whether a model claiming to be conditional produced identical output for different conditions. No tolerance to choose here, and deliberately so: the comparison holds the latent draw fixed, so a model that genuinely reads its conditions cannot @@ -124,12 +114,11 @@ def _declares_conditioning_it_does_not_use(row: dict[str, Any]) -> bool: return generator is not None and generator.conditional -def prepare_submission(frame: pd.DataFrame, spec: EvalSpec | None = None) -> pd.DataFrame: +def prepare_submission(frame: pd.DataFrame) -> pd.DataFrame: """Check a table of new rows and stamp their flags, ready to publish. Args: frame: Rows as produced by `Evaluator.leaderboard`. - spec: The spec they were scored under, for its gate thresholds. Returns: The same rows with `verified` forced False and `flags` filled in. @@ -157,7 +146,7 @@ def prepare_submission(frame: pd.DataFrame, spec: EvalSpec | None = None) -> pd. prepared["verified"] = False prepared["verified_by"] = None prepared["verified_at"] = None - prepared["flags"] = [",".join(integrity_flags(record, spec)) for record in prepared.to_dict("records")] + prepared["flags"] = [",".join(integrity_flags(record)) for record in prepared.to_dict("records")] return prepared diff --git a/engiopt/evaluation/verify.py b/engiopt/evaluation/verify.py index b531a2c8..05294251 100644 --- a/engiopt/evaluation/verify.py +++ b/engiopt/evaluation/verify.py @@ -155,7 +155,7 @@ def verify_row( } ) uncorroborated = _uncorroborated(row, rescored) - rescored["flags"] = ",".join(integrity_flags(rescored, evaluator.spec)) + rescored["flags"] = ",".join(integrity_flags(rescored)) return VerificationResult( key=key, status="verified", diff --git a/engiopt/metrics.py b/engiopt/metrics.py index de8b56d1..47a46cb3 100644 --- a/engiopt/metrics.py +++ b/engiopt/metrics.py @@ -15,6 +15,13 @@ if TYPE_CHECKING: from engibench import OptiStep +MIN_PRDC_SAMPLES = 2 +"""Below two samples per set there is no neighbourhood to measure.""" + +EIGENVALUE_FLOOR = 1e-12 +"""Eigenvalues below this are treated as zero: a PSD matrix can return +slightly negative ones, and exact zeros would send the entropy to -inf.""" + def mmd(x: np.ndarray, y: np.ndarray, sigma: float = 1.0) -> float: """Compute the Maximum Mean Discrepancy (MMD) between two sets of samples. @@ -60,6 +67,218 @@ def dpp_diversity(x: np.ndarray, sigma: float = 1.0) -> float: return 0.0 # fallback in case of numerical issues +def log_dpp_diversity(x: np.ndarray, sigma: float = 1.0) -> float: + """Log-determinant form of `dpp_diversity`, for spaces where the raw determinant underflows. + + The determinant of an `n x n` kernel matrix is a product of `n` numbers below + one, so at `n = 50` it routinely lands near 1e-11 and can reach the limit of + double precision -- at which point the metric stops distinguishing models and + starts reporting rounding error. `slogdet` measures the same quantity on a + scale that survives. + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The log-determinant. Larger means more diverse. Returns `-inf` for a + singular matrix, which is the honest limit of a degenerate sample set. + """ + x_flat = x.reshape(x.shape[0], -1) + similarity = np.exp(-cdist(x_flat, x_flat, "sqeuclidean") / (2 * sigma**2)) + reg_matrix = similarity + 1e-6 * np.eye(x.shape[0]) + + # On a near-singular kernel matrix NumPy's slogdet warns about intermediate + # overflow while still returning a correct result -- which is precisely the + # case this function exists to handle, so the warnings are noise here. + with np.errstate(divide="ignore", over="ignore", invalid="ignore"): + sign, logabsdet = np.linalg.slogdet(reg_matrix) + + if sign <= 0 or not np.isfinite(logabsdet): + return float("-inf") + return float(logabsdet) + + +def dpp_geometric_mean(x: np.ndarray, sigma: float = 1.0) -> float: + r"""`n`-th root of the DPP determinant: the geometric mean of its eigenvalues. + + The same quantity `dpp_diversity` reports, on the only scale of the three + that is readable *and* comparable across sample sizes. + + - The raw determinant is a product of `n` numbers below one, so it underflows + (1e-11 at n=50, and worse) and distinct models render as identical zeros. + - The log-determinant fixes the underflow but is unbounded below and still + scales with `n`, so a value means nothing without knowing the sample count + it was computed at -- and two papers reporting "log-DPP" at n=50 and n=200 + are not comparable. + - `det(K)^(1/n) = exp(logdet / n)` is the geometric mean of the eigenvalues. + Since `K` has a unit diagonal its eigenvalues sum to `n`, so their + arithmetic mean is exactly 1 and the geometric mean lands in `(0, 1]` by + AM-GM: **1 means a perfectly diverse set (`K = I`), and values approach 0 + as samples collapse onto each other.** That bound holds at every `n`. + + Interpretation is therefore fixed rather than relative: 0.5 is the same + statement about a set of 50 designs as about a set of 500. + + This does **not** fix the pathology `vendi_score` exists to resist -- a + determinant still rewards any perturbation that pushes samples apart, so + adding noise to a collapsed set raises this too. It fixes the *numerical* + failure only, which is a different and more embarrassing one. + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The geometric mean of the kernel eigenvalues, in `(0, 1]`. Returns 0.0 + for a degenerate set, which is the honest limit rather than `-inf`. + """ + n = x.shape[0] + if n == 0: + return 0.0 + + logdet = log_dpp_diversity(x, sigma=sigma) + if not np.isfinite(logdet): + return 0.0 + return float(np.exp(logdet / n)) + + +def vendi_score(x: np.ndarray, sigma: float = 1.0) -> float: + """Effective number of distinct samples in a set. + + The exponential of the von Neumann entropy of the normalized similarity + matrix, which reads directly as a count: `n` identical samples score 1, and + `n` mutually dissimilar ones score `n`. + + Preferred over `dpp_diversity` for diversity. A determinant rewards any + perturbation that makes samples less similar, so adding noise to a collapsed + set *raises* it -- exactly the gaming this measure is meant to resist. The + entropy of the eigenvalue spectrum does not move that way, because noise + spreads eigenvalues without adding modes. + + See Friedman & Dieng (2023), "The Vendi Score". + + Args: + x: Samples of shape `(n, ...)`; flattened internally. + sigma: Bandwidth of the Gaussian kernel. + + Returns: + The effective sample count, in `[1, n]`. + """ + x_flat = x.reshape(x.shape[0], -1) + n = x_flat.shape[0] + if n == 0: + return 0.0 + + kernel = np.exp(-cdist(x_flat, x_flat, "sqeuclidean") / (2 * sigma**2)) / n + eigenvalues = np.linalg.eigvalsh(kernel) + + # Clip: eigenvalues of a PSD matrix can come back slightly negative, and + # zero eigenvalues contribute nothing to entropy but would produce -inf. + positive = eigenvalues[eigenvalues > EIGENVALUE_FLOOR] + if positive.size == 0: + return 1.0 + return float(np.exp(-np.sum(positive * np.log(positive)))) + + +def compute_median_sigma(x: np.ndarray, y: np.ndarray | None = None) -> float: + """Choose a kernel bandwidth by the median heuristic. + + A fixed bandwidth cannot serve two spaces at once: pixel space and a pruned + latent space differ in scale by orders of magnitude, and a kernel sized for + one saturates in the other. A bandwidth taken from the data adapts to + whichever space it is handed. + + Precisely, this returns `sqrt(median(d^2) / 2)`, which is the **median + distance divided by sqrt(2)** -- the `gamma = 1 / median(d^2)` form of the + median heuristic. Under the kernel `exp(-d^2 / 2 sigma^2)` that puts a pair + at the median distance at `exp(-1)`, not `exp(-0.5)`. Both conventions are + called "the median heuristic"; this is the one in force, it is applied + identically in pixel, PCA and latent space, and every published column was + computed under it. Said explicitly because the previous wording read as + "sigma is the median distance", which it is not. + + Sampling is capped and seeded so the bandwidth is reproducible. + + Args: + x: Samples of shape `(n, d)`. + y: Optional second sample set; distances are then cross-set. + + Returns: + The bandwidth, floored at 1e-6 so a degenerate set cannot divide by zero. + """ + x = x.reshape(x.shape[0], -1) + n_sample = min(500, len(x)) + rng = np.random.default_rng(42) + idx_x = rng.choice(len(x), n_sample, replace=len(x) < n_sample) + + if y is not None: + y = y.reshape(y.shape[0], -1) + idx_y = rng.choice(len(y), n_sample, replace=len(y) < n_sample) + dists = cdist(x[idx_x], y[idx_y], "sqeuclidean") + else: + dists = cdist(x[idx_x], x[idx_x], "sqeuclidean") + dists = dists[np.triu_indices_from(dists, k=1)] + + sigma = np.sqrt(np.median(dists) / 2) if len(dists) > 0 else 1.0 + return max(float(sigma), 1e-6) + + +def compute_prdc(real_features: np.ndarray, fake_features: np.ndarray, nearest_k: int = 5) -> dict[str, float]: + """Compute precision, recall, density, and coverage between two sample sets. + + The four metrics of Naeem et al. 2020, "Reliable Fidelity and Diversity + Metrics for Generative Models" (ICML). They exist because a single + distribution distance conflates two different failures: generating + implausible designs, and generating too few distinct ones. MMD reports one + number for both; these separate them. + + All four lie in `[0, 1]`, higher is better. + + - **precision**: fraction of generated samples inside some real sample's + k-NN ball -- are the generations plausible? + - **recall**: fraction of real samples inside some generated sample's ball + -- is the data manifold covered? + - **density**: smoothed precision, robust to real outliers that would + otherwise inflate it. + - **coverage**: fraction of real samples whose own ball contains a generated + sample; more robust than recall when generations are noisy outliers. + + Args: + real_features: Reference samples, shape `(n_real, d)`. + fake_features: Generated samples, shape `(n_fake, d)`. + nearest_k: Neighbours defining the ball radius, clamped to fit the sets. + + Returns: + Mapping with keys `precision`, `recall`, `density`, `coverage`. All NaN + when either set has fewer than two samples. + """ + real = real_features.reshape(real_features.shape[0], -1) + fake = fake_features.reshape(fake_features.shape[0], -1) + n_real, n_fake = len(real), len(fake) + + if n_real < MIN_PRDC_SAMPLES or n_fake < MIN_PRDC_SAMPLES: + return dict.fromkeys(("precision", "recall", "density", "coverage"), float("nan")) + + k = max(1, min(nearest_k, n_real - 1, n_fake - 1)) + + # Self-distance sits at index 0 of the partition, so index k is the k-th + # non-self neighbour. + real_radii = np.partition(cdist(real, real, "euclidean"), k, axis=1)[:, k] + fake_radii = np.partition(cdist(fake, fake, "euclidean"), k, axis=1)[:, k] + d_rf = cdist(real, fake, "euclidean") + + inside_real = d_rf <= real_radii[:, None] + inside_fake = d_rf <= fake_radii[None, :] + + return { + "precision": float(inside_real.any(axis=0).mean()), + "recall": float(inside_fake.any(axis=1).mean()), + "density": float(inside_real.sum(axis=0).mean()) / k, + "coverage": float((d_rf.min(axis=1) <= real_radii).mean()), + } + + def optimality_gap(opt_history: list[OptiStep], baseline: float) -> list[float]: """Compute the optimality gap of an optimization history. @@ -71,6 +290,3 @@ def optimality_gap(opt_history: list[OptiStep], baseline: float) -> list[float]: list[float]: The optimality gap at each step in opt_history. """ return [opt.obj_values - baseline for opt in opt_history] - - -# Return the failure ratio diff --git a/engiopt/specs/beams2d/v1.json b/engiopt/specs/beams2d/v1.json index 1e3e5a43..f8926c05 100644 --- a/engiopt/specs/beams2d/v1.json +++ b/engiopt/specs/beams2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "f65e2c02551e7e08", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/beams_2d_50_100_v0", "dataset_revision": "ccb64d4f09cf66a5ed9ba2081438d695b9e2438d", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/beams2d/v2.json b/engiopt/specs/beams2d/v2.json new file mode 100644 index 00000000..5f51fbaa --- /dev/null +++ b/engiopt/specs/beams2d/v2.json @@ -0,0 +1,48 @@ +{ + "condition_digest": "f65e2c02551e7e08", + "condition_seed": 1, + "dataset_id": "IDEALLab/beams_2d_50_100_v0", + "dataset_revision": "ccb64d4f09cf66a5ed9ba2081438d695b9e2438d", + "engibench_version": "0.2.0+0a028c03d02b", + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate" + ], + "n_samples": 50, + "notes": "Feasibility = EngiBench constraints plus the volfrac budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "volfrac", + "rmin", + "forcedist", + "overhang_constraint" + ], + "problem_id": "beams2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": null, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volfrac", + "aggregation": "mean" +} diff --git a/engiopt/specs/heatconduction2d/v1.json b/engiopt/specs/heatconduction2d/v1.json index b2186a46..a0221ecd 100644 --- a/engiopt/specs/heatconduction2d/v1.json +++ b/engiopt/specs/heatconduction2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "40030d1b2482e003", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/heat_conduction_2d_v0", "dataset_revision": "9f07c1e70d244f783e34a71c96db621173871dc1", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/heatconduction2d/v2.json b/engiopt/specs/heatconduction2d/v2.json new file mode 100644 index 00000000..bdfed215 --- /dev/null +++ b/engiopt/specs/heatconduction2d/v2.json @@ -0,0 +1,46 @@ +{ + "condition_digest": "40030d1b2482e003", + "condition_seed": 1, + "dataset_id": "IDEALLab/heat_conduction_2d_v0", + "dataset_revision": "9f07c1e70d244f783e34a71c96db621173871dc1", + "engibench_version": "0.2.0+0a028c03d02b", + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate" + ], + "n_samples": 50, + "notes": "Feasibility = EngiBench constraints plus the volume budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "volume", + "length" + ], + "problem_id": "heatconduction2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": null, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volume", + "aggregation": "mean" +} diff --git a/engiopt/specs/photonics2d/v1.json b/engiopt/specs/photonics2d/v1.json index e00b6341..f0171df9 100644 --- a/engiopt/specs/photonics2d/v1.json +++ b/engiopt/specs/photonics2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "6b8330bee5a8718c", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/photonics_2d_120_120_v1", "dataset_revision": "be811ff69d67cd8c79148e29f2a68d293ec00955", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/photonics2d/v2.json b/engiopt/specs/photonics2d/v2.json new file mode 100644 index 00000000..a0fe2064 --- /dev/null +++ b/engiopt/specs/photonics2d/v2.json @@ -0,0 +1,50 @@ +{ + "condition_digest": "6b8330bee5a8718c", + "condition_seed": 1, + "dataset_id": "IDEALLab/photonics_2d_120_120_v1", + "dataset_revision": "be811ff69d67cd8c79148e29f2a68d293ec00955", + "engibench_version": "0.2.0+0a028c03d02b", + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate" + ], + "n_samples": 50, + "notes": "No volume budget: feasibility is EngiBench's declared constraints alone. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": null, + "objective_weights": null, + "problem_conditions": [ + "lambda1", + "lambda2", + "blur_radius", + "num_elems_x", + "num_elems_y", + "num_optimization_steps" + ], + "problem_id": "photonics2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": null, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": null, + "aggregation": "mean" +} diff --git a/engiopt/specs/thermoelastic2d/v1.json b/engiopt/specs/thermoelastic2d/v1.json index 38c4302f..0271b3d6 100644 --- a/engiopt/specs/thermoelastic2d/v1.json +++ b/engiopt/specs/thermoelastic2d/v1.json @@ -1,12 +1,9 @@ { "condition_digest": "818d93c2a2d5ac25", "condition_seed": 1, - "copy_corpus_size": 512, - "copy_tol": 0.01, "dataset_id": "IDEALLab/thermoelastic_2d_v1", "dataset_revision": "7d5b4a694bd1a8f3f25fd745ac9f669764e29ea5", "engibench_version": "0.2.0+0a028c03d02b", - "max_copy_rate": 0.5, "metrics": [ "mmd", "dpp", diff --git a/engiopt/specs/thermoelastic2d/v2.json b/engiopt/specs/thermoelastic2d/v2.json new file mode 100644 index 00000000..977cbd97 --- /dev/null +++ b/engiopt/specs/thermoelastic2d/v2.json @@ -0,0 +1,51 @@ +{ + "condition_digest": "818d93c2a2d5ac25", + "condition_seed": 1, + "dataset_id": "IDEALLab/thermoelastic_2d_v1", + "dataset_revision": "7d5b4a694bd1a8f3f25fd745ac9f669764e29ea5", + "engibench_version": "0.2.0+0a028c03d02b", + "metrics": [ + "mmd", + "coverage", + "vendi", + "dpp", + "viol", + "volume_error", + "per_condition_distance", + "cond_sens", + "train_distance", + "generation_seconds", + "n_parameters", + "train_minutes", + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate" + ], + "n_samples": 50, + "notes": "Objectives combined per-sample by the `weight` condition. Feasibility = EngiBench constraints plus the volume_fraction_target budget. v2: the full metric suite on the one-decorator contract; `dpp` is now the n-th-root (geometric-mean) form and is not comparable with v1's raw determinant; `novelty` became `train_distance`; aggregation is declared here and recorded in every row.", + "objective_weight_condition": "weight", + "objective_weights": null, + "problem_conditions": [ + "fixed_elements", + "force_elements_x", + "force_elements_y", + "heatsink_elements", + "volume_fraction_target", + "rmin", + "weight" + ], + "problem_id": "thermoelastic2d", + "required_seeds": [ + 1, + 2, + 3 + ], + "sigma": null, + "version": "v2", + "volfrac_tol": 0.01, + "volume_condition": "volume_fraction_target", + "aggregation": "mean" +} diff --git a/example_metrics_suite.ipynb b/example_metrics_suite.ipynb new file mode 100644 index 00000000..2237acfe --- /dev/null +++ b/example_metrics_suite.ipynb @@ -0,0 +1,1856 @@ +{ + "cells": [ + { + "cell_type": "markdown", + "id": "0c0d6c21", + "metadata": {}, + "source": [ + "# Using the metrics suite, on beams2d\n", + "\n", + "You have trained models and want to know what their designs are worth. This notebook\n", + "runs the whole evaluation surface on a real EngiBench problem, with published\n", + "checkpoints, the physics metrics included:\n", + "\n", + "| You want to... | You call |\n", + "|---|---|\n", + "| see every metric and what it measures | `METRICS.explain()` |\n", + "| score published checkpoints by name | `Board.load(\"beams2d\", [\"vqgan\", ...])` |\n", + "| score checkpoints and your own constructions under one spec | `Board.from_evaluator(evaluator, models, designs=...)` |\n", + "| read the result | `board.explain()`, `board.rank(\"fog\")` |\n", + "| ask the same questions in another space | `Board(...).evaluate(board.designs, space=\"pca\")` |\n", + "| add a metric of your own | `@register_metric(...)` on a function |\n", + "\n", + "It downloads four checkpoints and the beams2d dataset from the Hub, samples twelve\n", + "designs per model, and runs the optimizer from each of them. Expect a few minutes on a\n", + "laptop." + ] + }, + { + "cell_type": "markdown", + "id": "e8b6c327", + "metadata": {}, + "source": [ + "## 1. What metrics exist, and what does each one measure?" + ] + }, + { + "cell_type": "code", + "execution_count": 1, + "id": "95381ed9", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:19:46.072636Z", + "iopub.status.busy": "2026-09-29T09:19:46.072216Z", + "iopub.status.idle": "2026-09-29T09:19:48.724305Z", + "shell.execute_reply": "2026-09-29T09:19:48.724074Z" + } + }, + "outputs": [ + { + "name": "stderr", + "output_type": "stream", + "text": [ + "/opt/anaconda3/envs/EngiBench312/lib/python3.12/site-packages/gymnasium/spaces/box.py:235: UserWarning: \u001b[33mWARN: Box low's precision lowered by casting to float32, current low.dtype=float64\u001b[0m\n", + " gym.logger.warn(\n", + "/opt/anaconda3/envs/EngiBench312/lib/python3.12/site-packages/gymnasium/spaces/box.py:305: UserWarning: \u001b[33mWARN: Box high's precision lowered by casting to float32, current high.dtype=float64\u001b[0m\n", + " gym.logger.warn(\n" + ] + }, + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questionfamilydirectioncost
mmdDistance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.distributionlower is bettercheap
dppSpread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed.diversityhigher is bettercheap
train_distancePer-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.memorizationdiagnosticcheap
cond_sensHow much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.conditionsdiagnosticcheap
violFraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied.feasibilitylower is bettercheap
iogOptimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.performancelower is betterexpensive
cogOptimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted.performancelower is betterexpensive
fogOptimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.performancelower is betterexpensive
per_condition_distancePer-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.conditionslower is bettercheap
volume_errorGap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.conditionslower is bettercheap
coverageFraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.distributionhigher is bettercheap
vendiEffective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.diversityhigher is bettercheap
calls_to_near_optimumOptimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there.performancelower is betterexpensive
gap_after_callsOptimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.performancelower is betterexpensive
reaches_reference_rateFraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there.performancehigher is betterexpensive
generation_secondsWall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.costlower is bettercheap
n_parametersNumber of trainable parameters in the model. Blank for a model with no network.costlower is bettercheap
train_minutesWall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.costlower is bettercheap
\n", + "
" + ], + "text/plain": [ + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "calls_to_near_optimum Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. \n", + "gap_after_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", + "\n", + " family direction cost \n", + "mmd distribution lower is better cheap \n", + "dpp diversity higher is better cheap \n", + "train_distance memorization diagnostic cheap \n", + "cond_sens conditions diagnostic cheap \n", + "viol feasibility lower is better cheap \n", + "iog performance lower is better expensive \n", + "cog performance lower is better expensive \n", + "fog performance lower is better expensive \n", + "per_condition_distance conditions lower is better cheap \n", + "volume_error conditions lower is better cheap \n", + "coverage distribution higher is better cheap \n", + "vendi diversity higher is better cheap \n", + "calls_to_near_optimum performance lower is better expensive \n", + "gap_after_calls performance lower is better expensive \n", + "reaches_reference_rate performance higher is better expensive \n", + "generation_seconds cost lower is better cheap \n", + "n_parameters cost lower is better cheap \n", + "train_minutes cost lower is better cheap " + ] + }, + "execution_count": 1, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "import warnings\n", + "\n", + "import pandas as pd\n", + "\n", + "from engiopt.evaluation.registry import METRICS\n", + "\n", + "# Importing the metrics package registers the built-in metrics.\n", + "import engiopt.evaluation.metrics # noqa: F401 # isort: skip\n", + "\n", + "warnings.filterwarnings(\"ignore\")\n", + "pd.set_option(\"display.max_colwidth\", None)\n", + "pd.set_option(\"display.width\", 200)\n", + "\n", + "METRICS.explain()" + ] + }, + { + "cell_type": "markdown", + "id": "30340537", + "metadata": {}, + "source": [ + "Each row names the quantity, says how it is computed, and says what the two ends of\n", + "its scale mean. The direction column marks some metrics **diagnostic**: those columns\n", + "tell you whether the other columns can be trusted, and are never ranked on." + ] + }, + { + "cell_type": "markdown", + "id": "810b2c11", + "metadata": {}, + "source": [ + "## 2. The problem, and the evaluation spec" + ] + }, + { + "cell_type": "markdown", + "id": "a5e804dc", + "metadata": {}, + "source": [ + "Every published number on beams2d is computed under one frozen spec. The spec fixes\n", + "which fifty test conditions are scored, the reference optimum for each, and how\n", + "per-design values collapse to one number. The kernel bandwidth behind `mmd`, `vendi`\n", + "and `dpp` is the median pairwise distance of the training designs unless the spec pins\n", + "a value, and the resolved number is recorded in every row. Two\n", + "models scored under the same spec are comparable; two scored under different specs\n", + "are not, and every row records which spec produced it.\n", + "\n", + "The dataset has one optimal design per set of conditions. That is what makes a\n", + "lookup table a serious competitor here, and it is why the spec draws its conditions\n", + "from the withheld test split.\n", + "\n", + "This notebook departs from the published spec in two declared ways, so that it runs in\n", + "minutes. It scores twelve conditions rather than fifty, which cuts sampling and\n", + "optimizer time by the same factor, and so it drops the published spec's condition\n", + "digest, which pins the fifty. And it aggregates per-design values with the median rather\n", + "than the mean, because one diverged design drags a mean by orders of magnitude and the\n", + "physics columns are read as medians throughout EngiOpt. Numbers here are therefore not\n", + "comparable with the public leaderboard; the calls are identical." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "e6a82efe", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:19:48.726734Z", + "iopub.status.busy": "2026-09-29T09:19:48.726545Z", + "iopub.status.idle": "2026-09-29T09:19:50.397378Z", + "shell.execute_reply": "2026-09-29T09:19:50.397093Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "conditions scored: 12 | design shape: (50, 100)\n", + "condition keys: ('volfrac', 'rmin', 'forcedist', 'overhang_constraint')\n", + "kernel bandwidth: median distance of the training designs | aggregation: median\n" + ] + } + ], + "source": [ + "import dataclasses\n", + "\n", + "from engiopt.evaluation.evaluator import Evaluator\n", + "from engiopt.evaluation.spec import EvalSpec\n", + "\n", + "published = EvalSpec.load(\"beams2d/v2\")\n", + "spec = dataclasses.replace(published, n_samples=12, condition_digest=None, aggregation=\"median\")\n", + "evaluator = Evaluator.for_problem(\"beams2d\", spec=spec)\n", + "\n", + "REFERENCE = evaluator.resolved.ref_designs # the withheld optimum for each scored condition\n", + "print(\"conditions scored:\", len(REFERENCE), \"| design shape:\", REFERENCE.shape[1:])\n", + "print(\"condition keys:\", evaluator.resolved.condition_keys)\n", + "print(\"kernel bandwidth:\", spec.sigma or \"median distance of the training designs\", \"| aggregation:\", spec.aggregation)" + ] + }, + { + "cell_type": "markdown", + "id": "a42d95f7", + "metadata": {}, + "source": [ + "## 3. Three constructions built from the dataset" + ] + }, + { + "cell_type": "markdown", + "id": "9cd57fca", + "metadata": {}, + "source": [ + "Beside the trained models, three rows whose right score is known in advance. They are\n", + "not competitors. They are the probes that show what each column can and cannot see.\n", + "\n", + "| Row | What it does | What it should reveal |\n", + "|---|---|---|\n", + "| `lookup_table` | for each scored condition, the training design whose conditions are nearest | a model that never generalizes, yet answers each condition plausibly |\n", + "| `unconditional` | training designs drawn at random, ignoring the conditions | a model that ignores its conditions entirely |\n", + "| `permuted_conditions` | the reference optima, shuffled across the conditions | a perfect set of designs, every one for the wrong conditions |" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "ed671c87", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:19:50.399757Z", + "iopub.status.busy": "2026-09-29T09:19:50.399657Z", + "iopub.status.idle": "2026-09-29T09:19:53.339577Z", + "shell.execute_reply": "2026-09-29T09:19:53.339355Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "training designs: (3880, 50, 100)\n" + ] + } + ], + "source": [ + "import numpy as np\n", + "\n", + "train = evaluator.problem.dataset[\"train\"]\n", + "TRAIN = np.stack([np.asarray(design) for design in train[\"optimal_design\"]])\n", + "train_conditions = np.column_stack([np.asarray(train[key], dtype=float) for key in evaluator.resolved.condition_keys])\n", + "scored_conditions = evaluator.resolved.conditions_tensor.cpu().numpy().astype(float)\n", + "\n", + "# Nearest neighbour in condition space, each condition standardized so none dominates by its units.\n", + "scale = train_conditions.std(axis=0)\n", + "scale[scale == 0] = 1.0\n", + "nearest = np.argmin(np.linalg.norm((train_conditions[None] - scored_conditions[:, None]) / scale, axis=2), axis=1)\n", + "\n", + "rng = np.random.default_rng(0)\n", + "CONSTRUCTIONS = {\n", + " \"lookup_table\": TRAIN[nearest],\n", + " \"unconditional\": TRAIN[rng.integers(0, len(TRAIN), len(REFERENCE))],\n", + " \"permuted_conditions\": REFERENCE[rng.permutation(len(REFERENCE))],\n", + "}\n", + "print(\"training designs:\", TRAIN.shape)" + ] + }, + { + "cell_type": "markdown", + "id": "d7550774", + "metadata": {}, + "source": [ + "## 4. Score the trained models and the constructions under one spec" + ] + }, + { + "cell_type": "markdown", + "id": "cdaf60e3", + "metadata": {}, + "source": [ + "Four published checkpoints at training seed one. Two of them were trained with a\n", + "configuration that is not the script's default, so they are named by the fingerprint\n", + "of that configuration rather than by name alone. Everything else is one call.\n", + "\n", + "The physics columns run the optimizer from each design, so each cell below takes a\n", + "minute or two. The GAN and VQGAN family first, then the diffusion model, which samples\n", + "slowly, then the constructions." + ] + }, + { + "cell_type": "code", + "execution_count": 4, + "id": "13d92b40", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:19:53.341784Z", + "iopub.status.busy": "2026-09-29T09:19:53.341697Z", + "iopub.status.idle": "2026-09-29T09:22:29.216672Z", + "shell.execute_reply": "2026-09-29T09:22:29.216442Z" + } + }, + "outputs": [ + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "fdffa82ec9bd43a49684f69d954dbf18", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Fetching 8 files: 0%| | 0/8 [00:00\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmdcoveragevendidppviolvolume_errorper_condition_distancecond_senstrain_distancetrain_distance_ratio...train_minutesiogcogfogcalls_to_near_optimumgap_after_1_callsgap_after_2_callsgap_after_5_callsgap_after_10_callsreaches_reference_rate
cgan_cnn_2d0.53370.502.05040.02660.91670.07220.41440.24070.290214.2969...NaN1.816083e+091.735349e+093.075878.01.735347e+091018.6727174.616956.50270.1667
gan_cnn_2d0.82280.251.02900.00021.00000.24070.44470.00000.258112.7131...NaN8.950722e+098.689641e+090.001917.58.689632e+094121.7064189.387547.18830.4167
vqgan0.00961.005.86420.43460.33330.00580.08330.47040.03531.7389...NaN2.413100e+008.195000e-01-0.61941.52.423100e+005.60370.51940.01350.6667
\n", + "

3 rows × 22 columns

\n", + "" + ], + "text/plain": [ + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog fog \\\n", + "cgan_cnn_2d 0.5337 0.50 2.0504 0.0266 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... NaN 1.816083e+09 1.735349e+09 3.0758 \n", + "gan_cnn_2d 0.8228 0.25 1.0290 0.0002 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... NaN 8.950722e+09 8.689641e+09 0.0019 \n", + "vqgan 0.0096 1.00 5.8642 0.4346 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... NaN 2.413100e+00 8.195000e-01 -0.6194 \n", + "\n", + " calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "cgan_cnn_2d 78.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 \n", + "gan_cnn_2d 17.5 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 \n", + "vqgan 1.5 2.423100e+00 5.6037 0.5194 0.0135 0.6667 \n", + "\n", + "[3 rows x 22 columns]" + ] + }, + "execution_count": 4, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from engiopt.evaluation.board import Board\n", + "from engiopt.utils.all_generators import BUILTIN_GENERATORS\n", + "\n", + "EXPENSIVE = True # set False to skip the optimizer and get the cheap columns in seconds\n", + "\n", + "\n", + "def load(name, fingerprint=None):\n", + " \"\"\"The published beams2d checkpoint of a model family at training seed one.\"\"\"\n", + " return BUILTIN_GENERATORS[name].from_pretrained(\n", + " evaluator.problem, problem_id=\"beams2d\", seed=1, config_fingerprint=fingerprint\n", + " )\n", + "\n", + "\n", + "fast_models = {\n", + " \"cgan_cnn_2d\": load(\"cgan_cnn_2d\", \"825831f6\"),\n", + " \"gan_cnn_2d\": load(\"gan_cnn_2d\", \"6293adb3\"),\n", + " \"vqgan\": load(\"vqgan\"),\n", + "}\n", + "board = Board.from_evaluator(evaluator, fast_models, expensive=EXPENSIVE)\n", + "board.frame.round(4)" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "1c9be14a", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:22:29.219834Z", + "iopub.status.busy": "2026-09-29T09:22:29.219741Z", + "iopub.status.idle": "2026-09-29T09:27:42.199916Z", + "shell.execute_reply": "2026-09-29T09:27:42.199651Z" + } + }, + "outputs": [ + { + "data": { + "application/vnd.jupyter.widget-view+json": { + "model_id": "d116c8f9594549a9b8684d6ff7d865a6", + "version_major": 2, + "version_minor": 0 + }, + "text/plain": [ + "Fetching 3 files: 0%| | 0/3 [00:00\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmdcoveragevendidppviolvolume_errorper_condition_distancecond_senstrain_distancetrain_distance_ratio...train_minutesiogcogfogcalls_to_near_optimumgap_after_1_callsgap_after_2_callsgap_after_5_callsgap_after_10_callsreaches_reference_rate
diffusion_2d_cond0.10411.06.08260.47840.83330.12710.15320.39690.1437.0439...NaN-1.72116.0217-0.91284.0-1.712726.39796.19040.31090.9167
\n", + "

1 rows × 22 columns

\n", + "" + ], + "text/plain": [ + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog fog \\\n", + "diffusion_2d_cond 0.1041 1.0 6.0826 0.4784 0.8333 0.1271 0.1532 0.3969 0.143 7.0439 ... NaN -1.7211 6.0217 -0.9128 \n", + "\n", + " calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "diffusion_2d_cond 4.0 -1.7127 26.3979 6.1904 0.3109 0.9167 \n", + "\n", + "[1 rows x 22 columns]" + ] + }, + "execution_count": 5, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "diffusion = Board.from_evaluator(evaluator, {\"diffusion_2d_cond\": load(\"diffusion_2d_cond\")}, expensive=EXPENSIVE)\n", + "board.frame = pd.concat([board.frame, diffusion.frame])\n", + "board.designs.update(diffusion.designs)\n", + "diffusion.frame.round(4)" + ] + }, + { + "cell_type": "code", + "execution_count": 6, + "id": "3187d624", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:27:42.202759Z", + "iopub.status.busy": "2026-09-29T09:27:42.202669Z", + "iopub.status.idle": "2026-09-29T09:29:08.291208Z", + "shell.execute_reply": "2026-09-29T09:29:08.290997Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmdcoveragevendidppviolvolume_errorper_condition_distancecond_senstrain_distancetrain_distance_ratio...train_minutesiogcogfogcalls_to_near_optimumgap_after_1_callsgap_after_2_callsgap_after_5_callsgap_after_10_callsreaches_reference_rate
cgan_cnn_2d0.53370.502.05040.02660.91670.07220.41440.24070.290214.2969...NaN1.816083e+091.735349e+093.075878.01.735347e+091018.6727174.616956.50270.1667
gan_cnn_2d0.82280.251.02900.00021.00000.24070.44470.00000.258112.7131...NaN8.950722e+098.689641e+090.001917.58.689632e+094121.7064189.387547.18830.4167
vqgan0.00961.005.86420.43460.33330.00580.08330.47040.03531.7389...NaN2.413100e+008.195000e-01-0.61941.52.423100e+005.60370.51940.01350.6667
diffusion_2d_cond0.10411.006.08260.47840.83330.12710.15320.39690.14307.0439...NaN-1.721100e+006.021700e+00-0.91284.0-1.712700e+0026.39796.19040.31090.9167
lookup_table0.03951.005.36110.17960.75000.01250.0569NaN0.00000.0000...NaN9.574600e+007.314000e-01-0.73633.59.590800e+008.28253.33130.36760.5833
unconditional0.09021.007.01330.57790.91670.05620.3993NaN0.00000.0000...NaN4.805870e+032.817597e+041.4287101.04.806746e+031994.096366.67044.23960.1667
permuted_conditions0.00001.006.18700.20390.91670.05620.4231NaN0.02031.0000...NaN8.783402e+078.788590e+074.4569101.08.782727e+071865.304085.998227.05100.0000
\n", + "

7 rows × 22 columns

\n", + "
" + ], + "text/plain": [ + " mmd coverage vendi dpp viol volume_error per_condition_distance cond_sens train_distance train_distance_ratio ... train_minutes iog cog \\\n", + "cgan_cnn_2d 0.5337 0.50 2.0504 0.0266 0.9167 0.0722 0.4144 0.2407 0.2902 14.2969 ... NaN 1.816083e+09 1.735349e+09 \n", + "gan_cnn_2d 0.8228 0.25 1.0290 0.0002 1.0000 0.2407 0.4447 0.0000 0.2581 12.7131 ... NaN 8.950722e+09 8.689641e+09 \n", + "vqgan 0.0096 1.00 5.8642 0.4346 0.3333 0.0058 0.0833 0.4704 0.0353 1.7389 ... NaN 2.413100e+00 8.195000e-01 \n", + "diffusion_2d_cond 0.1041 1.00 6.0826 0.4784 0.8333 0.1271 0.1532 0.3969 0.1430 7.0439 ... NaN -1.721100e+00 6.021700e+00 \n", + "lookup_table 0.0395 1.00 5.3611 0.1796 0.7500 0.0125 0.0569 NaN 0.0000 0.0000 ... NaN 9.574600e+00 7.314000e-01 \n", + "unconditional 0.0902 1.00 7.0133 0.5779 0.9167 0.0562 0.3993 NaN 0.0000 0.0000 ... NaN 4.805870e+03 2.817597e+04 \n", + "permuted_conditions 0.0000 1.00 6.1870 0.2039 0.9167 0.0562 0.4231 NaN 0.0203 1.0000 ... NaN 8.783402e+07 8.788590e+07 \n", + "\n", + " fog calls_to_near_optimum gap_after_1_calls gap_after_2_calls gap_after_5_calls gap_after_10_calls reaches_reference_rate \n", + "cgan_cnn_2d 3.0758 78.0 1.735347e+09 1018.6727 174.6169 56.5027 0.1667 \n", + "gan_cnn_2d 0.0019 17.5 8.689632e+09 4121.7064 189.3875 47.1883 0.4167 \n", + "vqgan -0.6194 1.5 2.423100e+00 5.6037 0.5194 0.0135 0.6667 \n", + "diffusion_2d_cond -0.9128 4.0 -1.712700e+00 26.3979 6.1904 0.3109 0.9167 \n", + "lookup_table -0.7363 3.5 9.590800e+00 8.2825 3.3313 0.3676 0.5833 \n", + "unconditional 1.4287 101.0 4.806746e+03 1994.0963 66.6704 4.2396 0.1667 \n", + "permuted_conditions 4.4569 101.0 8.782727e+07 1865.3040 85.9982 27.0510 0.0000 \n", + "\n", + "[7 rows x 22 columns]" + ] + }, + "execution_count": 6, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "probes = Board.from_evaluator(evaluator, {}, designs=CONSTRUCTIONS, expensive=EXPENSIVE)\n", + "board.frame = pd.concat([board.frame, probes.frame])\n", + "board.designs.update(probes.designs)\n", + "board.frame.round(4)" + ] + }, + { + "cell_type": "markdown", + "id": "ae905c42", + "metadata": {}, + "source": [ + "## 5. Read the board" + ] + }, + { + "cell_type": "code", + "execution_count": 7, + "id": "c0434cce", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:29:08.293925Z", + "iopub.status.busy": "2026-09-29T09:29:08.293829Z", + "iopub.status.idle": "2026-09-29T09:29:08.298386Z", + "shell.execute_reply": "2026-09-29T09:29:08.298211Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questiondirectionpickssplit-half reference
mmdDistance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable.lower is betterpermuted_conditionsNone
coverageFraction of the reference optima that have at least one generated design close to them. One means every reference design is reached.higher is bettertie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditionsNone
vendiEffective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct.higher is betterunconditionalNone
dppSpread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed.higher is betterunconditionalNone
violFraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied.lower is bettervqganNone
volume_errorGap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.lower is bettervqganNone
per_condition_distancePer-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum.lower is betterlookup_tableNone
cond_sensHow much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored.diagnosticNone
train_distancePer-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.diagnosticNone
train_distance_ratioPer-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data.diagnosticNone
generation_secondsWall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine.lower is bettergan_cnn_2dNone
n_parametersNumber of trainable parameters in the model. Blank for a model with no network.lower is bettercgan_cnn_2dNone
train_minutesWall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does.lower is betterNone
iogOptimality gap of each generated design as generated, before any re-optimization. Zero means already optimal.lower is betterdiffusion_2d_condNone
cogOptimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted.lower is betterlookup_tableNone
fogOptimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum.lower is betterdiffusion_2d_condNone
calls_to_near_optimumOptimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there.lower is bettervqganNone
gap_after_1_callsOptimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.lower is betterdiffusion_2d_condNone
gap_after_2_callsOptimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.lower is bettervqganNone
gap_after_5_callsOptimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.lower is bettervqganNone
gap_after_10_callsOptimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget.lower is bettervqganNone
reaches_reference_rateFraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there.higher is betterdiffusion_2d_condNone
\n", + "
" + ], + "text/plain": [ + " question \\\n", + "mmd Distance between the generated designs and the reference optima, compared as whole sets. Zero means the two sets are indistinguishable. \n", + "coverage Fraction of the reference optima that have at least one generated design close to them. One means every reference design is reached. \n", + "vendi Effective number of distinct designs in the generated set. One means all identical, the sample size means all mutually distinct. \n", + "dpp Spread of the generated set, as the geometric mean of its similarity-kernel eigenvalues. One is perfectly spread, near zero is collapsed. \n", + "viol Fraction of the generated designs that break a constraint of the problem. Zero means every design is admissible; blank when the designs' conditions were not supplied. \n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. \n", + "per_condition_distance Per-element distance from each generated design to the reference optimum for its own conditions. Zero means every design matches its own optimum. \n", + "cond_sens How much each design changes, per element, when the model is re-run under another sample's conditions. Zero means the conditions are ignored. \n", + "train_distance Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "train_distance_ratio Per-element distance from each generated design to the nearest design in the training set. Zero means the model reproduces its training data. \n", + "generation_seconds Wall-clock seconds the model took to generate the evaluation batch. Comparable only between models run on the same machine. \n", + "n_parameters Number of trainable parameters in the model. Blank for a model with no network. \n", + "train_minutes Wall-clock minutes the model took to train, when the checkpoint records it. Blank where no checkpoint does. \n", + "iog Optimality gap of each generated design as generated, before any re-optimization. Zero means already optimal. \n", + "cog Optimality gap summed over every optimizer step when re-optimizing from each generated design, the area under its gap curve. Zero means no optimizer effort was wasted. \n", + "fog Optimality gap of each design once re-optimization from it has finished. Zero means the warm start reached the reference optimum. \n", + "calls_to_near_optimum Optimizer calls from each generated design until its gap stays within five percent of the reference optimum for its conditions. Fewer is faster; one more than the budget means it never got there. \n", + "gap_after_1_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_2_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_5_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "gap_after_10_calls Optimality gap remaining after one, two, five and ten optimizer calls from each generated design. Zero means the defect was repaired within that budget. \n", + "reaches_reference_rate Fraction of generated designs whose re-optimization reaches or beats the reference optimum. One means every warm start gets there. \n", + "\n", + " direction picks split-half reference \n", + "mmd lower is better permuted_conditions None \n", + "coverage higher is better tie: vqgan, diffusion_2d_cond, lookup_table, unconditional, permuted_conditions None \n", + "vendi higher is better unconditional None \n", + "dpp higher is better unconditional None \n", + "viol lower is better vqgan None \n", + "volume_error lower is better vqgan None \n", + "per_condition_distance lower is better lookup_table None \n", + "cond_sens diagnostic None \n", + "train_distance diagnostic None \n", + "train_distance_ratio diagnostic None \n", + "generation_seconds lower is better gan_cnn_2d None \n", + "n_parameters lower is better cgan_cnn_2d None \n", + "train_minutes lower is better None \n", + "iog lower is better diffusion_2d_cond None \n", + "cog lower is better lookup_table None \n", + "fog lower is better diffusion_2d_cond None \n", + "calls_to_near_optimum lower is better vqgan None \n", + "gap_after_1_calls lower is better diffusion_2d_cond None \n", + "gap_after_2_calls lower is better vqgan None \n", + "gap_after_5_calls lower is better vqgan None \n", + "gap_after_10_calls lower is better vqgan None \n", + "reaches_reference_rate higher is better diffusion_2d_cond None " + ] + }, + "execution_count": 7, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "board.explain()" + ] + }, + { + "cell_type": "markdown", + "id": "ab758e29", + "metadata": {}, + "source": [ + "Why these seven rows: four trained models and three constructions whose right\n", + "answer is known, so the board shows which columns can tell a good model from a\n", + "construction that games one metric. How: every row was scored under one spec, twelve\n", + "withheld conditions, medians over designs, and the physics columns re-optimized from\n", + "each design. What the numbers say:\n", + "\n", + "**`permuted_conditions` posts an MMD of exactly zero and full coverage. It is the\n", + "reference set.** It also posts the worst final gap on the board, 4.46, and reaches the\n", + "reference optimum from none of its designs. A perfect set of designs, each for the\n", + "wrong conditions, is the best model by the distribution columns and the worst by the\n", + "physics. `per_condition_distance` reads 0.42 for it, in the same tier as the GANs, and\n", + "that is the column that sees the problem before any optimizer runs.\n", + "\n", + "**`lookup_table` never generalizes, every design is a training design, and it beats\n", + "every trained model except diffusion on final gap.** It has the best cumulative gap on\n", + "the board and the smallest per-condition distance, 0.057. That is not a bug. With one\n", + "optimum per condition and a dense training sweep, the nearest training design is a very\n", + "good warm start. `train_distance` reads exactly zero for it, and so does the ratio: that\n", + "is the column that says what it is. `permuted_conditions` reads a ratio of exactly one,\n", + "by construction, because it is the reference set and one is what real unseen designs\n", + "read. The trained models read 1.7 for the VQGAN, 7 for diffusion and 13 to 14 for the\n", + "GANs: further from the training data than real designs are, which is what blurred or\n", + "noisy output looks like. Unlike the data is not the same as generalizing.\n", + "\n", + "**`diffusion_2d_cond` has the best final gap, minus 0.91, and reaches the reference\n", + "optimum from eleven designs in twelve.** It also violates a constraint in ten of twelve\n", + "designs and misses the requested material fraction by the widest margin of the trained\n", + "models, 0.127. Its initial gap is negative, meaning better than the reference optimum\n", + "before any optimization, which is what a design that spends more material than it was\n", + "allowed looks like. Read `iog` beside `viol` and `volume_error`, never alone.\n", + "\n", + "**`vqgan` is the most obedient model.** Fewest violations, smallest volume error, and\n", + "the largest response when its conditions change. Its physics is middling. Obedience\n", + "and optimality are different columns.\n", + "\n", + "**`cgan_cnn_2d` and `gan_cnn_2d` post initial gaps near 10^9 and 10^10.** The\n", + "optimizer repairs them; the final gaps are 3.1 and 0.002. These are designs that start\n", + "absurd and are cheap to fix, which is why `iog` and `fog` disagree about them. Cheap\n", + "in the end, but not fast: they need 78 and 17.5 optimizer calls to get within five\n", + "percent of the optimum, against 1.5 for the VQGAN, 4 for diffusion and 3.5 for the\n", + "lookup table. The unconditional and permuted rows read 101, one more than the budget,\n", + "because they never get there. That column is measured against the optimum for each\n", + "design's conditions, not against where the design started, so an absurd start does not\n", + "read as a fast one.\n", + "`gan_cnn_2d` has a `vendi` of 1.09, twelve designs that are effectively one, and a\n", + "`cond_sens` of exactly zero, because it is an unconditional model that was handed\n", + "conditions anyway. That zero is what stops it collecting a conditional model's score.\n", + "\n", + "**`viol` is high even for real training designs.** On beams2d it includes the volume\n", + "budget at a tolerance of 0.01, and a design made for a neighboring condition misses\n", + "its own budget by more than that. Seven in twelve of the lookup table's designs fail\n", + "on that alone.\n", + "\n", + "Two things this board cannot show. `cond_sens` and the cost columns are blank for the\n", + "constructions, because they have no model to re-run or inspect. And there is no\n", + "reference row here; that row comes from the saved-designs path in section 7, where at\n", + "twelve conditions it is six designs against six, a coarser measurement than the model\n", + "rows. The gap path is not always monotone: for the diffusion model it rises after the\n", + "first call, as the optimizer restores feasibility before it improves the objective,\n", + "which is why its gap after two calls is larger than after one." + ] + }, + { + "cell_type": "markdown", + "id": "86fa6b15", + "metadata": {}, + "source": [ + "## 6. Rank, but only on a column that can be ranked" + ] + }, + { + "cell_type": "code", + "execution_count": 8, + "id": "556ad06c", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:29:08.300718Z", + "iopub.status.busy": "2026-09-29T09:29:08.300641Z", + "iopub.status.idle": "2026-09-29T09:29:08.303791Z", + "shell.execute_reply": "2026-09-29T09:29:08.303584Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
fogtrain_distance_ratioper_condition_distance
diffusion_2d_cond-0.91287.04390.1532
lookup_table-0.73630.00000.0569
vqgan-0.61941.73890.0833
gan_cnn_2d0.001912.71310.4447
unconditional1.42870.00000.3993
cgan_cnn_2d3.075814.29690.4144
permuted_conditions4.45691.00000.4231
\n", + "
" + ], + "text/plain": [ + " fog train_distance_ratio per_condition_distance\n", + "diffusion_2d_cond -0.9128 7.0439 0.1532\n", + "lookup_table -0.7363 0.0000 0.0569\n", + "vqgan -0.6194 1.7389 0.0833\n", + "gan_cnn_2d 0.0019 12.7131 0.4447\n", + "unconditional 1.4287 0.0000 0.3993\n", + "cgan_cnn_2d 3.0758 14.2969 0.4144\n", + "permuted_conditions 4.4569 1.0000 0.4231" + ] + }, + "execution_count": 8, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "column = \"fog\" if EXPENSIVE else \"mmd\"\n", + "board.rank(column)[[column, \"train_distance_ratio\", \"per_condition_distance\"]].round(4)" + ] + }, + { + "cell_type": "code", + "execution_count": 9, + "id": "31513288", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:29:08.306192Z", + "iopub.status.busy": "2026-09-29T09:29:08.306129Z", + "iopub.status.idle": "2026-09-29T09:29:08.307671Z", + "shell.execute_reply": "2026-09-29T09:29:08.307506Z" + } + }, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "'train_distance_ratio' is a diagnostic: it says whether the other columns mean what they look like. Read it beside them; do not rank on it.\n" + ] + } + ], + "source": [ + "try:\n", + " board.rank(\"train_distance_ratio\")\n", + "except ValueError as e:\n", + " print(e)" + ] + }, + { + "cell_type": "markdown", + "id": "a53a253a", + "metadata": {}, + "source": [ + "## 7. The same questions in another space" + ] + }, + { + "cell_type": "markdown", + "id": "f08ba1a9", + "metadata": {}, + "source": [ + "Where a metric is computed is an argument, not a property of the metric. The board\n", + "kept every model's sampled designs, so they can be re-scored after projecting onto\n", + "the leading principal components of the reference designs. Metrics that need the\n", + "actual designs, such as the constraint check and the training-distance column, are\n", + "skipped there. The kernel bandwidth is the training designs' median distance in the\n", + "projected space, found the same way it was in pixel space.\n", + "\n", + "This is the control that asks whether a learned latent space is needed at all. If PCA\n", + "of the data answers the same questions, any rotation of the same width would do. A\n", + "learned space registers itself with `register_space` and appears here by name." + ] + }, + { + "cell_type": "code", + "execution_count": 10, + "id": "3ae7f494", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:29:08.309937Z", + "iopub.status.busy": "2026-09-29T09:29:08.309870Z", + "iopub.status.idle": "2026-09-29T09:29:09.062834Z", + "shell.execute_reply": "2026-09-29T09:29:09.062604Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
mmd@pcadpp@pcaper_condition_distance@pcacoverage@pcavendi@pca
cgan_cnn_2d0.47080.00087.16931.00001.2525
gan_cnn_2d0.75700.00009.21630.58331.0015
vqgan0.00300.37580.46171.00006.7750
diffusion_2d_cond0.05480.44402.02031.00006.5482
lookup_table0.04500.16590.56181.00005.8951
unconditional0.06880.22808.64221.00004.8833
permuted_conditions0.00000.209110.52381.00007.4443
reference (split-half)0.16960.7890NaN1.00004.9480
\n", + "
" + ], + "text/plain": [ + " mmd@pca dpp@pca per_condition_distance@pca coverage@pca vendi@pca\n", + "cgan_cnn_2d 0.4708 0.0008 7.1693 1.0000 1.2525\n", + "gan_cnn_2d 0.7570 0.0000 9.2163 0.5833 1.0015\n", + "vqgan 0.0030 0.3758 0.4617 1.0000 6.7750\n", + "diffusion_2d_cond 0.0548 0.4440 2.0203 1.0000 6.5482\n", + "lookup_table 0.0450 0.1659 0.5618 1.0000 5.8951\n", + "unconditional 0.0688 0.2280 8.6422 1.0000 4.8833\n", + "permuted_conditions 0.0000 0.2091 10.5238 1.0000 7.4443\n", + "reference (split-half) 0.1696 0.7890 NaN 1.0000 4.9480" + ] + }, + "execution_count": 10, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "pca_board = Board(evaluator.problem, reference=REFERENCE, train=TRAIN)\n", + "pca_board.evaluate(board.designs, space=\"pca\", width=8, aggregation=\"median\").round(4)" + ] + }, + { + "cell_type": "markdown", + "id": "3391d356", + "metadata": {}, + "source": [ + "## 8. Registering a metric of your own" + ] + }, + { + "cell_type": "markdown", + "id": "5ecea0fe", + "metadata": {}, + "source": [ + "A metric is a plain function. The evaluator calls it once per model and hands it one\n", + "object, the evaluation context, which holds everything that model's evaluation knows:\n", + "\n", + "| On the context | What it is |\n", + "|---|---|\n", + "| `gen_designs` | the designs the model produced, one per scored condition, as an array of shape `(n, 50, 100)` here |\n", + "| `ref_designs` | the reference optimum for each of those conditions, same shape |\n", + "| `conditions` | the scored conditions themselves, one row per design |\n", + "| `train_designs` | the training split, flattened, for anything that asks \"has the model seen this?\" |\n", + "| `optimization` | the physics results, one gap path per design; touching it runs the optimizer, so declare `cost=\"expensive\"` |\n", + "| `reduce(values)` | collapses one value per design to one number the way the spec asked, median here, mean by default |\n", + "\n", + "The decorator adds the four declarations a reader needs: the name, which family of\n", + "question it belongs to, whether it needs the simulator, and which direction is better,\n", + "with `None` for a column that is read but never ranked on. The first line of the\n", + "docstring is the sentence `explain()` shows. Nothing else has to be wired up." + ] + }, + { + "cell_type": "code", + "execution_count": 11, + "id": "277a5953", + "metadata": { + "execution": { + "iopub.execute_input": "2026-09-29T09:29:09.065670Z", + "iopub.status.busy": "2026-09-29T09:29:09.065580Z", + "iopub.status.idle": "2026-09-29T09:29:09.071959Z", + "shell.execute_reply": "2026-09-29T09:29:09.071778Z" + } + }, + "outputs": [ + { + "data": { + "text/html": [ + "
\n", + "\n", + "\n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + " \n", + "
questiondirectionpickssplit-half reference
volume_errorGap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target.lower is bettervqganNone
density_ratioMaterial fraction of each generated design over that of the reference optima. One means the same amount of material.diagnosticNone
\n", + "
" + ], + "text/plain": [ + " question direction picks split-half reference\n", + "volume_error Gap between each design's material fraction and the fraction its conditions requested. Zero means every design hit its volume target. lower is better vqgan None\n", + "density_ratio Material fraction of each generated design over that of the reference optima. One means the same amount of material. diagnostic None" + ] + }, + "execution_count": 11, + "metadata": {}, + "output_type": "execute_result" + } + ], + "source": [ + "from engiopt.evaluation.registry import register_metric\n", + "\n", + "\n", + "@register_metric(\"density_ratio\", family=\"feasibility\", cost=\"cheap\", higher_is_better=None)\n", + "def density_ratio(context):\n", + " \"\"\"Material fraction of each generated design over that of the reference optima. One means the same amount of material.\"\"\"\n", + " generated = context.gen_designs.reshape(len(context.gen_designs), -1).mean(axis=1) # one fraction per generated design\n", + " reference = context.ref_designs.reshape(len(context.ref_designs), -1).mean(axis=1) # one per reference optimum\n", + " return context.reduce(generated) / context.reduce(reference) # each collapsed the way the spec asked, then compared\n", + "\n", + "\n", + "Board.from_evaluator(evaluator, {}, designs=board.designs, metrics=[\"volume_error\", \"density_ratio\"]).explain()" + ] + }, + { + "cell_type": "markdown", + "id": "02e4c560", + "metadata": {}, + "source": [ + "It appears with its question and direction attached, scored under the same spec as\n", + "everything else. Declared diagnostic, because using more material is not better or\n", + "worse on its own; read it beside `volume_error`, which says whether each design hit\n", + "the fraction its conditions asked for." + ] + }, + { + "cell_type": "markdown", + "id": "5f536feb", + "metadata": {}, + "source": [ + "## 9. Not in this notebook" + ] + }, + { + "cell_type": "markdown", + "id": "f452d664", + "metadata": {}, + "source": [ + "- **Latent spaces.** `space=\"lv\"` needs a verified autoencoder pinned by the spec. It\n", + " arrives with the LVAE work and plugs into section 7 by name.\n", + "- **Uncertainty.** Every physics column here is a median over fifty designs whose\n", + " per-design values already exist. Error bars from resampling those values, and folds\n", + " of the full test split for the cheap metrics, are the next addition.\n", + "- **Seeds.** Every row is training seed one. The Hub holds eleven seeds of each\n", + " family's default configuration; the spread across them is larger than the sampling\n", + " spread for some families, and it belongs on an honest board beside these numbers." + ] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "name": "python", + "pygments_lexer": "ipython3" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/tests/conftest.py b/tests/conftest.py index 9afbcb6e..ab57faf3 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -131,6 +131,10 @@ def select(self, indices: Any) -> FakeDataset: order = list(indices) return FakeDataset({name: [values[i] for i in order] for name, values in self._columns.items()}) + def select_columns(self, names: Any) -> FakeDataset: + """Keep only the named columns, as `datasets.Dataset.select_columns` does.""" + return FakeDataset({name: self._columns[name] for name in names}) + @pytest.fixture def mixed_condition_dataset() -> FakeDataset: diff --git a/tests/test_board.py b/tests/test_board.py new file mode 100644 index 00000000..13dba630 --- /dev/null +++ b/tests/test_board.py @@ -0,0 +1,167 @@ +"""`Board`: from saved designs to a readable table in three calls.""" + +from __future__ import annotations + +from typing import Any + +import numpy as np +import pytest + +from engiopt.evaluation.board import Board +from engiopt.evaluation.board import REFERENCE_ROW +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.registry import METRICS + +# Importing the metrics package registers the built-ins. +import engiopt.evaluation.metrics # noqa: F401 # isort: skip + + +@pytest.fixture +def designs(fake_problem: Any) -> tuple[dict[str, np.ndarray], np.ndarray, np.ndarray]: + rng = np.random.default_rng(0) + shape = fake_problem.design_space.shape + train = rng.random((20, *shape)) + ref = rng.random((6, *shape)) + models = {"honest": np.clip(ref + rng.normal(0, 0.01, ref.shape), 0, 1), "copier": ref.copy()} + return models, ref, train + + +def test_evaluate_scores_every_model_on_the_cheap_metrics(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + frame = Board(fake_problem, reference=ref, train=train).evaluate(models) + assert list(frame.index) == ["honest", "copier", REFERENCE_ROW] + assert {"mmd", "train_distance", "viol"} <= set(frame.columns) + + +def test_explain_names_a_pick_only_for_ranked_columns(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + explained = board.explain() + assert explained.loc["mmd", "picks"] == "copier", "identical sets score the best MMD" + assert explained.loc["train_distance", "picks"] == "", "a diagnostic picks nobody" + assert explained.loc["mmd", "question"] == METRICS["mmd"].description + assert explained.loc["mmd", "split-half reference"] == board.frame.loc[REFERENCE_ROW, "mmd"] + + +def test_the_reference_row_is_measured_and_never_picked(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + assert board.frame.loc[REFERENCE_ROW, "mmd"] > 0.0, "half of the data against the other half is not identical" + assert REFERENCE_ROW not in set(board.explain()["picks"]) + assert REFERENCE_ROW not in board.rank("mmd").index, "a scale reference is never ranked" + assert REFERENCE_ROW not in board.evaluate(models, reference_row=False).index + + +def test_rank_refuses_a_diagnostic(fake_problem: Any, designs: Any) -> None: + models, ref, train = designs + board = Board(fake_problem, reference=ref, train=train) + board.evaluate(models) + assert board.rank("mmd").index[0] == "copier" + with pytest.raises(ValueError, match="diagnostic"): + board.rank("train_distance") + + +def test_space_is_an_argument_not_a_metric(fake_problem: Any, designs: Any) -> None: + """The same metric, asked in PCA space, gets a suffixed column; pixel-only metrics are skipped.""" + models, ref, train = designs + frame = Board(fake_problem, reference=ref, train=train).evaluate(models, space="pca", width=3) + assert "mmd@pca" in frame.columns + assert "viol@pca" not in frame.columns, "a constraint check on PCA codes would be meaningless" + assert frame.loc["copier", "mmd@pca"] == pytest.approx(0.0) + + +def test_aggregation_is_a_policy_on_the_context(fake_problem: Any) -> None: + """One diverged design moves the mean and not the median.""" + values = [0.1, 0.1, 0.1, 100.0] + zeros = np.zeros((4, 8, 10)) + mean_ctx = EvaluationContext(problem=fake_problem, problem_id="f", gen_designs=zeros, ref_designs=zeros) + median_ctx = EvaluationContext( + problem=fake_problem, problem_id="f", gen_designs=zeros, ref_designs=zeros, aggregation="median" + ) + assert mean_ctx.reduce(values) == pytest.approx(25.075) + assert median_ctx.reduce(values) == pytest.approx(0.1) + + +def test_volume_error_reaches_saved_designs(fake_problem: Any, designs: Any) -> None: + """A board told which condition is the volume budget scores it; one not told leaves it blank.""" + models, ref, _ = designs + conditions = {"volfrac": [float(design.mean()) for design in ref]} + told = Board(fake_problem, reference=ref, conditions=conditions, volume_condition="volfrac") + frame = told.evaluate({"exact": ref.copy()}, metrics=["volume_error"]) + assert frame.loc["exact", "volume_error"] == pytest.approx(0.0), "designs at their own budgets have no error" + untold = Board(fake_problem, reference=ref, conditions=conditions) + assert np.isnan(untold.evaluate(models, metrics=["volume_error"]).loc["honest", "volume_error"]) + + +def test_dict_conditions_survive_a_full_evaluate(fake_problem: Any, designs: Any) -> None: + """A plain dict of columns must feed row-reading metrics too. + + `viol` asks for one design's brief at a time (`conditions[0]`), which a bare + dict answers with `KeyError: 0`. The board wraps the dict, so the same input + that fed `volume_error` survives a default evaluate. + """ + models, ref, _ = designs + conditions = {"volfrac": [float(design.mean()) for design in ref]} + board = Board(fake_problem, reference=ref, conditions=conditions, volume_condition="volfrac") + frame = board.evaluate(models) + assert not np.isnan(frame.loc["honest", "viol"]), "the row-access path must work, not just the column path" + assert not np.isnan(frame.loc["honest", "volume_error"]) + + +def test_a_dataset_slice_is_designs_and_briefs_at_once(fake_problem: Any) -> None: + """Passing the test rows keeps design i and brief i aligned by construction. + + The briefs already sit beside the optimal designs in the dataset; slicing + the designs into an array is what loses them. Handing the board the rows + themselves means nothing is lost and nothing can be misaligned. + """ + rows = fake_problem.dataset["train"] + board = Board(fake_problem, reference=rows, volume_condition="volfrac") + assert board.reference.shape == (3, 8, 10), "the design column became the reference array" + assert board.conditions is not None + assert board.conditions["volfrac"] == [0.5, 0.5, 0.5], "the remaining columns became the briefs" + frame = board.evaluate({"exact": board.reference.copy()}, metrics=["volume_error"]) + # The dataset's designs are flat fields at 0.1, 0.2 and 0.3 against a 0.5 budget. + assert frame.loc["exact", "volume_error"] == pytest.approx(np.mean([0.4, 0.3, 0.2])) + + +def test_misaligned_conditions_are_refused(fake_problem: Any, designs: Any) -> None: + """Conditions are an answer key, not a filter; a short one is wrong labels, not a subset.""" + _, ref, _ = designs + with pytest.raises(ValueError, match="do not select"): + Board(fake_problem, reference=ref, conditions={"volfrac": [0.2, 0.3]}) + + +def test_every_row_shares_one_inferred_bandwidth(fake_problem: Any, designs: Any, monkeypatch: pytest.MonkeyPatch) -> None: + """The split-half row must not size its kernel from its own half. + + Left to each context, a half landing in one cluster of the reference set + infers a degenerate bandwidth (the 1e-6 floor), so its kernel columns would + differ from the model rows in bandwidth as well as in sample size. + """ + from engiopt import metrics as metrics_mod + + models, ref, _ = designs + calls: list[int] = [] + real = metrics_mod.compute_median_sigma + + def spy(x: Any, y: Any = None) -> float: + calls.append(len(x)) + return real(x, y) + + monkeypatch.setattr(metrics_mod, "compute_median_sigma", spy) + Board(fake_problem, reference=ref).evaluate(models) + assert calls == [len(ref)], "one bandwidth, sized from the full reference set, for every row" + + +def test_the_default_bandwidth_lets_diversity_metrics_see_anything(fake_problem: Any, designs: Any) -> None: + """At a fixed sigma=10 every design looks identical to every other; the training-data median does not.""" + models, ref, _ = designs + collapsed = np.repeat(models["honest"][:1], len(ref), axis=0) + frame = Board(fake_problem, reference=ref).evaluate({"collapsed": collapsed, "spread": models["honest"]}) + assert frame.loc["collapsed", "vendi"] == pytest.approx(1.0) + assert frame.loc["spread", "vendi"] > 3.0 + saturated = Board(fake_problem, reference=ref, sigma=10.0).evaluate({"spread": models["honest"]}) + assert saturated.loc["spread", "vendi"] < 1.5, "a pinned, too-wide kernel is still honored" diff --git a/tests/test_conditions_and_gaps.py b/tests/test_conditions_and_gaps.py index 05d93526..63b165b8 100644 --- a/tests/test_conditions_and_gaps.py +++ b/tests/test_conditions_and_gaps.py @@ -17,6 +17,7 @@ from engiopt.evaluation.registry import METRICS from engiopt.transforms import get_image_condition_keys from engiopt.transforms import get_scalar_condition_keys +from tests.conftest import FakeDataset from tests.conftest import FakeViolations # Importing the metrics package registers the built-ins. @@ -283,6 +284,8 @@ def test_signs_are_per_objective_for_mixed_directions() -> None: def _feasibility_context(problem: Any, design_value: float, **kwargs: Any) -> EvaluationContext: designs = np.full((1, *problem.design_space.shape), design_value) + # A feasibility check reads the design's conditions, so a context without them reports nothing. + kwargs.setdefault("conditions", FakeDataset({"volfrac": [design_value]})) return EvaluationContext( problem=problem, problem_id="fake", diff --git a/tests/test_evaluate_default_spec.py b/tests/test_evaluate_default_spec.py new file mode 100644 index 00000000..d3992b40 --- /dev/null +++ b/tests/test_evaluate_default_spec.py @@ -0,0 +1,38 @@ +"""The CLI must not carry its own default spec version. + +The library default moved to v2 while `evaluate.py` still built +`f"{problem_id}/v1"`, so a bare `python -m engiopt.evaluate` selected a spec +whose metric list no longer matched the registry (`KeyError: 'novelty'`). +The default lives in `EvalSpec.load` alone; the CLI passes `--spec` through. +""" + +from __future__ import annotations + +from typing import Any + +import pytest + +from engiopt import evaluate +from engiopt.evaluation.spec import EvalSpec + + +class _StopAtSpecError(Exception): + """Raised by the fake to stop `main` before it loads anything real.""" + + +def test_the_cli_without_a_spec_defers_to_the_library_default(monkeypatch: pytest.MonkeyPatch) -> None: + seen: dict[str, Any] = {} + + def fake_for_problem(problem_id: str, *, spec: Any = None, **_: Any) -> Any: + seen["spec"] = spec + raise _StopAtSpecError + + monkeypatch.setattr(evaluate.Evaluator, "for_problem", fake_for_problem) + with pytest.raises(_StopAtSpecError): + evaluate.main(evaluate.Args()) + assert seen["spec"] is None, "no --spec means the library default, not a version the CLI invents" + + +def test_a_bare_problem_reference_loads_the_current_spec_version() -> None: + """`EvalSpec.load("beams2d")` is what the CLI's no-flag path now resolves.""" + assert EvalSpec.load("beams2d").version == EvalSpec.version, "a bare reference follows the dataclass default" diff --git a/tests/test_evaluation.py b/tests/test_evaluation.py index e6e80d68..17dafcb9 100644 --- a/tests/test_evaluation.py +++ b/tests/test_evaluation.py @@ -89,12 +89,32 @@ def test_builtin_metrics_declare_their_cost() -> None: judged by a constraint check rather than by running the optimizer. The two integrity metrics are cheap too, and that matters more than it - sounds: `novelty` and `cond_sens` decide whether a row can be ranked at all, + sounds: `train_distance` and `cond_sens` decide whether a row can be ranked at all, so a board that could only afford the cheap pass would otherwise have to rank models it had never checked for memorization. """ - assert {spec.name for spec in METRICS.select(cost="cheap")} == {"mmd", "dpp", "viol", "novelty", "cond_sens"} - assert {spec.name for spec in METRICS.select(cost="expensive")} == {"iog", "cog", "fog"} + assert {spec.name for spec in METRICS.select(cost="cheap")} == { + "mmd", + "dpp", + "viol", + "train_distance", + "cond_sens", + "per_condition_distance", + "volume_error", + "coverage", + "vendi", + "generation_seconds", + "n_parameters", + "train_minutes", + } + assert {spec.name for spec in METRICS.select(cost="expensive")} == { + "iog", + "cog", + "fog", + "calls_to_near_optimum", + "gap_after_calls", + "reaches_reference_rate", + } def test_cheap_metrics_never_touch_the_solver(fake_problem: Any) -> None: @@ -127,11 +147,21 @@ def test_registry_rejects_colliding_output_columns() -> None: """Two metrics writing the same column would silently overwrite each other.""" registry = MetricRegistry() register_metric( - "first", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + "first", + family="diversity", + cost="cheap", + higher_is_better=True, + outputs=("shared",), + registry=registry, )(lambda ctx: {"shared": 1.0}) with pytest.raises(ValueError, match="already emitted"): register_metric( - "second", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + "second", + family="diversity", + cost="cheap", + higher_is_better=True, + outputs=("shared",), + registry=registry, )(lambda ctx: {"shared": 2.0}) @@ -166,7 +196,7 @@ def _board() -> pd.DataFrame: "seed": 1, "spec_version": "v1", "mmd": 0.1, - "dpp": 5.0, + "viol": 0.1, }, { "problem_id": "p", @@ -176,7 +206,7 @@ def _board() -> pd.DataFrame: "seed": 2, "spec_version": "v1", "mmd": 0.3, - "dpp": 7.0, + "viol": 0.1, }, { "problem_id": "p", @@ -186,16 +216,16 @@ def _board() -> pd.DataFrame: "seed": 1, "spec_version": "v1", "mmd": 0.05, - "dpp": 1.0, + "viol": 0.5, }, ] ) def test_rank_respects_each_metric_direction() -> None: - """Lower MMD ranks first; higher DPP ranks first.""" + """Both columns are lower-is-better, and they pick different winners.""" assert rank(_board(), "mmd").iloc[0]["algo_id"] == "b" - assert rank(_board(), "dpp").iloc[0]["algo_id"] == "a" + assert rank(_board(), "viol").iloc[0]["algo_id"] == "a" def test_rank_aggregates_seeds_with_the_median_by_default() -> None: @@ -207,9 +237,9 @@ def test_rank_aggregates_seeds_with_the_median_by_default() -> None: def test_disagreement_surfaces_conflicting_rankings() -> None: """The two metrics crown different winners; the view must show both.""" - table = disagreement(_board(), ["mmd", "dpp"]) + table = disagreement(_board(), ["mmd", "viol"]) assert table.loc[("p", "b", "cfg_b", "someone/engiopt-b", "v1"), "mmd"] == 1 - assert table.loc[("p", "a", "cfg_a", "someone/engiopt-a", "v1"), "dpp"] == 1 + assert table.loc[("p", "a", "cfg_a", "someone/engiopt-a", "v1"), "viol"] == 1 def _two_configs_board() -> pd.DataFrame: diff --git a/tests/test_evaluator_rows.py b/tests/test_evaluator_rows.py new file mode 100644 index 00000000..9db81ee4 --- /dev/null +++ b/tests/test_evaluator_rows.py @@ -0,0 +1,59 @@ +"""`Evaluator.for_rows`: score against a dataset slice you chose. + +The spec's own draw stays the leaderboard contract. This path swaps only the +draw: your rows' conditions drive the sampling, your rows' optimal designs are +the reference, and everything else the spec declares stays in force. +""" + +from __future__ import annotations + +import numpy as np +import pytest + +from engiopt.evaluation.evaluator import Evaluator +from engiopt.evaluation.spec import EvalSpec +from tests.conftest import FakeDataset +from tests.conftest import FakeProblem + + +@pytest.fixture +def rows() -> FakeDataset: + """Two hand-picked test rows with distinct briefs and known optima.""" + shape = (8, 10) + return FakeDataset( + { + "volfrac": [0.2, 0.3], + "rmin": [2.0, 2.0], + "optimal_design": [np.full(shape, 0.2), np.full(shape, 0.3)], + } + ) + + +@pytest.fixture +def evaluator(rows: FakeDataset, monkeypatch: pytest.MonkeyPatch) -> Evaluator: + from engibench.utils.all_problems import BUILTIN_PROBLEMS + + monkeypatch.setitem(BUILTIN_PROBLEMS, "fake2d", FakeProblem) + spec = EvalSpec(problem_id="fake2d", volume_condition="volfrac") + return Evaluator.for_rows("fake2d", rows, spec=spec) + + +def test_the_rows_become_the_reference_and_the_briefs(evaluator: Evaluator, rows: FakeDataset) -> None: + assert evaluator.resolved.ref_designs.shape == (2, 8, 10) + assert evaluator.resolved.ref_designs[0].mean() == pytest.approx(0.2) + assert evaluator.resolved.conditions["volfrac"] == [0.2, 0.3] + assert evaluator.resolved.conditions_tensor.shape == (2, 2), "one row per design, one column per scalar condition" + assert evaluator.spec.n_samples == len(rows), "generators sample one design per chosen row" + + +def test_a_custom_slice_is_not_the_frozen_contract(evaluator: Evaluator) -> None: + """The digest certifies the spec's own draw; a custom draw must not inherit it.""" + assert evaluator.spec.condition_digest is None + assert evaluator.spec.volume_condition == "volfrac", "everything but the draw survives from the spec" + + +def test_designs_are_scored_against_the_chosen_rows(evaluator: Evaluator) -> None: + """A batch equal to the rows' own optima has zero volume error under their briefs.""" + ctx = evaluator.context_for_designs(evaluator.resolved.ref_designs.copy()) + row = evaluator.score_context(ctx, only=["volume_error"], include_expensive=False) + assert row["volume_error"] == pytest.approx(0.0) diff --git a/tests/test_leaderboard_integrity.py b/tests/test_leaderboard_integrity.py index 26b73da0..859212d0 100644 --- a/tests/test_leaderboard_integrity.py +++ b/tests/test_leaderboard_integrity.py @@ -32,7 +32,6 @@ from engiopt.evaluation.spec import EvalSpec from engiopt.evaluation.submission import admission_problems from engiopt.evaluation.submission import FLAG_IGNORES_CONDITIONS -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.submission import FLAG_UNVERIFIED from engiopt.evaluation.submission import integrity_flags from engiopt.evaluation.submission import prepare_submission @@ -52,77 +51,59 @@ def _context(problem: Any, gen: np.ndarray, ref: np.ndarray, **kwargs: Any) -> E def test_a_generator_returning_the_reference_designs_is_caught(fake_problem: Any) -> None: - """The attack this metric exists for. + """A lookup table over the training split reads as distance zero. - A lookup table keyed on the condition vector returns the dataset-optimal - design for each scored condition. That is the *definition* of a perfect - `mmd` and a near-zero `iog`, so no amount of care in those metrics can - distinguish it from a model that learned the problem. This one can. + A lookup table keyed on the condition vector returns a training design for + each scored condition. Its distribution score is as good as the data's, so + no amount of care in those metrics can distinguish it from a model that + learned the problem. The distance to the nearest training design can. """ rng = np.random.default_rng(0) ref = rng.random((10, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, ref.copy(), ref) - - scores = METRICS["novelty"].fn(ctx) + train = rng.random((20, *fake_problem.design_space.shape)) + ctx = _context(fake_problem, train[:10].copy(), ref, train_designs_fn=lambda: train) - assert scores["copy_rate"] == 1.0 - assert scores["novelty"] == pytest.approx(0.0, abs=1e-12) - # And it does indeed post a perfect distribution score, which is the point. - assert METRICS["mmd"].fn(ctx) == pytest.approx(0.0, abs=1e-9) + scores = METRICS["train_distance"].fn(ctx) + assert scores["train_distance"] == pytest.approx(0.0, abs=1e-12) + assert scores["train_distance_ratio"] == pytest.approx(0.0, abs=1e-9) -def test_a_generator_producing_its_own_designs_is_not_flagged(fake_problem: Any) -> None: - """The check must not fire on a model that merely resembles the data.""" +def test_a_generator_producing_its_own_designs_reads_like_real_data(fake_problem: Any) -> None: + """The ratio's reference point: real unseen designs score about one, and so does a model that generates.""" rng = np.random.default_rng(1) ref = rng.random((10, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, rng.random((10, *fake_problem.design_space.shape)), ref) - - scores = METRICS["novelty"].fn(ctx) - - assert scores["copy_rate"] == 0.0 - assert scores["novelty"] > 0 - - -def test_copying_the_training_split_is_caught_too(fake_problem: Any) -> None: - """Not only the scored references are copyable -- the whole public dataset is. - - A model that memorized the training split and emits rows of it under - whatever conditions it is given never touches a reference design, so a check - that looked only at those would report it as perfectly novel. - """ - rng = np.random.default_rng(2) - ref = rng.random((6, *fake_problem.design_space.shape)) train = rng.random((20, *fake_problem.design_space.shape)) - ctx = _context(fake_problem, train[:6].copy(), ref, copy_corpus_fn=lambda: train) + ctx = _context(fake_problem, rng.random((10, *fake_problem.design_space.shape)), ref, train_designs_fn=lambda: train) - assert METRICS["novelty"].fn(ctx)["copy_rate"] == 1.0 + assert METRICS["train_distance"].fn(ctx)["train_distance_ratio"] == pytest.approx(1.0, abs=0.25) -def test_near_copies_count_as_copies(fake_problem: Any) -> None: +def test_near_copies_stay_near_zero(fake_problem: Any) -> None: """Adding imperceptible noise to a retrieved design must not launder it. - Otherwise the defense is defeated by one line, and the tolerance is what - decides how much perturbation counts as having generated something. + There is no tolerance to game: the distance simply stays tiny, and the ratio + against real held-out designs stays near zero. """ rng = np.random.default_rng(3) ref = rng.random((8, *fake_problem.design_space.shape)) - barely_perturbed = ref + rng.normal(scale=1e-4, size=ref.shape) - - scores = METRICS["novelty"].fn(_context(fake_problem, barely_perturbed, ref, copy_tol=0.01)) + train = rng.random((20, *fake_problem.design_space.shape)) + barely_perturbed = train[:8] + rng.normal(scale=1e-4, size=(8, *fake_problem.design_space.shape)) - assert scores["copy_rate"] == 1.0 + scores = METRICS["train_distance"].fn(_context(fake_problem, barely_perturbed, ref, train_designs_fn=lambda: train)) + assert scores["train_distance"] < 1e-3 + assert scores["train_distance_ratio"] < 0.01 -def test_novelty_is_diagnostic_and_cannot_be_ranked_on() -> None: - """Ranking on novelty would put pure noise in first place. +def test_memorization_metrics_are_diagnostic_and_cannot_be_ranked_on() -> None: + """Ranking on distance-to-training-data would put pure noise in first place. The metric has no good direction -- zero means retrieval, large means only "unlike the data", which is what a broken model also achieves. Registering a direction would have created a new thing to game while closing an old one. """ - assert METRICS["novelty"].higher_is_better is None + assert METRICS["train_distance"].higher_is_better is None with pytest.raises(ValueError, match="diagnostic"): - rank(pd.DataFrame([{"problem_id": "p", "algo_id": "a", "novelty": 0.5}]), "novelty") + rank(pd.DataFrame([{"problem_id": "p", "algo_id": "a", "train_distance": 0.5}]), "train_distance") # ---------------------------------------------------------------------- @@ -249,18 +230,6 @@ def test_every_rejected_row_is_reported_at_once() -> None: assert len(excinfo.value.problems_by_row) == 2 -def test_a_memorizing_row_is_published_but_flagged() -> None: - """Published, because hiding it throws away what the board just learned. - - Flagged, because calling it first place would be the board endorsing the - thing it exists to detect. - """ - prepared = prepare_submission(pd.DataFrame([_publishable(copy_rate=0.9)]), EvalSpec(problem_id="beams2d")) - - assert len(prepared) == 1 - assert FLAG_MEMORIZED in prepared["flags"].iloc[0] - - # ---------------------------------------------------------------------- # Eligibility: what gets ranked, as distinct from what gets published # ---------------------------------------------------------------------- @@ -280,8 +249,8 @@ def test_unverified_rows_are_published_but_never_ranked() -> None: def test_flagged_rows_stay_on_the_board_and_out_of_the_ranking() -> None: - """A retrieval system belongs in the table and not in the ordering.""" - frame = pd.DataFrame([_scored(algo_id="honest"), _scored(algo_id="copier", flags=FLAG_MEMORIZED)]) + """A model that ignores the conditions it declares belongs in the table and not in the ordering.""" + frame = pd.DataFrame([_scored(algo_id="honest"), _scored(algo_id="blind", flags=FLAG_IGNORES_CONDITIONS)]) eligible = eligible_rows(frame) @@ -398,28 +367,28 @@ def _sample(self, conditions: Any, n: int) -> Any: def test_the_evaluator_detects_a_lookup_table_end_to_end(fake_problem: Any) -> None: - """The full path: sample, build the corpus, score, and report `copy_rate`. + """The full path: sample, fetch the training split, score, and report `train_distance`. - This is the claim the whole memorization defense rests on, so it is asserted + This is the claim the memorization column rests on, so it is asserted through the evaluator rather than against a hand-built context -- a metric that works but is never fed would pass every other test in this file. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "novelty")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "train_distance")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) + train = np.asarray(fake_problem.dataset["train"]["optimal_design"]) evaluator = _evaluator(fake_problem, spec, ref, conditions) - lookup_table = _generator(fake_problem, lambda _conditions, _n: ref) + lookup_table = _generator(fake_problem, lambda _conditions, n: train[np.arange(n) % len(train)]) row = evaluator.score(lookup_table) - assert row["copy_rate"] == 1.0 - assert row["mmd"] == pytest.approx(0.0, abs=1e-9) - assert FLAG_MEMORIZED in integrity_flags(row, spec) + assert row["train_distance"] == pytest.approx(0.0, abs=1e-12) + assert row["train_distance_ratio"] == pytest.approx(0.0, abs=1e-9) def test_the_evaluator_leaves_an_honest_model_unflagged(fake_problem: Any) -> None: """The same path, for a model that generates rather than retrieves.""" - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "novelty", "cond_sens")) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("mmd", "train_distance", "cond_sens")) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) rng = np.random.default_rng(7) @@ -435,9 +404,9 @@ def test_the_evaluator_leaves_an_honest_model_unflagged(fake_problem: Any) -> No ) row = evaluator.score(honest) - assert row["copy_rate"] == 0.0 + assert row["train_distance"] > 0.0 assert row["cond_sens"] > 0 - assert not [flag for flag in integrity_flags(row, spec) if flag != FLAG_UNVERIFIED] + assert not [flag for flag in integrity_flags(row) if flag != FLAG_UNVERIFIED] def test_the_evaluator_catches_a_conditional_model_that_ignores_conditions(fake_problem: Any) -> None: @@ -478,26 +447,26 @@ def _pure_noise(_conditions: Any, n: int) -> Any: assert row["cond_sens"] == pytest.approx(0.0, abs=1e-12) -def test_the_copy_corpus_is_drawn_once_for_a_whole_sweep(fake_problem: Any) -> None: - """Every model in a sweep is checked against the same corpus, fetched once. +def test_the_training_split_is_fetched_once_for_a_whole_sweep(fake_problem: Any) -> None: + """Every model in a sweep is checked against the same training designs, fetched once. - Both halves matter: a per-model corpus would be slow, and a corpus that - varied between models would make their `copy_rate` values incomparable. + Both halves matter: a per-model fetch would be slow, and a split that varied + between models would make their `train_distance` values incomparable. """ - spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("novelty",)) + spec = EvalSpec(problem_id="fake", n_samples=4, metrics=("train_distance",)) ref = np.stack([np.full(fake_problem.design_space.shape, v) for v in (0.4, 0.5, 0.6, 0.7)]) conditions = fake_problem.dataset["train"].select([0, 1, 2, 0]) evaluator = _evaluator(fake_problem, spec, ref, conditions) draws = 0 - original = evaluator._draw_copy_corpus + original = evaluator._load_train_designs def _counting_draw() -> Any: nonlocal draws draws += 1 return original() - evaluator._draw_copy_corpus = _counting_draw # type: ignore[method-assign] + evaluator._load_train_designs = _counting_draw # type: ignore[method-assign] for _ in range(3): evaluator.score(_generator(fake_problem, lambda _c, n: np.zeros((n, *fake_problem.design_space.shape)))) diff --git a/tests/test_metric_registry.py b/tests/test_metric_registry.py new file mode 100644 index 00000000..de6c4cc5 --- /dev/null +++ b/tests/test_metric_registry.py @@ -0,0 +1,62 @@ +"""The metric declaration contract: one decorator, one direction, one sentence.""" + +from __future__ import annotations + +import pytest + +from engiopt.evaluation.registry import MetricRegistry +from engiopt.evaluation.registry import METRICS +from engiopt.evaluation.registry import register_metric + + +@pytest.fixture +def registry() -> MetricRegistry: + """A registry of its own, so a test never publishes into the global one.""" + return MetricRegistry() + + +def test_a_custom_registry_is_not_bypassed_when_empty(registry: MetricRegistry) -> None: + """An empty registry is falsy, which `registry or METRICS` silently ignored.""" + before = len(METRICS) + + @register_metric("solo", family="diversity", cost="cheap", higher_is_better=True, registry=registry) + def solo(ctx: object) -> float: + """A metric registered into a registry of its own.""" + return 0.0 + + assert "solo" in registry + assert len(METRICS) == before, "registration leaked into the global registry" + + +def test_the_docstring_is_the_question(registry: MetricRegistry) -> None: + """No separate text to maintain: the first docstring line is what readers see.""" + + @register_metric("share", family="feasibility", cost="cheap", higher_is_better=False, registry=registry) + def share(ctx: object) -> float: + """What fraction of the designs are infeasible? + + Longer explanation lives below the first line. + """ + return 0.0 + + assert registry["share"].description == "What fraction of the designs are infeasible?" + assert registry.explain().loc["share", "direction"] == "lower is better" + + +def test_a_diagnostic_declares_no_direction(registry: MetricRegistry) -> None: + @register_metric("diag", family="conditions", cost="cheap", higher_is_better=None, registry=registry) + def diag(ctx: object) -> float: + """A metric read beside the others, never ranked on.""" + return 0.0 + + assert registry["diag"].direction == "diagnostic" + + +def test_two_metrics_cannot_claim_one_column(registry: MetricRegistry) -> None: + register_metric( + "first", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + )(lambda ctx: {"shared": 1.0}) + with pytest.raises(ValueError, match="already emitted"): + register_metric( + "second", family="diversity", cost="cheap", higher_is_better=True, outputs=("shared",), registry=registry + )(lambda ctx: {"shared": 2.0}) diff --git a/tests/test_metrics_suite.py b/tests/test_metrics_suite.py new file mode 100644 index 00000000..a68f88f4 --- /dev/null +++ b/tests/test_metrics_suite.py @@ -0,0 +1,273 @@ +"""The ported metrics: each one on the toy structure every EngiBench dataset shares.""" + +from __future__ import annotations + +from typing import Any + +import numpy as np +import pytest + +from engiopt.evaluation.board import Board +from engiopt.evaluation.board import register_space +from engiopt.evaluation.board import SPACES +from engiopt.evaluation.context import EvaluationContext +from engiopt.evaluation.context import OptimizationResults +from engiopt.evaluation.registry import METRICS +from engiopt.evaluation.spec import EvalSpec + +# Importing the metrics package registers the built-ins. +import engiopt.evaluation.metrics # noqa: F401 # isort: skip + + +def _ctx(problem: Any, gen: np.ndarray, ref: np.ndarray, **kwargs: Any) -> EvaluationContext: + return EvaluationContext(problem=problem, problem_id="fake", gen_designs=gen, ref_designs=ref, **kwargs) + + +@pytest.fixture +def sets(fake_problem: Any) -> tuple[np.ndarray, np.ndarray]: + rng = np.random.default_rng(0) + shape = fake_problem.design_space.shape + return rng.random((12, *shape)), rng.random((12, *shape)) + + +# ---------------------------------------------------------------- diversity + + +def test_vendi_counts_distinct_designs(fake_problem: Any, sets: Any) -> None: + gen, _ = sets + collapsed = np.repeat(gen[:1], 12, axis=0) + assert METRICS["vendi"].fn(_ctx(fake_problem, collapsed, gen)) == pytest.approx(1.0) + spread = METRICS["vendi"].fn(_ctx(fake_problem, gen, gen, sigma=0.05)) + assert spread == pytest.approx(12.0, rel=0.05), "mutually dissimilar designs count as n" + + +def test_dpp_is_on_a_unit_scale_and_prefers_spread(fake_problem: Any, sets: Any) -> None: + gen, _ = sets + collapsed = np.repeat(gen[:1], 12, axis=0) + np.random.default_rng(1).normal(0, 1e-3, gen.shape) + low = METRICS["dpp"].fn(_ctx(fake_problem, collapsed, gen, sigma=1.0)) + high = METRICS["dpp"].fn(_ctx(fake_problem, gen, gen, sigma=1.0)) + assert 0.0 <= low < high <= 1.0 + + +# ------------------------------------------------------------- distribution + + +def test_coverage_is_one_for_the_reference_itself_and_zero_far_away(fake_problem: Any, sets: Any) -> None: + _, ref = sets + assert METRICS["coverage"].fn(_ctx(fake_problem, ref.copy(), ref)) == pytest.approx(1.0) + assert METRICS["coverage"].fn(_ctx(fake_problem, ref + 50.0, ref)) == pytest.approx(0.0) + + +# --------------------------------------------------------------- conditions + + +def test_per_condition_distance_sees_what_mmd_cannot(fake_problem: Any, sets: Any) -> None: + """The correct designs in the wrong order: a perfect set, every one for the wrong conditions.""" + _, ref = sets + permuted = ref[np.random.default_rng(2).permutation(len(ref))] + assert METRICS["mmd"].fn(_ctx(fake_problem, permuted, ref)) == pytest.approx(0.0, abs=1e-12) + assert METRICS["per_condition_distance"].fn(_ctx(fake_problem, permuted, ref)) > 0.1 + assert METRICS["per_condition_distance"].fn(_ctx(fake_problem, ref.copy(), ref)) == pytest.approx(0.0) + + +def test_volume_error_reads_the_material_fraction_off_the_design(fake_problem: Any) -> None: + shape = fake_problem.design_space.shape + gen = np.stack([np.full(shape, 0.3), np.full(shape, 0.5)]) + ctx = _ctx(fake_problem, gen, gen, conditions={"volfrac": np.array([0.3, 0.4])}, volume_condition="volfrac") + assert METRICS["volume_error"].fn(ctx) == pytest.approx(0.05) + assert np.isnan(METRICS["volume_error"].fn(_ctx(fake_problem, gen, gen))) + + +# ------------------------------------------------------------- memorization + + +def test_train_distance_reads_zero_for_copies_and_one_for_data_like_designs(fake_problem: Any, sets: Any) -> None: + """The ratio's scale comes from the reference designs, so a real held-out design scores about one.""" + gen, ref = sets + train = np.random.default_rng(3).random((20, *fake_problem.design_space.shape)) + copies = METRICS["train_distance"].fn(_ctx(fake_problem, train[:12], ref, train_designs_fn=lambda: train)) + assert copies["train_distance"] == pytest.approx(0.0) + assert copies["train_distance_ratio"] == pytest.approx(0.0) + fresh = METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref, train_designs_fn=lambda: train)) + assert fresh["train_distance"] > 0.1 + assert fresh["train_distance_ratio"] == pytest.approx(1.0, abs=0.25), "as far from training as real unseen designs" + without_split = METRICS["train_distance"].fn(_ctx(fake_problem, gen, ref)) + assert np.isnan(without_split["train_distance"]) + assert np.isnan(without_split["train_distance_ratio"]) + + +def test_the_bandwidth_defaults_to_the_training_designs_median(fake_problem: Any, sets: Any) -> None: + """No number anywhere: the kernel scale comes from the training split, or the reference set without one.""" + from engiopt import metrics as metrics_mod + + gen, ref = sets + train = np.random.default_rng(5).random((30, *fake_problem.design_space.shape)) + with_train = _ctx(fake_problem, gen, ref, train_designs_fn=lambda: train) + assert with_train.kernel_sigma == pytest.approx(metrics_mod.compute_median_sigma(train.reshape(30, -1))) + without = _ctx(fake_problem, gen, ref) + assert without.kernel_sigma == pytest.approx(metrics_mod.compute_median_sigma(ref.reshape(len(ref), -1))) + assert _ctx(fake_problem, gen, ref, sigma=3.0).kernel_sigma == 3.0, "a pinned value wins" + + +def test_the_v2_specs_pin_no_bandwidth() -> None: + assert EvalSpec.load("beams2d/v2").sigma is None + + +# -------------------------------------------------------------- performance + + +@pytest.fixture +def trajectories(fake_problem: Any, sets: Any) -> EvaluationContext: + """Two re-optimization paths: one reaches the reference optimum, one never does.""" + gen, ref = sets + ctx = _ctx(fake_problem, gen[:2], ref[:2]) + reaches = np.array([5.0, 4.0, 3.0, 0.0, 0.0]) + never = np.array([5.0, 4.0, 3.0, 2.0, 1.0]) + # Reference objective 10 for both, so "near optimal" means a gap of 0.5 or less. + ctx.__dict__["optimization"] = OptimizationResults(trajectories=[reaches, never], reference_objectives=[10.0, 10.0]) + return ctx + + +def test_calls_to_near_optimum_is_set_by_the_optimum_not_the_start(trajectories: EvaluationContext) -> None: + # reaches: last outside the 0.5 band at call 3; never: still outside at its last call, so budget + 1 = 6. + assert METRICS["calls_to_near_optimum"].fn(trajectories) == pytest.approx((3 + 6) / 2) + + +def test_an_absurd_start_is_not_credited_with_settling(fake_problem: Any, sets: Any) -> None: + """A gap of 1e9 that drops to 1e3 in one call has not arrived; the old relative band said it had.""" + gen, ref = sets + ctx = _ctx(fake_problem, gen[:1], ref[:1]) + path = np.array([1e9, 1e3, 1e2, 10.0, 3.0]) + ctx.__dict__["optimization"] = OptimizationResults(trajectories=[path], reference_objectives=[10.0]) + assert METRICS["calls_to_near_optimum"].fn(ctx) == pytest.approx(6.0), "never within 0.5 of the optimum" + + +def test_gap_after_calls_carries_a_short_path_forward(trajectories: EvaluationContext) -> None: + gaps = METRICS["gap_after_calls"].fn(trajectories) + assert gaps["gap_after_1_calls"] == pytest.approx(5.0) + assert gaps["gap_after_2_calls"] == pytest.approx(4.0) + assert gaps["gap_after_5_calls"] == pytest.approx(0.5) + assert gaps["gap_after_10_calls"] == pytest.approx(0.5), "a five-call path is done at call ten" + + +def test_reaches_reference_rate_is_a_rate_not_a_censored_count(trajectories: EvaluationContext) -> None: + assert METRICS["reaches_reference_rate"].fn(trajectories) == pytest.approx(0.5) + + +def test_trajectory_metrics_respect_the_aggregation_policy(trajectories: EvaluationContext) -> None: + trajectories.aggregation = "median" + assert METRICS["calls_to_near_optimum"].fn(trajectories) == pytest.approx(4.5), "median of two equals their mean" + + +# --------------------------------------------------------------------- cost + + +def test_cost_metrics_are_nan_without_a_live_model(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + ctx = _ctx(fake_problem, gen, ref) + assert all(np.isnan(METRICS[name].fn(ctx)) for name in ("generation_seconds", "n_parameters", "train_minutes")) + timed = _ctx(fake_problem, gen, ref, sample_seconds=1.5, model_params=1000, train_minutes=30.0) + assert METRICS["generation_seconds"].fn(timed) == 1.5 + assert METRICS["n_parameters"].fn(timed) == 1000 + assert METRICS["train_minutes"].fn(timed) == 30.0 + + +# -------------------------------------------------------------------- board + + +def test_a_space_is_a_registered_projection(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + register_space("halves", lambda reference, width: lambda designs: designs.reshape(len(designs), -1)[:, ::2]) + frame = Board(fake_problem, reference=ref).evaluate({"m": gen}, space="halves") + assert "mmd@halves" in frame.columns + assert "viol@halves" not in frame.columns + del SPACES["halves"] + with pytest.raises(KeyError, match="Unknown space"): + Board(fake_problem, reference=ref).evaluate({"m": gen}, space="halves") + + +def test_the_reference_row_leaves_per_condition_metrics_blank(fake_problem: Any, sets: Any) -> None: + gen, ref = sets + frame = Board(fake_problem, reference=ref).evaluate({"m": gen}) + assert np.isnan(frame.loc["reference (split-half)", "per_condition_distance"]) + assert not np.isnan(frame.loc["m", "per_condition_distance"]) + + +def test_from_evaluator_scores_models_and_saved_designs_under_one_spec() -> None: + class StubContext: + def __init__(self, designs: Any) -> None: + self.gen_designs = designs + + class StubEvaluator: + problem = None + registry = METRICS + spec = type("Spec", (), {"sigma": 1.0, "volume_condition": "volfrac"})() + resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() + + def context_for(self, generator: Any) -> StubContext: + return StubContext(generator) + + def context_for_designs(self, designs: Any) -> StubContext: + return StubContext(designs) + + def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: + return {"mmd": float(np.mean(ctx.gen_designs)), "iog": 1.0 if include_expensive else float("nan")} + + board = Board.from_evaluator( + StubEvaluator(), {"a": np.full(3, 0.2), "b": np.full(3, 0.1)}, designs={"probe": np.full(3, 0.3)}, expensive=True + ) + assert list(board.frame.index) == ["a", "b", "probe"] + assert board.rank("mmd").index[0] == "b" + assert board.frame.loc["probe", "iog"] == 1.0 + assert set(board.designs) == {"a", "b", "probe"} + assert board.volume_condition == "volfrac", "re-scoring the kept designs must still see the volume budget" + + +def test_from_evaluator_keeps_the_evaluators_registry() -> None: + """A custom metric the evaluator scored must be readable and rankable on the returned board.""" + from engiopt.evaluation.registry import MetricRegistry + from engiopt.evaluation.registry import register_metric + + custom = MetricRegistry() + register_metric("mine", family="diversity", cost="cheap", higher_is_better=True, registry=custom)(lambda ctx: 0.0) + + class StubEvaluator: + problem = None + registry = custom + spec = type("Spec", (), {"sigma": None, "volume_condition": None})() + resolved = type("Resolved", (), {"ref_designs": np.zeros((2, 3)), "conditions": None})() + + def context_for(self, generator: Any) -> Any: + return type("Ctx", (), {"gen_designs": generator})() + + def score_context(self, ctx: Any, *, only: Any, include_expensive: bool) -> dict[str, float]: + return {"mine": float(np.mean(ctx.gen_designs))} + + board = Board.from_evaluator(StubEvaluator(), {"a": np.full(3, 0.2), "b": np.full(3, 0.4)}) + assert board.rank("mine").index[0] == "b" + assert "mine" in board.explain().index + + +def test_viol_is_blank_without_the_designs_conditions(fake_problem: Any, sets: Any) -> None: + """A constraint check that reads the conditions cannot run on bare designs; blank beats raising.""" + gen, ref = sets + assert np.isnan(METRICS["viol"].fn(_ctx(fake_problem, gen, ref))) + + +def test_the_default_spec_metrics_are_all_registered() -> None: + assert set(EvalSpec(problem_id="beams2d").metrics) <= set(METRICS) + + +# -------------------------------------------------------------------- specs + + +@pytest.mark.parametrize("problem_id", ["beams2d", "heatconduction2d", "photonics2d", "thermoelastic2d"]) +def test_every_v2_spec_names_only_registered_metrics(problem_id: str) -> None: + spec = EvalSpec.load(f"{problem_id}/v2") + assert spec.version == "v2" + assert spec.aggregation == "mean" + assert set(spec.metrics) <= set(METRICS) + + +def test_the_default_spec_is_v2() -> None: + assert EvalSpec.load("beams2d").version == "v2" diff --git a/tests/test_specs.py b/tests/test_specs.py index 4279d10f..b2bf4974 100644 --- a/tests/test_specs.py +++ b/tests/test_specs.py @@ -24,6 +24,14 @@ SPEC_PATHS = sorted(SPEC_ROOT.glob("*/*.json")) SPEC_IDS = [f"{path.parent.name}/{path.stem}" for path in SPEC_PATHS] +CURRENT_PATHS = [ + max(SPEC_ROOT.glob(f"{problem}/*.json"), key=lambda p: p.stem) + for problem in sorted({p.parent.name for p in SPEC_PATHS}) +] +CURRENT_IDS = [f"{path.parent.name}/{path.stem}" for path in CURRENT_PATHS] +"""The newest spec per problem. Older versions stay committed so published rows can +be read under the protocol that produced them, and may name metrics since retired.""" + class _FakeConditions: """A stand-in for the sampled-conditions dataset the digest hashes.""" @@ -50,9 +58,9 @@ def test_committed_spec_loads_and_round_trips(path: Any) -> None: assert dataclasses.asdict(spec) == dataclasses.asdict(EvalSpec(**json.loads(path.read_text()))) -@pytest.mark.parametrize("path", SPEC_PATHS, ids=SPEC_IDS) +@pytest.mark.parametrize("path", CURRENT_PATHS, ids=CURRENT_IDS) def test_committed_spec_requests_registered_metrics(path: Any) -> None: - """A spec asking for a metric nobody registered would fail at evaluation time.""" + """A current spec asking for a metric nobody registered would fail at evaluation time.""" for metric in EvalSpec.load(str(path)).metrics: assert metric in METRICS, f"{path.name} requests unregistered metric {metric!r}" @@ -237,17 +245,11 @@ def _capture(self: EvalSpec, problem: Any) -> EvalSpec: volume_condition="volfrac", volfrac_tol=0.05, required_seeds=(1, 2, 3, 4), - copy_tol=0.02, - max_copy_rate=0.25, - copy_corpus_size=64, ) assert captured["volume_condition"] == "volfrac" assert captured["volfrac_tol"] == 0.05 assert captured["required_seeds"] == (1, 2, 3, 4) - assert captured["copy_tol"] == 0.02 - assert captured["max_copy_rate"] == 0.25 - assert captured["copy_corpus_size"] == 64 def test_freeze_spec_defaults_track_the_dataclass() -> None: @@ -262,5 +264,5 @@ def test_freeze_spec_defaults_track_the_dataclass() -> None: from engiopt.evaluation import spec as spec_mod parameters = inspect.signature(spec_mod.freeze_spec).parameters - for field_name in ("metrics", "volfrac_tol", "required_seeds", "copy_tol", "max_copy_rate", "copy_corpus_size"): + for field_name in ("metrics", "volfrac_tol", "required_seeds"): assert parameters[field_name].default == getattr(EvalSpec, field_name), field_name diff --git a/tests/test_verification.py b/tests/test_verification.py index 8359e233..a289b512 100644 --- a/tests/test_verification.py +++ b/tests/test_verification.py @@ -17,7 +17,6 @@ from engiopt.evaluation import verify as verify_mod from engiopt.evaluation.spec import EvalSpec -from engiopt.evaluation.submission import FLAG_MEMORIZED from engiopt.evaluation.verify import verify_board from engiopt.evaluation.verify import verify_row @@ -38,7 +37,7 @@ class _FakeEvaluator: def __init__(self, scores: dict[str, Any] | None = None): self.spec = EvalSpec(problem_id="beams2d") - self.scores = scores or {"mmd": 0.04, "copy_rate": 0.0} + self.scores = scores or {"mmd": 0.04, "train_distance": 0.5} self.calls = 0 def score(self, generator: Any, *, include_expensive: bool = False) -> dict[str, Any]: @@ -139,7 +138,7 @@ def test_the_verified_row_carries_the_runners_numbers(loads_fake_generator: None fudging. Re-scoring sidesteps that: the runner's number is the number, and the submitted one becomes a claim reported as corroborated or not. """ - evaluator = _FakeEvaluator({"mmd": 0.42, "copy_rate": 0.0}) + evaluator = _FakeEvaluator({"mmd": 0.42, "train_distance": 0.5}) result = verify_row(_row(mmd=0.001), verifier="ideallab-ci", evaluator=evaluator) @@ -174,16 +173,6 @@ def test_a_reproduced_claim_says_so(loads_fake_generator: None) -> None: assert "corroborated" in result.detail -def test_verification_re_derives_the_flags_rather_than_trusting_them(loads_fake_generator: None) -> None: - """A submitter who cleared their own `memorized` flag must not keep it cleared.""" - evaluator = _FakeEvaluator({"mmd": 0.04, "copy_rate": 0.95}) - - result = verify_row(_row(flags=""), verifier="ci", evaluator=evaluator) - - assert result.row is not None - assert FLAG_MEMORIZED in result.row["flags"] - - # ---------------------------------------------------------------------- # Fetching the package the row actually names # ----------------------------------------------------------------------